Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3723bf5fe6 | ||
|
|
a348fc1ccc | ||
|
|
677680034a | ||
|
|
222126db1d | ||
|
|
418ceaf8c1 | ||
|
|
c7e23690e0 | ||
|
|
4d8bfa103b | ||
|
|
a68ac0c49a | ||
|
|
9d05f3721d | ||
|
|
16fc717091 | ||
|
|
925b0d1ab5 | ||
|
|
e65d74bc6f | ||
|
|
4858942c55 | ||
|
|
23cc009879 | ||
|
|
4d470b3dbb | ||
|
|
2483200963 | ||
|
|
1ac76a9472 | ||
|
|
9a59ad73dd | ||
|
|
6373aba80f | ||
|
|
fc956b9db6 | ||
|
|
98ae28657f | ||
|
|
4cf9d80fba | ||
|
|
20b1f4f3fb | ||
|
|
a6364bfa08 | ||
|
|
5b3fa26366 | ||
|
|
c6b0eb4de0 | ||
|
|
0441b7f19f | ||
|
|
7cd71de1f4 | ||
|
|
b1d6a57883 | ||
|
|
0b6b29a30c | ||
|
|
55cb4103be | ||
|
|
67233d1c2a | ||
|
|
0f75e9904a | ||
|
|
98c294126b | ||
|
|
0a8f3e21b8 | ||
|
|
2dbc3bd937 | ||
|
|
9d2ec8d35a | ||
|
|
423d24780b | ||
|
|
37d717054e | ||
|
|
1cb75b7971 | ||
|
|
5bfed0fe07 | ||
|
|
5a0e0d35b9 | ||
|
|
d2b34e89b0 | ||
|
|
6dde7d4657 | ||
|
|
c7513df4f9 | ||
|
|
946d3eaf95 | ||
|
|
bf45aa3a45 | ||
|
|
16e408f3f0 | ||
|
|
4108fe6014 | ||
|
|
b0fb2b8b05 | ||
|
|
1ddf7a1021 | ||
|
|
a70f7f3b7b | ||
|
|
6e3c393ef9 | ||
|
|
c7e5215b50 | ||
|
|
25686feebf | ||
|
|
8e3b320eb8 | ||
|
|
f5823277dc | ||
|
|
33924c074c | ||
|
|
21c64f90aa | ||
|
|
8e853e3ff8 | ||
|
|
76eab10b14 | ||
|
|
5c5a1fec4b | ||
|
|
7016fa4902 | ||
|
|
74cb03423e | ||
|
|
5988e21ed7 | ||
|
|
965226fd52 | ||
|
|
353a8c1c8f | ||
|
|
5747d9a2d8 | ||
|
|
e01b04de46 | ||
|
|
ef9232a2f7 | ||
|
|
5268027e6b | ||
|
|
b6598017c8 | ||
|
|
5af3a81490 | ||
|
|
7b7ab279f2 | ||
|
|
01669f2f12 | ||
|
|
338b5275be | ||
|
|
ef94562125 | ||
|
|
3616b813ec | ||
|
|
e1067dbbe5 | ||
|
|
ab37440ce6 | ||
|
|
cf3002664b | ||
|
|
fc8d5f203a | ||
|
|
c5806b9ad9 | ||
|
|
b145607029 | ||
|
|
4a3b755162 | ||
|
|
16dbcbe85d | ||
|
|
c04aaecb51 | ||
|
|
eaa069e322 | ||
|
|
2d7616121b | ||
|
|
1544813bfe | ||
|
|
2708c33c75 | ||
|
|
23a7458acf | ||
|
|
8afb7bc570 | ||
|
|
94765e48ff | ||
|
|
31916539af | ||
|
|
79d1b58afe | ||
|
|
99feb03607 | ||
|
|
b091b4eaeb | ||
|
|
d7dfeed6dc | ||
|
|
7e01a96e53 | ||
|
|
bb5cb32838 | ||
|
|
31e0adc681 | ||
|
|
8445995321 | ||
|
|
abba43eb63 | ||
|
|
4a0991c1d2 | ||
|
|
364b93a4b9 | ||
|
|
85546bb9e2 | ||
|
|
ba3fe7027c | ||
|
|
5999cd2848 | ||
|
|
639a9cb9a7 | ||
|
|
a09fa9df42 | ||
|
|
4e69fdb3be | ||
|
|
b3efafcc73 | ||
|
|
036e863e4a | ||
|
|
e3cdedbf0f | ||
|
|
38eb9bb19a | ||
|
|
0bb58b65ec | ||
|
|
fb30ff218d | ||
|
|
e1edbb0e89 | ||
|
|
72118b049f | ||
|
|
c007d08419 | ||
|
|
73b261b94f | ||
|
|
f4bb617f62 | ||
|
|
4c630d3e7b | ||
|
|
bc71c57ba9 | ||
|
|
ab5d422835 | ||
|
|
72ee55ed53 | ||
|
|
df4bdc9d58 | ||
|
|
eaad47a6f6 | ||
|
|
8c3060342f | ||
|
|
c9c6cfc0ee | ||
|
|
6438acec60 | ||
|
|
76a8bba15f | ||
|
|
ad220b9d93 | ||
|
|
ca791f4000 | ||
|
|
0dafcdd9e3 | ||
|
|
6e62489d9e | ||
|
|
076aebc7e6 | ||
|
|
b6dc49200d | ||
|
|
0a5b0780f5 | ||
|
|
c4348480f3 | ||
|
|
f76df0688c | ||
|
|
84cbf5c1f3 | ||
|
|
0f92a3cf63 | ||
|
|
60cbc4c68b | ||
|
|
ae11a636dc | ||
|
|
25567919ea | ||
|
|
dcd8ba2a0d | ||
|
|
f205dc2a3b | ||
|
|
c40d3172ac | ||
|
|
52533bea09 | ||
|
|
9eb36fd697 | ||
|
|
f8f4b3044a | ||
|
|
0dc257d610 | ||
|
|
20865a2653 | ||
|
|
da07e67efd | ||
|
|
b36001940a | ||
|
|
f4d944c49c | ||
|
|
e4652b99e2 | ||
|
|
6a73b09d15 | ||
|
|
af82979d43 | ||
|
|
3d87abcf1c | ||
|
|
4cb9aa6664 | ||
|
|
bc79644f16 | ||
|
|
e36b2d1519 | ||
|
|
fe15a9bb00 | ||
|
|
af577a4c5a | ||
|
|
9e81be7228 | ||
|
|
28a2f95631 | ||
|
|
07fcb3282c | ||
|
|
41a5bbf3e8 | ||
|
|
84b77f68e5 | ||
|
|
c29402d731 | ||
|
|
76e9271dce | ||
|
|
60c5a82c85 | ||
|
|
eb4821127c | ||
|
|
028bd89959 | ||
|
|
0f53d67ee4 | ||
|
|
aa5489e804 | ||
|
|
2a86f039ea | ||
|
|
79c6896153 | ||
|
|
46293f618c | ||
|
|
080440bd9c | ||
|
|
bd3c253420 | ||
|
|
82e13ed949 | ||
|
|
fb04e85a14 | ||
|
|
06762a0f5e | ||
|
|
92f35fab19 | ||
|
|
cd11ed7a04 | ||
|
|
4407fee49f | ||
|
|
48d0c70f61 | ||
|
|
4061e635d3 | ||
|
|
741a4c23ca | ||
|
|
c6e72a8454 | ||
|
|
180fe665cb | ||
|
|
6a24249e7c | ||
|
|
ee211d087b | ||
|
|
aec752faa3 | ||
|
|
c9540570ae | ||
|
|
6e3915fbc1 | ||
|
|
c3d2d87a74 | ||
|
|
f423aebb80 | ||
|
|
37b74f4df3 | ||
|
|
247604cdde | ||
|
|
2bb61a7d09 | ||
|
|
1ecec7a9bc | ||
|
|
325350d192 | ||
|
|
1be5bd92fa | ||
|
|
93793b6af5 | ||
|
|
26f6929eb4 | ||
|
|
808ef152e5 | ||
|
|
e7d7e0157f | ||
|
|
edc4164704 | ||
|
|
7412cd5c78 | ||
|
|
3f54152191 | ||
|
|
3fe7709b86 | ||
|
|
abce50e34d | ||
|
|
c704d384f4 | ||
|
|
4f2bb7e52f | ||
|
|
1bc376921f | ||
|
|
6cefb7c5b5 | ||
|
|
59fbc05031 | ||
|
|
4b81ded58b | ||
|
|
927c902785 | ||
|
|
d3943fe37d | ||
|
|
2bd9c9b881 |
+21
-11
@@ -1,12 +1,14 @@
|
||||
FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source
|
||||
# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x
|
||||
# which reached EOL in April 2026 — we copy node + npm + corepack from the
|
||||
# upstream node:22 image instead so we can stay on a supported LTS without
|
||||
# waiting for Debian 14 (forky, ~mid-2027). Bookworm-based slim image used
|
||||
# so the produced binary links against glibc 2.36, which runs cleanly on
|
||||
# our Debian 13 (trixie, glibc 2.41) runtime. Bumping to a new Node major
|
||||
# is a one-line ARG change; see #4977.
|
||||
FROM node:22-bookworm-slim@sha256:7af03b14a13c8cdd38e45058fd957bf00a72bbe17feac43b1c15a689c029c732 AS node_source
|
||||
# Node 26 source stage. Debian trixie's bundled nodejs is pinned to 20.x
|
||||
# (EOL April 2026), so we copy node + npm + corepack from the upstream node:26
|
||||
# image instead. Node 26 (Current; LTS promotion ~Oct 2026) is REQUIRED by the
|
||||
# native OpenTUI TUI engine, which loads its renderer via the experimental
|
||||
# `node:ffi` API that only exists on Node 26.3+ (the Ink engine + web build run
|
||||
# on it too). Bookworm-based slim image used so the produced binary links
|
||||
# against glibc 2.36, which runs cleanly on our Debian 13 (trixie, glibc 2.41)
|
||||
# runtime. The pinned tag ships v26.3.0. Bumping Node is a one-line change here.
|
||||
# NOTE: verify the full image build + Ink/web/Playwright on Node 26 in CI.
|
||||
FROM node:26-bookworm-slim@sha256:79723b41edbedf595f62e943a9f8b0ba9af5b1e61045c5f8f59c2c02c1212a16 AS node_source
|
||||
FROM debian:13.4
|
||||
|
||||
# Disable Python stdout buffering to ensure logs are printed immediately
|
||||
@@ -90,7 +92,7 @@ RUN useradd -u 10000 -m -d /opt/data hermes
|
||||
|
||||
COPY --chmod=0755 --from=uv_source /usr/local/bin/uv /usr/local/bin/uvx /usr/local/bin/
|
||||
|
||||
# Node 22 LTS: copy the node binary plus the bundled npm + corepack JS
|
||||
# Node 26: copy the node binary plus the bundled npm + corepack JS
|
||||
# installs from the upstream image. npm and npx are recreated as symlinks
|
||||
# because they're symlinks in the source image (and need to live on PATH).
|
||||
# See node_source stage at the top of the file for the version-bump
|
||||
@@ -119,7 +121,7 @@ COPY ui-tui/packages/hermes-ink/ ui-tui/packages/hermes-ink/
|
||||
|
||||
# `npm_config_install_links=false` forces npm to install `file:` deps as
|
||||
# symlinks instead of copies. This is the default since npm 10+, which is
|
||||
# what the image ships now (via the node:22 source stage). We set it
|
||||
# what the image ships now (via the node:26 source stage). We set it
|
||||
# explicitly anyway as defense-in-depth: the previous Debian-bundled npm
|
||||
# 9.x defaulted to install-as-copy, which produced a hidden
|
||||
# node_modules/.package-lock.json that permanently disagreed with the root
|
||||
@@ -181,8 +183,16 @@ RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra
|
||||
# invalidate the (relatively slow) web + ui-tui build layer.
|
||||
COPY web/ web/
|
||||
COPY ui-tui/ ui-tui/
|
||||
COPY ui-opentui/ ui-opentui/
|
||||
# ui-opentui is the opt-in native OpenTUI engine (HERMES_TUI_ENGINE=opentui;
|
||||
# default stays Ink). .dockerignore strips its node_modules/dist, so install +
|
||||
# esbuild-build it here -> dist/main.js, then prune devDeps (esbuild/babel/
|
||||
# vitest); the runtime only needs the prod deps (the external @opentui/core +
|
||||
# its native blob -- the bundle inlines solid/effect). Build needs Node 26.3
|
||||
# (node:ffi floor), which this image ships.
|
||||
RUN cd web && npm run build && \
|
||||
cd ../ui-tui && npm run build
|
||||
cd ../ui-tui && npm run build && \
|
||||
cd ../ui-opentui && npm install --no-audit --no-fund && npm run build && npm prune --omit=dev
|
||||
|
||||
# ---------- Source code ----------
|
||||
# .dockerignore excludes node_modules, so the installs above survive.
|
||||
|
||||
@@ -107,6 +107,8 @@ You can still bring your own keys per-tool whenever you want — the gateway is
|
||||
|
||||
Hermes has two entry points: start the terminal UI with `hermes`, or run the gateway and talk to it from Telegram, Discord, Slack, WhatsApp, Signal, or Email. Once you're in a conversation, many slash commands are shared across both interfaces.
|
||||
|
||||
> **TUI engine:** On supported hosts (Linux/macOS with Node 26.3+), the terminal UI defaults to the native **OpenTUI** engine, which the installer provisions for you. The legacy **Ink** engine remains the fallback — it's used automatically on Windows, Termux, or when the native engine can't run, and you can select it explicitly with `HERMES_TUI_ENGINE=ink hermes`. Ink is not going away; it's the kept fallback.
|
||||
|
||||
| Action | CLI | Messaging platforms |
|
||||
| ------------------------------ | --------------------------------------------- | -------------------------------------------------------------------------------- |
|
||||
| Start chatting | `hermes` | Run `hermes gateway setup` + `hermes gateway start`, then send the bot a message |
|
||||
|
||||
@@ -299,6 +299,7 @@ def init_agent(
|
||||
# would mangle the escape sequences. None = use builtins.print.
|
||||
agent._print_fn = None
|
||||
agent.background_review_callback = None # Optional sync callback for gateway delivery
|
||||
agent.memory_notifications = "on" # Memory update notifications: "off", "on", "verbose"
|
||||
agent.skip_context_files = skip_context_files
|
||||
agent.load_soul_identity = load_soul_identity
|
||||
agent.pass_session_id = pass_session_id
|
||||
|
||||
+159
-27
@@ -3079,23 +3079,20 @@ def _try_configured_fallback_chain(
|
||||
if not fb_provider or fb_provider.lower() == skip:
|
||||
continue
|
||||
fb_model = str(entry.get("model", "")).strip() or None
|
||||
fb_base_url = str(entry.get("base_url", "")).strip() or None
|
||||
fb_api_key = str(entry.get("api_key", "")).strip() or None
|
||||
|
||||
label = f"fallback_chain[{i}]({fb_provider})"
|
||||
|
||||
try:
|
||||
fb_client = _resolve_single_provider(
|
||||
fb_provider, fb_model, fb_base_url, fb_api_key)
|
||||
fb_client, resolved_model = _resolve_fallback_entry(entry)
|
||||
except Exception:
|
||||
fb_client = None
|
||||
fb_client, resolved_model = None, None
|
||||
|
||||
if fb_client is not None:
|
||||
logger.info(
|
||||
"Auxiliary %s: %s on %s — configured fallback to %s (%s)",
|
||||
task, reason, failed_provider, label, fb_model or "default",
|
||||
task, reason, failed_provider, label, resolved_model or fb_model or "default",
|
||||
)
|
||||
return fb_client, fb_model, label
|
||||
return fb_client, resolved_model or fb_model, label
|
||||
tried.append(label)
|
||||
|
||||
if tried:
|
||||
@@ -3106,6 +3103,103 @@ def _try_configured_fallback_chain(
|
||||
return None, None, ""
|
||||
|
||||
|
||||
def _fallback_entry_api_key(entry: Dict[str, Any]) -> Optional[str]:
|
||||
"""Resolve inline or env-backed API key from a fallback-chain entry."""
|
||||
explicit = str(entry.get("api_key") or "").strip()
|
||||
if explicit:
|
||||
return explicit
|
||||
key_env = str(entry.get("key_env") or entry.get("api_key_env") or "").strip()
|
||||
if key_env:
|
||||
return os.getenv(key_env, "").strip() or None
|
||||
return None
|
||||
|
||||
|
||||
def _resolve_fallback_entry(entry: Dict[str, Any]) -> Tuple[Optional[Any], Optional[str]]:
|
||||
"""Resolve one fallback entry through the central provider router."""
|
||||
provider = str(entry.get("provider") or "").strip()
|
||||
model = str(entry.get("model") or "").strip() or None
|
||||
if not provider or not model:
|
||||
return None, None
|
||||
base_url = str(entry.get("base_url") or "").strip() or None
|
||||
api_key = _fallback_entry_api_key(entry)
|
||||
api_mode = str(entry.get("api_mode") or entry.get("transport") or "").strip() or None
|
||||
return resolve_provider_client(
|
||||
provider,
|
||||
model=model,
|
||||
explicit_base_url=base_url,
|
||||
explicit_api_key=api_key,
|
||||
api_mode=api_mode,
|
||||
)
|
||||
|
||||
|
||||
def _try_main_fallback_chain(
|
||||
task: Optional[str],
|
||||
failed_provider: str = "",
|
||||
reason: str = "error",
|
||||
) -> Tuple[Optional[Any], Optional[str], str]:
|
||||
"""Try the top-level main-agent fallback chain for an auxiliary call.
|
||||
|
||||
``provider: auto`` auxiliary tasks should respect the user's declared
|
||||
main fallback policy before dropping into Hermes' built-in discovery
|
||||
chain. The top-level chain is read through ``get_fallback_chain`` so
|
||||
both modern ``fallback_providers`` and legacy ``fallback_model`` entries
|
||||
participate in the same order as the main agent.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.fallback_config import get_fallback_chain
|
||||
|
||||
chain = get_fallback_chain(load_config())
|
||||
except Exception as exc:
|
||||
logger.debug("Auxiliary %s: could not load main fallback chain: %s", task or "call", exc)
|
||||
return None, None, ""
|
||||
|
||||
if not chain:
|
||||
return None, None, ""
|
||||
|
||||
failed_norm = (failed_provider or "").strip().lower()
|
||||
main_norm = (_read_main_provider() or "").strip().lower()
|
||||
skip = {p for p in (failed_norm, main_norm, "auto") if p}
|
||||
tried: List[str] = []
|
||||
|
||||
for i, entry in enumerate(chain):
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
fb_provider = str(entry.get("provider") or "").strip()
|
||||
fb_model = str(entry.get("model") or "").strip()
|
||||
if not fb_provider or not fb_model:
|
||||
continue
|
||||
fb_norm = fb_provider.lower()
|
||||
label = f"fallback_providers[{i}]({fb_provider})"
|
||||
if fb_norm in skip:
|
||||
tried.append(f"{label} (skipped)")
|
||||
continue
|
||||
if _is_provider_unhealthy(fb_norm):
|
||||
_log_skip_unhealthy(fb_norm, task)
|
||||
tried.append(f"{label} (unhealthy)")
|
||||
continue
|
||||
try:
|
||||
fb_client, resolved_model = _resolve_fallback_entry(entry)
|
||||
except Exception as exc:
|
||||
logger.debug("Auxiliary %s: main fallback %s failed to resolve: %s", task or "call", label, exc)
|
||||
fb_client, resolved_model = None, None
|
||||
if fb_client is not None:
|
||||
logger.info(
|
||||
"Auxiliary %s: %s on %s — main fallback chain to %s (%s)",
|
||||
task or "call", reason, failed_provider or "auto", label,
|
||||
resolved_model or fb_model,
|
||||
)
|
||||
return fb_client, resolved_model or fb_model, fb_provider
|
||||
tried.append(label)
|
||||
|
||||
if tried:
|
||||
logger.debug(
|
||||
"Auxiliary %s: main fallback chain exhausted (tried: %s)",
|
||||
task or "call", ", ".join(tried),
|
||||
)
|
||||
return None, None, ""
|
||||
|
||||
|
||||
def _resolve_single_provider(
|
||||
provider: str,
|
||||
model: Optional[str] = None,
|
||||
@@ -3116,16 +3210,19 @@ def _resolve_single_provider(
|
||||
|
||||
Uses the existing provider resolution infrastructure where possible.
|
||||
"""
|
||||
# Reuse resolve_provider_client which handles provider→client mapping
|
||||
# Reuse resolve_provider_client which handles provider→client mapping.
|
||||
client, resolved_model = resolve_provider_client(
|
||||
provider=provider,
|
||||
model=model,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
explicit_base_url=base_url,
|
||||
explicit_api_key=api_key,
|
||||
)
|
||||
return client
|
||||
|
||||
def _resolve_auto(main_runtime: Optional[Dict[str, Any]] = None) -> Tuple[Optional[OpenAI], Optional[str]]:
|
||||
def _resolve_auto(
|
||||
main_runtime: Optional[Dict[str, Any]] = None,
|
||||
task: Optional[str] = None,
|
||||
) -> Tuple[Optional[OpenAI], Optional[str]]:
|
||||
"""Full auto-detection chain.
|
||||
|
||||
Priority:
|
||||
@@ -3223,7 +3320,22 @@ def _resolve_auto(main_runtime: Optional[Dict[str, Any]] = None) -> Tuple[Option
|
||||
main_provider, resolved or main_model)
|
||||
return client, resolved or main_model
|
||||
|
||||
# ── Step 2: aggregator / fallback chain ──────────────────────────────
|
||||
# ── Step 2: user-configured fallback policy ─────────────────────────
|
||||
# In auto mode, respect the task-specific fallback chain first, then the
|
||||
# main agent's top-level fallback_providers/fallback_model chain. The
|
||||
# hardcoded provider discovery chain below is only the convenience default
|
||||
# for users who have not declared a fallback policy.
|
||||
if task:
|
||||
fb_client, fb_model, _fb_label = _try_configured_fallback_chain(
|
||||
task, main_provider or "auto", reason="main provider unavailable")
|
||||
if fb_client is not None:
|
||||
return fb_client, fb_model
|
||||
fb_client, fb_model, _fb_label = _try_main_fallback_chain(
|
||||
task, main_provider or "auto", reason="main provider unavailable")
|
||||
if fb_client is not None:
|
||||
return fb_client, fb_model
|
||||
|
||||
# ── Step 3: aggregator / fallback chain ──────────────────────────────
|
||||
tried = []
|
||||
for label, try_fn in _get_provider_chain():
|
||||
if _is_provider_unhealthy(label):
|
||||
@@ -3344,6 +3456,7 @@ def resolve_provider_client(
|
||||
api_mode: str = None,
|
||||
main_runtime: Optional[Dict[str, Any]] = None,
|
||||
is_vision: bool = False,
|
||||
task: Optional[str] = None,
|
||||
) -> Tuple[Optional[Any], Optional[str]]:
|
||||
"""Central router: given a provider name and optional model, return a
|
||||
configured client with the correct auth, base URL, and API format.
|
||||
@@ -3464,7 +3577,7 @@ def resolve_provider_client(
|
||||
|
||||
# ── Auto: try all providers in priority order ────────────────────
|
||||
if provider == "auto":
|
||||
client, resolved = _resolve_auto(main_runtime=main_runtime)
|
||||
client, resolved = _resolve_auto(main_runtime=main_runtime, task=task)
|
||||
if client is None:
|
||||
return None, None
|
||||
# When auto-detection lands on a non-OpenRouter provider (e.g. a
|
||||
@@ -4357,11 +4470,16 @@ def _client_cache_key(
|
||||
api_mode: Optional[str] = None,
|
||||
main_runtime: Optional[Dict[str, Any]] = None,
|
||||
is_vision: bool = False,
|
||||
task: Optional[str] = None,
|
||||
) -> tuple:
|
||||
runtime = _normalize_main_runtime(main_runtime)
|
||||
runtime_key = tuple(runtime.get(field, "") for field in _MAIN_RUNTIME_FIELDS) if provider == "auto" else ()
|
||||
# `auto` can now resolve through task-specific or main fallback policy,
|
||||
# so the task participates in the cache key. Non-auto providers keep the
|
||||
# old cache shape because the explicit provider/model tuple is sufficient.
|
||||
task_key = (task or "") if provider == "auto" else ""
|
||||
pool_hint = _pool_cache_hint(provider, main_runtime=main_runtime)
|
||||
return (provider, async_mode, base_url or "", api_key or "", api_mode or "", runtime_key, is_vision, pool_hint)
|
||||
return (provider, async_mode, base_url or "", api_key or "", api_mode or "", runtime_key, is_vision, task_key, pool_hint)
|
||||
|
||||
|
||||
def _store_cached_client(cache_key: tuple, client: Any, default_model: Optional[str], *, bound_loop: Any = None) -> None:
|
||||
@@ -4554,6 +4672,7 @@ def _get_cached_client(
|
||||
api_mode: str = None,
|
||||
main_runtime: Optional[Dict[str, Any]] = None,
|
||||
is_vision: bool = False,
|
||||
task: Optional[str] = None,
|
||||
) -> Tuple[Optional[Any], Optional[str]]:
|
||||
"""Get or create a cached client for the given provider.
|
||||
|
||||
@@ -4591,6 +4710,7 @@ def _get_cached_client(
|
||||
api_mode=api_mode,
|
||||
main_runtime=main_runtime,
|
||||
is_vision=is_vision,
|
||||
task=task,
|
||||
)
|
||||
with _client_cache_lock:
|
||||
if cache_key in _client_cache:
|
||||
@@ -4635,6 +4755,7 @@ def _get_cached_client(
|
||||
api_mode=api_mode,
|
||||
main_runtime=runtime,
|
||||
is_vision=is_vision,
|
||||
task=task,
|
||||
)
|
||||
if client is not None:
|
||||
# For async clients, remember which loop they were created on so we
|
||||
@@ -5140,7 +5261,7 @@ def call_llm(
|
||||
if not resolved_base_url:
|
||||
logger.info("Auxiliary %s: provider %s unavailable, trying auto-detection chain",
|
||||
task or "call", resolved_provider)
|
||||
client, final_model = _get_cached_client("auto", main_runtime=main_runtime)
|
||||
client, final_model = _get_cached_client("auto", main_runtime=main_runtime, task=task)
|
||||
if client is None:
|
||||
raise RuntimeError(
|
||||
f"No LLM provider configured for task={task} provider={resolved_provider}. "
|
||||
@@ -5466,14 +5587,19 @@ def call_llm(
|
||||
|
||||
# Fallback order (#26882, #26803):
|
||||
# 1. User-configured fallback_chain (per-task) if set
|
||||
# 2. Main agent model (last-resort safety net)
|
||||
# For auto users (no explicit aux provider), use the full
|
||||
# auto-detection chain instead — its Step 1 IS the main agent
|
||||
# model, so users on `auto` already get main-model fallback.
|
||||
# 2. For auto: top-level main fallback_providers/fallback_model
|
||||
# 3. For auto: built-in auxiliary discovery chain
|
||||
# 4. For explicit aux providers: main agent model safety net
|
||||
fb_client, fb_model, fb_label = (None, None, "")
|
||||
if is_auto:
|
||||
fb_client, fb_model, fb_label = _try_payment_fallback(
|
||||
resolved_provider, task, reason=reason)
|
||||
fb_client, fb_model, fb_label = _try_configured_fallback_chain(
|
||||
task, resolved_provider or "auto", reason=reason)
|
||||
if fb_client is None:
|
||||
fb_client, fb_model, fb_label = _try_main_fallback_chain(
|
||||
task, resolved_provider or "auto", reason=reason)
|
||||
if fb_client is None:
|
||||
fb_client, fb_model, fb_label = _try_payment_fallback(
|
||||
resolved_provider, task, reason=reason)
|
||||
else:
|
||||
fb_client, fb_model, fb_label = _try_configured_fallback_chain(
|
||||
task, resolved_provider or "auto", reason=reason)
|
||||
@@ -5636,7 +5762,7 @@ async def async_call_llm(
|
||||
if not resolved_base_url:
|
||||
logger.info("Auxiliary %s: provider %s unavailable, trying auto-detection chain",
|
||||
task or "call", resolved_provider)
|
||||
client, final_model = _get_cached_client("auto", async_mode=True)
|
||||
client, final_model = _get_cached_client("auto", async_mode=True, main_runtime=main_runtime, task=task)
|
||||
if client is None:
|
||||
raise RuntimeError(
|
||||
f"No LLM provider configured for task={task} provider={resolved_provider}. "
|
||||
@@ -5904,13 +6030,19 @@ async def async_call_llm(
|
||||
|
||||
# Fallback order (#26882, #26803):
|
||||
# 1. User-configured fallback_chain (per-task) if set
|
||||
# 2. Main agent model (last-resort safety net)
|
||||
# Auto users get the full auto-detection chain instead — its
|
||||
# Step 1 IS the main agent model.
|
||||
# 2. For auto: top-level main fallback_providers/fallback_model
|
||||
# 3. For auto: built-in auxiliary discovery chain
|
||||
# 4. For explicit aux providers: main agent model safety net
|
||||
fb_client, fb_model, fb_label = (None, None, "")
|
||||
if is_auto:
|
||||
fb_client, fb_model, fb_label = _try_payment_fallback(
|
||||
resolved_provider, task, reason=reason)
|
||||
fb_client, fb_model, fb_label = _try_configured_fallback_chain(
|
||||
task, resolved_provider or "auto", reason=reason)
|
||||
if fb_client is None:
|
||||
fb_client, fb_model, fb_label = _try_main_fallback_chain(
|
||||
task, resolved_provider or "auto", reason=reason)
|
||||
if fb_client is None:
|
||||
fb_client, fb_model, fb_label = _try_payment_fallback(
|
||||
resolved_provider, task, reason=reason)
|
||||
else:
|
||||
fb_client, fb_model, fb_label = _try_configured_fallback_chain(
|
||||
task, resolved_provider or "auto", reason=reason)
|
||||
|
||||
+121
-19
@@ -237,18 +237,25 @@ _COMBINED_REVIEW_PROMPT = (
|
||||
def summarize_background_review_actions(
|
||||
review_messages: List[Dict],
|
||||
prior_snapshot: List[Dict],
|
||||
notification_mode: str = "on",
|
||||
) -> List[str]:
|
||||
"""Build the human-facing action summary for a background review pass.
|
||||
|
||||
Walks the review agent's session messages and collects "successful tool
|
||||
action" descriptions to surface to the user (e.g. "Memory updated").
|
||||
Tool messages already present in ``prior_snapshot`` are skipped so we
|
||||
don't re-surface stale results from the prior conversation that the
|
||||
review agent inherited via ``conversation_history`` (issue #14944).
|
||||
Walks the review agent's session messages and collects successful memory
|
||||
and skill-management actions to surface to the user. Tool messages already
|
||||
present in ``prior_snapshot`` are skipped so stale inherited results are
|
||||
not re-surfaced as fresh background work (issue #14944).
|
||||
|
||||
Matching is by ``tool_call_id`` when available, with a content-equality
|
||||
fallback for tool messages that lack one.
|
||||
``notification_mode`` controls display detail:
|
||||
- ``off``: return no actions.
|
||||
- ``on``: generic "Memory updated"/tool messages.
|
||||
- ``verbose``: include compact content previews from tool-call arguments.
|
||||
"""
|
||||
mode = str(notification_mode or "on").lower()
|
||||
if mode == "off":
|
||||
return []
|
||||
verbose = mode == "verbose"
|
||||
|
||||
existing_tool_call_ids = set()
|
||||
existing_tool_contents = set()
|
||||
for prior in prior_snapshot or []:
|
||||
@@ -262,6 +269,42 @@ def summarize_background_review_actions(
|
||||
if isinstance(content, str):
|
||||
existing_tool_contents.add(content)
|
||||
|
||||
# Map review-agent tool results back to the calls that produced them. The
|
||||
# result JSON only says "Entry added"; the call arguments contain action,
|
||||
# target, and content previews. Restricting to notify_tools also prevents
|
||||
# helper tools from surfacing as memory work just because they succeeded.
|
||||
notify_tools = {"memory", "skill_manage"}
|
||||
all_tool_call_ids: set = set()
|
||||
call_details: dict = {}
|
||||
for msg in review_messages or []:
|
||||
if not isinstance(msg, dict) or msg.get("role") != "assistant":
|
||||
continue
|
||||
for tc in msg.get("tool_calls", []) or []:
|
||||
if not isinstance(tc, dict):
|
||||
continue
|
||||
fn = tc.get("function", {}) or {}
|
||||
fn_name = fn.get("name", "")
|
||||
tcid = tc.get("id")
|
||||
if tcid:
|
||||
all_tool_call_ids.add(tcid)
|
||||
if fn_name not in notify_tools:
|
||||
continue
|
||||
try:
|
||||
args = json.loads(fn.get("arguments", "{}"))
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
args = {}
|
||||
if tcid:
|
||||
call_details[tcid] = {
|
||||
"tool": fn_name,
|
||||
"action": args.get("action", "?"),
|
||||
"target": args.get("target", "memory"),
|
||||
"content": args.get("content", ""),
|
||||
"old_text": args.get("old_text", ""),
|
||||
"name": args.get("name", ""),
|
||||
"old_string": args.get("old_string", ""),
|
||||
"new_string": args.get("new_string", ""),
|
||||
}
|
||||
|
||||
actions: List[str] = []
|
||||
for msg in review_messages or []:
|
||||
if not isinstance(msg, dict) or msg.get("role") != "tool":
|
||||
@@ -273,6 +316,8 @@ def summarize_background_review_actions(
|
||||
content_str = msg.get("content")
|
||||
if isinstance(content_str, str) and content_str in existing_tool_contents:
|
||||
continue
|
||||
if tcid and all_tool_call_ids and tcid not in call_details:
|
||||
continue
|
||||
try:
|
||||
data = json.loads(msg.get("content", "{}"))
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
@@ -280,19 +325,75 @@ def summarize_background_review_actions(
|
||||
if not isinstance(data, dict) or not data.get("success"):
|
||||
continue
|
||||
message = data.get("message", "")
|
||||
target = data.get("target", "")
|
||||
if "created" in message.lower():
|
||||
actions.append(message)
|
||||
elif "updated" in message.lower():
|
||||
actions.append(message)
|
||||
elif "added" in message.lower() or (target and "add" in message.lower()):
|
||||
label = "Memory" if target == "memory" else "User profile" if target == "user" else target
|
||||
actions.append(f"{label} updated")
|
||||
elif "Entry added" in message:
|
||||
label = "Memory" if target == "memory" else "User profile" if target == "user" else target
|
||||
actions.append(f"{label} updated")
|
||||
elif "removed" in message.lower() or "replaced" in message.lower():
|
||||
detail = call_details.get(tcid, {})
|
||||
target = data.get("target", "") or detail.get("target", "")
|
||||
is_skill = detail.get("tool") == "skill_manage"
|
||||
|
||||
message_lower = message.lower()
|
||||
if not verbose:
|
||||
if "created" in message_lower:
|
||||
actions.append(message)
|
||||
continue
|
||||
if "updated" in message_lower:
|
||||
actions.append(message)
|
||||
continue
|
||||
if is_skill and "patched" in message_lower:
|
||||
actions.append(message)
|
||||
continue
|
||||
|
||||
if is_skill:
|
||||
label = "Skill"
|
||||
elif target:
|
||||
label = "Memory" if target == "memory" else "User profile" if target == "user" else target
|
||||
else:
|
||||
continue
|
||||
|
||||
if verbose:
|
||||
action = detail.get("action", "")
|
||||
content = detail.get("content", "")
|
||||
old_text = detail.get("old_text", "")
|
||||
skill_name = detail.get("name", "")
|
||||
max_preview = 120
|
||||
if is_skill:
|
||||
change = data.get("_change", {})
|
||||
old_string = change.get("old", "") or detail.get("old_string", "")
|
||||
new_string = change.get("new", "") or detail.get("new_string", "")
|
||||
description = change.get("description", "")
|
||||
if action == "patch" and (old_string or new_string):
|
||||
old_preview = old_string[:80].replace("\n", " ") + (
|
||||
"…" if len(old_string) > 80 else ""
|
||||
)
|
||||
new_preview = new_string[:80].replace("\n", " ") + (
|
||||
"…" if len(new_string) > 80 else ""
|
||||
)
|
||||
actions.append(
|
||||
f"📝 Skill '{skill_name}' patched: "
|
||||
f"\"{old_preview}\" → \"{new_preview}\""
|
||||
)
|
||||
elif action == "create" and description:
|
||||
actions.append(f"📝 Skill '{skill_name}' created: {description}")
|
||||
elif action == "edit" and description:
|
||||
actions.append(f"📝 Skill '{skill_name}' rewritten: {description}")
|
||||
else:
|
||||
actions.append(f"📝 {message}" if message else f"Skill {action}")
|
||||
elif action == "add" and content:
|
||||
preview = content[:max_preview] + ("…" if len(content) > max_preview else "")
|
||||
actions.append(f"{label} ➕ {preview}")
|
||||
elif action == "replace" and content:
|
||||
preview = content[:max_preview] + ("…" if len(content) > max_preview else "")
|
||||
actions.append(f"{label} ✏️ {preview}")
|
||||
elif action == "remove" and old_text:
|
||||
preview = old_text[:60] + ("…" if len(old_text) > 60 else "")
|
||||
actions.append(f"{label} ➖ {preview}")
|
||||
else:
|
||||
actions.append(f"{label} updated")
|
||||
elif (
|
||||
"added" in message_lower
|
||||
or "replaced" in message_lower
|
||||
or "removed" in message_lower
|
||||
or (target and "add" in message.lower())
|
||||
or "Entry added" in message
|
||||
):
|
||||
actions.append(f"{label} updated")
|
||||
return actions
|
||||
|
||||
@@ -522,6 +623,7 @@ def _run_review_in_thread(
|
||||
actions = summarize_background_review_actions(
|
||||
review_messages,
|
||||
messages_snapshot,
|
||||
notification_mode=getattr(agent, "memory_notifications", "on"),
|
||||
)
|
||||
|
||||
if actions:
|
||||
|
||||
@@ -166,6 +166,39 @@ function profileRemoteOverride(config, profile) {
|
||||
return { url, authMode: normAuthMode(entry.authMode), token: entry.token }
|
||||
}
|
||||
|
||||
/**
|
||||
* In global-remote mode one backend serves every Desktop profile, so REST calls
|
||||
* that are scoped by renderer-side `request.profile` must carry that scope as a
|
||||
* query parameter. Local pooled backends and per-profile remote overrides do not
|
||||
* need this: they already run against a backend scoped to the target profile.
|
||||
*/
|
||||
function pathWithGlobalRemoteProfile(path, profile, opts = {}) {
|
||||
const scopedProfile = connectionScopeKey(profile)
|
||||
if (!scopedProfile || !opts.globalRemote || opts.profileRemoteOverride) {
|
||||
return path
|
||||
}
|
||||
|
||||
const rawPath = String(path || '')
|
||||
if (!rawPath) {
|
||||
return path
|
||||
}
|
||||
|
||||
let parsed
|
||||
try {
|
||||
parsed = new URL(rawPath, 'http://hermes.local')
|
||||
} catch {
|
||||
return path
|
||||
}
|
||||
|
||||
if (parsed.searchParams.has('profile')) {
|
||||
return path
|
||||
}
|
||||
|
||||
parsed.searchParams.set('profile', scopedProfile)
|
||||
|
||||
return `${parsed.pathname}${parsed.search}${parsed.hash}`
|
||||
}
|
||||
|
||||
function tokenPreview(value) {
|
||||
const raw = String(value || '')
|
||||
|
||||
@@ -247,6 +280,7 @@ module.exports = {
|
||||
cookiesHaveLiveSession,
|
||||
normAuthMode,
|
||||
normalizeRemoteBaseUrl,
|
||||
pathWithGlobalRemoteProfile,
|
||||
profileRemoteOverride,
|
||||
resolveAuthMode,
|
||||
resolveTestWsUrl,
|
||||
|
||||
@@ -24,6 +24,7 @@ const {
|
||||
cookiesHaveLiveSession,
|
||||
normAuthMode,
|
||||
normalizeRemoteBaseUrl,
|
||||
pathWithGlobalRemoteProfile,
|
||||
profileRemoteOverride,
|
||||
resolveAuthMode,
|
||||
resolveTestWsUrl,
|
||||
@@ -90,6 +91,72 @@ test('profileRemoteOverride tolerates a missing/!object profiles map', () => {
|
||||
assert.equal(profileRemoteOverride(null, 'coder'), null)
|
||||
})
|
||||
|
||||
// --- pathWithGlobalRemoteProfile ---
|
||||
|
||||
test('pathWithGlobalRemoteProfile appends profile in global remote mode', () => {
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('/api/model/info', 'iris', {
|
||||
globalRemote: true,
|
||||
profileRemoteOverride: false
|
||||
}),
|
||||
'/api/model/info?profile=iris'
|
||||
)
|
||||
})
|
||||
|
||||
test('pathWithGlobalRemoteProfile preserves existing query params', () => {
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('/api/model/options?force=1', 'iris', {
|
||||
globalRemote: true,
|
||||
profileRemoteOverride: false
|
||||
}),
|
||||
'/api/model/options?force=1&profile=iris'
|
||||
)
|
||||
})
|
||||
|
||||
test('pathWithGlobalRemoteProfile does not replace an explicit profile query', () => {
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('/api/model/info?profile=default', 'iris', {
|
||||
globalRemote: true,
|
||||
profileRemoteOverride: false
|
||||
}),
|
||||
'/api/model/info?profile=default'
|
||||
)
|
||||
})
|
||||
|
||||
test('pathWithGlobalRemoteProfile skips local and per-profile remote override paths', () => {
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('/api/model/info', 'iris', {
|
||||
globalRemote: false,
|
||||
profileRemoteOverride: false
|
||||
}),
|
||||
'/api/model/info'
|
||||
)
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('/api/model/info', 'iris', {
|
||||
globalRemote: true,
|
||||
profileRemoteOverride: true
|
||||
}),
|
||||
'/api/model/info'
|
||||
)
|
||||
})
|
||||
|
||||
test('pathWithGlobalRemoteProfile skips empty profile/path safely', () => {
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('/api/model/info', '', {
|
||||
globalRemote: true,
|
||||
profileRemoteOverride: false
|
||||
}),
|
||||
'/api/model/info'
|
||||
)
|
||||
assert.equal(
|
||||
pathWithGlobalRemoteProfile('', 'iris', {
|
||||
globalRemote: true,
|
||||
profileRemoteOverride: false
|
||||
}),
|
||||
''
|
||||
)
|
||||
})
|
||||
|
||||
// --- normalizeRemoteBaseUrl ---
|
||||
|
||||
test('normalizeRemoteBaseUrl strips trailing slashes, hash, and query', () => {
|
||||
|
||||
@@ -63,6 +63,7 @@ const {
|
||||
cookiesHaveLiveSession,
|
||||
normAuthMode,
|
||||
normalizeRemoteBaseUrl,
|
||||
pathWithGlobalRemoteProfile,
|
||||
profileRemoteOverride,
|
||||
resolveAuthMode,
|
||||
resolveTestWsUrl,
|
||||
@@ -5083,65 +5084,75 @@ function focusWindow(win) {
|
||||
win.focus()
|
||||
}
|
||||
|
||||
function spawnSecondaryWindow({ sessionId, watch, newSession } = {}) {
|
||||
const icon = getAppIconPath()
|
||||
const win = new BrowserWindow({
|
||||
width: SESSION_WINDOW_MIN_WIDTH,
|
||||
height: SESSION_WINDOW_MIN_HEIGHT,
|
||||
minWidth: SESSION_WINDOW_MIN_WIDTH,
|
||||
minHeight: SESSION_WINDOW_MIN_HEIGHT,
|
||||
title: 'Hermes',
|
||||
titleBarStyle: 'hidden',
|
||||
titleBarOverlay: getTitleBarOverlayOptions(),
|
||||
trafficLightPosition: IS_MAC ? WINDOW_BUTTON_POSITION : undefined,
|
||||
vibrancy: IS_MAC ? 'sidebar' : undefined,
|
||||
opacity: windowOpacity(),
|
||||
icon,
|
||||
// Don't show until the renderer's first themed paint is ready. macOS
|
||||
// `vibrancy` ignores `backgroundColor` and paints a translucent OS
|
||||
// material (which follows the OS appearance, not the app theme), so a
|
||||
// dark-themed app on a light-mode Mac flashes white until the renderer
|
||||
// covers it. ready-to-show fires after the boot-time paint in
|
||||
// themes/context.tsx, so the window appears already themed.
|
||||
show: false,
|
||||
backgroundColor: getWindowBackgroundColor(),
|
||||
webPreferences: {
|
||||
preload: path.join(__dirname, 'preload.cjs'),
|
||||
contextIsolation: true,
|
||||
webviewTag: true,
|
||||
sandbox: true,
|
||||
nodeIntegration: false,
|
||||
devTools: true
|
||||
}
|
||||
})
|
||||
|
||||
if (IS_MAC) {
|
||||
win.setWindowButtonPosition?.(WINDOW_BUTTON_POSITION)
|
||||
}
|
||||
|
||||
win.once('ready-to-show', () => {
|
||||
if (!win.isDestroyed()) win.show()
|
||||
})
|
||||
|
||||
win.on('will-enter-full-screen', () => sendWindowStateChanged(true))
|
||||
win.on('enter-full-screen', () => sendWindowStateChanged(true))
|
||||
win.on('will-leave-full-screen', () => sendWindowStateChanged(false))
|
||||
win.on('leave-full-screen', () => sendWindowStateChanged(false))
|
||||
|
||||
wireCommonWindowHandlers(win)
|
||||
|
||||
win.loadURL(
|
||||
buildSessionWindowUrl(sessionId, {
|
||||
devServer: DEV_SERVER,
|
||||
rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex(),
|
||||
watch,
|
||||
newSession
|
||||
})
|
||||
)
|
||||
|
||||
return win
|
||||
}
|
||||
|
||||
// Open (or focus) a standalone window for a single chat session.
|
||||
function createSessionWindow(sessionId, { watch = false } = {}) {
|
||||
return sessionWindows.openOrFocus(sessionId, () => {
|
||||
const icon = getAppIconPath()
|
||||
const win = new BrowserWindow({
|
||||
width: SESSION_WINDOW_MIN_WIDTH,
|
||||
height: SESSION_WINDOW_MIN_HEIGHT,
|
||||
minWidth: SESSION_WINDOW_MIN_WIDTH,
|
||||
minHeight: SESSION_WINDOW_MIN_HEIGHT,
|
||||
title: 'Hermes',
|
||||
titleBarStyle: 'hidden',
|
||||
titleBarOverlay: getTitleBarOverlayOptions(),
|
||||
trafficLightPosition: IS_MAC ? WINDOW_BUTTON_POSITION : undefined,
|
||||
vibrancy: IS_MAC ? 'sidebar' : undefined,
|
||||
opacity: windowOpacity(),
|
||||
icon,
|
||||
// Don't show until the renderer's first themed paint is ready. macOS
|
||||
// `vibrancy` ignores `backgroundColor` and paints a translucent OS
|
||||
// material (which follows the OS appearance, not the app theme), so a
|
||||
// dark-themed app on a light-mode Mac flashes white until the renderer
|
||||
// covers it. ready-to-show fires after the boot-time paint in
|
||||
// themes/context.tsx, so the window appears already themed.
|
||||
show: false,
|
||||
backgroundColor: getWindowBackgroundColor(),
|
||||
webPreferences: {
|
||||
preload: path.join(__dirname, 'preload.cjs'),
|
||||
contextIsolation: true,
|
||||
webviewTag: true,
|
||||
sandbox: true,
|
||||
nodeIntegration: false,
|
||||
devTools: true
|
||||
}
|
||||
})
|
||||
return sessionWindows.openOrFocus(sessionId, () => spawnSecondaryWindow({ sessionId, watch }))
|
||||
}
|
||||
|
||||
if (IS_MAC) {
|
||||
win.setWindowButtonPosition?.(WINDOW_BUTTON_POSITION)
|
||||
}
|
||||
|
||||
win.once('ready-to-show', () => {
|
||||
if (!win.isDestroyed()) win.show()
|
||||
})
|
||||
|
||||
win.on('will-enter-full-screen', () => sendWindowStateChanged(true))
|
||||
win.on('enter-full-screen', () => sendWindowStateChanged(true))
|
||||
win.on('will-leave-full-screen', () => sendWindowStateChanged(false))
|
||||
win.on('leave-full-screen', () => sendWindowStateChanged(false))
|
||||
|
||||
wireCommonWindowHandlers(win)
|
||||
|
||||
win.loadURL(
|
||||
buildSessionWindowUrl(sessionId, {
|
||||
devServer: DEV_SERVER,
|
||||
rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex(),
|
||||
watch
|
||||
})
|
||||
)
|
||||
|
||||
return win
|
||||
})
|
||||
// Open a fresh compact window on the new-session draft (#/). Not registry-keyed:
|
||||
// like ⌘N in a browser, every press opens a new window — and a draft window that
|
||||
// later converts to a real session must not get refocused as if it were blank.
|
||||
function createNewSessionWindow() {
|
||||
return spawnSecondaryWindow({ newSession: true })
|
||||
}
|
||||
|
||||
function createWindow() {
|
||||
@@ -5328,6 +5339,11 @@ ipcMain.handle('hermes:window:openSession', async (_event, sessionId, opts) => {
|
||||
|
||||
return { ok: true }
|
||||
})
|
||||
ipcMain.handle('hermes:window:openNewSession', async () => {
|
||||
createNewSessionWindow()
|
||||
|
||||
return { ok: true }
|
||||
})
|
||||
ipcMain.handle('hermes:bootstrap:reset', async () => {
|
||||
// Renderer's "Reload and retry" path. Clear the latched failure and
|
||||
// reset connection state so the next startHermes() call restarts the
|
||||
@@ -5597,9 +5613,14 @@ ipcMain.handle('hermes:api', async (_event, request) => {
|
||||
|
||||
await prepareProfileDeleteRequest(request)
|
||||
|
||||
const connection = await ensureBackend(request?.profile)
|
||||
const profile = request?.profile
|
||||
const connection = await ensureBackend(profile)
|
||||
const timeoutMs = resolveTimeoutMs(request?.timeoutMs, DEFAULT_FETCH_TIMEOUT_MS)
|
||||
const url = `${connection.baseUrl}${request.path}`
|
||||
const requestPath = pathWithGlobalRemoteProfile(request.path, profile, {
|
||||
globalRemote: globalRemoteActive(),
|
||||
profileRemoteOverride: profileHasRemoteOverride(profile)
|
||||
})
|
||||
const url = `${connection.baseUrl}${requestPath}`
|
||||
// OAuth gateways authenticate REST via the HttpOnly session cookie held in
|
||||
// the OAuth partition — route through Electron's net stack bound to that
|
||||
// session so the cookie attaches automatically. Token/local modes keep using
|
||||
|
||||
@@ -6,6 +6,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
|
||||
touchBackend: profile => ipcRenderer.invoke('hermes:backend:touch', profile),
|
||||
getGatewayWsUrl: profile => ipcRenderer.invoke('hermes:gateway:ws-url', profile),
|
||||
openSessionWindow: (sessionId, opts) => ipcRenderer.invoke('hermes:window:openSession', sessionId, opts),
|
||||
openNewSessionWindow: () => ipcRenderer.invoke('hermes:window:openNewSession'),
|
||||
getBootProgress: () => ipcRenderer.invoke('hermes:boot-progress:get'),
|
||||
getConnectionConfig: profile => ipcRenderer.invoke('hermes:connection-config:get', profile),
|
||||
saveConnectionConfig: payload => ipcRenderer.invoke('hermes:connection-config:save', payload),
|
||||
|
||||
@@ -15,12 +15,13 @@ const SESSION_WINDOW_MIN_HEIGHT = 620
|
||||
// flag MUST sit in the query string BEFORE the '#': anything after the '#' is
|
||||
// treated as the route by HashRouter and would break routeSessionId(). The
|
||||
// renderer reads the flag from window.location.search to suppress the install /
|
||||
// onboarding overlays and the global session sidebar. `watch=1` marks a
|
||||
// spectator window (e.g. a running subagent's session): the renderer resumes
|
||||
// it lazily so the gateway never builds an agent just to stream into it.
|
||||
function buildSessionWindowUrl(sessionId, { devServer, rendererIndexPath, watch } = {}) {
|
||||
const query = `?win=secondary${watch ? '&watch=1' : ''}`
|
||||
const route = `#/${encodeURIComponent(sessionId)}`
|
||||
// onboarding overlays and the global session sidebar. `new=1` marks the compact
|
||||
// scratch window; `watch=1` marks a spectator window (e.g. a running subagent's
|
||||
// session): the renderer resumes it lazily so the gateway never builds an agent
|
||||
// just to stream into it.
|
||||
function buildSessionWindowUrl(sessionId, { devServer, rendererIndexPath, watch, newSession } = {}) {
|
||||
const query = `?win=secondary${newSession ? '&new=1' : ''}${watch ? '&watch=1' : ''}`
|
||||
const route = newSession ? '#/' : `#/${encodeURIComponent(sessionId)}`
|
||||
|
||||
if (devServer) {
|
||||
const base = devServer.endsWith('/') ? devServer.slice(0, -1) : devServer
|
||||
|
||||
@@ -82,6 +82,12 @@ test('buildSessionWindowUrl adds the watch flag for spectator windows, before th
|
||||
assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1#/abc')
|
||||
})
|
||||
|
||||
test('buildSessionWindowUrl routes new-session windows to the draft (#/)', () => {
|
||||
const url = buildSessionWindowUrl(null, { devServer: 'http://localhost:5173', newSession: true })
|
||||
|
||||
assert.equal(url, 'http://localhost:5173/?win=secondary&new=1#/')
|
||||
})
|
||||
|
||||
test('registry opens one window per session and focuses on re-open', () => {
|
||||
const registry = createSessionWindowRegistry()
|
||||
let built = 0
|
||||
|
||||
@@ -23,6 +23,7 @@ import { type Translations, useI18n } from '@/i18n'
|
||||
import { sessionTitle } from '@/lib/chat-runtime'
|
||||
import { ExternalLink, ExternalLinkIcon, hostPathLabel, urlSlugTitleLabel, useLinkTitle } from '@/lib/external-link'
|
||||
import { FileImage, FileText, FolderOpen, Link2 } from '@/lib/icons'
|
||||
import { mediaExternalUrl } from '@/lib/media'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { notifyError } from '@/store/notifications'
|
||||
import type { SessionInfo, SessionMessage } from '@/types/hermes'
|
||||
@@ -124,17 +125,12 @@ function artifactKind(value: string): ArtifactKind {
|
||||
}
|
||||
|
||||
function artifactHref(value: string): string {
|
||||
if (
|
||||
value.startsWith('http://') ||
|
||||
value.startsWith('https://') ||
|
||||
value.startsWith('file://') ||
|
||||
value.startsWith('data:')
|
||||
) {
|
||||
if (value.startsWith('http://') || value.startsWith('https://') || value.startsWith('data:')) {
|
||||
return value
|
||||
}
|
||||
|
||||
if (value.startsWith('/')) {
|
||||
return `file://${encodeURI(value)}`
|
||||
if (value.startsWith('file://') || value.startsWith('/')) {
|
||||
return mediaExternalUrl(value)
|
||||
}
|
||||
|
||||
return value
|
||||
|
||||
@@ -42,6 +42,7 @@ import {
|
||||
$sessions,
|
||||
sessionPinId
|
||||
} from '@/store/session'
|
||||
import { isNewSessionWindow, isSecondaryWindow } from '@/store/windows'
|
||||
import type { ModelOptionsResponse } from '@/types/hermes'
|
||||
|
||||
import { routeSessionId } from '../routes'
|
||||
@@ -122,7 +123,7 @@ function ChatHeader({
|
||||
// A brand-new session has no session to pin/delete/rename, so the header is
|
||||
// just a dead "New session" label + chevron. Drop it (and its border)
|
||||
// entirely until there's a real session to act on.
|
||||
if (!selectedSessionId && !activeSessionId && !isRoutedSessionView) {
|
||||
if (isNewSessionWindow() || (!selectedSessionId && !activeSessionId && !isRoutedSessionView)) {
|
||||
return null
|
||||
}
|
||||
|
||||
@@ -302,7 +303,10 @@ export function ChatView({
|
||||
// waiting for the resume effect (which paints a frame later) to clear them.
|
||||
const routeSessionMismatch = isRoutedSessionView && routedSessionId !== selectedSessionId
|
||||
|
||||
const showIntro = freshDraftReady && !isRoutedSessionView && !selectedSessionId && !activeSessionId && messagesEmpty
|
||||
// The compact new-session pop-out skips the wordmark/tagline intro — it's a
|
||||
// scratch window, not the full-height empty state.
|
||||
const showIntro =
|
||||
!isSecondaryWindow() && freshDraftReady && !isRoutedSessionView && !selectedSessionId && !activeSessionId && messagesEmpty
|
||||
|
||||
// Session is still loading if the route references a session we haven't
|
||||
// resumed yet. Once `activeSessionId` is set (runtime has resumed), the
|
||||
|
||||
@@ -77,6 +77,7 @@ import {
|
||||
setSessionsLoading,
|
||||
setSessionsTotal
|
||||
} from '../store/session'
|
||||
import { onSessionsChanged } from '../store/session-sync'
|
||||
import { clearSessionTodos, setSessionTodos, todoListActive } from '../store/todos'
|
||||
import { openUpdatesWindow, startUpdatePoller, stopUpdatePoller } from '../store/updates'
|
||||
import { isSecondaryWindow } from '../store/windows'
|
||||
@@ -464,6 +465,17 @@ export function DesktopController() {
|
||||
void refreshSessions()
|
||||
}, [refreshSessions])
|
||||
|
||||
// Another window mutated the shared session list (e.g. a chat started in the
|
||||
// pop-out). Re-pull so the sidebar reflects it. Pop-outs have no sidebar, so
|
||||
// only real windows bother.
|
||||
useEffect(() => {
|
||||
if (isSecondaryWindow()) {
|
||||
return
|
||||
}
|
||||
|
||||
return onSessionsChanged(() => void refreshSessions().catch(() => undefined))
|
||||
}, [refreshSessions])
|
||||
|
||||
// ALL-profiles view pages one profile at a time: fetch that profile's next
|
||||
// page and merge it in place, leaving every other profile's rows untouched.
|
||||
const loadMoreSessionsForProfile = useCallback(async (profile: string) => {
|
||||
|
||||
@@ -37,6 +37,7 @@ import {
|
||||
switcherActive,
|
||||
switcherJustClosed
|
||||
} from '@/store/session-switcher'
|
||||
import { openNewSessionInNewWindow } from '@/store/windows'
|
||||
import { useTheme } from '@/themes/context'
|
||||
|
||||
import { requestComposerFocus } from '../chat/composer/focus'
|
||||
@@ -132,6 +133,7 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void {
|
||||
deps.startFreshSession()
|
||||
window.dispatchEvent(new CustomEvent('hermes:new-session-shortcut'))
|
||||
},
|
||||
'session.newWindow': () => void openNewSessionInNewWindow(),
|
||||
'session.next': () => stepSession(1),
|
||||
'session.prev': () => stepSession(-1),
|
||||
...sessionSlotHandlers,
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import type { HermesReadDirResult } from '@/global'
|
||||
import { $connection, setCurrentCwd } from '@/store/session'
|
||||
|
||||
import { resetProjectTreeState } from './files/use-project-tree'
|
||||
|
||||
import { RightSidebarPane } from './index'
|
||||
|
||||
const readDir = vi.fn<(path: string) => Promise<HermesReadDirResult>>()
|
||||
const selectPaths = vi.fn()
|
||||
|
||||
function ok(entries: { name: string; path: string; isDirectory: boolean }[]): HermesReadDirResult {
|
||||
return { entries }
|
||||
}
|
||||
|
||||
function installBridge() {
|
||||
;(
|
||||
window as unknown as {
|
||||
hermesDesktop: {
|
||||
readDir: typeof readDir
|
||||
selectPaths: typeof selectPaths
|
||||
}
|
||||
}
|
||||
).hermesDesktop = { readDir, selectPaths }
|
||||
}
|
||||
|
||||
describe('RightSidebarPane', () => {
|
||||
beforeEach(() => {
|
||||
$connection.set(null)
|
||||
resetProjectTreeState()
|
||||
setCurrentCwd('/repo')
|
||||
readDir.mockReset()
|
||||
selectPaths.mockReset()
|
||||
readDir.mockResolvedValue(ok([{ name: 'README.md', path: '/repo/README.md', isDirectory: false }]))
|
||||
selectPaths.mockResolvedValue(['/repo-next'])
|
||||
installBridge()
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
$connection.set(null)
|
||||
setCurrentCwd('')
|
||||
resetProjectTreeState()
|
||||
delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop
|
||||
})
|
||||
|
||||
it('refreshes the current tree without opening the folder picker', async () => {
|
||||
const onChangeCwd = vi.fn()
|
||||
|
||||
render(<RightSidebarPane onActivateFile={vi.fn()} onActivateFolder={vi.fn()} onChangeCwd={onChangeCwd} />)
|
||||
|
||||
await waitFor(() => expect(screen.getByRole('button', { name: 'Refresh tree' }).hasAttribute('disabled')).toBe(false))
|
||||
|
||||
readDir.mockClear()
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Refresh tree' }))
|
||||
|
||||
await waitFor(() => expect(readDir).toHaveBeenCalledWith('/repo'))
|
||||
expect(selectPaths).not.toHaveBeenCalled()
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Open folder' }))
|
||||
|
||||
await waitFor(() =>
|
||||
expect(selectPaths).toHaveBeenCalledWith({
|
||||
defaultPath: '/repo',
|
||||
directories: true,
|
||||
multiple: false,
|
||||
title: 'Change working directory'
|
||||
})
|
||||
)
|
||||
await waitFor(() => expect(onChangeCwd).toHaveBeenCalledWith('/repo-next'))
|
||||
})
|
||||
})
|
||||
@@ -126,12 +126,12 @@ interface FilesystemTabProps extends FileTreeBodyProps {
|
||||
onRefresh: () => void
|
||||
}
|
||||
|
||||
// Sidebar palette + hover-reveal: refresh tracks label hover; collapse-all
|
||||
// stays visible while any folder is expanded.
|
||||
// Sidebar palette + hover-reveal: header actions stay reachable while moving
|
||||
// from the project label to the action buttons.
|
||||
const HEADER_ACTION_CLASS =
|
||||
'text-sidebar-foreground/70 hover:bg-sidebar-accent! hover:text-sidebar-accent-foreground! focus-visible:ring-sidebar-ring'
|
||||
|
||||
const HEADER_ACTION_LABEL_REVEAL = `${HEADER_ACTION_CLASS} pointer-events-none opacity-0 transition-opacity focus-visible:pointer-events-auto focus-visible:opacity-100 peer-focus-visible/project-label:pointer-events-auto peer-focus-visible/project-label:opacity-100 peer-hover/project-label:pointer-events-auto peer-hover/project-label:opacity-100`
|
||||
const HEADER_ACTION_LABEL_REVEAL = `${HEADER_ACTION_CLASS} pointer-events-none opacity-0 transition-opacity focus-visible:pointer-events-auto focus-visible:opacity-100 group-focus-within/project-header:pointer-events-auto group-focus-within/project-header:opacity-100 group-hover/project-header:pointer-events-auto group-hover/project-header:opacity-100`
|
||||
|
||||
function FilesystemTab({
|
||||
canCollapse,
|
||||
@@ -158,7 +158,7 @@ function FilesystemTab({
|
||||
return (
|
||||
<div className="flex min-h-0 flex-1 flex-col">
|
||||
<RightSidebarSectionHeader>
|
||||
<div className="peer/project-label flex min-w-0 flex-1">
|
||||
<div className="flex min-w-0 flex-1">
|
||||
<button
|
||||
className="flex w-full min-w-0 items-center rounded-md text-left hover:text-(--ui-text-secondary)"
|
||||
onClick={() => void onChangeFolder()}
|
||||
@@ -216,7 +216,7 @@ function FilesystemTab({
|
||||
}
|
||||
|
||||
export function RightSidebarSectionHeader({ children }: { children: ReactNode }) {
|
||||
return <div className="flex h-7 shrink-0 items-center px-2.5">{children}</div>
|
||||
return <div className="group/project-header flex h-7 shrink-0 items-center px-2.5">{children}</div>
|
||||
}
|
||||
|
||||
interface FileTreeBodyProps {
|
||||
|
||||
@@ -47,6 +47,7 @@ import {
|
||||
setTurnStartedAt,
|
||||
setYoloActive
|
||||
} from '@/store/session'
|
||||
import { broadcastSessionsChanged } from '@/store/session-sync'
|
||||
import { clearSessionSubagents, pruneDelegateFallbackSubagents, upsertSubagent } from '@/store/subagents'
|
||||
import { setSessionTodos } from '@/store/todos'
|
||||
import { recordToolDiff } from '@/store/tool-diffs'
|
||||
@@ -641,6 +642,9 @@ export function useMessageStream({
|
||||
})
|
||||
|
||||
void refreshSessions().catch(() => undefined)
|
||||
// Sync the freshly-titled row to other windows (e.g. main, when the turn
|
||||
// ran in the pop-out).
|
||||
broadcastSessionsChanged()
|
||||
|
||||
if (compactedTurnRef.current.delete(sessionId)) {
|
||||
shouldHydrate = false
|
||||
|
||||
@@ -58,6 +58,7 @@ import { clearSessionTodos } from '@/store/todos'
|
||||
|
||||
import type {
|
||||
ClientSessionState,
|
||||
BrowserManageResponse,
|
||||
FileAttachResponse,
|
||||
HandoffFailResponse,
|
||||
HandoffRequestResponse,
|
||||
@@ -1141,6 +1142,81 @@ export function usePromptActions({
|
||||
} catch (err) {
|
||||
renderSlashOutput(`error: ${err instanceof Error ? err.message : String(err)}`)
|
||||
}
|
||||
},
|
||||
// /browser connect|disconnect|status manages the live CDP connection on
|
||||
// the gateway host, mirroring the TUI's browser.manage RPC. It mutates
|
||||
// BROWSER_CDP_URL (and may launch Chrome) in the gateway process — only
|
||||
// meaningful when that process runs on this machine, so it's gated to
|
||||
// local connections. A remote gateway would act on the wrong host.
|
||||
browser: async ctx => {
|
||||
const resolved = await withSlashOutput(ctx)
|
||||
|
||||
if (!resolved) {
|
||||
return
|
||||
}
|
||||
|
||||
const { render: renderSlashOutput, sessionId } = resolved
|
||||
|
||||
if ($connection.get()?.mode === 'remote') {
|
||||
renderSlashOutput(
|
||||
'/browser manages a Chromium-family browser on the gateway host — only available when connected to a local gateway.'
|
||||
)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
const [rawAction = 'status', ...rest] = ctx.arg.trim().split(/\s+/).filter(Boolean)
|
||||
const cmdAction = rawAction.toLowerCase()
|
||||
|
||||
if (!['connect', 'disconnect', 'status'].includes(cmdAction)) {
|
||||
renderSlashOutput(
|
||||
'usage: /browser [connect|disconnect|status] [url] · persistent: set browser.cdp_url in config.yaml'
|
||||
)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
const url = cmdAction === 'connect' ? rest.join(' ').trim() || 'http://127.0.0.1:9222' : undefined
|
||||
|
||||
if (url) {
|
||||
renderSlashOutput(`checking Chromium-family browser remote debugging at ${url}...`)
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await requestGateway<BrowserManageResponse>('browser.manage', {
|
||||
action: cmdAction,
|
||||
session_id: sessionId,
|
||||
...(url && { url })
|
||||
})
|
||||
|
||||
// Without a streamed session subscription, the gateway bundles its
|
||||
// progress lines into `messages` — flush them inline.
|
||||
result?.messages?.forEach(message => renderSlashOutput(message))
|
||||
|
||||
if (cmdAction === 'status') {
|
||||
renderSlashOutput(
|
||||
result?.connected
|
||||
? `browser connected: ${result.url || '(url unavailable)'}`
|
||||
: 'browser not connected (try /browser connect <url> or set browser.cdp_url in config.yaml)'
|
||||
)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if (cmdAction === 'disconnect') {
|
||||
renderSlashOutput('browser disconnected')
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if (result?.connected) {
|
||||
renderSlashOutput('Browser connected to live Chromium-family browser via CDP')
|
||||
renderSlashOutput(`Endpoint: ${result.url || '(url unavailable)'}`)
|
||||
renderSlashOutput('next browser tool call will use this CDP endpoint')
|
||||
}
|
||||
} catch (err) {
|
||||
renderSlashOutput(`error: ${err instanceof Error ? err.message : String(err)}`)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -42,6 +42,7 @@ import {
|
||||
setYoloActive,
|
||||
workspaceCwdForNewSession
|
||||
} from '@/store/session'
|
||||
import { broadcastSessionsChanged } from '@/store/session-sync'
|
||||
import { reportBackendContract } from '@/store/updates'
|
||||
import { isWatchWindow } from '@/store/windows'
|
||||
import type { SessionCreateResponse, SessionInfo, SessionResumeResponse, SessionRuntimeInfo, UsageStats } from '@/types/hermes'
|
||||
@@ -472,6 +473,9 @@ export function useSessionActions({
|
||||
// server later returns its own preview/title and supersedes this.
|
||||
upsertOptimisticSession(created, stored, null, preview?.trim() || null)
|
||||
navigate(sessionRoute(stored), { replace: true })
|
||||
// Other windows (e.g. the main window when this is the pop-out) can't
|
||||
// see this session until they re-pull the shared list.
|
||||
broadcastSessionsChanged()
|
||||
}
|
||||
|
||||
setFreshDraftReady(false)
|
||||
|
||||
@@ -16,7 +16,7 @@ import {
|
||||
} from '@/store/layout'
|
||||
import { $paneWidthOverride } from '@/store/panes'
|
||||
import { $connection } from '@/store/session'
|
||||
import { isSecondaryWindow } from '@/store/windows'
|
||||
import { isNewSessionWindow, isSecondaryWindow } from '@/store/windows'
|
||||
|
||||
import { SIDEBAR_COLLAPSE_MEDIA_QUERY } from '../layout-constants'
|
||||
|
||||
@@ -80,6 +80,7 @@ export function AppShell({
|
||||
const connection = useStore($connection)
|
||||
const viewportFullscreen = useSyncExternalStore(subscribeWindowSize, viewportIsFullscreen, () => false)
|
||||
const isFullscreen = Boolean(connection?.isFullscreen) || viewportFullscreen
|
||||
const hideTitlebarControls = isNewSessionWindow()
|
||||
const titlebarControls = titlebarControlsPosition(connection?.windowButtonPosition, isFullscreen)
|
||||
// Width Windows/Linux reserve for the OS-painted min/max/close overlay (zero
|
||||
// on macOS, where window controls sit on the left and are reported via
|
||||
@@ -162,7 +163,9 @@ export function AppShell({
|
||||
} as CSSProperties
|
||||
}
|
||||
>
|
||||
<TitlebarControls leftTools={leftTitlebarTools} onOpenSettings={onOpenSettings} tools={titlebarTools} />
|
||||
{!hideTitlebarControls && (
|
||||
<TitlebarControls leftTools={leftTitlebarTools} onOpenSettings={onOpenSettings} tools={titlebarTools} />
|
||||
)}
|
||||
|
||||
<main className="relative z-3 flex min-h-0 w-full flex-1 flex-col overflow-hidden transition-none">
|
||||
<PaneShell className="min-h-0 flex-1">
|
||||
@@ -183,7 +186,9 @@ export function AppShell({
|
||||
the panes' z-20 resize handles, keeping every pane resizable. */}
|
||||
{mainOverlays}
|
||||
|
||||
<StatusbarControls items={statusbarItems} leftItems={leftStatusbarItems} />
|
||||
{/* The compact pop-out drops the statusbar — it's a scratch window, not
|
||||
the full shell. */}
|
||||
{!isSecondaryWindow() && <StatusbarControls items={statusbarItems} leftItems={leftStatusbarItems} />}
|
||||
</main>
|
||||
|
||||
{overlays}
|
||||
|
||||
@@ -46,6 +46,12 @@ export interface SlashExecResponse {
|
||||
warning?: string
|
||||
}
|
||||
|
||||
export interface BrowserManageResponse {
|
||||
connected?: boolean
|
||||
url?: string
|
||||
messages?: string[]
|
||||
}
|
||||
|
||||
export interface SessionSteerResponse {
|
||||
// 'queued' == accepted into the live turn's steer slot (injected at the next
|
||||
// tool-result boundary); 'rejected' == no live tool window, caller queues.
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { ThreadPrimitive, useAuiEvent, useAuiState } from '@assistant-ui/react'
|
||||
import {
|
||||
type CSSProperties,
|
||||
type ComponentProps,
|
||||
type FC,
|
||||
memo,
|
||||
@@ -21,6 +22,7 @@ import {
|
||||
resetThreadScroll,
|
||||
setThreadAtBottom
|
||||
} from '@/store/thread-scroll'
|
||||
import { isNewSessionWindow, isSecondaryWindow } from '@/store/windows'
|
||||
|
||||
import { MessageRenderBoundary } from './message-render-boundary'
|
||||
|
||||
@@ -132,6 +134,13 @@ const ThreadMessageListInner: FC<ThreadMessageListProps> = ({
|
||||
const hiddenCount = firstVisible
|
||||
const visibleGroups = hiddenCount > 0 ? groups.slice(hiddenCount) : groups
|
||||
const restoreFromBottomRef = useRef<number | null>(null)
|
||||
const newSessionWindow = isNewSessionWindow()
|
||||
const newSessionTitlebarGap = 'calc(var(--titlebar-height)+0.75rem)'
|
||||
const threadContentTopPad = newSessionWindow
|
||||
? 'pt-[calc(var(--titlebar-height)+0.75rem)]'
|
||||
: isSecondaryWindow()
|
||||
? 'pt-6'
|
||||
: 'pt-[calc(var(--titlebar-height)+1.5rem)]'
|
||||
|
||||
useEffect(() => setThreadAtBottom(isAtBottom), [isAtBottom])
|
||||
useEffect(() => () => resetThreadScroll(), [])
|
||||
@@ -235,7 +244,12 @@ const ThreadMessageListInner: FC<ThreadMessageListProps> = ({
|
||||
return (
|
||||
<div
|
||||
className="relative min-h-0 max-w-full overflow-hidden contain-[layout_paint]"
|
||||
style={{ height: clampToComposer ? 'var(--thread-viewport-height)' : '100%' }}
|
||||
style={
|
||||
{
|
||||
height: clampToComposer ? 'var(--thread-viewport-height)' : '100%',
|
||||
...(newSessionWindow ? { '--sticky-human-top': newSessionTitlebarGap } : {})
|
||||
} as CSSProperties
|
||||
}
|
||||
>
|
||||
<div
|
||||
className="size-full overflow-x-hidden overflow-y-auto overscroll-contain"
|
||||
@@ -252,9 +266,7 @@ const ThreadMessageListInner: FC<ThreadMessageListProps> = ({
|
||||
</div>
|
||||
) : (
|
||||
<div
|
||||
className={cn(
|
||||
'mx-auto flex w-full max-w-(--composer-width) min-w-0 flex-col px-6 pt-[calc(var(--titlebar-height)+1.5rem)]'
|
||||
)}
|
||||
className={cn('mx-auto flex w-full max-w-(--composer-width) min-w-0 flex-col px-6', threadContentTopPad)}
|
||||
data-slot="aui_thread-content"
|
||||
ref={contentRef as React.RefCallback<HTMLDivElement>}
|
||||
>
|
||||
|
||||
Vendored
+2
@@ -24,6 +24,8 @@ declare global {
|
||||
// a spectator window (lazy resume — no agent build) for live-streaming
|
||||
// a running subagent's session.
|
||||
openSessionWindow: (sessionId: string, opts?: { watch?: boolean }) => Promise<{ ok: boolean; error?: string }>
|
||||
// Open (or focus) a compact secondary window on the new-session draft.
|
||||
openNewSessionWindow: () => Promise<{ ok: boolean; error?: string }>
|
||||
getBootProgress: () => Promise<DesktopBootProgress>
|
||||
getConnectionConfig: (profile?: null | string) => Promise<DesktopConnectionConfig>
|
||||
saveConnectionConfig: (payload: DesktopConnectionConfigInput) => Promise<DesktopConnectionConfig>
|
||||
|
||||
@@ -189,6 +189,7 @@ export const en: Translations = {
|
||||
'nav.cron': 'Open scheduled jobs',
|
||||
'nav.agents': 'Open agents',
|
||||
'session.new': 'New session',
|
||||
'session.newWindow': 'New session in window',
|
||||
'session.next': 'Next session',
|
||||
'session.prev': 'Previous session',
|
||||
'session.slot.1': 'Switch to recent session 1',
|
||||
|
||||
@@ -185,6 +185,7 @@ export const zh: Translations = {
|
||||
'nav.cron': '打开定时任务',
|
||||
'nav.agents': '打开智能体',
|
||||
'session.new': '新建会话',
|
||||
'session.newWindow': '在新窗口中新建会话',
|
||||
'session.next': '下一个会话',
|
||||
'session.prev': '上一个会话',
|
||||
'session.slot.1': '切换到最近会话 1',
|
||||
|
||||
@@ -3,6 +3,7 @@ import { describe, expect, it } from 'vitest'
|
||||
import type { ChatMessage, ChatMessagePart } from './chat-messages'
|
||||
import {
|
||||
appendAssistantTextPart,
|
||||
appendReasoningPart,
|
||||
chatMessageText,
|
||||
preserveLocalAssistantErrors,
|
||||
renderMediaTags,
|
||||
@@ -175,6 +176,52 @@ describe('renderMediaTags', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('interleaved reasoning/text coalescing', () => {
|
||||
it('keeps narration contiguous when reasoning interrupts mid-sentence', () => {
|
||||
// Models that interleave reasoning_content + content deltas emit
|
||||
// text → reasoning → text within one tool-bounded segment. The two text
|
||||
// fragments are really one sentence and must not be split by the
|
||||
// "Thinking" block between them.
|
||||
let parts: ChatMessagePart[] = appendAssistantTextPart([], 'Let me ')
|
||||
parts = appendReasoningPart(parts, 'checking the file...')
|
||||
parts = appendAssistantTextPart(parts, 'verify the full file is correct:')
|
||||
|
||||
expect(parts.map(p => p.type)).toEqual(['text', 'reasoning'])
|
||||
expect((parts[0] as { text: string }).text).toBe('Let me verify the full file is correct:')
|
||||
expect((parts[1] as { text: string }).text).toBe('checking the file...')
|
||||
})
|
||||
|
||||
it('merges reasoning bursts that straddle a narration fragment', () => {
|
||||
let parts: ChatMessagePart[] = appendReasoningPart([], 'first thought ')
|
||||
parts = appendAssistantTextPart(parts, 'Working on it.')
|
||||
parts = appendReasoningPart(parts, 'second thought')
|
||||
|
||||
expect(parts.map(p => p.type)).toEqual(['reasoning', 'text'])
|
||||
expect((parts[0] as { text: string }).text).toBe('first thought second thought')
|
||||
expect((parts[1] as { text: string }).text).toBe('Working on it.')
|
||||
})
|
||||
|
||||
it('starts a fresh text part after a tool call (segment boundary)', () => {
|
||||
let parts: ChatMessagePart[] = appendAssistantTextPart([], 'Let me check.')
|
||||
parts = upsertToolPart(parts, { name: 'read_file', tool_id: 'tc-1' }, 'running')
|
||||
parts = appendAssistantTextPart(parts, 'Now editing.')
|
||||
|
||||
expect(parts.map(p => p.type)).toEqual(['text', 'tool-call', 'text'])
|
||||
expect((parts[0] as { text: string }).text).toBe('Let me check.')
|
||||
expect((parts[2] as { text: string }).text).toBe('Now editing.')
|
||||
})
|
||||
|
||||
it('does not merge reasoning across a tool call', () => {
|
||||
let parts: ChatMessagePart[] = appendReasoningPart([], 'before tool')
|
||||
parts = upsertToolPart(parts, { name: 'read_file', tool_id: 'tc-1' }, 'running')
|
||||
parts = appendReasoningPart(parts, 'after tool')
|
||||
|
||||
expect(parts.map(p => p.type)).toEqual(['reasoning', 'tool-call', 'reasoning'])
|
||||
expect((parts[0] as { text: string }).text).toBe('before tool')
|
||||
expect((parts[2] as { text: string }).text).toBe('after tool')
|
||||
})
|
||||
})
|
||||
|
||||
describe('preserveLocalAssistantErrors', () => {
|
||||
it('preserves a local user+error pair when hydration omits the failed turn', () => {
|
||||
const nextMessages: ChatMessage[] = [
|
||||
|
||||
@@ -178,50 +178,70 @@ function displayContentForMessage(role: SessionMessage['role'], content: unknown
|
||||
return [refs.join('\n'), visibleText].filter(Boolean).join('\n\n') || visibleText
|
||||
}
|
||||
|
||||
export function appendTextPart(parts: ChatMessagePart[], delta: string): ChatMessagePart[] {
|
||||
const next = [...parts]
|
||||
const last = next.at(-1)
|
||||
|
||||
if (last?.type === 'text') {
|
||||
next[next.length - 1] = { ...last, text: `${last.text}${delta}` }
|
||||
|
||||
return next
|
||||
}
|
||||
|
||||
next.push(textPart(delta))
|
||||
|
||||
return next
|
||||
const STREAM_PART: Record<'reasoning' | 'text', (text: string) => ChatMessagePart> = {
|
||||
reasoning: reasoningPart,
|
||||
text: textPart
|
||||
}
|
||||
|
||||
export function appendAssistantTextPart(parts: ChatMessagePart[], delta: string): ChatMessagePart[] {
|
||||
const next = appendTextPart(parts, delta)
|
||||
const last = next.at(-1)
|
||||
// Coalesce a streaming delta into the most recent same-type part within the
|
||||
// current segment, where a segment is bounded by any non-streaming part (a
|
||||
// tool call, image, …). The opposite streaming channel (text <-> reasoning) is
|
||||
// transparent, so a reasoning burst between two content deltas can't shred one
|
||||
// sentence into text / Thinking / text — the fragmentation models that
|
||||
// interleave reasoning_content + content otherwise produce. Tool calls still
|
||||
// open a fresh part, preserving narration order across steps.
|
||||
function appendStreamPart(
|
||||
parts: ChatMessagePart[],
|
||||
type: 'reasoning' | 'text',
|
||||
delta: string
|
||||
): { index: number; parts: ChatMessagePart[] } {
|
||||
const next = [...parts]
|
||||
|
||||
if (last?.type === 'text') {
|
||||
const current = last.text
|
||||
for (let i = next.length - 1; i >= 0; i--) {
|
||||
const part = next[i]
|
||||
|
||||
const deltaMayContainMedia =
|
||||
delta.includes('MEDIA:') || delta.includes('DIA:') || delta.includes('EDIA:') || delta.includes('IA:')
|
||||
if (part.type === type) {
|
||||
next[i] = { ...part, text: `${(part as { text: string }).text}${delta}` } as ChatMessagePart
|
||||
|
||||
const needsMediaPass = deltaMayContainMedia || current.includes('MEDIA:')
|
||||
const nextText = needsMediaPass ? renderMediaTags(current) : current
|
||||
next[next.length - 1] = nextText === current ? last : { ...last, text: nextText }
|
||||
return { index: i, parts: next }
|
||||
}
|
||||
|
||||
if (part.type !== 'text' && part.type !== 'reasoning') {
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
return next
|
||||
next.push(STREAM_PART[type](delta))
|
||||
|
||||
return { index: next.length - 1, parts: next }
|
||||
}
|
||||
|
||||
export function appendTextPart(parts: ChatMessagePart[], delta: string): ChatMessagePart[] {
|
||||
return appendStreamPart(parts, 'text', delta).parts
|
||||
}
|
||||
|
||||
export function appendReasoningPart(parts: ChatMessagePart[], delta: string): ChatMessagePart[] {
|
||||
const next = [...parts]
|
||||
const last = next.at(-1)
|
||||
return appendStreamPart(parts, 'reasoning', delta).parts
|
||||
}
|
||||
|
||||
if (last?.type === 'reasoning') {
|
||||
next[next.length - 1] = { ...last, text: `${last.text}${delta}` }
|
||||
export function appendAssistantTextPart(parts: ChatMessagePart[], delta: string): ChatMessagePart[] {
|
||||
const { index, parts: next } = appendStreamPart(parts, 'text', delta)
|
||||
const part = next[index]
|
||||
|
||||
if (part?.type !== 'text') {
|
||||
return next
|
||||
}
|
||||
|
||||
next.push(reasoningPart(delta))
|
||||
const mayContainMedia =
|
||||
delta.includes('MEDIA:') || delta.includes('DIA:') || delta.includes('EDIA:') || delta.includes('IA:')
|
||||
|
||||
if (mayContainMedia || part.text.includes('MEDIA:')) {
|
||||
const rendered = renderMediaTags(part.text)
|
||||
|
||||
if (rendered !== part.text) {
|
||||
next[index] = { ...part, text: rendered }
|
||||
}
|
||||
}
|
||||
|
||||
return next
|
||||
}
|
||||
|
||||
@@ -52,6 +52,17 @@ describe('desktop slash command curation', () => {
|
||||
expect(desktopSlashUnavailableMessage('/personality')).toBeNull()
|
||||
})
|
||||
|
||||
it('treats /browser as an executable action command (local-gateway connect)', () => {
|
||||
// /browser used to be terminal-only; it now resolves to a desktop action
|
||||
// handler that routes browser.manage RPC when the gateway is local.
|
||||
expect(isDesktopSlashCommand('/browser')).toBe(true)
|
||||
expect(isDesktopSlashSuggestion('/browser')).toBe(true)
|
||||
expect(desktopSlashUnavailableMessage('/browser')).toBeNull()
|
||||
expect(resolveDesktopCommand('/browser')?.surface).toEqual({ kind: 'action', action: 'browser' })
|
||||
// Bare /browser expands to its sub-action options in the popover.
|
||||
expect(resolveDesktopCommand('/browser')?.args).toBe(true)
|
||||
})
|
||||
|
||||
it('allows aliases to execute without cluttering the popover', () => {
|
||||
expect(isDesktopSlashSuggestion('/reset')).toBe(false)
|
||||
expect(isDesktopSlashCommand('/reset')).toBe(true)
|
||||
|
||||
@@ -30,6 +30,7 @@ export interface DesktopThemeCommandOption {
|
||||
*/
|
||||
export type DesktopActionId =
|
||||
| 'branch'
|
||||
| 'browser'
|
||||
| 'handoff'
|
||||
| 'help'
|
||||
| 'new'
|
||||
@@ -103,6 +104,12 @@ const DESKTOP_COMMAND_SPECS: readonly DesktopCommandSpec[] = [
|
||||
{ name: '/skin', description: 'Switch desktop theme or cycle to the next one', surface: action('skin'), args: true },
|
||||
{ name: '/title', description: 'Rename the current session', surface: action('title') },
|
||||
{ name: '/help', description: 'Show desktop slash commands', aliases: ['/commands'], surface: action('help') },
|
||||
{
|
||||
name: '/browser',
|
||||
description: 'Manage browser CDP connection [connect|disconnect|status] (local gateway only)',
|
||||
surface: action('browser'),
|
||||
args: true
|
||||
},
|
||||
|
||||
// Overlay pickers
|
||||
{ name: '/model', description: 'Switch the model for this session', surface: picker('model'), hidden: true },
|
||||
@@ -142,7 +149,7 @@ const DESKTOP_COMMAND_SPECS: readonly DesktopCommandSpec[] = [
|
||||
// per reason beats 40 identical object literals.
|
||||
const NO_DESKTOP_SURFACE: Record<DesktopUnavailableReason, readonly string[]> = {
|
||||
terminal: [
|
||||
'/browser', '/busy', '/clear', '/compact', '/config', '/copy', '/cron', '/details',
|
||||
'/busy', '/clear', '/compact', '/config', '/copy', '/cron', '/details',
|
||||
'/exit', '/footer', '/gateway', '/gquota', '/history', '/image', '/indicator', '/logs',
|
||||
'/mouse', '/paste', '/platforms', '/plugins', '/quit', '/redraw', '/reload', '/restart',
|
||||
'/sb', '/set-home', '/sethome', '/snap', '/snapshot', '/statusbar', '/toolsets', '/update', '/verbose'
|
||||
|
||||
@@ -66,6 +66,7 @@ export const KEYBIND_ACTIONS: readonly KeybindActionMeta[] = [
|
||||
|
||||
// ── Session ──────────────────────────────────────────────────────────────
|
||||
{ id: 'session.new', category: 'session', defaults: ['mod+n', 'shift+n'] },
|
||||
{ id: 'session.newWindow', category: 'session', defaults: ['mod+shift+n'] },
|
||||
// ⌃Tab / ⌃⇧Tab — the universal tab-cycle chord. Literally Control, not Cmd
|
||||
// (macOS reserves Cmd+Tab for app switching); see `ctrl` in combo.ts.
|
||||
{ id: 'session.next', category: 'session', defaults: ['ctrl+tab'] },
|
||||
|
||||
@@ -2,7 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { $connection } from '@/store/session'
|
||||
|
||||
import { filePathFromMediaPath, gatewayMediaDataUrl, isRemoteGateway } from './media'
|
||||
import { filePathFromMediaPath, gatewayMediaDataUrl, isRemoteGateway, mediaExternalUrl } from './media'
|
||||
|
||||
describe('isRemoteGateway', () => {
|
||||
afterEach(() => {
|
||||
@@ -35,6 +35,38 @@ describe('filePathFromMediaPath', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('mediaExternalUrl', () => {
|
||||
afterEach(() => {
|
||||
$connection.set(null)
|
||||
})
|
||||
|
||||
it('passes through http(s) URLs untouched', () => {
|
||||
$connection.set({ mode: 'remote', baseUrl: 'https://gw', token: 't' } as never)
|
||||
expect(mediaExternalUrl('https://example.com/a.png')).toBe('https://example.com/a.png')
|
||||
})
|
||||
|
||||
it('keeps file:// form in local mode', () => {
|
||||
$connection.set({ mode: 'local' } as never)
|
||||
expect(mediaExternalUrl('/tmp/a.png')).toBe('file:///tmp/a.png')
|
||||
expect(mediaExternalUrl('file:///tmp/a.png')).toBe('file:///tmp/a.png')
|
||||
})
|
||||
|
||||
it('rewrites gateway-local paths to an authenticated download URL', () => {
|
||||
$connection.set({ mode: 'remote', baseUrl: 'https://gw', token: 's e/cret' } as never)
|
||||
expect(mediaExternalUrl('file:///tmp/a b.png')).toBe(
|
||||
'https://gw/api/files/download?path=%2Ftmp%2Fa%20b.png&token=s%20e%2Fcret'
|
||||
)
|
||||
expect(mediaExternalUrl('/tmp/a b.png')).toBe(
|
||||
'https://gw/api/files/download?path=%2Ftmp%2Fa%20b.png&token=s%20e%2Fcret'
|
||||
)
|
||||
})
|
||||
|
||||
it('falls back to file:// when remote connection lacks a token', () => {
|
||||
$connection.set({ mode: 'remote', baseUrl: 'https://gw' } as never)
|
||||
expect(mediaExternalUrl('/tmp/a.png')).toBe('file:///tmp/a.png')
|
||||
})
|
||||
})
|
||||
|
||||
describe('gatewayMediaDataUrl', () => {
|
||||
const api = vi.fn(async () => ({ data_url: 'data:image/png;base64,ZHVtbXk=' }))
|
||||
|
||||
|
||||
@@ -56,8 +56,25 @@ export function mediaMarkdownHref(path: string): string {
|
||||
return `#media:${encodeURIComponent(path)}`
|
||||
}
|
||||
|
||||
// Resolve a media path to a URL the shell can open. Remote mode rewrites
|
||||
// gateway-local paths to an authenticated /api/files/download URL (the file
|
||||
// lives on the gateway, not this disk); local mode keeps the file:// form.
|
||||
export function mediaExternalUrl(path: string): string {
|
||||
return /^(?:https?|file):/i.test(path) ? path : `file://${path}`
|
||||
if (/^https?:/i.test(path)) {
|
||||
return path
|
||||
}
|
||||
|
||||
if (isRemoteGateway()) {
|
||||
const conn = $connection.get()
|
||||
|
||||
if (conn?.baseUrl && conn.token) {
|
||||
const file = encodeURIComponent(filePathFromMediaPath(path))
|
||||
|
||||
return `${conn.baseUrl}/api/files/download?path=${file}&token=${encodeURIComponent(conn.token)}`
|
||||
}
|
||||
}
|
||||
|
||||
return /^file:/i.test(path) ? path : `file://${path}`
|
||||
}
|
||||
|
||||
// Custom Electron scheme (registered in electron/main.cjs) that streams a local
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
// Cross-window session-list sync. Each desktop window is its own renderer
|
||||
// process with its own gateway socket and session store, so a mutation in one
|
||||
// (e.g. a new chat started in the compact pop-out) never reaches another
|
||||
// window. This bus pings every window to re-pull the shared session list; the
|
||||
// data already lives in the backend, the other window just doesn't know to look.
|
||||
const CHANNEL = 'hermes:sessions'
|
||||
|
||||
const channel = typeof BroadcastChannel === 'undefined' ? null : new BroadcastChannel(CHANNEL)
|
||||
|
||||
// A window that mutated the session list (created / titled a chat) tells the
|
||||
// others to refresh. A BroadcastChannel never delivers to its own poster, so the
|
||||
// caller refreshes locally as it already does.
|
||||
export function broadcastSessionsChanged(): void {
|
||||
channel?.postMessage(1)
|
||||
}
|
||||
|
||||
export function onSessionsChanged(handler: () => void): () => void {
|
||||
if (!channel) {
|
||||
return () => {}
|
||||
}
|
||||
|
||||
channel.addEventListener('message', handler)
|
||||
|
||||
return () => channel.removeEventListener('message', handler)
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { canOpenSessionWindow, openSessionInNewWindow } from './windows'
|
||||
import { canOpenSessionWindow, openNewSessionInNewWindow, openSessionInNewWindow } from './windows'
|
||||
|
||||
const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] }
|
||||
const initialHermesDesktop = desktopWindow.hermesDesktop
|
||||
@@ -11,9 +11,13 @@ vi.mock('./notifications', () => ({
|
||||
notifyError: (...args: unknown[]) => notifyError(...args)
|
||||
}))
|
||||
|
||||
function installBridge(openSessionWindow?: Window['hermesDesktop']['openSessionWindow']) {
|
||||
function installBridge(
|
||||
openSessionWindow?: Window['hermesDesktop']['openSessionWindow'],
|
||||
openNewSessionWindow?: Window['hermesDesktop']['openNewSessionWindow']
|
||||
) {
|
||||
desktopWindow.hermesDesktop = {
|
||||
...(openSessionWindow ? { openSessionWindow } : {})
|
||||
...(openSessionWindow ? { openSessionWindow } : {}),
|
||||
...(openNewSessionWindow ? { openNewSessionWindow } : {})
|
||||
} as unknown as Window['hermesDesktop']
|
||||
}
|
||||
|
||||
@@ -101,3 +105,39 @@ describe('openSessionInNewWindow', () => {
|
||||
expect(notifyError).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
})
|
||||
|
||||
describe('openNewSessionInNewWindow', () => {
|
||||
it('no-ops gracefully when the bridge is absent (web fallback)', async () => {
|
||||
delete desktopWindow.hermesDesktop
|
||||
|
||||
await openNewSessionInNewWindow()
|
||||
|
||||
expect(notifyError).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('no-ops when openNewSessionWindow is missing', async () => {
|
||||
installBridge(vi.fn().mockResolvedValue({ ok: true }))
|
||||
|
||||
await openNewSessionInNewWindow()
|
||||
|
||||
expect(notifyError).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('invokes the bridge', async () => {
|
||||
const openNew = vi.fn().mockResolvedValue({ ok: true })
|
||||
installBridge(vi.fn().mockResolvedValue({ ok: true }), openNew)
|
||||
|
||||
await openNewSessionInNewWindow()
|
||||
|
||||
expect(openNew).toHaveBeenCalledTimes(1)
|
||||
expect(notifyError).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('notifies on an ok:false result', async () => {
|
||||
installBridge(vi.fn().mockResolvedValue({ ok: true }), vi.fn().mockResolvedValue({ ok: false, error: 'nope' }))
|
||||
|
||||
await openNewSessionInNewWindow()
|
||||
|
||||
expect(notifyError).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -6,6 +6,7 @@ import { notifyError } from './notifications'
|
||||
// never from the router. A "secondary" window renders a single chat without the
|
||||
// global session sidebar or the install / onboarding overlays.
|
||||
const SECONDARY_WINDOW_FLAG = 'secondary'
|
||||
const NEW_SESSION_WINDOW_FLAG = '1'
|
||||
|
||||
let secondaryWindowCache: boolean | null = null
|
||||
|
||||
@@ -27,6 +28,26 @@ export function isSecondaryWindow(): boolean {
|
||||
return result
|
||||
}
|
||||
|
||||
let newSessionWindowCache: boolean | null = null
|
||||
|
||||
export function isNewSessionWindow(): boolean {
|
||||
if (newSessionWindowCache !== null) {
|
||||
return newSessionWindowCache
|
||||
}
|
||||
|
||||
let result = false
|
||||
|
||||
try {
|
||||
result = new URLSearchParams(window.location.search).get('new') === NEW_SESSION_WINDOW_FLAG
|
||||
} catch {
|
||||
result = false
|
||||
}
|
||||
|
||||
newSessionWindowCache = result
|
||||
|
||||
return result
|
||||
}
|
||||
|
||||
let watchWindowCache: boolean | null = null
|
||||
|
||||
// A "watch" window spectates a session that is being driven elsewhere (a
|
||||
@@ -57,6 +78,22 @@ export function canOpenSessionWindow(): boolean {
|
||||
return typeof window !== 'undefined' && typeof window.hermesDesktop?.openSessionWindow === 'function'
|
||||
}
|
||||
|
||||
type WindowOpenResult = { ok: boolean; error?: string } | undefined
|
||||
|
||||
// Run a window-open bridge call, surfacing any failure as a toast. Shared by the
|
||||
// session pop-out and the new-session pop-out.
|
||||
async function openWindow(call: () => Promise<WindowOpenResult>, failMessage: string): Promise<void> {
|
||||
try {
|
||||
const result = await call()
|
||||
|
||||
if (!result?.ok) {
|
||||
notifyError(new Error(result?.error || 'unknown error'), failMessage)
|
||||
}
|
||||
} catch (err) {
|
||||
notifyError(err, failMessage)
|
||||
}
|
||||
}
|
||||
|
||||
// Open (or focus) a standalone OS window for a single chat session. No-ops
|
||||
// gracefully outside Electron so callers can wire it unconditionally.
|
||||
// `watch: true` opens a spectator window (lazy resume, live-mirror stream).
|
||||
@@ -65,13 +102,14 @@ export async function openSessionInNewWindow(sessionId: string, opts?: { watch?:
|
||||
return
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await window.hermesDesktop.openSessionWindow(sessionId, opts)
|
||||
await openWindow(() => window.hermesDesktop.openSessionWindow(sessionId, opts), 'Could not open chat in a new window')
|
||||
}
|
||||
|
||||
if (!result?.ok) {
|
||||
notifyError(new Error(result?.error || 'unknown error'), 'Could not open chat in a new window')
|
||||
}
|
||||
} catch (err) {
|
||||
notifyError(err, 'Could not open chat in a new window')
|
||||
// Open a fresh compact window on the new-session draft.
|
||||
export async function openNewSessionInNewWindow(): Promise<void> {
|
||||
if (!canOpenSessionWindow() || typeof window.hermesDesktop.openNewSessionWindow !== 'function') {
|
||||
return
|
||||
}
|
||||
|
||||
await openWindow(() => window.hermesDesktop.openNewSessionWindow(), 'Could not open new session window')
|
||||
}
|
||||
|
||||
@@ -724,7 +724,7 @@ platform_toolsets:
|
||||
# # allowed_chats: ["-1001234567890"]
|
||||
# extra:
|
||||
# disable_link_previews: false # Set true to suppress Telegram URL previews in bot messages
|
||||
# rich_messages: false # Opt in to Bot API 10.1 rich messages; default uses legacy MarkdownV2
|
||||
# rich_messages: false # Bot API 10.1 rich messages (tables/task lists/details/math); default true, set false to force legacy MarkdownV2
|
||||
#
|
||||
# Discord-specific settings (config.yaml top-level, not under platforms:):
|
||||
#
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
# Ink TUI — diagnostic environment flags
|
||||
|
||||
Non-secret behavioral knobs for the Ink engine (`ui-tui/`). These are
|
||||
**environment overrides**, not `.env` secrets — set them in your shell for a
|
||||
session, or `export` them in your shell rc to make them sticky. They mirror the
|
||||
OpenTUI engine's flags (`docs/opentui-env-flags.md`) so a single switch covers
|
||||
both engines.
|
||||
|
||||
| Flag | Default | What it does |
|
||||
|---|---|---|
|
||||
| `HERMES_TUI_DIAGNOSTICS` | off | Master diagnostics switch. Turning it on enables the developer/profiling surface across the TUI — including the memory self-sampler below. One `export HERMES_TUI_DIAGNOSTICS=1` in your shell rc covers **every** session you start, on **either** engine. |
|
||||
| `HERMES_TUI_MEMLOG` | = `HERMES_TUI_DIAGNOSTICS` | In-process 1Hz memory self-sampling (`ui-tui/src/lib/memlog.ts`) → `~/.hermes/logs/memwatch/<boot>-<pid>.jsonl`. Defaults to the master switch; set `=1` / `=0` to force it on/off independently. |
|
||||
|
||||
## What the memory trace captures
|
||||
|
||||
Each Ink session, when sampling is enabled, appends one JSON line per second to
|
||||
its own file under `~/.hermes/logs/memwatch/`, keyed by boot time + pid:
|
||||
|
||||
```json
|
||||
{"t":1781514892,"rss_kb":92148,"heap_used_kb":7234,"external_kb":2378}
|
||||
```
|
||||
|
||||
- `t` — unix seconds.
|
||||
- `rss_kb` — resident set size (the number that matters for the native-RSS-gap
|
||||
story: rss climbing while heap stays flat is the #15141-class signal).
|
||||
- `heap_used_kb` — V8 heap in use.
|
||||
- `external_kb` — off-heap (buffers, native allocations).
|
||||
|
||||
**Ink emits no `mounted` / `peak_mounted` field.** Those are OpenTUI's
|
||||
windowing dev counters; Ink has no windowing, so it logs the rss/heap/external
|
||||
core only. `memwatch-report.mjs` treats `mounted` as optional, so Ink lines
|
||||
aggregate cleanly alongside OpenTUI's.
|
||||
|
||||
## Why this exists — cross-engine memory comparison
|
||||
|
||||
The filename scheme, directory, and line schema are **byte-compatible with
|
||||
OpenTUI's collector** (`ui-opentui/src/boundary/memlog.ts`). Both engines write
|
||||
to the same `~/.hermes/logs/memwatch/` directory, so one aggregator reads both:
|
||||
|
||||
```sh
|
||||
# enable on either/both engines (master switch covers both)
|
||||
export HERMES_TUI_DIAGNOSTICS=1
|
||||
HERMES_TUI_ENGINE=ink hermes --tui # Ink session → its own .jsonl
|
||||
HERMES_TUI_ENGINE=opentui hermes --tui # OpenTUI session → its own .jsonl
|
||||
|
||||
# fleet table across BOTH engines' sessions:
|
||||
cd ~/github/tui-bench && node memwatch-report.mjs
|
||||
```
|
||||
|
||||
This is what makes a true side-by-side **real-world** memory arc possible —
|
||||
cold floor → load → plateau/leak — instead of comparing OpenTUI dogfood traces
|
||||
against an Ink harness with no equivalent data.
|
||||
|
||||
## Cost & safety
|
||||
|
||||
- ~50 bytes/s when on; one `process.memoryUsage()` + one short append per
|
||||
second. The interval is **unref'd** — it never keeps the process alive.
|
||||
- 14-day retention: older traces are pruned (best-effort) at start.
|
||||
- **Every failure path disables the logger silently.** Diagnostics must never
|
||||
break the TUI — this is the one place the "errors propagate" rule is
|
||||
intentionally inverted, matching the OpenTUI collector.
|
||||
- Off by default: regular users write nothing.
|
||||
|
||||
## Getting a meaningful trace
|
||||
|
||||
A short scroll-through won't show growth. For a comparison against OpenTUI's
|
||||
4–5h sessions, drive a tool-heavy 2–3h Ink session as the floor (see
|
||||
`docs/plans/opentui-ink-asymmetry-note.md` for why the harness ≠ dogfood data).
|
||||
@@ -0,0 +1,120 @@
|
||||
# Handoff — OpenTUI memory + UX, continuing on the canonical branch
|
||||
|
||||
**You are continuing the Hermes OpenTUI engine work.** This is the base operating manual; the
|
||||
user (glitch) appends specific tasks on top. Read it, then read the repo docs it points to. It
|
||||
assumes NO prior transcript/memory.
|
||||
|
||||
## Where things are
|
||||
|
||||
- **Canonical branch: `feat/opentui-native-engine`** (the draft PR to main, #42922).
|
||||
`feat/opentui-memory-window` is a synonym at the *same tip* — they were consolidated. Treat
|
||||
native-engine as canonical; if you work from memory-window, periodically
|
||||
`git push origin HEAD:feat/opentui-native-engine` to keep them in sync, or just use native-engine.
|
||||
- The native engine source is **`ui-opentui/`**; the legacy Ink engine is `ui-tui/` (shipping
|
||||
default, untouched by this campaign). The Python gateway is `tui_gateway/`, launcher
|
||||
`hermes_cli/main.py`.
|
||||
- **The worktree is often the user's LIVE global `hermes`** (`~/.local/bin/hermes` symlinks into a
|
||||
worktree's `.venv`). Consequences: (1) NEVER leave the worktree in a half-merged/conflicted state
|
||||
— a new `hermes` session would fail to build; (2) after you land source changes, rebuild
|
||||
`dist/main.js` so the next session picks them up; (3) `hermes-stable` is the flip-back to the
|
||||
stock `~/.hermes/hermes-agent` install if you need to bypass the worktree.
|
||||
- Backups of pre-merge branch states exist as `backup/*` refs (recoverable via `git reset`).
|
||||
|
||||
## Runtime, build, gate (Node 26 — NOT Bun; the port is done)
|
||||
|
||||
```sh
|
||||
export PATH="$HOME/.local/share/fnm/node-versions/v26.3.0/installation/bin:$PATH"
|
||||
cd ui-opentui && node scripts/build.mjs # → dist/main.js (esbuild + Solid/JSX)
|
||||
HERMES_TUI_MOUSE=1 node --experimental-ffi --no-warnings dist/main.js # launch; quit = double Ctrl+C
|
||||
cd ui-opentui && npm run check # THE GATE: prettier+eslint(typed)+vitest (~700). Judge by `echo $?`, never a piped tail.
|
||||
```
|
||||
|
||||
Never run bun here. Never run `hermes update` in the worktree (it flips the branch — recovery is
|
||||
painful). Never broad-pkill tui_gateway (other live sessions). Host RAM ~15GB, often <5GB free —
|
||||
run benches SEQUENTIALLY (the harness already wraps SUTs in `systemd-run … MemoryMax=2G`).
|
||||
|
||||
## The docs that are the source of truth (read, and KEEP UPDATED as you change things)
|
||||
|
||||
- `docs/opentui-memory-story.md` — ELI5 of the whole memory architecture (primitives + every decision).
|
||||
- `docs/plans/opentui-transcript-windowing.md` — windowing design (S1 spacers, S2 append-time), the
|
||||
`correctionIsLegal` zero-jank law, pre-registered gates, SHIPPED status + S3 backlog.
|
||||
- `docs/opentui-env-flags.md` — the consolidated env-flag ledger (master switch / user / dev / plumbing).
|
||||
- `docs/opentui-upstream-alignment.md` — forkless invariant, `boundary/` shim ledger, the per-release
|
||||
OpenTUI upgrade playbook (native-yoga is coming upstream — re-tune windowing margins when it lands).
|
||||
- the bench suite (cells, harness, live-attach, memwatch) now lives in its own
|
||||
repo: **tui-bench** (`github.com/NousResearch/tui-bench`); see its `README.md`.
|
||||
- `ui-opentui/README.md` — Node 26 onboarding (fnm setup that doesn't disturb other projects).
|
||||
- `docs/plans/ink-memory-adversarial-review.md` — Ink's memory weaknesses (F1–F10, the turnabout).
|
||||
- `docs/plans/gateway-death-forensics.md`, `docs/plans/workorder-2026-06-11-results.md`,
|
||||
`docs/plans/rebase-from-main-spec.md` — forensics, the merge-bar verdict, the rebase plan.
|
||||
|
||||
## Workflow (this is how the last 60+ commits were produced with ~zero rework)
|
||||
|
||||
1. **Subagent-driven** (skill: `subagent-driven-development`): one implementer per task with a TIGHT
|
||||
file fence ("you own exactly these files; `git diff --cached --stat` before commit, abort on
|
||||
out-of-fence"), a mandatory `opentui` skill read FIRST for any renderable work, and a gate judged
|
||||
by exit code. Verify the self-report YOURSELF (re-run the gate, read the riskiest hunks, check the
|
||||
commit file-list) — a subagent "✅ done" is a claim, not a fact.
|
||||
2. **Adversarial review** after a task: a fresh read-only reviewer (Explore-type) with NAMED attack
|
||||
surfaces. Then ADJUDICATE in code — reviewers over-flag; ~half of "blockers" don't survive a read.
|
||||
3. **Parallel implementers are safe ONLY with disjoint file fences.** Read-only recon agents
|
||||
parallelize freely.
|
||||
4. **Live smoke catches what headless can't** — tmux + the `tmux-pane-screenshot` skill for real
|
||||
colored frames. The demo: `node scripts/build.mjs scripts/demo.tsx .demo` then
|
||||
`DEMO_TOTAL=2000 … node --experimental-ffi --no-warnings .demo/demo.js`.
|
||||
5. Commit format `opentui(v6): …`, **NO attribution lines**. The user's standing instruction is
|
||||
"commit + push as you land things" — honor it; otherwise don't push without asking. Edit large
|
||||
load-bearing files (the Python launcher, `store.ts`) DIRECTLY, never via subagent.
|
||||
|
||||
## Dogfooding (the user works on this FROM the hermes TUI)
|
||||
|
||||
`export HERMES_TUI_DIAGNOSTICS=1` in the shell rc turns on, for every session: the `/mem` +
|
||||
`/heapdump` slash commands, window-stats, and **fleet memory self-logging** to
|
||||
`~/.hermes/logs/memwatch/<boot>-<pid>.jsonl`. Aggregate all sessions with
|
||||
`node memwatch-report.mjs` from the **tui-bench** repo
|
||||
(`github.com/NousResearch/tui-bench`) (per-session baseline/peak/slope + SLOPE/PEAK/MOUNTED anomaly
|
||||
flags). Chase a flagged session with tui-bench's `live-attach.sh <pid> --heap`. The discipline: live
|
||||
anomaly → encode as a bench cell → fix → validate against live sessions again.
|
||||
|
||||
## Current state (2026-06) + the ranked backlog
|
||||
|
||||
Windowing SHIPPED: 2k-msg peak ~300MB (was 686; Ink 234), scroll p99 6ms, cap restored 1000→3000,
|
||||
determinism digest unchanged, peak mounted ~31 rows. Live sessions peak <200MB. The transcript is no
|
||||
longer the biggest lever — the ~160MB floor is ≈104MB Node+OpenTUI runtime + **≈55MB tool/skill
|
||||
catalogs hydrated at boot**. Ranked next levers:
|
||||
|
||||
1. **W3 — 1GB V8 heap default** (small, ~free): set the unconstrained default in
|
||||
`_resolve_tui_heap_mb`; both engines are Node now so both inherit it. Ink half = separate gated
|
||||
commit (shipping engine). Measured −90MB at bench scale.
|
||||
2. **cg_peak harness fix** (small): the cgroup `memory.peak` field is polluted (shared across runs) —
|
||||
reset/scope it before quoting tui-bench's `report.html` again. Trust `vmhwm_kb` + `samples[].rss_kb`.
|
||||
3. **New bench cells** (before W1, as its baselines): `resume-1900` (real p99 shape: time-to-first-
|
||||
paint + post-hydration RSS) and `10MB-tool-output` (the F1 byte-unbounded class). Run BOTH engines.
|
||||
4. **Catalog lazy-load** (new, promoted by live data): don't hydrate 1,185 tools at boot — fetch on
|
||||
picker-open. Attacks the ≈55MB floor; pays on EVERY session (median is 20 msgs). Likely cheaper
|
||||
than W1.
|
||||
5. **W1 thin renderer** (structural, biggest): bodies live in the gateway (SQLite); TUI keeps ~300B
|
||||
stubs + fetches bodies for the window only. Design the gateway windowed-read RPC FIRST. WATCH: `/copy`
|
||||
and the ⧉ block-copy read store parts — they need a fetch-on-demand fallback or W1 ships a copy regression.
|
||||
6. **Standing**: when native-yoga OpenTUI ships, run the upgrade playbook (re-bench, re-tune margins,
|
||||
audit the shim ledger). Three questions to relay to the OpenTUI maintainer are in the alignment doc.
|
||||
|
||||
## What NOT to do
|
||||
- Don't copy opencode's 100-msg store cap (user's p90 session is 182 msgs — it would truncate normal use).
|
||||
- Don't reintroduce estimate-correction scroll jank (the user explicitly vetoed it; `correctionIsLegal` forbids it).
|
||||
- Don't cite the obsolete "~210MB bun renderer / +120MB" memory figures — pre-port, pre-windowing, wrong.
|
||||
- Don't push/PR without the standing OK; don't commit `.plans/` scratch unless asked.
|
||||
|
||||
## Suggested skills
|
||||
(All available from the Hermes TUI agent too — this is the dogfooding surface. Curated to the load-bearing set, not the full ~40-skill catalog.)
|
||||
- `opentui-tui-engineering` — the workflow/architecture/pitfalls layer for `ui-opentui/` (just updated).
|
||||
- `hermes-tui-architecture` — the Hermes-specific TUI facts (launch pipeline, both engines; just updated).
|
||||
- `opentui` — the offline renderable-API doc set; mandatory `skill_view` before any view/renderable code.
|
||||
- `subagent-driven-development` — the process spine for parallel/heavy work.
|
||||
- `tmux-pane-screenshot` — real colored PNG of a tmux pane for visual verification (ported
|
||||
into hermes skills 2026-06-13). Use: `bash ~/.hermes/skills/software-development/
|
||||
tmux-pane-screenshot/scripts/tshot.sh <session:win.pane> out.png 2`, then Read the PNG.
|
||||
`freeze` (~/go/bin) + the resvg rasterizer are shared/system-wide — works as-is.
|
||||
- `effect-ts` — for the Effect-at-boundary entry/lifecycle code.
|
||||
- `superpowers:brainstorming` — before committing to a memory-architecture design (e.g. W1's store split).
|
||||
- `systematic-debugging` — if a gate fails; root-cause before patching.
|
||||
@@ -0,0 +1,81 @@
|
||||
# OpenTUI env flags — the consolidated ledger
|
||||
|
||||
Every environment variable the OpenTUI TUI reads (grep-verified 2026-06-12),
|
||||
classified by who should ever touch it. The design rule shipped with this doc:
|
||||
**regular users see zero diagnostic surface by default; one master switch
|
||||
(`HERMES_TUI_DIAGNOSTICS=1`) turns all of it on when needed.**
|
||||
|
||||
## 1. The master switch
|
||||
|
||||
| var | default | effect |
|
||||
|---|---|---|
|
||||
| `HERMES_TUI_DIAGNOSTICS` | **off** | Enables the diagnostic slash commands (`/mem`, `/heapdump`). While off they're hidden from `/help` (client-side filter) and invoking them prints the enable hint rather than executing. They never appear in slash *completion* in either state — completion is gateway-driven and these are client-only commands the gateway doesn't know (an adversarial review confirmed there's no bypass path; if a SERVER command named `mem`/`heapdump` is ever added it must be gated gateway-side too — the client gate would shadow but not hide it). Also flips the *default* of `HERMES_TUI_WINDOW_STATS` to on. Not a secret — support flows are "relaunch with `HERMES_TUI_DIAGNOSTICS=1`". |
|
||||
|
||||
## 2. User-facing configuration (fine to document publicly)
|
||||
|
||||
| var | default | effect |
|
||||
|---|---|---|
|
||||
| `HERMES_TUI_ENGINE` | auto (`opentui` if Node≥26.3 + built, else `ink`) | Engine pick; also `display.tui_engine` in config.yaml. |
|
||||
| `HERMES_TUI_MOUSE` / `HERMES_TUI_MOUSE_TRACKING` / `HERMES_TUI_DISABLE_MOUSE` | on | Mouse support (wheel scroll, selection, click-to-expand). **Defers to Ink's env surface (`logic/env.ts` `resolveMouseEnabled`):** precedence is `HERMES_TUI_MOUSE_TRACKING` (toggle, force knob) > `HERMES_TUI_DISABLE_MOUSE=1` (legacy kill switch) > `HERMES_TUI_MOUSE` (OpenTUI-native alias, kept — also what the launcher sets) > default on. OpenTUI's renderer mouse is a single boolean, so Ink's granular off\|wheel\|buttons\|all collapses to on/off (the granular mode lives in `display.mouse_tracking` config). |
|
||||
| `HERMES_TUI_SCROLL_SPEED` (alias `CLAUDE_CODE_SCROLL_SPEED`) | native | Wheel-scroll speed multiplier (Ink parity). UNSET → OpenTUI's native scroll acceleration (untouched). A positive value (clamped to (0,20]) installs a constant-multiplier `ScrollAcceleration` on the transcript scrollbox (`view/transcript.tsx`). |
|
||||
| `HERMES_TUI_NO_CONFIRM` | off | Skip the destructive-action confirm step (`/clear`, `/new`) and run immediately (Ink parity, `NO_CONFIRM_DESTRUCTIVE`). Wired at the `confirm` seam (`entry/main.tsx`). |
|
||||
| `HERMES_TUI_MAX_MESSAGES` | ceiling | Scrollback rows kept in the TUI. Can LOWER the ceiling, never raise: 3000 with windowing, 1000 with windowing off (handle-table safety). |
|
||||
| `HERMES_TUI_TOOL_OUTPUT_LINES` | unlimited | Cap expanded tool-output lines (set a number to restore a cap). |
|
||||
| `HERMES_TUI_TOOL_OUTPUTS` | **on** | Keep rich tool-call OUTPUTS (full result body + raw result/args dicts). `=off` drops both the RENDER and the STORE of those bodies (Ink parity: only a one-line context preview + name/duration/error/diff survive) — the memory lever for the OpenTUI-vs-Ink retention asymmetry, and what the bench launches OpenTUI with for the fair engine-overhead comparison (W3). Diffs (file-edit) are KEPT either way. |
|
||||
| `HERMES_TUI_HEAP_MB` | cgroup-aware (default 8192) | V8 `--max-old-space-size` (MB) for BOTH engines. Highest precedence (then `display.tui_heap_mb` config, then the cgroup-75% fallback). Set it LOW for a low-mem session (still cgroup-clamped on top so it never exceeds the container); raise it to lift the ceiling. The low-mem opt-in signal that also arms `HERMES_TUI_PROACTIVE_GC` (W1). |
|
||||
| `HERMES_TUI_PROACTIVE_GC` | = low-`HERMES_TUI_HEAP_MB` (≤4096) | Idle-gated `global.gc()` for the low-mem path. Defaults ON only when a low heap cap is set (so the knobs compose); `=on`/`=off` forces it. Needs `--expose-gc` (the OpenTUI argv now carries it). Never runs mid-stream; tightens cadence above 400MB RSS but stays idle-gated. OpenTUI-only — Ink never GCs proactively (W2). |
|
||||
| `HERMES_TUI_COMPOSER_ROWS` | default rows | Composer height. |
|
||||
|
||||
## 3. Escape hatches & tuning (dev-facing, individually settable)
|
||||
|
||||
| var | default | effect |
|
||||
|---|---|---|
|
||||
| `HERMES_TUI_WINDOWING` | **on** | `0` = bit-exact pre-windowing renderer (every row mounts; cap clamps back to 1000). The A/B + regression escape hatch. |
|
||||
| `HERMES_TUI_WINDOW_IDLE_MS` | ~1000 | Idle-measure pulse cadence (the spacer-exactness march). Test knob. |
|
||||
| `HERMES_TUI_WINDOW_STATS` | = `HERMES_TUI_DIAGNOSTICS` | Exposes live/peak mounted-row counters (`globalThis.__hermesTuiWindowStats`) for tui-bench's live-attach reads. |
|
||||
| `HERMES_TUI_MEMLOG` | = `HERMES_TUI_DIAGNOSTICS` | In-process 1Hz memory self-sampling (`boundary/memlog.ts`) → `~/.hermes/logs/memwatch/<boot>-<pid>.jsonl` (rss/heap/external + mounted rows; 14-day retention). Fleet view: `node memwatch-report.mjs` from the tui-bench repo (`github.com/NousResearch/tui-bench`). The "monitor all my sessions" answer: one `export HERMES_TUI_DIAGNOSTICS=1` in your shell rc covers every session. |
|
||||
| `HERMES_TUI_LOG_LEVEL` / `HERMES_TUI_LOG_FILE` | engine defaults | Logging verbosity/destination (`/logs` reads the ring buffer regardless). Deliberately independent of the master switch — support often wants logs without the full diag surface. |
|
||||
| `HERMES_HEAPDUMP_ON_START` | off | Write one V8 heap snapshot at boot (Ink parity). A deliberate baseline-capture escape hatch that BYPASSES the diagnostics master switch; lands at `$HERMES_HOME/logs/opentui-heap-<ts>.heapsnapshot` and echoes the path as a system line (`entry/main.tsx`). |
|
||||
| `HERMES_TUI_NOTIFY` | on | Desktop-notification kill switch (`=0`/`false`/`off` silences the "waiting on you" pings). The ping itself goes through the renderer's native `triggerNotification` (protocol detection + tmux/Zellij wrapping); the window title is not gated by this. |
|
||||
|
||||
## 4. Internal plumbing (set by the launcher/tui-bench/tests — humans never set these)
|
||||
|
||||
| var | set by | effect |
|
||||
|---|---|---|
|
||||
| `HERMES_PYTHON`, `HERMES_PYTHON_SRC_ROOT`, `HERMES_CWD` | launcher / bench | Which gateway python + repo root + cwd the TUI spawns against (the bench's fake-gateway seam). |
|
||||
| `HERMES_TUI_ACTIVE_SESSION_FILE` | launcher/bench | Session handoff file. |
|
||||
| `HERMES_TUI_RESUME`, `HERMES_TUI_QUERY`, `HERMES_TUI_PROMPT`, `HERMES_TUI_IMAGE`, `HERMES_TUI_FAKE` | launcher/tests | Resume-at-boot; seeded prompt (`--tui "prompt"`: launcher sets `HERMES_TUI_QUERY`, the engine reads QUERY > the `HERMES_TUI_PROMPT` alias > a bare argv tail — `logic/env.ts` `startupPrompt`); seeded image PATH (`--image`: `HERMES_TUI_IMAGE`, `image.attach`ed before the prompt — `startupImage`, attach in `postSessionSetup`); fake-mode. |
|
||||
| `HERMES_AUTO_HEAPDUMP*` (`_COOLDOWN_MS`/`_MAX_BYTES`), `HERMES_HEAPDUMP_DIR`, `HERMES_HEAPDUMP_MAX_BYTES` | — | **NOT read by the OpenTUI engine (deliberate).** The engine ports Ink's #34095 silent-death early-WARNING (a transcript system line, `boundary/memoryMonitor.ts`) but NOT the auto heap-SNAPSHOT capture — the always-on memlog NDJSON trace is the diagnosis path, and its rss-vs-heap divergence is the better diagnostic for the native-RSS leak class (#15141) a V8 snapshot captures poorly. So the #41948 disk-fill safety set (gate/cooldown/byte-cap/dir) has no consumer here. `HERMES_HEAPDUMP_ON_START` (manual one-shot, §3) is the only heapdump knob the engine honors. |
|
||||
| `HERMES_TUI_RPC_TIMEOUT_MS`, `HERMES_TUI_STARTUP_TIMEOUT_MS` | tests/CI | Protocol timeouts. |
|
||||
| (`ui-tui` only) `HERMES_TUI_MEMSAMPLE_FD/MS` | bench | Ink fd-3 node sampler. |
|
||||
|
||||
## 5. Ink flags NOT ported — handled natively or out of scope
|
||||
|
||||
These exist on the legacy Ink TUI (`ui-tui/`) and are deliberately **not** read
|
||||
by the OpenTUI engine. Documented so a missing flag reads as a decision, not a gap.
|
||||
|
||||
| Ink flag | why not ported |
|
||||
|---|---|
|
||||
| `HERMES_TUI_TRUECOLOR` | OpenTUI core does COLORTERM/truecolor detection natively — the Ink force-truecolor hack is a fork workaround we shed. |
|
||||
| `HERMES_TUI_FORCE_OSC52` | OpenTUI core owns OSC52 clipboard as a primitive; no fallback hint needed. |
|
||||
| `HERMES_TUI_INLINE` / `HERMES_TUI_TERMUX_MODE` / `HERMES_TUI_TERMUX_FAST_ECHO` | Termux/primary-buffer accommodations. OpenTUI's native FFI floor (Node ≥26.3 + `--experimental-ffi`) is absent on Termux, so those sessions stay on **Ink** — these are correctly N/A for the OpenTUI engine. |
|
||||
| `HERMES_TUI_FPS` | Ink FPS overlay; the OpenTUI equivalent is the diag/window-stats surface (`HERMES_TUI_WINDOW_STATS`). Not parity-critical. |
|
||||
| `HERMES_DEV_CREDITS` / `HERMES_DEV_PERF*` | Dev-only throwaway scaffolding (live-spend readout, perf logging) — not user parity. |
|
||||
| `HERMES_BIN` / `HERMES_TUI_GATEWAY_URL` / `HERMES_TUI_SIDECAR_URL` | External-CLI / remote-gateway-URL overrides. OpenTUI spawns its gateway via the Effect boundary (`liveGateway.ts`) and does not shell out to `hermes` or take an external gateway URL. |
|
||||
| `HERMES_VOICE` | Voice mode is tracked on the OpenTUI parity backlog separately, not here. |
|
||||
|
||||
## How the pieces compose (the support script)
|
||||
|
||||
- Regular user, normal day: zero flags, zero diagnostic commands visible.
|
||||
- "My TUI feels heavy" support flow: `HERMES_TUI_DIAGNOSTICS=1 hermes` → `/mem`
|
||||
for the live numbers, `/heapdump` for a snapshot to attach, window stats
|
||||
exposed for tui-bench's `live-attach.sh <pid>` to read.
|
||||
- Developer profiling: same master switch + the individual knobs
|
||||
(`HERMES_TUI_WINDOWING=0` A/B, `WINDOW_IDLE_MS` tuning) as needed.
|
||||
- Anything in section 4 appearing in a user-facing doc is a bug.
|
||||
|
||||
Gating implementation: `logic/env.ts` (`diagnosticsEnabled()`),
|
||||
`logic/slash.ts` (`DIAGNOSTIC_COMMANDS` — dispatch hint, help + completion
|
||||
filtering), `view/transcript.tsx` (stats default). Tests:
|
||||
`slash.test.ts` (gating both states), `utilityCommands.test.ts` (commands
|
||||
themselves, gate enabled suite-wide).
|
||||
@@ -0,0 +1,207 @@
|
||||
# How the OpenTUI transcript got from 686MB to ~300MB — the full story
|
||||
|
||||
*For: glitch. Branch: `feat/opentui-memory-window`. Everything here is measured,
|
||||
not vibes; every number has a result JSON in the **tui-bench** repo's `results/` (`github.com/NousResearch/tui-bench`).*
|
||||
|
||||
---
|
||||
|
||||
## 1. The cast of characters (the primitives, bottom-up)
|
||||
|
||||
To understand where the memory went, you need to know who's holding it. Six
|
||||
layers, from the screen up:
|
||||
|
||||
**The terminal grid.** Your terminal is a spreadsheet of character cells.
|
||||
Nobody pays per-message here — tmux holds ~5MB flat no matter how long the
|
||||
session is (we measured). The terminal is never the problem.
|
||||
|
||||
**The OpenTUI native renderer (Zig).** A compiled library that owns the
|
||||
"frame buffer" — the grid of cells about to be painted. Every piece of text the
|
||||
TUI shows lives in a native **TextBuffer** (the characters + their colors),
|
||||
viewed through a **TextBufferView**, styled by a **SyntaxStyle**. Each of those
|
||||
is a **native handle** — a ticket into one global table that has only **65,535
|
||||
slots, total, ever** (16-bit indices — like a coat check with 65k hooks).
|
||||
Destroying a renderable returns its tickets, so the constraint is not "how much
|
||||
have you ever created" but **"how much is alive right now."**
|
||||
|
||||
**Renderables.** OpenTUI's UI objects — `<text>`, `<box>`, `<markdown>`,
|
||||
`<code>`, `<scrollbox>`. One transcript row (a message with its tool calls,
|
||||
markdown, code blocks, copy chips) is a *tree* of these: **~16 text renderables
|
||||
≈ 47 native handles ≈ ~250–340KB of RSS, per row.** This is the number that
|
||||
drives everything. 1,400 mounted rows × 47 handles = table full = the crash we
|
||||
root-caused last week.
|
||||
|
||||
**Yoga (the layout engine, WASM).** Every renderable also has a Yoga node —
|
||||
Yoga is the flexbox calculator that decides where boxes go. OpenTUI ships it
|
||||
compiled to **WebAssembly**, and WASM has a brutal property: its memory can
|
||||
**grow but never shrink** back to the OS. So the peak number of
|
||||
*simultaneously-mounted* renderables sets a high-water mark you pay **forever**,
|
||||
even after everything is destroyed. (Fun fact from this week's forensics: we
|
||||
spent two days believing Ink had this disease. It doesn't — our Ink fork swapped
|
||||
Yoga-WASM for a plain TypeScript port at fork creation. **We** are the ones
|
||||
running layout in WASM. The accusation was true; we just had the defendant
|
||||
wrong.)
|
||||
|
||||
**Solid (the view framework).** Renders each store message into a row via
|
||||
`<For>`. The property we exploit: Solid mounts/unmounts *surgically* — remove a
|
||||
row from what the component returns and Solid destroys exactly that row's
|
||||
renderables (returning its handles and freeing its Yoga nodes), touching
|
||||
nothing else. No virtual-DOM diffing, no collateral re-renders.
|
||||
|
||||
**V8 (the JavaScript engine) + the store.** The store keeps every message as JS
|
||||
strings/objects. V8's garbage collector is *lazy by design*: with the default
|
||||
8GB ceiling we launch with, it sees no reason to clean up aggressively, so RSS
|
||||
includes a lot of "collectible but not yet collected" garbage. Cheap to fix,
|
||||
worth real MB (measured below).
|
||||
|
||||
**The scrollbox.** One detail that fooled everyone at some point:
|
||||
`viewportCulling` (on by default) skips *drawing* offscreen rows — but they stay
|
||||
fully **mounted**: handles held, Yoga nodes alive, memory paid. Culling saves
|
||||
paint time, not memory. That misunderstanding is half the reason the "rolling
|
||||
store cap" was expected to be enough, and wasn't.
|
||||
|
||||
## 2. Why it was 686MB
|
||||
|
||||
Simple arithmetic. The old TUI mounted **every message in the store** as a full
|
||||
renderable tree. 2,000 messages × ~16 renderables × (handles + Yoga nodes +
|
||||
text buffers + V8 objects) ≈ 670–690MB, growing ~300MB per 1,000 messages. And
|
||||
at ~1,400 rows the handle table filled: first a hard crash (exit 7), then —
|
||||
after our containment fix — survival with **unstyled text** past that point,
|
||||
plus a cap clamped from 3,000 rows down to 1,000 as the price of not crashing.
|
||||
|
||||
Ink, meanwhile, sat at ~234MB at the same workload, because Ink only ever
|
||||
mounts the rows near your viewport (~84–400 live nodes). Its memory is the
|
||||
*data* plus some caches — not the *view*.
|
||||
|
||||
## 3. The decisions, in order
|
||||
|
||||
### Decision 1: virtualize the view, don't starve the store
|
||||
|
||||
Two ways to cut view memory: keep fewer messages (opencode's answer — they keep
|
||||
100 and delete the rest from memory; transcript truth lives on their server), or
|
||||
keep all messages but only *materialize* the ones near the viewport. You vetoed
|
||||
the first (your p90 session is 182 messages — a 100-row store truncates normal
|
||||
sessions), so: **windowing**. Notably the OpenTUI devs confirmed this week that
|
||||
framework-level virtualization is the intended path — the engine doesn't ship
|
||||
it out of the box, and opencode never built it. We did.
|
||||
|
||||
### Decision 2: exact heights, recorded at unmount — never estimates in your face
|
||||
|
||||
This is the load-bearing idea, and it's where we beat Ink at its own game.
|
||||
|
||||
The hard problem of any virtualized list: an unmounted row still needs to
|
||||
occupy its correct *height*, or the scrollbar lies and content jumps. Ink
|
||||
solves it by **guessing** heights and correcting after measurement — those
|
||||
corrections are precisely the 83–101ms scroll stutters you hate. You explicitly
|
||||
vetoed "estimate-correction jank" as a model.
|
||||
|
||||
Our advantage: OpenTUI lays out with real, queryable heights. So when a row
|
||||
scrolls out of the window, we record its **exact laid-out height** (an
|
||||
`onSizeChange` hook fires inside layout, pre-paint) and replace the row with an
|
||||
empty `<box height={exactly-that}/>` — a **spacer**: one Yoga node, zero text
|
||||
buffers, zero native handles. Think of a bookshelf where books you're not
|
||||
reading are swapped for cardboard sleeves cut to *exactly* the book's
|
||||
thickness: the shelf never shifts, and you can't tell from across the room.
|
||||
|
||||
The window is your viewport ± one viewport of margin (plus hysteresis so it
|
||||
doesn't thrash at the edges). Scroll near a spacer and the real row remounts —
|
||||
at the recorded height, so nothing moves.
|
||||
|
||||
And one **law**, written into the code as `correctionIsLegal`: a spacer's
|
||||
height may only ever be corrected where you *cannot see it* — fully above the
|
||||
viewport (with the scroll position compensated in the same frame, so the world
|
||||
doesn't move) or fully below it. A correction that would shift visible content
|
||||
is forbidden, structurally. Jank isn't tuned down; it's outlawed.
|
||||
|
||||
### Decision 3 (the S2 insight): adjudicate on *append*, not just on scroll
|
||||
|
||||
S1 alone got 686 → 518MB. Why not more? Because of *when* windowing decided.
|
||||
S1 re-decided the window when you **scrolled**. But during a streaming burst —
|
||||
an agent turn dumping hundreds of rows — you don't scroll; rows arrive, each
|
||||
mounting fully, and only get demoted later. That transient pile-up is mostly
|
||||
invisible in steady-state numbers… except for Yoga-WASM, where **the transient
|
||||
peak is permanent** (memory never shrinks). The burst was quietly ratcheting
|
||||
the floor.
|
||||
|
||||
S2 makes the window recompute on **transcript growth**: while you're pinned at
|
||||
the bottom, the window anchors to the content *bottom*, so a row that falls
|
||||
more than a margin behind the live edge becomes a spacer the moment it's
|
||||
measured — not whenever you next scroll. Measured result: across a 1,500-row
|
||||
burst, the peak number of simultaneously-mounted rows is **31**.
|
||||
|
||||
Same trick for **resume**: opening a 2,000-message session used to mount all of
|
||||
it (transient peak again — paid forever). Now resume mounts only the bottom
|
||||
window; everything above starts as spacers using a line-count estimate, and an
|
||||
idle-time "measure march" quietly mounts ten rows at a time near the window
|
||||
edge, records their true heights, and swaps them back — all outside the
|
||||
viewport, all invisible by the law above.
|
||||
|
||||
### Decision 4: rows that must never be windowed
|
||||
|
||||
Windowing has to know what it's not allowed to touch:
|
||||
- **Streaming rows** — the native markdown renderer streams incrementally;
|
||||
unmounting mid-stream would restart it visibly.
|
||||
- **The bottom 30 rows** — the region you actually live in.
|
||||
- **Rows under a mouse selection** — the review caught that a lingering
|
||||
highlight originally froze windowing *forever* (memory regrowing silently).
|
||||
Fixed: only an active drag pauses swaps, and selected rows get pinned, so
|
||||
copy is byte-exact while everything else keeps windowing.
|
||||
|
||||
### Decision 5: give back the scrollback (cap 1,000 → 3,000)
|
||||
|
||||
The 1,000-row clamp existed only because mounted-rows == stored-rows and the
|
||||
handle table dies at ~1,400. With windowing, mounted ≈ 31 regardless of store
|
||||
size — so the cap went back to the originally-shipped 3,000. It's
|
||||
windowing-aware: the `HERMES_TUI_WINDOWING=0` escape hatch (which mounts
|
||||
everything again) keeps the safe 1,000.
|
||||
|
||||
### Decision 6 (measured, not yet shipped as default): right-size the V8 heap
|
||||
|
||||
Running the windowed TUI with a 512MB heap ceiling instead of 8GB forced V8 to
|
||||
actually collect: another −90MB with zero latency cost. That's queued as a
|
||||
launcher default change (~1GB), for both engines.
|
||||
|
||||
## 4. The scoreboard
|
||||
|
||||
At 2,000 messages (your real p99 session size — yes, we checked your DB:
|
||||
median session is 20 messages, p99 is 1,941):
|
||||
|
||||
| | peak memory | scroll p99 (slowest 1-in-100) |
|
||||
|---|---|---|
|
||||
| OpenTUI before | 686MB | 16ms |
|
||||
| + S1 windowing | 518MB | 16ms |
|
||||
| + S2 append/resume windowing | **300–375MB** | **6ms** |
|
||||
| Ink (reference) | 229–246MB | ~100ms |
|
||||
|
||||
At the **3,000-message stress** with the restored triple-size scrollback:
|
||||
**360MB, fully styled, scroll p99 8ms** — a workload that six days ago crashed
|
||||
the process, and three days ago survived only by dropping syntax colors.
|
||||
|
||||
Scroll got *faster* because there are simply fewer live renderables to walk.
|
||||
The determinism gate stayed **byte-identical** — the windowed TUI's settled
|
||||
frame is provably the same pixels as before. And the live smoke (2,000-message
|
||||
session: full sweep to the top, resize storm, back to bottom) returned a frame
|
||||
pixel-identical to boot, with deep history fully syntax-highlighted — something
|
||||
the pre-windowing TUI literally could not do.
|
||||
|
||||
## 5. What's honestly still open
|
||||
|
||||
- The remaining ~60–120MB over Ink is mostly the **store's JS strings** and
|
||||
process baseline — the view is no longer the problem. The structural fix is
|
||||
the **thin renderer** (W1): bodies live in the Python gateway (which already
|
||||
has them in SQLite); the TUI keeps ~300-byte stubs and fetches bodies only
|
||||
for the window. That also fixes the class of problem neither engine handles
|
||||
today: a single 10MB tool output.
|
||||
- Two accepted, documented limits: scrollbar-*jumping* deep into a freshly
|
||||
resumed session can land on estimate-height rows that snap to true height as
|
||||
they enter view (normal scrolling doesn't — the margin pre-measures; the idle
|
||||
march erodes the exposure over time), and a tool you expanded, scrolled far
|
||||
away from, then returned to will have re-collapsed (state is component-local;
|
||||
hoisting it to the store is queued).
|
||||
- Everything is behind `HERMES_TUI_WINDOWING` (default on, `0` = bit-exact old
|
||||
behavior) — a one-env escape hatch if anything feels off in real use.
|
||||
|
||||
*Where to verify: the **tui-bench** repo's `results/` (`github.com/NousResearch/tui-bench`; every number above), the design+gates doc
|
||||
`docs/plans/opentui-transcript-windowing.md`, tests in
|
||||
`ui-opentui/src/test/window.test.ts` and `transcriptWindow.test.tsx` (the
|
||||
zero-jank invariants are literal assertions: identical scrollHeight windowed
|
||||
vs not, byte-stable frames across corrections).*
|
||||
@@ -0,0 +1,432 @@
|
||||
# OpenTUI native engine — PR documentation
|
||||
|
||||
**Branch:** `feat/opentui-native-engine` · **Base:** `origin/main` (merged in; HEAD is at `~main`)
|
||||
**New engine root:** `ui-opentui/` (Node 26 + `@opentui/core` 0.4.1 + `@opentui/solid`, Effect at the boundary)
|
||||
**Legacy engine root:** `ui-tui/` (React + the `@hermes/ink` fork at `ui-tui/packages/hermes-ink/`)
|
||||
|
||||
> This is the canonical in-repo doc for the PR. The companion interactive HTML
|
||||
> write-up (`~/projects/opentui-perf-writeup/index.html`) is the case/benchmark
|
||||
> deep-dive; this doc is the reviewable text version + the four things review
|
||||
> actually needs: **(1) the LoC reduction math, (2) the measured perf deltas,
|
||||
> (3) the real UI divergence (with screenshots), (4) the non-core / kitchen-sink
|
||||
> change audit.**
|
||||
|
||||
This PR adds a from-scratch native terminal UI built on OpenTUI, intended to
|
||||
replace the React/Ink TUI **and the Ink fork we maintain alone**. It currently
|
||||
ships as a parallel engine (Ink untouched, auto-fallback), selected by
|
||||
`HERMES_TUI_ENGINE` env > `display.tui_engine` config > auto (OpenTUI when the
|
||||
host is Node ≥ 26.3 with the built bundle, else Ink). **100% parity with the Ink
|
||||
TUI is the bar.**
|
||||
|
||||
---
|
||||
|
||||
## 1. Line-of-code reduction (the headline maintenance win)
|
||||
|
||||
All counts are **git-tracked files only** (respects `.gitignore`; `dist/` and
|
||||
`node_modules/` are untracked and excluded). Measured live on this branch at
|
||||
`~HEAD`. "Code" = `.ts/.tsx/.js/.jsx` only; "total" includes config/json/md.
|
||||
|
||||
### What gets *removed* when Ink is retired
|
||||
|
||||
| Area | Files | Total lines | Code lines (ts/tsx/js) | Non-blank code |
|
||||
|---|---:|---:|---:|---:|
|
||||
| `ui-tui/src/` — Ink **consumer app** (our React/Ink view code) | 204 | 40,422 | 40,422 | 33,550 |
|
||||
| `ui-tui/packages/hermes-ink/` — **the fork** (`@hermes/ink`) | 148 | 28,167 | 28,113 | 23,718 |
|
||||
| **`ui-tui/` whole tree (tracked)** | **362** | **69,320** | **68,831** | **57,545** |
|
||||
|
||||
The `ui-tui/` whole-tree number (69,320) also folds in a handful of build
|
||||
scripts, `.prettierrc`, `package.json`, etc. The two rows above it are the
|
||||
load-bearing split:
|
||||
|
||||
- **The fork alone is 28,167 LOC across 148 files** — code we own and can never
|
||||
sync from upstream. Upstream Ink v6.8.0 `src/` is ~7,259 LOC, so the fork's
|
||||
renderer core is **~3.2× the size of stock Ink**. (Cross-checked against the
|
||||
HTML write-up's `ink-fork-analysis.json`: 28,111 LOC / 148 files — the 56-line
|
||||
delta is a single tracked JSON the file-level count includes.)
|
||||
- **The consumer app is another 40,422 LOC** — React components/hooks that only
|
||||
exist to drive Ink.
|
||||
|
||||
### What gets *added*
|
||||
|
||||
| Area | Files | Total lines | Code lines | Non-blank code |
|
||||
|---|---:|---:|---:|---:|
|
||||
| `ui-opentui/src/` — new engine (app code **+ its own tests**) | 153 | 28,763 | 28,763 | 26,495 |
|
||||
| ↳ non-test (app code only) | 97 | 16,628 | 16,628 | 15,450 |
|
||||
| ↳ tests (`src/test/`) | 56 | 12,135 | 12,135 | 11,045 |
|
||||
| Tree-sitter grammars (`python`…`toml`) | 0 | 0 | 0 | 0 |
|
||||
| **`ui-opentui/` whole tree (tracked)** | **~170** | **~34,800** | **29,614** | **27,283** |
|
||||
|
||||
> Tree-sitter grammars carry **zero repo lines**: the engine declares the 10
|
||||
> extra grammars as remote URLs (`src/boundary/parsers.manifest.json`) and
|
||||
> OpenTUI fetches+caches each `.wasm`/`.scm` on first use into
|
||||
> `~/.hermes/cache/opentui-parsers/` (à la opencode, which vendors none). An
|
||||
> earlier revision vendored them as 37,302 checked-in binary lines (10 `.wasm` +
|
||||
> 10 `.scm`); that's gone — code lines and total lines now move together.
|
||||
|
||||
### The net reduction (code lines, the honest comparison)
|
||||
|
||||
| Comparison | Removed (ts/tsx/js) | Added (ts/tsx/js) | Net change |
|
||||
|---|---:|---:|---:|
|
||||
| **Incl. fork** — retire all of `ui-tui/` vs add `ui-opentui/src` | −68,831 | +28,763 | **−40,068 LOC (−58%)** |
|
||||
| **Incl. fork, app-vs-app** (exclude both test suites) | −56,463¹ | +16,628 | **−39,835 LOC (−71%)** |
|
||||
| **Excl. fork** — only the Ink *consumer app* vs new engine | −40,422 | +28,763 | **−11,659 LOC (−29%)** |
|
||||
| **The fork in isolation** (the unsyncable liability we shed) | −28,113 | — | **−28,113 code lines deleted outright (28,167 incl. its 1 config file)** |
|
||||
|
||||
¹ `ui-tui/src` non-test = 28,350 LOC + fork (≈ all 28,113 code lines are non-test;
|
||||
it carries only ~54 config lines) = 56,463. (`ui-tui/src` carries 80 test files /
|
||||
12,072 LOC; the new engine carries 56 test files / 12,135 LOC.)
|
||||
|
||||
**Read it this way:**
|
||||
|
||||
- **The cleanest single number: ~−40k code lines net** (retire all of `ui-tui/`,
|
||||
add `ui-opentui/src`). That is a **~58% reduction in the TUI's
|
||||
hand-maintained surface**, and it *includes* the new engine's full 56-file test
|
||||
suite.
|
||||
- **The most important number is the fork: −28,167 LOC of unsyncable engine
|
||||
code** disappears. That is the load-bearing maintenance win — it's not just
|
||||
fewer lines, it's lines we are the *sole* maintainer of (own reconciler, ANSI
|
||||
parser, scrollbox, selection/OSC52, hand-rolled memory eviction, Yoga binding).
|
||||
- **Even excluding the fork** — i.e. if you imagine upstream Ink were free — the
|
||||
app rewrite is still a net reduction (−11,659 LOC) because the new engine
|
||||
mounts OpenTUI built-ins instead of hand-building components.
|
||||
|
||||
### Caveat on the comparison (keep it honest for review)
|
||||
|
||||
- These are **whole-tree retirements vs a single source dir add.** If/when Ink is
|
||||
deleted, the `ui-tui/` `package.json`, lockfile, and build scripts go too; the
|
||||
table counts `ui-tui/src` + the fork as the apples-to-apples "hand-maintained
|
||||
TS" figure.
|
||||
- **Tree-sitter grammars are NOT vendored.** The 10 extra grammars are declared
|
||||
as remote URLs (`src/boundary/parsers.manifest.json`); OpenTUI fetches each
|
||||
`.wasm`/`.scm` on first use of a language and caches it under
|
||||
`~/.hermes/cache/opentui-parsers/` (profile-aware, set via
|
||||
`HERMES_TUI_PARSER_CACHE` by the launcher). Registration does **zero** network;
|
||||
the fetch is lazy and off the boot critical path, and an unreachable
|
||||
GitHub/air-gapped env degrades that language to plain text — never a throw. This
|
||||
replaces an earlier revision that vendored 37k binary lines, so the repo no
|
||||
longer grows on disk for syntax highlighting. (Trade-off: first-use-per-language
|
||||
needs network to `github.com`/`raw.githubusercontent.com`; pre-seed the cache in
|
||||
a Docker build if you need offline highlighting.)
|
||||
- Python/backend LoC is **not** part of this reduction: `tui_gateway/` (~12k LOC)
|
||||
is **shared by both engines** and stays. See §4.
|
||||
|
||||
---
|
||||
|
||||
## 2. Performance (CPU / latency / memory)
|
||||
|
||||
Measured with the `tui-bench` harness driving **both engines on a real PTY
|
||||
120×40**, fake gateway feeding deterministic events, `/proc`-sampled identically,
|
||||
each SUT under `systemd-run --scope -p MemoryMax=2G -p MemorySwapMax=0`,
|
||||
sequential with a load-gate + 10s cooldown. Determinism gate **GREEN**, 71 result
|
||||
files, 0 cell errors, 3 reps/cell, `@opentui/core` 0.4.1 native-yoga
|
||||
(`libopentui.so`, no `yoga.wasm`). Every number traces to a `summary.<field>` in
|
||||
a result dir. Source: `~/projects/opentui-html/bench-numbers.json` (frozen
|
||||
2026-06-14, build under test `1ddf7a102` + WIP).
|
||||
|
||||
### Scorecard
|
||||
|
||||
| Dimension | Winner | Margin | Source cell |
|
||||
|---|---|---|---|
|
||||
| Streaming frame rate | **OpenTUI** | **~3×** (43 vs 14 fps) | `cpu800.frame_pacing` |
|
||||
| Streaming smoothness (interframe p95) | **OpenTUI** | **40ms vs ~220ms** (no ¼-second stalls) | `cpu800.frame_pacing` |
|
||||
| Scroll CPU | **OpenTUI** | **~2.7× cheaper** (134–155 vs 403–416 ticks) | `scroll3000.scroll.cpu_ticks` |
|
||||
| Cold-start floor | **OpenTUI** | ~97–103 vs ~107–109 MB | `startup.vmhwm_kb` |
|
||||
| Session-create latency | **OpenTUI** | ~151–177 vs ~204–229 ms | `startup.session_create_ms` |
|
||||
| First-byte paint | Ink | ~93 vs ~122 ms | `startup.first_byte_ms` |
|
||||
| Memory @ small/typical | Ink | OpenTUI +30–50 MB | `mem50/100/300.vmhwm` |
|
||||
| Memory @ heavy tool output | **OpenTUI** | **crossover** (258–265 vs 280–290 MB) | `results-fat-mem-*` |
|
||||
| Layout reflow latency | **Ink** | **~0ms vs ~13ms** (OpenTUI's one honest loss) | `resize3000.resize.reflow_ms` |
|
||||
|
||||
### The honest reading
|
||||
|
||||
- **OpenTUI wins everything you feel continuously** — frame rate (~3×), scroll
|
||||
CPU (~2.7×), and smoothness (no 200ms hitches; p95 40ms vs ~220ms). This is the
|
||||
lead. The single most user-perceptible difference is the stall-free stream.
|
||||
- **Memory: lead with smoothness, not raw RSS.** Ink is lighter at small/typical
|
||||
sizes (OpenTUI carries a ~102 MB irreducible Node+V8+`libopentui.so` floor, so
|
||||
it sits +30–50 MB above Ink there). But it **crosses over** under heavy tool
|
||||
output (mem300: 258–265 MB OpenTUI vs 280–290 MB Ink) because windowing beats
|
||||
Ink's mount-every-row. Real-world: 20 memwatch sessions show a flat ~108 MB
|
||||
floor and ~0 MB/h on long sessions (one 15h session, 0 MB/h; one 4.4h session
|
||||
plateaus flat at ~237 MB with mounted rows pinned at 33).
|
||||
- **The one outright loss is layout reflow** (~13ms p50 vs Ink's ~0ms; under a
|
||||
resize storm OpenTUI degrades to ~14fps/~197ms vs Ink ~26fps/~100ms). Heavier
|
||||
native renderables vs Ink's string nodes. This is a real, quantified
|
||||
optimization target — **not** a regression vs current behavior, and **not** the
|
||||
"halved 0.4.0→0.4.1" delta (we measured the absolute 12–15ms only; do not quote
|
||||
"halved" from this run).
|
||||
- **The memory fix is engine-agnostic** — a rolling display cap
|
||||
(`HERMES_TUI_MAX_MESSAGES=3000` default) that is display-only and never touches
|
||||
the model's context. Uncapped is a stress config, not real usage (10k msgs
|
||||
uncapped: 793 MB; capped sessions are flat MB/h).
|
||||
- **Gut-check vs upstream/opencode: no bugs.** Exactly one frame callback
|
||||
(early-exits cheaply), zero `writeToScrollback` for the transcript (one sticky
|
||||
`<scrollbox>` + reactive `<For>`), native `<markdown streaming>` byte-for-byte
|
||||
parity with live opencode, no reactive-read-outside-tracking-scope (the #1 Solid
|
||||
trap). Source: `docs/plans/opentui-gutcheck-verification.md`.
|
||||
|
||||
Full methodology + every cell: see the HTML write-up's benchmark sections and
|
||||
`docs/plans/opentui-endgame-benchmark-report.md`.
|
||||
|
||||
---
|
||||
|
||||
## 3. UI parity — and where the two engines genuinely diverge visually
|
||||
|
||||
100% *feature* parity is the bar (matrix in §6), but the two engines are **not**
|
||||
visually identical. The Ink TUI renders the transcript as a **box-drawing tree**;
|
||||
OpenTUI renders it **flat and marker-based**. This is a deliberate design
|
||||
divergence, captured in `ui-opentui/src/view/messageLine.tsx`:
|
||||
|
||||
> *"the view is a dark room and gold is the single lamp — it sits on the NEWEST
|
||||
> answer's `⚕` and the user's `❯`, nowhere else (older assistant glyphs demote to
|
||||
> grey: they merely happened)."*
|
||||
|
||||
Real screenshots (saved under `docs/research/opentui-screenshots/`), captured live
|
||||
on a real PTY 120×40 via the `tmux-pane-screenshot` workflow — **same session
|
||||
resumed in both engines** where possible.
|
||||
|
||||
### Legacy Ink — `docs/research/opentui-screenshots/ink-transcript.png`
|
||||
|
||||

|
||||
|
||||
- **Box-drawing tree layout.** Each turn is a nested structure: `└─ Response`,
|
||||
`└─ ▾ Tool calls (1)`, ` └─ ● Terminal("…")` — explicit corner rails and
|
||||
disclosure triangles.
|
||||
- **`┊` dotted quote-bar** prefixes assistant prose.
|
||||
- **Tool calls collapse by default** behind a `▾ Tool calls (N)` disclosure,
|
||||
nested one rail deeper.
|
||||
- **Whole assistant message tinted gold/amber** (body text is colored, not just
|
||||
the marker).
|
||||
- Right-edge scrollbar: thin `│` track + `┃`/orange thumb.
|
||||
- Status bar: `─ ready │ opus 4.8 fast high │ 0/1m │ [░░░░░░] 0% │ 25s │ voice off │ 1 session ─ ~`
|
||||
— leading dash, pipe-delimited fields, trailing `~`.
|
||||
- **No top header bar.**
|
||||
|
||||
### New OpenTUI — `docs/research/opentui-screenshots/opentui-transcript.png` (+ `opentui-toolcall.png`)
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
- **Flat, marker-based layout.** No tree rails. Assistant = `⚕` (caduceus, gold
|
||||
only on the newest answer), user = `❯` (gold chevron + gold text). Older
|
||||
assistant glyphs demote to grey.
|
||||
- **Neutral body text.** Gold is reserved for markers and inline-code accents;
|
||||
prose is grey/white (the "single lamp" rule), so the screen reads calmer than
|
||||
Ink's all-amber blocks.
|
||||
- **Tool calls render inline, expanded, on one header line:**
|
||||
`⚕ ▶ delegate_task Run the shell command `…` (/agents to monitor) · 41s (11 lines)`
|
||||
— marker, `▶` collapse triangle, bold tool name, grey arg preview, hint,
|
||||
`· duration`, `(N lines)` — and the result flows flat directly below (no nesting
|
||||
rail). Per-tool renderers exist (`view/tools/registry.tsx`) — bash/file+diff/
|
||||
read/search/skill/clarify/todo each render differently, not a uniform dump.
|
||||
- **Per-block `⧉ copy` affordance** on a quiet footer line under every settled
|
||||
assistant block and user prompt (click → copies that block's source).
|
||||
- **Top header bar:** `⚕ Hermes Agent · opentui · ready` + a gold horizontal rule
|
||||
(Ink has none).
|
||||
- Status bar (real backend): `● claude-fable-5 │ [▒▒▒] 4% │ …/lively-thrush/hermes-agent (feat/opentui-native-engine)`
|
||||
— green status dot, model, context/token bar, **right-pinned cwd + branch**.
|
||||
|
||||
### Divergence summary table
|
||||
|
||||
| Aspect | Ink (legacy) | OpenTUI (new) |
|
||||
|---|---|---|
|
||||
| Transcript structure | Box-drawing **tree** (`└─`, rails) | **Flat**, indented, marker-based |
|
||||
| Assistant marker | `└─ Response` rail + `┊` quote-bar | `⚕` caduceus glyph |
|
||||
| User marker | (rail) | `❯` gold chevron |
|
||||
| Assistant body color | Tinted gold/amber | Neutral grey/white (gold = accents only) |
|
||||
| Tool calls | Collapsed `▾ Tool calls (N)`, nested | Inline expanded header + flat result |
|
||||
| Per-tool rendering | Largely uniform | Dedicated renderers per tool |
|
||||
| Copy affordance | `/copy` command | `/copy` **+ per-block `⧉ copy`** |
|
||||
| Header bar | None | `⚕ Hermes Agent · opentui · ready` + rule |
|
||||
| Status bar | `─`/`│`-delimited, trailing `~` | dot + bars + right-pinned cwd/branch |
|
||||
|
||||
**For review:** the divergence is intentional (a design pass, not an accident),
|
||||
but it means "drop-in replacement" is true at the *feature* level, not the
|
||||
*pixel* level. A user switching engines will immediately notice the flatter,
|
||||
calmer transcript. Worth calling out explicitly so the swap isn't sold as
|
||||
visually invisible.
|
||||
|
||||
---
|
||||
|
||||
## 4. Non-core / kitchen-sink change audit (what review should scrutinize)
|
||||
|
||||
Full report: **`docs/research/opentui-noncore-change-audit.md`** (file-by-file,
|
||||
commit-by-commit, with `file:line` evidence). Summary below.
|
||||
|
||||
This PR's net footprint vs `origin/main` (two-dot diff = exactly this PR's adds,
|
||||
no main work re-included):
|
||||
|
||||
| Bucket | Files | Net diff |
|
||||
|---|---:|---:|
|
||||
| UI (`ui-opentui/`, the engine + tests) | 197 | +36,001 / −1 |
|
||||
| Docs | 8 | +1,164 / −0 |
|
||||
| **Other (the review-flag surface)** | **28** | **+3,218 / −204** |
|
||||
|
||||
The 28 "other" files are the only place this PR touches shared Hermes core. They
|
||||
classify as:
|
||||
|
||||
### ✅ CORE-OPENTUI-NECESSARY (the engine can't work without these; Ink path provably untouched)
|
||||
|
||||
- **`hermes_cli/main.py`** (+382/−5) — dual-engine launcher (engine resolution,
|
||||
Node 26 / fnm detection, `_make_opentui_argv`, heap override). Default falls
|
||||
back to Ink unless the host is OpenTUI-ready (`main.py:1685`); OpenTUI is
|
||||
dispatched *around* the Ink bootstrap, never through it (`main.py:1914-1922`).
|
||||
- **`scripts/install.sh`** (+78/−1) — `install_opentui` stage, **strictly
|
||||
best-effort** (every failure returns 0; falls back to Ink; Windows/Termux
|
||||
skipped). Ink install path unchanged.
|
||||
- **`Dockerfile`** (+21/−11) — Node 22→**26** bump (required by the `node:ffi`
|
||||
renderer) + `ui-opentui` build step. Opt-in; Ink build line preserved. **Caveat:
|
||||
the Node major bump affects the whole image (Ink + web + Playwright)** — the
|
||||
diff self-flags "verify the full image build on Node 26 in CI."
|
||||
- **`hermes_cli/_parser.py`** (+16/−2) — bare `--resume` → OpenTUI session picker;
|
||||
`--resume <id>` unchanged.
|
||||
- **`tui_gateway/server.py`** (+612/−40) — predominantly opt-in RPCs/fields the
|
||||
new engine calls (`session.peek`, `session.list` filters, `startup.catalog`,
|
||||
`diff_unified`, window-title, skin keys). Each is gated so **the Ink path is
|
||||
byte-for-byte unchanged** (`server.py:3930`, `:4254`, `:10447`). *Note:* this
|
||||
file also carries some of the cost-accounting code (below) — separable.
|
||||
|
||||
> `tui_gateway/` (~12k LOC Python) is **shared by both engines** and is **not**
|
||||
> removed when Ink is retired. Only the `ui-tui/` frontend tree goes.
|
||||
|
||||
### 🚩 FLAG FOR REVIEW — Category C, separable from an OpenTUI PR
|
||||
|
||||
These do **not** need to ship with the engine and a reviewer should ask to split
|
||||
them out:
|
||||
|
||||
1. **Provider-reported-cost accounting** (commits `85546bb9e` + `364b93a4b` +
|
||||
`e01b04de4`) — a coherent feature spanning **11 files**: `agent/usage_pricing.py`,
|
||||
`plugins/model-providers/openrouter/__init__.py`,
|
||||
`agent/transports/chat_completions.py`, `agent/agent_init.py`, `run_agent.py`,
|
||||
`agent/conversation_loop.py`, `agent/account_usage.py`, `hermes_state.py`,
|
||||
`gateway/slash_commands.py`, the cost half of `cli.py`, and the
|
||||
`_get_usage`/`_compact_usage_text` blocks of `tui_gateway/server.py` (+ 5 test
|
||||
files). Strongest evidence: commit `85546bb9e` *"gateway: capture real
|
||||
provider-reported cost (openrouter usage accounting)"* — a provider-accounting
|
||||
rework, not a renderer.
|
||||
2. **`plugins/model-providers/openrouter/__init__.py`** — sends
|
||||
`usage:{include:true}`, a provider request-shape change affecting *all*
|
||||
interfaces, not just the TUI (`openrouter/__init__.py:85-90` cites the
|
||||
OpenRouter usage-accounting docs).
|
||||
3. **Worktree lock / dirty-tree preservation** (commit `94765e48f`,
|
||||
`cli.py` + `tests/cli/test_worktree.py`, ~145 lines) — git-worktree lifecycle
|
||||
safety plumbing with **zero TUI references** (`cli.py:1391-1545`, `:1635-1713`).
|
||||
4. **`tools/clarify_tool.py`** (+16/−4) — docstring/schema-description-only fix
|
||||
(commit `16e408f3f`); applies to every interface, trivially separable.
|
||||
|
||||
### ✅ Conversation-loop / role-alternation / prompt-cache correctness verdict: **NO RISK**
|
||||
|
||||
Verified: none of `run_agent.py`, `agent/conversation_loop.py`,
|
||||
`agent/agent_init.py`, `agent/transports/chat_completions.py` touch
|
||||
message-role alternation or the prompt-cache prefix. The
|
||||
`conversation_loop.py` added lines grep clean for
|
||||
`cache_control|alternation|prompt_cach|api_messages`; the cache/alternation
|
||||
machinery (`:57`, `:660-674`, `:759`) is untouched; the PR's insertion at
|
||||
`:1809-1879` is purely additive cost bookkeeping after `cost_result`. **Prompt
|
||||
caching and strict role alternation are preserved.**
|
||||
|
||||
---
|
||||
|
||||
## 5. What this does and does NOT fix
|
||||
|
||||
**Fixes (structurally, by replacing the rendering substrate):** the renderer bug
|
||||
class — layout/scroll/input/copy/mouse/markdown/resize — plus the
|
||||
hand-maintained memory-eviction problem (windowing + Solid keyed `<For>`
|
||||
unmount→`destroy()`→`free()`), and several long-open feature requests (mouse,
|
||||
collapsible tool calls, session title/status bar, double-ESC, chronological
|
||||
thinking/tool ordering).
|
||||
|
||||
**Does NOT fix:** the gateway is unchanged — the biggest single hotspot file in
|
||||
triage is `tui_gateway/server.py`, and whole bug clusters are gateway/Python-side
|
||||
(WS write-timeout/RPC pool, MCP-failure startup freezes, shell.exec denylist).
|
||||
The engine swap addresses rendering/input/scroll/memory; **gateway bugs ride
|
||||
along.** The Effect-boundary hardening does make those failures *visible* (typed
|
||||
events → system lines instead of a frozen spinner) and the TUI auto-heals
|
||||
(crash → backoff → respawn → resume, capped 3/60s).
|
||||
|
||||
---
|
||||
|
||||
## 6. Feature parity matrix (vs the Ink TUI)
|
||||
|
||||
Verbatim, detailed, surface-by-surface with `file:line` evidence:
|
||||
**`docs/plans/opentui-ink-parity-matrix.md`** (interactive/filterable version in
|
||||
the HTML write-up). Headline state:
|
||||
|
||||
| Surface | State |
|
||||
|---|---|
|
||||
| Transcript rendering (scrollbox, markdown, code, diffs, collapsible tools, reasoning, chronological order, windowing) | **full parity (9/9)** |
|
||||
| Blocking prompts (approval/clarify/sudo/secret/confirm) | **full parity (5/5)** |
|
||||
| Theming (skins, light/dark, ANSI-256 norm) | **full parity** |
|
||||
| Mouse / copy (tracking, selection, multi-click, OSC52, click-to-expand, wheel accel) | **full parity** |
|
||||
| Resilience (crash auto-heal + resume) | **parity++ (exponential backoff)** |
|
||||
| Composer / input | near parity — **missing: external editor (Ctrl+G → `$EDITOR`)**; ghost-text autosuggest partial |
|
||||
| Slash commands | core parity — **missing: `/setup`, `/redraw`, `/plugins`, `/voice`**; `/undo` prefill + `/image` partial |
|
||||
| Status bar / header chrome | almost all closed — **missing: MCP-servers panel, profile-in-prompt** |
|
||||
| Agent surfaces | most shipped — **missing: voice indicators, browser/CDP indicator** |
|
||||
| Utility commands | **missing: `/redraw`, `/setup`**; rest present |
|
||||
|
||||
> The original PR-draft gap list was **substantially stale** — the WIP since
|
||||
> shipped context %/token bar, cost, compressions, duration, update banner, todos
|
||||
> panel, activity feed, notifications, background-task indicator, **and per-tool
|
||||
> renderers** (the "every tool renders the same" claim is false:
|
||||
> `view/tools/registry.tsx` has dedicated renderers).
|
||||
|
||||
### Genuinely-remaining parity gaps
|
||||
|
||||
- [ ] **External editor (Ctrl+G → `$EDITOR`)** — highest-impact missing composer affordance
|
||||
- [ ] MCP-servers detail panel; profile-in-prompt marker
|
||||
- [ ] Voice indicators (listening/transcribing/REC/STT) + `/voice`
|
||||
- [ ] Browser/CDP connection indicator + `/browser`
|
||||
- [ ] `/setup` wizard handoff, `/redraw`, `/plugins` hub
|
||||
- [ ] Draggable scrollbar; sticky-prompt line
|
||||
- [ ] `/undo` prefill into composer; model-picker persist-global toggle; skills-hub install/manage
|
||||
|
||||
---
|
||||
|
||||
## 7. Rollout, runtime & risks
|
||||
|
||||
- **Runtime:** plain Node 26 (FFI floor 26.3+) — one runtime, no Bun. (Note: the
|
||||
upstream OpenTUI docs say "requires Bun"; this engine deliberately runs on Node
|
||||
26's experimental `node:ffi` instead — that's the load-bearing runtime decision.)
|
||||
- **Rollback:** Ink is untouched and remains the fallback; reverting is a launcher
|
||||
decision, not a code revert.
|
||||
- **Default-engine selection:** auto-picks OpenTUI only when the host is genuinely
|
||||
set up (Node ≥ 26.3 + built bundle), else Ink; explicit env/config bypasses the
|
||||
probe.
|
||||
- **Known sharp edges:** `libopentui.so` native-lib distribution (P1 upstream:
|
||||
copies can fill `/tmp`); the Dockerfile Node major bump needs full-image CI
|
||||
verification; tree-sitter grammars are fetched from GitHub on first use and
|
||||
cached in `~/.hermes/cache/opentui-parsers/` — air-gapped hosts get plain-text
|
||||
highlighting until the cache is pre-seeded (the fetch never blocks boot and
|
||||
never throws).
|
||||
|
||||
## 8. Try it
|
||||
|
||||
```bash
|
||||
hermes # auto-selects OpenTUI when the host supports it
|
||||
HERMES_TUI_ENGINE=opentui hermes # force the native engine
|
||||
HERMES_TUI_ENGINE=ink hermes # force the legacy Ink engine
|
||||
# preview standalone (no backend), Node 26:
|
||||
cd ui-opentui && npm install
|
||||
node scripts/build.mjs scripts/demo.tsx .demo
|
||||
DEMO_TOTAL=120 HERMES_TUI_MAX_MESSAGES=80 \
|
||||
node --experimental-ffi --no-warnings .demo/demo.js # inside a TTY
|
||||
```
|
||||
|
||||
Requires Node 26.3+. On older Node / Windows / Termux it auto-falls-back to Ink.
|
||||
|
||||
---
|
||||
|
||||
## Appendix — source-of-truth files in this repo
|
||||
|
||||
| Topic | File |
|
||||
|---|---|
|
||||
| Non-core change audit (full) | `docs/research/opentui-noncore-change-audit.md` |
|
||||
| Feature parity matrix (verbatim) | `docs/plans/opentui-ink-parity-matrix.md` |
|
||||
| Benchmark report | `docs/plans/opentui-endgame-benchmark-report.md` |
|
||||
| Gut-check verification | `docs/plans/opentui-gutcheck-verification.md` |
|
||||
| Ink↔OpenTUI capture asymmetry | `docs/plans/opentui-ink-asymmetry-note.md` |
|
||||
| UI screenshots | `docs/research/opentui-screenshots/{ink,opentui}-*.png` |
|
||||
| PR description (prose) | `docs/pr-description-main-doc.md` |
|
||||
| Interactive write-up | `~/projects/opentui-perf-writeup/index.html` (out-of-repo) |
|
||||
@@ -0,0 +1,73 @@
|
||||
# Upstream alignment — how we inherit OpenTUI's performance work for free
|
||||
|
||||
Context (maintainer, 2026-06-11): opencode's 100-message cap was a November-era
|
||||
performance workaround, since obsoleted; the **next OpenTUI version ships
|
||||
native yoga** (≥2× layout performance, more improvements building on it);
|
||||
opencode does not use virtualization.
|
||||
|
||||
## The invariant that makes alignment free
|
||||
|
||||
**We are forkless and public-API-only.** The windowing layer (S1+S2) drives the
|
||||
STOCK `<scrollbox>` through documented surface only — `onSizeChange`,
|
||||
`setFrameCallback`, `scrollTop`/`viewport`/`scrollHeight`, Solid `<Show>`
|
||||
mount/unmount. Zero patches to `@opentui/core`. Every upstream release
|
||||
therefore drops in by bumping three pinned versions in `ui-opentui/package.json`
|
||||
(`@opentui/{core,keymap,solid}`, currently 0.4.0). Keep it that way: any new
|
||||
code that needs core behavior goes through a `boundary/` wrapper, never a
|
||||
patched dependency.
|
||||
|
||||
## What native yoga changes for us (and what it doesn't)
|
||||
|
||||
- **Kills the WASM ratchet** (grow-only linear memory → freeable native
|
||||
allocations). This retro-justifies S2 less, but S2's append-time windowing
|
||||
remains correct: transient mounted peaks still cost handles and RSS.
|
||||
- **Does NOT obsolete windowing.** The binding constraint is the 65,535-slot
|
||||
native handle table: ~47 handles/row × 3,000 stored rows ≈ 141k handles —
|
||||
over the table at ANY layout speed. Windowing is what makes the 3,000-row
|
||||
scrollback possible; yoga's backend is irrelevant to that math.
|
||||
- **Makes windowing feel even better**: 2× layout = cheaper margin remounts =
|
||||
smaller window margins viable and less exposure for the one accepted limit
|
||||
(estimate-height snap under scrollbar jumps). After the bump, re-tune margin/
|
||||
hysteresis against the scroll cell.
|
||||
|
||||
## The shim ledger (delete-on-upstream-fix; all in `ui-opentui/src/boundary/`)
|
||||
|
||||
| shim | what it papers over | delete when |
|
||||
|---|---|---|
|
||||
| `ffiSafe.ts` | u32 draw coords go negative under Node FFI (Bun silently wraps) — ERR_INVALID_ARG_VALUE loop | upstream clamps, or Node FFI path is officially supported |
|
||||
| `nativeHandles.ts` | SyntaxStyle exhaustion crashes mid-mount; degrade-to-unstyled | handle table widened (INDEX_BITS>16) or per-kind tables |
|
||||
| `renderer.ts` exit-signal guard | core 0.4.0 treats SIGPIPE (clipboard spawn) as an exit signal; its own uncaughtException handler allocates a handle and dies (exit-7 masking) | both fixed upstream |
|
||||
| `clipboard.ts` hardening | same SIGPIPE incident class | with the above |
|
||||
|
||||
Each is (a) isolated, (b) inert if upstream fixes the behavior, (c) worth
|
||||
reporting upstream — four concrete, reproduced, root-caused issues. Filing them
|
||||
is the cheapest alignment lever we have: it converts our workarounds into
|
||||
upstream regression tests. (Needs glitch's go-ahead — public repo activity.)
|
||||
|
||||
## The upgrade playbook (per upstream release)
|
||||
|
||||
1. Branch `chore/opentui-X.Y.Z`, bump the three pins, `npm ci`.
|
||||
2. `npm run check` (648 tests; the windowing invariants — identical
|
||||
scrollHeight ON/OFF, byte-stable frames across corrections — are literal
|
||||
assertions and will catch behavioral drift).
|
||||
3. Bench acceptance, sequential: `--cell gate` (determinism digest; EXPECT a
|
||||
new digest if upstream changed rendering — eyeball the frame, re-bless),
|
||||
`--cell mem3000 --msgs 2000` + `--cell scroll --msgs 3000` vs current
|
||||
numbers (300–375MB / p99 6–8ms), `--cell pipeline` (frame pacing ≥22fps).
|
||||
4. Shim audit: try each boundary shim OFF; delete the ones upstream fixed.
|
||||
5. Live tmux smoke (scroll sweep / resize / selection-copy), screenshots.
|
||||
6. Windowing re-tune if layout got faster: margins up or hysteresis down,
|
||||
re-run scroll cell, keep p99 ≤ 17ms gate.
|
||||
|
||||
The bench suite IS the upgrade contract — it's exactly the harness that lets
|
||||
us take every upstream improvement within a day of release, with proof.
|
||||
|
||||
## Questions worth relaying to the maintainer
|
||||
|
||||
1. Any plan to widen the 16-bit native handle table (or split per-kind)?
|
||||
That's our hard ceiling, independent of yoga.
|
||||
2. Is the Node `--experimental-ffi` path on their support radar, or Bun-only?
|
||||
(Native yoga adds new FFI surface; we run Node.)
|
||||
3. Would they take the windowing layer's core-agnostic pieces (exact-height
|
||||
spacer pattern, correction-legality rule) as a documented recipe or
|
||||
framework-level utility? We have it production-shaped with tests.
|
||||
@@ -0,0 +1,150 @@
|
||||
# OpenTUI — Background Activity: agents inspection, background panel, notifications + density
|
||||
|
||||
**Status:** SPEC (brainstormed with glitch 2026-06-13) · target branch `feat/opentui-native-engine`
|
||||
**Hard constraint:** TUI-LAYER ONLY (`ui-opentui/`). **Zero changes to `tui_gateway/server.py` or
|
||||
`run_agent.py` core.** Build only on gateway events/RPCs that already exist. Everything below was
|
||||
feasibility-checked against the live gateway surface (see "Gateway surface" §).
|
||||
|
||||
## Why
|
||||
|
||||
Dogfeedback (screenshots `iznq/qxpe/rpiw/rplj`):
|
||||
1. **Agents dashboard is too crowded** (`rplj`) — master rows dump each subagent's full multi-line
|
||||
prompt; the trace pane is squished. Inspection + transcript reading is "not great."
|
||||
2. **Background processes are basically invisible** (`qxpe`) — completions leak into the transcript
|
||||
as plain lines that read like model output; no panel, no badge, notifications are non-existent.
|
||||
3. **Input zone is too crowded** (`rpiw`) — status bar + composer + agents tray + completion menu +
|
||||
shell note stack under the transcript.
|
||||
|
||||
## Design decisions (from the brainstorm)
|
||||
|
||||
- **Two SEPARATE surfaces, ONE shared substrate.** Background *agents* (delegated subagents) and
|
||||
background *work* (detached runs + OS processes) are visually/feature-wise distinct, but share the
|
||||
underlying tracking + notification + badge plumbing.
|
||||
- **Notifications are multi-channel** on every relevant state change:
|
||||
- **(C) inline card** in the transcript — a distinct, colored, collapsed *system card*, clearly
|
||||
NOT model output (replaces today's plain-line leak).
|
||||
- **(A) ambient badge** — a live count in chrome (status-bar `bg:`/the `⚡ N agents` tray) that
|
||||
flashes on change; you pull-to-inspect. Stays visible while things run.
|
||||
- **OSC desktop** — reuse the EXISTING `boundary/termChrome.ts` (`notify`, OSC 9/99/777, already
|
||||
focus-gated so it only fires when the terminal is blurred).
|
||||
- **Agents surface = inspection only.** No foregrounding / "become the subagent" (that would change
|
||||
core subagent UX — explicitly out of scope). Scannable list + a faithful render of the *already-
|
||||
tracked* live activity (goal/model/reasoning/tool calls/progress/summary). No new fetch.
|
||||
- **Background surface = view + stop.** List runs + OS processes with status/uptime; cancel a run
|
||||
(`session.interrupt`/`subagent.interrupt`); **stop-all** OS processes (`process.stop`). Per-process
|
||||
kill and per-process logs are NOT exposed as RPCs → out of scope under the no-core rule (noted).
|
||||
- **Input density is in scope** (own phase).
|
||||
|
||||
## Gateway surface we build on (verified — all already exist)
|
||||
|
||||
| Need | Mechanism (existing) |
|
||||
|---|---|
|
||||
| Background-run lifecycle | `prompt.background` (start), `background.complete` (event) |
|
||||
| Notifications | `notification.show` / `notification.clear` events — payload `{text, level, kind, ttl_ms, key, id}` |
|
||||
| Subagent stream | `subagent.spawn_requested/start/thinking/tool/progress/complete` events (store already consumes) |
|
||||
| List OS processes | `agents.list` RPC → `{processes:[{session_id, command, status, uptime_seconds}]}` |
|
||||
| Stop OS processes | `process.stop` RPC → `kill_all()` (**all**, not per-process) |
|
||||
| Cancel a run / subagent | `session.interrupt`, `subagent.interrupt` |
|
||||
| List active sessions/runs | `session.active_list`, `session.status` |
|
||||
| Subagent trace (archived) | `spawn_tree.list/load` (already used by `/replay`) |
|
||||
| OSC desktop notify | `boundary/termChrome.ts` `notify(TermNotification)` |
|
||||
|
||||
**Honest limits (no-core constraint):** OS processes get list + stop-all only — no per-process kill
|
||||
(`process_registry.kill_process` exists but isn't an RPC) and no per-process log tail
|
||||
(`read_log` isn't an RPC). If the no-core rule is ever relaxed, each is a ~5-line additive `@method`.
|
||||
|
||||
## Architecture (Approach 1 — substrate-first)
|
||||
|
||||
```
|
||||
gateway events ──► store: backgroundActivity slice ──► derived counts/state
|
||||
│ │
|
||||
├─► notificationDispatcher ─────────┼─► (C) inline card (transcript)
|
||||
│ (card + badge + OSC) ├─► (A) ambient badge (statusBar/tray)
|
||||
│ └─► OSC via termChrome.notify
|
||||
├─► Surface 1: AgentsDashboard (revamp) — list + rich activity pane
|
||||
└─► Surface 2: BackgroundPanel (new) — runs + processes, stop
|
||||
```
|
||||
|
||||
### Shared substrate (the "underneath" both surfaces use)
|
||||
|
||||
- **`logic/backgroundActivity.ts`** (new) — pure model + reducers. Types:
|
||||
- `BackgroundRun` (from `prompt.background`/`background.complete`/`session.active_list`):
|
||||
`{ id, label, status: 'running'|'complete'|'failed'|'cancelled', startedAt, summary? }`
|
||||
- `BackgroundProcess` (from `agents.list`): `{ sessionId, command, status, uptimeSeconds }`
|
||||
- `Notification` (from `notification.show`): `{ id, key?, text, level, kind, ttlMs?, at }`
|
||||
- Pure helpers: `applyNotification`, `clearNotification(key)`, counts (`runningCount`),
|
||||
`mergeProcessList`, dedupe by `key`/`id`. Fully unit-testable (no renderer).
|
||||
- **`store.ts`** — a `backgroundActivity` slice + event handlers for `notification.show/clear`,
|
||||
`background.complete`, and a polled `agents.list` snapshot (poll only while a panel/badge is live,
|
||||
or piggyback existing cadence). Existing `subagent.*` handling is untouched.
|
||||
- **`logic/notificationDispatcher.ts`** (new, pure) — given a state-change, decide the channels:
|
||||
returns `{ card?: SystemCard, badge: delta, osc?: TermNotification }`. The boundary calls
|
||||
`termChrome.notify` for the OSC part; the store appends the card + bumps the badge.
|
||||
|
||||
### Surface 1 — Agents inspection overlay (revamp `view/overlays/agentsDashboard.tsx`)
|
||||
|
||||
- **Master list rows = ONE line each:** `<statusGlyph> <truncated goal (truncRight to width)> · <model>`.
|
||||
No multi-line prompt dump. Selected row highlighted (existing `▸` + accent).
|
||||
- **Detail pane = faithful activity transcript** of the selected agent, styled like the main
|
||||
transcript (not flat dumped lines): goal+model header, then the trace rendered by *type*
|
||||
(reasoning / tool-call+result / progress / final summary), newest last, sticky-bottom, PgUp/PgDn.
|
||||
- Requires giving `SubagentInfo.trace` light typing (`{ kind:'tool'|'reasoning'|'progress'|'summary', text }`)
|
||||
instead of `string[]`, populated where `subagent.*` events are reduced. Internal data-shape
|
||||
change only; no gateway change.
|
||||
- Keep Esc/q close, ↑↓ select. Reuse theme + `truncRight` from statusBar.
|
||||
|
||||
### Surface 2 — Background panel (new `view/overlays/backgroundPanel.tsx`)
|
||||
|
||||
- **Two sections:** *Runs* (background agent runs) and *Processes* (OS processes from `agents.list`).
|
||||
- Each row: status glyph + label/command (truncated) + uptime/elapsed + status.
|
||||
- **Actions:** `↑↓` select; on a *run* → `c` cancel (`session.interrupt`/`subagent.interrupt`);
|
||||
global **stop-all processes** (`x` → `process.stop`, confirm). Esc/q close.
|
||||
- **Access:** new client slash `/bg` (alias `/background`, `/jobs`) in `logic/slash.ts` CLIENT set →
|
||||
`store.openBackgroundPanel()`. Also reachable from the ambient badge.
|
||||
- Poll `agents.list` on open + on a light interval while open; stop polling on close.
|
||||
|
||||
### Notifications (the (C)+(A)+OSC wiring)
|
||||
|
||||
- **(C) inline card** — a new transcript element `view/notificationCard.tsx`: a bordered/colored,
|
||||
`selectable:false` system card keyed by `notification.id`, level-tinted (`info/warn/error`),
|
||||
collapsed to one line by default with the `kind` + `text`; clearable by `notification.clear` key.
|
||||
Appended into the message stream as a distinct row type (NOT a plain `system` text line). Replaces
|
||||
the current plain-line leak. (`/details` interplay: cards are chrome, always shown, never windowed.)
|
||||
- **(A) ambient badge** — `statusBar.tsx` `bg: N` segment (already reserved) bound to
|
||||
`runningCount()`; the `agentsTray.tsx` count already exists — extend it to "agents + background."
|
||||
Flash/recolor on a fresh notification (brief).
|
||||
- **OSC** — on `notification.show` with a terminal level (complete/failed), call
|
||||
`termChrome.notify({title, body})` (already focus-gated). No new escape-sequence code.
|
||||
|
||||
### Input-zone density pass (`view/composer.tsx` / `view/App.tsx`)
|
||||
|
||||
- Audit what stacks under the transcript and collapse/gate: the `⚡ N agents` tray line folds into
|
||||
the ambient badge (shrinks one line); ensure the shell-mode note, completion menu, and status bar
|
||||
don't co-stack more than necessary. Concrete rules decided with a tmux density pass (ASCII-mocked,
|
||||
approved) — kept minimal; no behavior change, just fewer competing chrome lines.
|
||||
|
||||
## Phases (implementation order — each gated + tmux-smoked + committed)
|
||||
|
||||
- **P1 — Notification substrate** (`backgroundActivity.ts` + `notificationDispatcher.ts` + store
|
||||
slice + `notificationCard.tsx` + badge wiring + OSC call). Highest visible win; the shared core.
|
||||
- **P2 — Agents inspection revamp** (`agentsDashboard.tsx` + typed `trace`). De-crowds `rplj`.
|
||||
- **P3 — Background panel** (`backgroundPanel.tsx` + `/bg` + actions). New surface.
|
||||
- **P4 — Input density pass.** Folds the tray into the badge; trims co-stacked chrome.
|
||||
|
||||
## Testing / gates (per phase)
|
||||
|
||||
- **Pure logic** (`backgroundActivity`, `notificationDispatcher`, slash `/bg` routing,
|
||||
trace-typing) → vitest unit tests, TDD where natural.
|
||||
- **Views** → headless frame tests (`renderProbe`) for the card, the de-crowded dashboard row
|
||||
format, the background panel sections; + **live tmux smoke** (`tmux-pane-screenshot`) for each
|
||||
surface using a seeded-store harness (the `uxSmoke` pattern: `store.apply`/`applyInfo`/
|
||||
`commitSnapshot` + canned events).
|
||||
- **Gate** `cd ui-opentui && npm run check` green (judge by real exit, not a piped tail) after each
|
||||
phase; rebuild `dist/main.js`; commit `opentui(v6): …` (no attribution) and push per standing instr.
|
||||
|
||||
## Out of scope (explicit)
|
||||
|
||||
- Foregrounding / "becoming" a subagent (B/C from the brainstorm) — would change core subagent UX.
|
||||
- Per-process kill + per-process log tail for OS processes — needs additive gateway RPCs (no-core veto).
|
||||
- "Collect result into transcript" for finished runs — deferred (Q6=B, view+stop only).
|
||||
- Any change to `tui_gateway/server.py` / `run_agent.py`.
|
||||
@@ -0,0 +1,248 @@
|
||||
# Plan — OpenTUI composer/UX batch (10 features)
|
||||
|
||||
> **STATUS: SHIPPED (2026-06-13).** All 10 features implemented, gate green
|
||||
> (ui-opentui 714 tests + 316 gateway + 25 cost tests), F5/F6 verified live via
|
||||
> tmux screenshot. Commits: `f4dacc68e` (F1/F2/F7/F8/F8b/F9/F10), `20d516ae9`
|
||||
> (F4/F5/F6), `9aa5e54be` (F3). Decisions taken: **D1 = cursor-aware onType**
|
||||
> (threaded `ta.cursorOffset`); **D2 = chrome cost is Nous-header-only via a new
|
||||
> `nous_header_cost_usd`, `/usage` page kept full via `real_session_cost_usd`**.
|
||||
> F10 (right-pinned cwd) was added mid-session by the user.
|
||||
|
||||
**Branch:** `feat/opentui-native-engine` · **Engine:** `ui-opentui/` (Node 26)
|
||||
**Gate:** `cd ui-opentui && PATH="$HOME/.local/share/fnm/node-versions/v26.3.0/installation/bin:$PATH" npm run check` → exit 0.
|
||||
|
||||
## TL;DR
|
||||
|
||||
Nine UX fixes for the native composer + clarify prompt. **8 of 9 are front-end-only**
|
||||
in `ui-opentui/`; only F3 (cost) touches the Python gateway. Every backend the new
|
||||
behaviour needs (`shell.exec`, `complete.path` with `@file:`/`@folder:`/fuzzy) **already
|
||||
exists** — most of this is client wiring, not new RPC surface. No new core tools, no new
|
||||
`HERMES_*` env vars, no prompt-cache impact (composer/prompt are client-render only).
|
||||
|
||||
| # | Symptom | Fix site | Backend |
|
||||
|---|---|---|---|
|
||||
| F1 | bare `/` opens the modal | `logic/slash.ts:115` `planCompletion` | none |
|
||||
| F2 | `/abs/path` text triggers slash | `logic/slash.ts:115` + `logic/skillMatch.ts` | none |
|
||||
| F3 | cost wrong / shows for non-Nous | `tui_gateway/server.py` + `agent/usage_pricing.py` | gateway |
|
||||
| F4 | can't paste until composer focused | `view/composer.tsx` onPaste/focus | none |
|
||||
| F5 | clarify ugly (no wrap, weak diff, "Other" is a row) | `view/prompts/clarifyPrompt.tsx` rewrite | none |
|
||||
| F6 | clarify arrows scroll the transcript | same rewrite (preventDefault) | none |
|
||||
| F7 | slash highlight/menu dies after line 1 | `logic/slash.ts:114` | none |
|
||||
| F8 | file mention dies after line 1 | `logic/slash.ts:114` | none |
|
||||
| F8b | `@` should be the ONLY file-mention trigger | `logic/slash.ts:93` `isPathLike` | none |
|
||||
| F9 | `!cmd` → run bash, show result | `entry/main.tsx` submit + new system render | uses existing `shell.exec` |
|
||||
|
||||
---
|
||||
|
||||
## F1 + F2 + F7 + F8 + F8b — the completion trigger (`logic/slash.ts`)
|
||||
|
||||
All five live in one ~10-line function, `planCompletion` (slash.ts:113-121). Current:
|
||||
|
||||
```ts
|
||||
export function planCompletion(text: string): CompletionPlan | null {
|
||||
if (text.includes('\n')) return null // ← F7/F8 die here
|
||||
if (text.startsWith('/')) return { from: 0, method: 'complete.slash', params: { text } } // ← F1/F2
|
||||
const word = /(\S+)$/.exec(text)?.[1]
|
||||
if (word && isPathLike(word)) { ... complete.path ... } // ← F8b: too many triggers
|
||||
return null
|
||||
}
|
||||
```
|
||||
|
||||
### F1/F2 — slash only for a real command token
|
||||
- A bare `/` (no char yet) must **not** query. Require `/` + at least one name char.
|
||||
- A `/abs/path` (slash followed by a path with more `/`) is **not** a command — it's
|
||||
text. The slash menu should only fire when the FIRST token matches the command
|
||||
grammar (`/[A-Za-z0-9][\w.-]*` — the `NAME_RE` already in skillMatch.ts:51, which
|
||||
excludes `/`). `/usr/bin` fails NAME_RE → no slash menu.
|
||||
- Concretely: replace `text.startsWith('/')` with: the text starts with `/`, and the
|
||||
first whitespace-delimited token after the `/` is non-empty AND matches `NAME_RE`
|
||||
(i.e. `/m`, `/model foo` → yes; `/`, `/usr/bin`, `/./x` → no). Reuse `slashTokens`
|
||||
/`NAME_RE` from skillMatch.ts so the trigger and the highlighter share one grammar.
|
||||
|
||||
### F7/F8 — completion must survive newlines (shift+enter)
|
||||
- `if (text.includes('\n')) return null` is the bug. It was a blunt guard so a multi-line
|
||||
paste wouldn't spam path-completion. The right rule operates on the **current line /
|
||||
current token at the cursor**, not the whole buffer.
|
||||
- The composer passes the full `plainText` to `onType`. We don't currently pass the
|
||||
cursor offset. **Decision D1 (below):** either (a) thread the cursor offset into
|
||||
`onType` and complete the token under the cursor, or (b) cheap interim — slice to the
|
||||
**last line** (`text.slice(text.lastIndexOf('\n')+1)`) and run the existing logic on
|
||||
that. (a) is correct (mid-buffer edits), (b) is 1 line and covers the reported case
|
||||
(typing at the end on line N). Recommend (a) for correctness; it also future-proofs
|
||||
@-mention mid-line.
|
||||
- Slash *highlighting* (skillMatch.ts `slashTokens`) **already scans multi-line text
|
||||
correctly** (it iterates the whole string, newline-aware via `nativeCharOffset`). So
|
||||
F7's "highlighting stopped" is really the same `planCompletion` newline bail starving
|
||||
the menu; the highlight token itself still styles. Verify in the live smoke.
|
||||
|
||||
### F8b — `@` is the only mention trigger
|
||||
- `isPathLike` (slash.ts:93) currently returns true for `@`, `~`, `./`, `../`, `/`, or
|
||||
any word containing `/`. The user wants **`@`-only** (drop `~`/`./`/bare paths as
|
||||
mention triggers). Narrow it to `word.startsWith('@')`.
|
||||
- The gateway `complete.path` (server.py:8543) already special-cases `@` richly
|
||||
(`@file:`, `@folder:`, `@diff`, `@staged`, `@url:`, `@git:`, fuzzy basename search).
|
||||
Its `~`/`./` branches become dead trigger paths from this TUI — leave the gateway code
|
||||
(Ink still uses the path forms; it's shared) but stop emitting those queries from
|
||||
ui-opentui. **No gateway change.**
|
||||
- Net: typing `@` (even bare) opens the mention menu via the `@`-bare branch at
|
||||
server.py:8555. Picking splices `@file:rel/path` etc. (existing accept path,
|
||||
`completionFrom` honoured).
|
||||
|
||||
**Tests:** extend `test/slash.test.ts` — `planCompletion('/')` → null; `planCompletion('/usr/bin')`
|
||||
→ null; `planCompletion('/model')` → complete.slash; multi-line `"a\n/mod"` → complete.slash
|
||||
on the trailing token; `"~/foo"` / `"./x"` → null (no longer path-like); `"@foo"` → complete.path.
|
||||
Keep them as behaviour assertions, not snapshots.
|
||||
|
||||
---
|
||||
|
||||
## F3 — cost: Nous-portal headers only (`tui_gateway` + `agent/usage_pricing.py`)
|
||||
|
||||
**Current:** `_get_usage` (server.py:2157-2167) sets `cost_usd` from
|
||||
`real_session_cost_usd(agent)` (usage_pricing.py:887), which sums **two** provider-reported
|
||||
sources:
|
||||
1. `agent.session_actual_cost_usd` — OpenRouter `usage.cost` accumulator.
|
||||
2. `agent.get_credits_spent_micros()` — Nous `x-nous-credits-*` header delta.
|
||||
|
||||
The TUI already **hides** the cost segment when `cost_usd` is absent (statusBar.tsx:241-243,
|
||||
`costText` returns '' when `costUsd === undefined`) — so this is purely "which sources count."
|
||||
|
||||
**User's intent (F3):** cost should come **only from the Nous portal headers**; suppress it
|
||||
for every other route (cache-token pricing is unreliable across the model long tail).
|
||||
|
||||
**Change:** make the OpenRouter accumulator source conditional on the route being Nous, OR
|
||||
drop source #1 entirely so only the header delta (source #2) feeds `cost_usd`. Source #2 is
|
||||
intrinsically Nous-only (the header only exists on Nous-portal responses), so dropping #1
|
||||
achieves "Nous-header-only" with one edit.
|
||||
|
||||
> **DECISION D2 (needs glitch's confirm):** Drop OpenRouter's `session_actual_cost_usd`
|
||||
> source from `real_session_cost_usd`? Trade-off: OpenRouter's `usage.cost` is itself
|
||||
> *provider-reported* (the real charged number, not a Hermes estimate), so OR users lose an
|
||||
> accurate readout. But it removes the cache-token guesswork the user is worried about and
|
||||
> matches "only via the headers when using nous portal" literally.
|
||||
> **Recommended default (implementing unless told otherwise):** gate source #1 so it only
|
||||
> contributes when the active route is the Nous portal (base_url == nous inference api),
|
||||
> else it's dropped. This keeps the segment Nous-only AND avoids touching shared OR/CLI
|
||||
> behaviour for the `/usage` page. If even Nous-route OR-accumulator is unwanted, collapse
|
||||
> to header-only.
|
||||
|
||||
**Scope guard:** `real_session_cost_usd` is also consumed by `/usage` page rendering
|
||||
(server.py:2237) and DB usage totals. Prefer a NEW, status-bar-specific helper
|
||||
(e.g. `nous_header_cost_usd(agent)`) wired only into `_get_usage`'s `cost_usd`, leaving the
|
||||
`/usage` accounting page untouched — so we don't regress the full cost report. Confirm with
|
||||
the gate + a gateway unit test (`tui_gateway` tests) that a non-Nous session yields no
|
||||
`cost_usd`.
|
||||
|
||||
---
|
||||
|
||||
## F4 — paste while composer unfocused (`view/composer.tsx`)
|
||||
|
||||
**Current:** the global keyboard handler reclaims focus on a *printable keystroke*
|
||||
(`isPrintableKey`, composer.tsx:415-417). A **bracketed-paste event is not a keystroke** —
|
||||
it arrives at `onPaste` only if the textarea is focused, so an unfocused composer drops it;
|
||||
the user has to click/type first.
|
||||
|
||||
**Fix:** the renderer delivers paste through the focused renderable. Two options:
|
||||
- (a) Keep focus on the composer more aggressively (opencode keeps the prompt focused via a
|
||||
reactive effect). Risky — fights transcript scroll focus.
|
||||
- (b) **Recommended:** handle paste at the renderer/global level. Check whether OpenTUI
|
||||
exposes a global paste hook (`renderer.on('paste')` or a keyboard event with
|
||||
`key.name === 'paste'` / a paste event type). If a global paste signal exists, on paste:
|
||||
`ta.focus()` then route the bytes into the existing `onPaste` logic (image / placeholder /
|
||||
insert). **Must verify the API in the `opentui` skill before coding** (skill_view
|
||||
references/docs). If only the focused-renderable paste exists, fall back to (a) scoped:
|
||||
refocus the composer whenever no overlay/prompt is open and focus drifted (a
|
||||
`createEffect` watching focus + `store.state.prompt`/overlay state).
|
||||
|
||||
**Verify in live smoke** (tmux + tmux-pane-screenshot): scroll the transcript to drop focus,
|
||||
then paste — text must land without a prior click.
|
||||
|
||||
---
|
||||
|
||||
## F5 + F6 — clarify prompt rewrite (`view/prompts/clarifyPrompt.tsx`)
|
||||
|
||||
Screenshot `/tmp/screenshots/SCR-20260613-iznq.png` confirms: long options run off the right
|
||||
edge (no wrap), options differ only by `▶`/`—` glyphs (no numbers, weak), and "✎ Other…" is
|
||||
a `<select>` row that *switches* to an input on Enter rather than being an inline input.
|
||||
|
||||
**Current:** one native `<select>` over `[...choices, {Other}]` (clarifyPrompt.tsx:61-75).
|
||||
Native `<select>` doesn't wrap long rows and (F6) doesn't `preventDefault` arrows, so they
|
||||
leak to the transcript scrollbox.
|
||||
|
||||
**Rewrite plan (verify renderable API in `opentui` skill first):**
|
||||
- Replace native `<select>` with a **custom keyboard-driven list** (a `For` over options +
|
||||
a `selected` signal + `useKeyboard` with `key.preventDefault()` on up/down/enter — same
|
||||
pattern the composer's `routeMenuKey` uses; F6 fixed by preventDefault so arrows never
|
||||
reach the scrollbox).
|
||||
- **Wrapping (F5):** render each option as a `<text>` that wraps to the box width (no fixed
|
||||
single-line). Indent continuation lines under the option label. Confirm `<text>` soft-wrap
|
||||
behaviour in the opentui skill (it wraps by default within a flex box of bounded width).
|
||||
- **Differentiation (F5):** number every option `1.` `2.` … (digit hotkeys optional, nice-to-
|
||||
have), and give the selected row the themed `selectionBg` + accent fg (the composer's
|
||||
`completionCurrentBg` model), not just a glyph. Number + background + accent = three signals.
|
||||
- **Inline custom answer (F5):** render the `<input>` **inside the same screen, always
|
||||
present** as the last "row" (an `Other:` labeled input), instead of an item that toggles.
|
||||
Selecting/focusing it lets the user type; Enter in it submits the free text. Keep the
|
||||
existing `clarify.respond {answer}` wiring. Arrow-down past the last choice lands on the
|
||||
input; arrow-up from the input returns to the list (focus handoff like the composer↔tray).
|
||||
- Keep Esc/Ctrl+C → cancel (clarifyPrompt.tsx:31-33).
|
||||
|
||||
**Reference:** opencode's selection/list components in `~/github/opencode/packages/tui` for
|
||||
the wrap + highlight + hotkey idiom; the composer dropdown (composer.tsx:441-458) for the
|
||||
in-repo highlight/selectable pattern.
|
||||
|
||||
**Tests:** `test/render.test.tsx`-style headless frame — long option wraps (frame contains the
|
||||
tail of a long choice on a 2nd line), selected row shows numbered + highlighted, custom input
|
||||
present in the same frame, arrow keys don't change scrollTop (assert transcript scroll
|
||||
unchanged), Enter on a choice → onAnswer(choice), Enter in input → onAnswer(typed).
|
||||
|
||||
---
|
||||
|
||||
## F9 — `!cmd` runs bash (`entry/main.tsx` + a system render)
|
||||
|
||||
**Backend exists:** `shell.exec` (server.py:10301) runs the command (30s timeout, dangerous/
|
||||
hardline-command guards, returns `{stdout, stderr, code}`).
|
||||
**Ink parity reference:** `ui-tui/src/app/useSubmission.ts:291` — `full.startsWith('!')` →
|
||||
`shellExec(full.slice(1).trim())` → appends a user line `!cmd` + a system line with output;
|
||||
the prompt glyph flips while the buffer starts with `!` (appLayout.tsx:178).
|
||||
|
||||
**Plan (ui-opentui):**
|
||||
- In the entry `submit` (main.tsx:517-520), add a branch BEFORE the slash check:
|
||||
`if (text.startsWith('!')) { runShell(text.slice(1).trim()); return }`.
|
||||
- `runShell(cmd)`: `store.pushUser('!' + cmd)` (echo the invocation in the transcript), then
|
||||
`gateway.request('shell.exec', { command: cmd })`; on resolve, `store.pushSystem` the
|
||||
combined `stdout`/`stderr` (or the error message / non-zero `code`); on reject,
|
||||
pushSystem the error. Detached `runFork` like `submitPrompt`. No session turn, no model call.
|
||||
- Empty `!` (just the bang) → no-op (or a hint), matching Ink.
|
||||
- **Optional polish (parity, not required):** flip the composer prompt glyph (or tint) while
|
||||
the buffer starts with `!`, like Ink's appLayout. Low-risk; do only if cheap.
|
||||
|
||||
**Tests:** entry-level/logic test that a `!`-prefixed submit routes to `shell.exec` (not
|
||||
`prompt.submit`), and the system line renders stdout. Mirror the slashMenu.test harness
|
||||
(fake gateway capturing the method).
|
||||
|
||||
---
|
||||
|
||||
## Sequencing & fences (subagent-driven; disjoint files)
|
||||
|
||||
Parallel-safe groups (disjoint file fences):
|
||||
1. **slash trigger** — `logic/slash.ts` (+ `logic/skillMatch.ts` reuse) + `test/slash.test.ts`. (F1/F2/F7/F8/F8b)
|
||||
2. **clarify** — `view/prompts/clarifyPrompt.tsx` + a clarify test. (F5/F6)
|
||||
3. **shell-exec** — `entry/main.tsx` (edit DIRECTLY — load-bearing) + system render + test. (F9)
|
||||
4. **paste focus** — `view/composer.tsx` (edit directly; verify opentui paste API first). (F4)
|
||||
5. **cost** — `tui_gateway/server.py` + `agent/usage_pricing.py` + gateway test. (F3) — Python, isolated.
|
||||
|
||||
`entry/main.tsx` and `store.ts` are edited directly, never via subagent (handoff rule).
|
||||
Each renderable change: `skill_view(opentui, references/docs/...)` FIRST. Verify every
|
||||
subagent self-report (re-run `npm run check` exit code, read the diff).
|
||||
|
||||
## Open decisions (need glitch)
|
||||
- **D1 (F7/F8):** thread cursor offset into `onType` (correct) vs. last-line slice (cheap)?
|
||||
Recommend cursor offset.
|
||||
- **D2 (F3):** drop OpenRouter cost source entirely, or gate it to the Nous route? Recommend
|
||||
Nous-route gate via a status-bar-only helper, leaving `/usage` accounting intact.
|
||||
|
||||
## Invariants to preserve
|
||||
- Per-conversation prompt caching untouched (all client-render or post-hoc gateway usage).
|
||||
- No new `HERMES_*` env var (these are behaviour, not secrets).
|
||||
- Strict no change-detector tests — assert behaviour/invariants.
|
||||
- Don't regress the `/usage` accounting page when narrowing the chrome cost source.
|
||||
@@ -0,0 +1,217 @@
|
||||
# OpenTUI — usage/credits notice in the composer chrome
|
||||
|
||||
**Status:** spec (not started) · **Engine:** `ui-opentui/` · **Author:** glitch · 2026-06-14
|
||||
|
||||
## Goal
|
||||
|
||||
Render the gateway's **usage / credits notices** as a persistent, level-tinted
|
||||
**chrome banner pinned at the top of the input zone** (directly above the status
|
||||
bar), with the same lifecycle the Ink engine already has — sticky vs TTL,
|
||||
mid-turn hold + turn-end reveal, and "flash-and-yield" for the usage bands.
|
||||
|
||||
Today the OpenTUI engine **receives** these notices but mis-renders them as
|
||||
scrolling inline transcript cards with no lifecycle. This spec fixes that without
|
||||
touching the gateway or the agent (the data already flows correctly).
|
||||
|
||||
## What already exists (verified)
|
||||
|
||||
### The wire (source of truth — do NOT change)
|
||||
The gateway emits one event for every notice, snake_case payload:
|
||||
|
||||
```
|
||||
notification.show payload { text, level, kind, ttl_ms, key, id } # tui_gateway/server.py:2878
|
||||
notification.clear payload { key } # tui_gateway/server.py:2890
|
||||
```
|
||||
|
||||
These come from `AgentNotice` (`agent/credits_tracker.py:177`). The credits
|
||||
policy (`evaluate_credits_notices`, `agent/credits_tracker.py:245`) emits exactly
|
||||
four notices — the full catalog this feature renders:
|
||||
|
||||
| `key` | `text` (already glyphed by policy) | `level` | `kind` | `ttl_ms` | lifecycle |
|
||||
|-----------------------|-------------------------------------------------|-----------|----------|----------|----------------|
|
||||
| `credits.usage` | `⚠/• Credits N% used · $X cap` (bands 50/75/90) | info/warn | `sticky` | — | flash-and-yield |
|
||||
| `credits.grant_spent` | `• Grant spent · $X top-up left` | info | `sticky` | — | flash-and-yield |
|
||||
| `credits.depleted` | `✕ Credit access paused · run /usage for balance` | error | `sticky` | — | sticky |
|
||||
| `credits.restored` | `✓ Credit access restored` | success | `ttl` | `8000` | TTL self-expire |
|
||||
|
||||
**Load-bearing facts:**
|
||||
- `text` is **already glyphed** (⚠ • ✕ ✓) by the Python policy — the renderer
|
||||
**must not** prepend another glyph. It only tints by `level`.
|
||||
- `level` includes **`success`** (green) — a level the current OpenTUI parser
|
||||
silently drops to `info`.
|
||||
- `kind` is the **lifecycle marker** (`sticky` | `ttl`), NOT a display label.
|
||||
`id` == `key` (stable per kind, not unique per emission).
|
||||
- Notices are **reconciled**: the policy emits `to_clear` (a `notification.clear`)
|
||||
then `to_show`. A band change clears `credits.usage` then re-shows it.
|
||||
|
||||
### The Ink reference behavior (what we're matching)
|
||||
`ui-tui/src/app/turnController.ts` + `appChrome.tsx`:
|
||||
- `showNotice` (`:181`): if **busy**, hold in `pendingNotice` (latest-wins);
|
||||
if idle, apply now.
|
||||
- `applyNotice` (`:213`): set the visible notice; for `kind: 'ttl'` with
|
||||
`ttl_ms > 0`, arm a self-expiry timer (clearing any prior timer first).
|
||||
- `clearNotice(key)` (`:198`): drop the visible **and** pending notice only when
|
||||
the key matches (a stale clear must not wipe a newer notice).
|
||||
- `flushPendingNotice` (`:245`): at **turn end** (only the real end sites) apply
|
||||
the held notice — its TTL clock starts here, when it first becomes visible.
|
||||
- **Flash-and-yield** (`startMessage`, `:917`): at **turn start**, if the visible
|
||||
notice's key is `credits.usage` or `credits.grant_spent`, clear it — "show
|
||||
once, then get out of the way." `credits.depleted` and others stay sticky. The
|
||||
Python `active` latch keeps the key so it won't re-fire next turn.
|
||||
- Session reset clears all notice state so session A's notice can't bleed into B.
|
||||
- Color by level: `error→error`, `warn→warn`, `success→statusGood`,
|
||||
`info→accent` (`noticeColor`, `appChrome.tsx:192`).
|
||||
|
||||
### The OpenTUI side (what we change)
|
||||
- `notification.show` → `parseNotification` → `pushNotification` → **inline card**
|
||||
in the transcript (`store.ts:832`, `notificationCard.tsx`). All kinds, no
|
||||
lifecycle. The Option B process-completion card (`kind: 'process.complete'`)
|
||||
and `background.complete` (`kind: 'background task complete'`) also use this
|
||||
path — **they must keep working unchanged.**
|
||||
- `parseNotification` coerces `level` to `info|warn|error` only
|
||||
(`backgroundActivity.ts:48`) — drops `success`.
|
||||
- Store carries `lastNotification` (OSC seam), `bgTasks`; **no** `notice` slot.
|
||||
- Theme has `accent`, `warn`, `error`, `ok`/`statusGood`, `muted`
|
||||
(`logic/theme.ts`) — `success` maps to `statusGood`.
|
||||
- Input zone layout (`view/App.tsx:140-211`): a top-bordered column —
|
||||
`<StatusBar>` → composer `<Switch>` → `<AgentsTray>`. The new banner mounts at
|
||||
`App.tsx:144`, **directly above `<StatusBar>`** (the topmost line of the chrome).
|
||||
- Turn lifecycle hooks: `case 'message.start'` (`store.ts:779`, sets
|
||||
`info.running = true`) and `case 'message.complete'` (`store.ts:811`, sets
|
||||
`info.running = false`). `clearTranscript` (`store.ts:631`) is the reset site.
|
||||
- `Date.now()` is used freely in the store (`:877`) — `setTimeout` for TTL is fine.
|
||||
|
||||
## The one design decision: routing
|
||||
|
||||
`kind` is the discriminator. **`notification.show` with `kind === 'sticky'` or
|
||||
`kind === 'ttl'` → the new chrome-notice path; every other kind → the existing
|
||||
inline-card path, untouched.** This mirrors Ink's `Notice.kind: 'sticky' | 'ttl'`
|
||||
exactly, and the credits policy sets `kind` to one of those for all four notices,
|
||||
while the process/background cards use label-strings (`process.complete`,
|
||||
`background task complete`) that are neither — so they stay inline cards. No
|
||||
gateway change, no key-prefix sniffing.
|
||||
|
||||
**Divergence from Ink (intentional):** Ink hides the notice while busy because the
|
||||
FaceTicker shares its one status slot. OpenTUI's busy face (`StatusLine`) lives in
|
||||
the transcript area, so the banner has a **dedicated row** and stays visible
|
||||
through a turn (a depletion warning shouldn't vanish mid-turn). We still **hold
|
||||
new notices** that arrive mid-turn (`pendingNotice`) and reveal them at turn end —
|
||||
matching Ink's "don't pop a fresh banner mid-stream" intent.
|
||||
|
||||
## Implementation
|
||||
|
||||
### Phase 1 — parser + type (`logic/backgroundActivity.ts`)
|
||||
1. Widen `ActivityNotification.level` to `'info' | 'warn' | 'error' | 'success'`.
|
||||
2. `coerceLevel`: also accept `'success'` (still fall back to `'info'`).
|
||||
3. Add `export function isChromeNotice(n: ActivityNotification): boolean` →
|
||||
`n.kind === 'sticky' || n.kind === 'ttl'`.
|
||||
4. `parseNotification` already maps `ttl_ms → ttlMs` and preserves `key`/`id` —
|
||||
no shape change beyond the widened level.
|
||||
|
||||
**Tests** (`backgroundActivity.test.ts` or `notificationCard.test.tsx`):
|
||||
`success` survives parse; `kind: 'ttl'` + `ttl_ms` → `ttlMs`; `isChromeNotice`
|
||||
true for sticky/ttl, false for `process.complete`/`''`.
|
||||
|
||||
### Phase 2 — store lifecycle (`logic/store.ts`)
|
||||
Add state + a private (non-reactive) timer handle in `createSessionStore`:
|
||||
- `notice: ActivityNotification | null` (visible chrome notice) — new state field,
|
||||
init `null`.
|
||||
- `pendingNotice: ActivityNotification | null` — held mid-turn, init `null`.
|
||||
- `let noticeTimer: ReturnType<typeof setTimeout> | undefined` (closure var).
|
||||
|
||||
Functions (port of `turnController`):
|
||||
- `showNotice(n)`: `state.info.running ? setState('pendingNotice', n) : applyNotice(n)`
|
||||
(latest-wins — assigning replaces any prior pending).
|
||||
- `applyNotice(n)`: clear `noticeTimer`; `setState('notice', n)`; if
|
||||
`n.kind === 'ttl' && n.ttlMs && n.ttlMs > 0`, arm `setTimeout(n.ttlMs)` that
|
||||
clears `notice` only if `state.notice?.id === n.id` (defensive guard).
|
||||
- `clearNotice(key)`: if `state.pendingNotice?.key === key` → null it; if
|
||||
`state.notice?.key === key` → clear timer + null `notice`.
|
||||
- `flushPendingNotice()`: if `state.pendingNotice` → `applyNotice` it, null pending.
|
||||
- `clearNoticeState()`: null `notice` + `pendingNotice`, clear timer.
|
||||
|
||||
Wire into the event reducer:
|
||||
- `notification.show` (`store.ts:832`): route —
|
||||
`const n = parseNotification(...); if (!n) break; if (isChromeNotice(n)) showNotice(n); else pushNotification(n)`.
|
||||
(Still record `lastNotification` for the OSC seam in **both** paths — extract
|
||||
the `setState('lastNotification', {...n})` so a chrome notice also pings a
|
||||
blurred terminal, matching the inline-card behavior.)
|
||||
- `notification.clear` (`store.ts:837`): call **both** `clearNotificationCards(key)`
|
||||
(cards) **and** `clearNotice(key)` (chrome) — a key only ever lives in one, so
|
||||
calling both is safe and avoids guessing.
|
||||
- `message.start` (`store.ts:779`): flash-and-yield — if
|
||||
`state.notice?.key === 'credits.usage' || === 'credits.grant_spent'` →
|
||||
`clearNotice(state.notice.key)`. (Do this **before** flipping `running` true so
|
||||
the read is clean.)
|
||||
- `message.complete` (`store.ts:811`): call `flushPendingNotice()` (after the
|
||||
`running = false` set, so a held notice reveals on the now-idle bar).
|
||||
- `clearTranscript` (`store.ts:631`) and any session-switch reset:
|
||||
`clearNoticeState()`.
|
||||
|
||||
Export `notice` via the store's state and `showNotice`/`clearNotice` if a test or
|
||||
future slash command needs them.
|
||||
|
||||
**Tests** (`statusNotice.test.ts`, new):
|
||||
- idle `showNotice` → `state.notice` set, no card pushed.
|
||||
- routing: `notification.show` `kind:'sticky'` → `notice` set, **no** transcript
|
||||
card; `kind:'process.complete'` → card pushed, `notice` still null.
|
||||
- mid-turn hold: `message.start` → `showNotice` → `notice` stays null,
|
||||
`pendingNotice` set → `message.complete` → `notice` revealed.
|
||||
- `clearNotice` by key drops visible + pending; non-matching key is a no-op.
|
||||
- TTL: `kind:'ttl', ttlMs:50` auto-clears (vitest fake timers).
|
||||
- flash-and-yield: visible `credits.usage` cleared on `message.start`;
|
||||
`credits.depleted` persists across a start/complete cycle.
|
||||
- `clearTranscript` resets `notice` + `pendingNotice`.
|
||||
- `success` notice keeps its level.
|
||||
|
||||
### Phase 3 — view (`view/noticeBanner.tsx` + `App.tsx`)
|
||||
New `NoticeBanner` (sibling style to `notificationCard.tsx`):
|
||||
- Props: `notice: ActivityNotification | null`, plus terminal width for truncation.
|
||||
- `<Show when={notice}>` — renders nothing when null.
|
||||
- One row, `flexShrink: 0`, `paddingLeft: 1`, `selectable={false}`.
|
||||
- Text rendered **verbatim** (glyph already present), tinted by level:
|
||||
`error→error`, `warn→warn`, `success→statusGood`, `info→accent`.
|
||||
- Truncate to width with `truncRight` (`logic/truncate.ts`) so a long notice can
|
||||
never push the composer or wrap.
|
||||
|
||||
Mount in `App.tsx:144`, the first child of the top-bordered input zone, directly
|
||||
above `<StatusBar store={...} />`:
|
||||
```tsx
|
||||
<box border={['top']} ...>
|
||||
<NoticeBanner notice={props.store.state.notice} /> {/* new */}
|
||||
<StatusBar store={props.store} />
|
||||
...
|
||||
```
|
||||
|
||||
**Tests** (`noticeBanner.test.tsx`, frame): renders the text without adding a
|
||||
glyph; warn→warn color, success→statusGood color; truncates at narrow width;
|
||||
renders an empty frame when `notice` is null.
|
||||
|
||||
### Phase 4 — parity verification + docs
|
||||
- `npm run check` green (prettier + eslint + vitest).
|
||||
- Headless frame dump: a `credits.usage` warn banner above the status bar; a
|
||||
`credits.depleted` error banner surviving a turn; a `credits.restored` success
|
||||
banner that disappears after its TTL.
|
||||
- tmux smoke per `docs/opentui-dev-handoff.md` (inject the three notices via the
|
||||
test harness / a scripted gateway event; screenshot the chrome).
|
||||
- Cross-check the four-notice catalog renders identically in tone to Ink's
|
||||
`appChromeStatusRule` (color-by-level, no double glyph, truncation).
|
||||
|
||||
## Non-goals
|
||||
- No gateway/agent changes — the wire and the policy are the source of truth.
|
||||
- No new notice kinds — render exactly the four the policy emits.
|
||||
- The inline-card path (process/background completions) is **unchanged**.
|
||||
- No status-bar segment changes — the banner is its own row above the bar.
|
||||
|
||||
## Risk / footguns
|
||||
- **Schema decode-at-boundary**: `notification.show` payload is a loose Record
|
||||
read by `parseNotification`, not strict-decoded — a wrong-typed field won't blank
|
||||
the bar (unlike `applyInfo`). Keep the loose reads.
|
||||
- **createStore reference-aliasing**: store `notice` and `pendingNotice` distinct
|
||||
objects; when applying pending, it's already its own object — don't alias it to
|
||||
`lastNotification`. (See `[[solid-createstore-reference-aliasing]]`.)
|
||||
- **Timer leak**: `clearNoticeState` must clear `noticeTimer`; ensure session
|
||||
reset and store dispose clear it so a TTL callback can't fire into a dead store.
|
||||
- **Routing regression**: assert in tests that `process.complete` /
|
||||
`background task complete` still produce **cards**, not banners — the whole
|
||||
feature hinges on the `kind` discriminator.
|
||||
@@ -32,6 +32,7 @@ from typing import Any
|
||||
|
||||
_GLOBAL_DEFAULTS: dict[str, Any] = {
|
||||
"tool_progress": "all",
|
||||
"tool_progress_grouping": "accumulate", # "accumulate" = edit one bubble; "separate" = one msg per tool
|
||||
"show_reasoning": False,
|
||||
"tool_preview_length": 0,
|
||||
"streaming": None, # None = follow top-level streaming config
|
||||
@@ -238,6 +239,9 @@ def _normalise(setting: str, value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return value.lower() in {"true", "1", "yes", "on"}
|
||||
return bool(value)
|
||||
if setting == "tool_progress_grouping":
|
||||
val = str(value).lower()
|
||||
return val if val in ("accumulate", "separate") else "accumulate"
|
||||
if setting == "tool_preview_length":
|
||||
try:
|
||||
return int(value)
|
||||
|
||||
+25
-24
@@ -77,6 +77,13 @@ def _thread_metadata_for_source(source, reply_to_message_id: str | None = None)
|
||||
return metadata
|
||||
|
||||
|
||||
def _mark_notify_metadata(metadata: dict | None) -> dict:
|
||||
"""Clone metadata and mark a user-visible reply as notify-worthy."""
|
||||
notify_metadata = dict(metadata) if metadata else {}
|
||||
notify_metadata["notify"] = True
|
||||
return notify_metadata
|
||||
|
||||
|
||||
def _reply_anchor_for_event(event) -> str | None:
|
||||
"""Return reply_to id for platforms that need reply semantics.
|
||||
|
||||
@@ -3889,7 +3896,7 @@ class BasePlatformAdapter(ABC):
|
||||
chat_id=event.source.chat_id,
|
||||
content=_text,
|
||||
reply_to=_reply_anchor_for_event(event),
|
||||
metadata=thread_meta,
|
||||
metadata=_mark_notify_metadata(thread_meta),
|
||||
)
|
||||
if _eph_ttl > 0 and _r.success and _r.message_id:
|
||||
self._schedule_ephemeral_delete(
|
||||
@@ -3995,7 +4002,7 @@ class BasePlatformAdapter(ABC):
|
||||
chat_id=event.source.chat_id,
|
||||
content=_text,
|
||||
reply_to=_reply_anchor_for_event(event),
|
||||
metadata=_thread_meta,
|
||||
metadata=_mark_notify_metadata(_thread_meta),
|
||||
)
|
||||
if _eph_ttl > 0 and _r.success and _r.message_id:
|
||||
self._schedule_ephemeral_delete(
|
||||
@@ -4045,7 +4052,7 @@ class BasePlatformAdapter(ABC):
|
||||
chat_id=event.source.chat_id,
|
||||
content=_text,
|
||||
reply_to=_reply_anchor_for_event(event),
|
||||
metadata=_thread_meta,
|
||||
metadata=_mark_notify_metadata(_thread_meta),
|
||||
)
|
||||
if _eph_ttl > 0 and _r.success and _r.message_id:
|
||||
self._schedule_ephemeral_delete(
|
||||
@@ -4268,6 +4275,12 @@ class BasePlatformAdapter(ABC):
|
||||
)
|
||||
text_content = _recovered
|
||||
|
||||
# Final user-visible content (text, TTS, media, files) gets
|
||||
# the existing notify=True marker. Clone once so typing/status
|
||||
# metadata stays unmarked and progress bubbles remain
|
||||
# thread-strict.
|
||||
_final_thread_metadata = _mark_notify_metadata(_thread_metadata)
|
||||
|
||||
# Auto-TTS: if voice message, generate audio FIRST (before sending text)
|
||||
# Gated via ``_should_auto_tts_for_chat``: fires when the chat has
|
||||
# an explicit ``/voice on|tts`` opt-in OR when ``voice.auto_tts`` is
|
||||
@@ -4307,7 +4320,7 @@ class BasePlatformAdapter(ABC):
|
||||
chat_id=event.source.chat_id,
|
||||
audio_path=_tts_path,
|
||||
caption=telegram_tts_caption,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
_tts_caption_delivered = bool(
|
||||
telegram_tts_caption and getattr(tts_result, "success", False)
|
||||
@@ -4322,23 +4335,11 @@ class BasePlatformAdapter(ABC):
|
||||
if text_content and not _tts_caption_delivered:
|
||||
logger.info("[%s] Sending response (%d chars) to %s", self.name, len(text_content), event.source.chat_id)
|
||||
_reply_anchor = _reply_anchor_for_event(event)
|
||||
# Mark final response messages for notification delivery.
|
||||
# Platform adapters that support per-message notification
|
||||
# control (e.g. Telegram's disable_notification) use this
|
||||
# flag to override silent-mode and ensure the final
|
||||
# response triggers a push notification.
|
||||
# Clone to avoid mutating the metadata shared with the
|
||||
# typing-indicator task (which must remain unmarked).
|
||||
if _thread_metadata is not None:
|
||||
_thread_metadata = dict(_thread_metadata)
|
||||
_thread_metadata["notify"] = True
|
||||
else:
|
||||
_thread_metadata = {"notify": True}
|
||||
result = await self._send_with_retry(
|
||||
chat_id=event.source.chat_id,
|
||||
content=text_content,
|
||||
reply_to=_reply_anchor,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
_record_delivery(result)
|
||||
|
||||
@@ -4367,7 +4368,7 @@ class BasePlatformAdapter(ABC):
|
||||
await self.send_multiple_images(
|
||||
chat_id=event.source.chat_id,
|
||||
images=images,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
human_delay=human_delay,
|
||||
)
|
||||
except Exception as batch_err:
|
||||
@@ -4409,7 +4410,7 @@ class BasePlatformAdapter(ABC):
|
||||
await self.send_multiple_images(
|
||||
chat_id=event.source.chat_id,
|
||||
images=_batch,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
human_delay=human_delay,
|
||||
)
|
||||
except Exception as batch_err:
|
||||
@@ -4424,19 +4425,19 @@ class BasePlatformAdapter(ABC):
|
||||
media_result = await self.send_voice(
|
||||
chat_id=event.source.chat_id,
|
||||
audio_path=media_path,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
elif ext in _VIDEO_EXTS:
|
||||
media_result = await self.send_video(
|
||||
chat_id=event.source.chat_id,
|
||||
video_path=media_path,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
else:
|
||||
media_result = await self.send_document(
|
||||
chat_id=event.source.chat_id,
|
||||
file_path=media_path,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
|
||||
if not media_result.success:
|
||||
@@ -4454,13 +4455,13 @@ class BasePlatformAdapter(ABC):
|
||||
await self.send_video(
|
||||
chat_id=event.source.chat_id,
|
||||
video_path=file_path,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
else:
|
||||
await self.send_document(
|
||||
chat_id=event.source.chat_id,
|
||||
file_path=file_path,
|
||||
metadata=_thread_metadata,
|
||||
metadata=_final_thread_metadata,
|
||||
)
|
||||
except Exception as file_err:
|
||||
logger.error("[%s] Error sending local file %s: %s", self.name, file_path, file_err)
|
||||
|
||||
@@ -678,8 +678,13 @@ class EmailAdapter(BasePlatformAdapter):
|
||||
image_url: str,
|
||||
caption: Optional[str] = None,
|
||||
reply_to: Optional[str] = None,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> SendResult:
|
||||
"""Send an image URL as part of an email body."""
|
||||
"""Send an image URL as part of an email body.
|
||||
|
||||
``metadata`` is accepted to honor the base-class contract; the
|
||||
email body send doesn't use it.
|
||||
"""
|
||||
text = caption or ""
|
||||
text += f"\n\nImage: {image_url}"
|
||||
return await self.send(chat_id, text.strip(), reply_to)
|
||||
|
||||
+140
-21
@@ -419,11 +419,13 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
self._mention_patterns = self._compile_mention_patterns()
|
||||
self._reply_to_mode: str = getattr(config, 'reply_to_mode', 'first') or 'first'
|
||||
self._disable_link_previews: bool = self._coerce_bool_extra("disable_link_previews", False)
|
||||
# Bot API 10.1 Rich Messages: when explicitly enabled, send final
|
||||
# replies via sendRichMessage with the raw agent markdown so
|
||||
# tables/task lists/etc. render natively. Disabled by default because
|
||||
# several Telegram clients accept but render rich messages poorly.
|
||||
self._rich_messages_enabled: bool = self._coerce_bool_extra("rich_messages", False)
|
||||
# Bot API 10.1 Rich Messages: render constructs the legacy MarkdownV2
|
||||
# path degrades (tables → bullet lists, task lists, <details>, block
|
||||
# math) via sendRichMessage / editMessageText's rich_message param using
|
||||
# the raw agent markdown. Enabled by default; users can opt out for
|
||||
# clients that accept but render rich messages poorly via
|
||||
# platforms.telegram.extra.rich_messages: false.
|
||||
self._rich_messages_enabled: bool = self._coerce_bool_extra("rich_messages", True)
|
||||
# Latched off after a capability failure on sendRichMessage /
|
||||
# sendRichMessageDraft (e.g. older python-telegram-bot without the
|
||||
# endpoint) so later sends skip the doomed rich attempt entirely.
|
||||
@@ -979,18 +981,54 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
return True
|
||||
return False
|
||||
|
||||
def _needs_rich_rendering(self, content: str) -> bool:
|
||||
"""Return True for markdown constructs that the legacy path degrades.
|
||||
|
||||
Keep ordinary replies on the pre-rich MarkdownV2 path so Telegram
|
||||
clients render a consistent font weight/spacing. The rich endpoint is
|
||||
reserved for constructs where raw markdown materially improves output:
|
||||
pipe tables (MarkdownV2 has no table syntax and rewrites them into
|
||||
bullet lists), GFM task lists, collapsible ``<details>`` blocks, and
|
||||
block math. Adapted from #45995 (@YonganZhang).
|
||||
"""
|
||||
if not content:
|
||||
return False
|
||||
if any(_TABLE_SEPARATOR_RE.match(line) for line in content.splitlines()):
|
||||
return True
|
||||
if re.search(r"(?m)^\s*[-*]\s+\[[ xX]\]\s+", content):
|
||||
return True
|
||||
if re.search(r"(?m)^<details\b|^</details>|^<summary\b|^</summary>", content):
|
||||
return True
|
||||
if "$$" in content:
|
||||
return True
|
||||
return False
|
||||
|
||||
def _rich_eligible(self, content: str) -> bool:
|
||||
"""Capability/content eligibility for rich, ignoring ``expect_edits``.
|
||||
|
||||
Shared core of :meth:`_should_attempt_rich` minus the per-call
|
||||
``expect_edits`` metadata gate. The rich EDIT-finalize path
|
||||
(:meth:`_try_edit_rich`) needs this: a streamed preview is sent with
|
||||
``expect_edits=True`` to stay on the editable path mid-stream, but the
|
||||
FINAL edit should still upgrade to rich when the content warrants it.
|
||||
"""
|
||||
return bool(
|
||||
getattr(self, "_rich_messages_enabled", True)
|
||||
and not getattr(self, "_rich_send_disabled", False)
|
||||
and content
|
||||
and content.strip()
|
||||
and self._needs_rich_rendering(content)
|
||||
and not self._has_telegram_desktop_details_math_crash_shape(content)
|
||||
and self._content_fits_rich_limits(content)
|
||||
and self._bot_supports_rich()
|
||||
)
|
||||
|
||||
def _should_attempt_rich(
|
||||
self, content: str, metadata: Optional[Dict[str, Any]] = None
|
||||
) -> bool:
|
||||
return bool(
|
||||
getattr(self, "_rich_messages_enabled", False)
|
||||
and not getattr(self, "_rich_send_disabled", False)
|
||||
and not (metadata or {}).get("expect_edits")
|
||||
and content
|
||||
and content.strip()
|
||||
and not self._has_telegram_desktop_details_math_crash_shape(content)
|
||||
and self._content_fits_rich_limits(content)
|
||||
and self._bot_supports_rich()
|
||||
not (metadata or {}).get("expect_edits")
|
||||
and self._rich_eligible(content)
|
||||
)
|
||||
|
||||
def prefers_fresh_final_streaming(
|
||||
@@ -998,12 +1036,13 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
) -> bool:
|
||||
"""Whether to replace a streamed preview with a fresh rich final.
|
||||
|
||||
Keep this disabled for Telegram. The fresh-final path briefly shows two
|
||||
copies of the final answer, then deletes the streaming preview after the
|
||||
rich send succeeds. That is especially visible on clients that support
|
||||
rich messages well, and it looks like duplicate delivery at the end of
|
||||
every streamed turn. Until Telegram rich edits are wired directly, final
|
||||
streamed replies should edit the existing preview in place.
|
||||
Disabled for Telegram. The fresh-final path briefly shows two copies of
|
||||
the final answer, then deletes the streaming preview after the rich send
|
||||
succeeds — it looks like duplicate delivery at the end of every streamed
|
||||
turn (the reason #46206 reverted it). Rich finalize is instead handled
|
||||
by editing the existing preview in place via Bot API 10.1's
|
||||
``editMessageText`` ``rich_message`` parameter (see
|
||||
:meth:`_try_edit_rich`), so no fresh re-send / delete is needed.
|
||||
"""
|
||||
return False
|
||||
|
||||
@@ -1019,7 +1058,7 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
streams split exactly as before.
|
||||
"""
|
||||
if (
|
||||
getattr(self, "_rich_messages_enabled", False)
|
||||
getattr(self, "_rich_messages_enabled", True)
|
||||
and not getattr(self, "_rich_send_disabled", False)
|
||||
and self._bot_supports_rich()
|
||||
):
|
||||
@@ -1207,9 +1246,74 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
message_id=str(message_id) if message_id is not None else None,
|
||||
)
|
||||
|
||||
async def _try_edit_rich(
|
||||
self,
|
||||
chat_id: str,
|
||||
message_id: str,
|
||||
content: str,
|
||||
) -> Optional[SendResult]:
|
||||
"""Edit an existing message in place as a rich message (Bot API 10.1).
|
||||
|
||||
Uses ``editMessageText`` with the ``rich_message`` parameter so a
|
||||
streamed preview can finalize as rich (tables/task lists/details/math)
|
||||
WITHOUT a fresh send + delete — no duplicate preview. Mirrors
|
||||
:meth:`_try_send_rich`'s error contract:
|
||||
|
||||
- success → ``SendResult(success=True, message_id=...)``
|
||||
- permanent / capability error → ``None`` (caller falls back to the
|
||||
legacy MarkdownV2 edit; capability errors latch rich off)
|
||||
- transient / unknown → ``SendResult(success=False)`` with retry
|
||||
semantics (the message may already be edited; do NOT legacy-resend)
|
||||
"""
|
||||
payload: Dict[str, Any] = {
|
||||
"chat_id": int(chat_id),
|
||||
"message_id": int(message_id),
|
||||
"rich_message": self._rich_message_payload(content),
|
||||
}
|
||||
if getattr(self, "_disable_link_previews", False):
|
||||
payload["link_preview_options"] = {"is_disabled": True}
|
||||
try:
|
||||
# Raw Bot API result; do not request return_type=Message (PTB does
|
||||
# not fully model the 10.1 response shape yet — a post-edit parse
|
||||
# error must not be mistaken for a failed edit).
|
||||
await self._bot.do_api_request("editMessageText", api_kwargs=payload)
|
||||
except Exception as exc:
|
||||
if self._is_rich_fallback_error(exc):
|
||||
if self._is_rich_capability_error(exc):
|
||||
self._rich_send_disabled = True
|
||||
# "Message is not modified" — content identical to the current
|
||||
# rich message; treat as a successful no-op so the caller does
|
||||
# not fall through to a redundant legacy edit.
|
||||
if "not modified" in str(exc).lower():
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
logger.debug(
|
||||
"[%s] rich editMessageText rejected (%s) — falling back to MarkdownV2 edit",
|
||||
self.name, exc,
|
||||
)
|
||||
return None
|
||||
if "not modified" in str(exc).lower():
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
err_str = str(exc).lower()
|
||||
try:
|
||||
from telegram.error import TimedOut as _TimedOut
|
||||
except (ImportError, AttributeError):
|
||||
_TimedOut = None
|
||||
is_timeout = (_TimedOut and isinstance(exc, _TimedOut)) or "timed out" in err_str
|
||||
is_connect_timeout = self._looks_like_connect_timeout(exc)
|
||||
logger.warning(
|
||||
"[%s] rich editMessageText transient failure (no legacy resend): %s",
|
||||
self.name, exc,
|
||||
)
|
||||
return SendResult(
|
||||
success=False,
|
||||
error=str(exc),
|
||||
retryable=(is_connect_timeout or not is_timeout),
|
||||
)
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
def _should_attempt_rich_draft(self, content: str) -> bool:
|
||||
return bool(
|
||||
getattr(self, "_rich_messages_enabled", False)
|
||||
getattr(self, "_rich_messages_enabled", True)
|
||||
and not getattr(self, "_rich_send_disabled", False)
|
||||
and not getattr(self, "_rich_draft_disabled", False)
|
||||
and content
|
||||
@@ -2555,6 +2659,21 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
if not self._bot:
|
||||
return SendResult(success=False, error="Not connected")
|
||||
|
||||
# Rich finalize (Bot API 10.1): when the completed content has
|
||||
# constructs the legacy MarkdownV2 edit degrades (tables → bullet
|
||||
# lists, task lists, <details>, block math) and rich is available,
|
||||
# edit the preview IN PLACE via editMessageText's rich_message param.
|
||||
# No fresh send + delete → no duplicate preview (the problem #46206
|
||||
# reverted the fresh-final path for). Attempted before the 4,096
|
||||
# overflow pre-flight because the rich text cap is 32,768 — a rich
|
||||
# table that exceeds the MarkdownV2 limit must not be split into legacy
|
||||
# chunks. Falls back to the legacy edit path (overflow split included)
|
||||
# on capability/permanent rejection.
|
||||
if finalize and self._rich_eligible(content):
|
||||
rich_result = await self._try_edit_rich(chat_id, message_id, content)
|
||||
if rich_result is not None:
|
||||
return rich_result
|
||||
|
||||
# Pre-flight: if content already exceeds the limit, split-and-deliver
|
||||
# without round-tripping a doomed edit.
|
||||
if utf16_len(content) > self.MAX_MESSAGE_LENGTH:
|
||||
|
||||
@@ -846,13 +846,20 @@ class WhatsAppAdapter(WhatsAppBehaviorMixin, BasePlatformAdapter):
|
||||
image_url: str,
|
||||
caption: Optional[str] = None,
|
||||
reply_to: Optional[str] = None,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> SendResult:
|
||||
"""Download image URL to cache, send natively via bridge."""
|
||||
"""Download image URL to cache, send natively via bridge.
|
||||
|
||||
``metadata`` is accepted to honor the base-class contract — the
|
||||
batch sender ``send_multiple_images`` passes it through to every
|
||||
send path. The bridge media call doesn't use it, matching the
|
||||
sibling overrides (send_video / send_voice / send_document).
|
||||
"""
|
||||
try:
|
||||
local_path = await cache_image_from_url(image_url)
|
||||
return await self._send_media_to_bridge(chat_id, local_path, "image", caption)
|
||||
except Exception:
|
||||
return await super().send_image(chat_id, image_url, caption, reply_to)
|
||||
return await super().send_image(chat_id, image_url, caption, reply_to, metadata)
|
||||
|
||||
async def send_image_file(
|
||||
self,
|
||||
@@ -1136,6 +1143,15 @@ class WhatsAppAdapter(WhatsAppBehaviorMixin, BasePlatformAdapter):
|
||||
body = data.get("body", "")
|
||||
if data.get("isGroup"):
|
||||
body = self._clean_bot_mention_text(body, data)
|
||||
|
||||
# If this is a reply, include the quoted message text so the agent
|
||||
# knows exactly what the user is responding to (fixes "approve" context issue)
|
||||
quoted_text = str(data.get("quotedText") or "").strip()
|
||||
if quoted_text and data.get("hasQuotedMessage"):
|
||||
# Truncate long quoted text to keep prompts reasonable
|
||||
if len(quoted_text) > 300:
|
||||
quoted_text = quoted_text[:297] + "..."
|
||||
body = f"[Replying to: \"{quoted_text}\"]\n{body}"
|
||||
MAX_TEXT_INJECT_BYTES = 100 * 1024
|
||||
if msg_type == MessageType.DOCUMENT and cached_urls:
|
||||
for doc_path in cached_urls:
|
||||
|
||||
+133
-19
@@ -402,6 +402,68 @@ async def _send_or_update_status_coro(adapter, chat_id, status_key, content, met
|
||||
return await adapter.send(chat_id, content, metadata=metadata)
|
||||
|
||||
|
||||
def _resolve_progress_thread_id(platform: Any, source_thread_id: Any, event_message_id: Any) -> Optional[str]:
|
||||
"""Return thread/root ID that progress/status bubbles should target."""
|
||||
platform_value = getattr(platform, "value", platform)
|
||||
platform_key = str(platform_value or "").lower()
|
||||
if source_thread_id:
|
||||
return str(source_thread_id)
|
||||
if platform_key in {"slack", "mattermost"} and event_message_id:
|
||||
return str(event_message_id)
|
||||
return None
|
||||
|
||||
|
||||
def _has_platform_display_override(user_config: dict, platform_key: str, setting: str) -> bool:
|
||||
"""Return True when display.platforms.<platform> explicitly sets setting."""
|
||||
display = user_config.get("display") if isinstance(user_config, dict) else None
|
||||
if not isinstance(display, dict):
|
||||
return False
|
||||
platforms = display.get("platforms")
|
||||
if not isinstance(platforms, dict):
|
||||
return False
|
||||
platform_cfg = platforms.get(platform_key)
|
||||
return isinstance(platform_cfg, dict) and setting in platform_cfg
|
||||
|
||||
|
||||
def _resolve_gateway_display_bool(
|
||||
user_config: dict,
|
||||
platform_key: str,
|
||||
setting: str,
|
||||
*,
|
||||
default: bool = False,
|
||||
platform: Any = None,
|
||||
require_platform_override_for: set[Any] | None = None,
|
||||
) -> bool:
|
||||
"""Resolve a boolean display setting with optional platform-only opt-in.
|
||||
|
||||
Some display features expose assistant scratch text rather than deliberate
|
||||
user-facing output. For high-noise threaded chat surfaces such as
|
||||
Mattermost, a global opt-in is too broad: they must be enabled with an
|
||||
explicit display.platforms.<platform>.<setting> override.
|
||||
"""
|
||||
current_platform = _gateway_platform_value(platform or platform_key)
|
||||
platform_only = {
|
||||
_gateway_platform_value(candidate)
|
||||
for candidate in (require_platform_override_for or set())
|
||||
}
|
||||
if (
|
||||
current_platform in platform_only
|
||||
and not _has_platform_display_override(user_config, platform_key, setting)
|
||||
):
|
||||
return False
|
||||
|
||||
from gateway.display_config import resolve_display_setting
|
||||
|
||||
value = resolve_display_setting(user_config, platform_key, setting, default)
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
if isinstance(value, str):
|
||||
return value.strip().lower() in {"true", "yes", "1", "on"}
|
||||
if value is None:
|
||||
return bool(default)
|
||||
return bool(value)
|
||||
|
||||
|
||||
def _telegramize_command_mentions(text: str, platform: Any) -> str:
|
||||
"""Rewrite slash-command mentions to Telegram-valid command names.
|
||||
|
||||
@@ -8978,17 +9040,24 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
source, session_entry, reason="agent-result-compression",
|
||||
)
|
||||
|
||||
# Prepend reasoning/thinking if display is enabled (per-platform)
|
||||
# Prepend reasoning/thinking if display is enabled (per-platform).
|
||||
# Mattermost requires explicit per-platform opt-in because this is
|
||||
# scratch text, not ordinary final-answer content.
|
||||
try:
|
||||
from gateway.display_config import resolve_display_setting as _rds
|
||||
_show_reasoning_effective = _rds(
|
||||
_show_reasoning_effective = _resolve_gateway_display_bool(
|
||||
_load_gateway_config(),
|
||||
_platform_config_key(source.platform),
|
||||
"show_reasoning",
|
||||
getattr(self, "_show_reasoning", False),
|
||||
default=bool(getattr(self, "_show_reasoning", False)),
|
||||
platform=source.platform,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
)
|
||||
except Exception:
|
||||
_show_reasoning_effective = getattr(self, "_show_reasoning", False)
|
||||
_show_reasoning_effective = (
|
||||
False
|
||||
if source.platform == Platform.MATTERMOST
|
||||
else getattr(self, "_show_reasoning", False)
|
||||
)
|
||||
if _show_reasoning_effective and response and not _intentional_silence:
|
||||
last_reasoning = agent_result.get("last_reasoning")
|
||||
if last_reasoning:
|
||||
@@ -13613,6 +13682,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
if _env_tp and not _tool_progress_configured
|
||||
else (_resolved_tp or _env_tp or "all")
|
||||
)
|
||||
# Tool progress grouping: "accumulate" (edit one bubble) or "separate" (one msg per tool)
|
||||
progress_grouping = resolve_display_setting(user_config, platform_key, "tool_progress_grouping") or "accumulate"
|
||||
# Disable tool progress for webhooks - they don't support message editing,
|
||||
# so each progress line would be sent as a separate message.
|
||||
from gateway.config import Platform
|
||||
@@ -13622,18 +13693,32 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
# in chat platforms while opting into concise mid-turn updates.
|
||||
interim_assistant_messages_enabled = (
|
||||
source.platform != Platform.WEBHOOK
|
||||
and bool(
|
||||
resolve_display_setting(
|
||||
user_config,
|
||||
platform_key,
|
||||
"interim_assistant_messages",
|
||||
True,
|
||||
)
|
||||
and _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
platform_key,
|
||||
"interim_assistant_messages",
|
||||
default=True,
|
||||
platform=source.platform,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
)
|
||||
)
|
||||
|
||||
# thinking_progress is independent — if enabled, we need the progress
|
||||
# queue even when tool_progress is off (thinking relay uses same infra).
|
||||
# Mattermost requires a per-platform opt-in: global scratch-text display
|
||||
# is too easy to leak into busy public threads.
|
||||
_thinking_enabled = _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
platform_key,
|
||||
"thinking_progress",
|
||||
default=False,
|
||||
platform=source.platform,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
)
|
||||
needs_progress_queue = tool_progress_enabled or _thinking_enabled
|
||||
|
||||
|
||||
# Queue for progress messages (thread-safe)
|
||||
progress_queue = queue.Queue() if tool_progress_enabled else None
|
||||
progress_queue = queue.Queue() if needs_progress_queue else None
|
||||
last_tool = [None] # Mutable container for tracking in closure
|
||||
last_progress_msg = [None] # Track last message for dedup
|
||||
repeat_count = [0] # How many times the same message repeated
|
||||
@@ -13739,6 +13824,24 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
logger.debug("tool-progress onboarding hint failed: %s", _hint_err)
|
||||
return
|
||||
|
||||
# "_thinking" is assistant scratch text between tool calls. It
|
||||
# is never ordinary tool progress: only relay it when the platform
|
||||
# explicitly opted into thinking_progress. Handle both legacy
|
||||
# callback shapes: ("_thinking", text) and
|
||||
# ("reasoning.available", "_thinking", text, ...).
|
||||
if event_type == "_thinking" or tool_name == "_thinking":
|
||||
if not _thinking_enabled:
|
||||
return
|
||||
thinking_text = preview if tool_name == "_thinking" else tool_name
|
||||
msg = f"💬 {thinking_text}" if thinking_text else None
|
||||
if msg:
|
||||
progress_queue.put(msg)
|
||||
return
|
||||
|
||||
# If tool_progress is off, only _thinking passes through (above).
|
||||
# Regular tool calls are suppressed.
|
||||
if not tool_progress_enabled:
|
||||
return
|
||||
|
||||
# Only act on tool.started events (ignore tool.completed, reasoning.available, etc.)
|
||||
if event_type not in {"tool.started",}:
|
||||
@@ -13884,10 +13987,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
# - Feishu only honors reply_in_thread when sending a reply, so topic
|
||||
# progress uses the triggering event message as the reply target
|
||||
# - Other platforms should use explicit source.thread_id only
|
||||
if source.platform == Platform.SLACK:
|
||||
_progress_thread_id = source.thread_id or event_message_id
|
||||
else:
|
||||
_progress_thread_id = source.thread_id
|
||||
_progress_thread_id = _resolve_progress_thread_id(
|
||||
source.platform, source.thread_id, event_message_id,
|
||||
)
|
||||
_progress_metadata = (
|
||||
self._thread_metadata_for_source(source, event_message_id)
|
||||
if _progress_thread_id == source.thread_id
|
||||
@@ -13920,7 +14022,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
|
||||
progress_lines = [] # Accumulated tool lines for the CURRENT editable bubble
|
||||
progress_msg_id = None # ID of the current progress message to edit
|
||||
can_edit = True # False once an edit fails (platform doesn't support it)
|
||||
can_edit = progress_grouping != "separate" # "separate" = one message per tool (pre-v0.9 behavior)
|
||||
_last_edit_ts = 0.0 # Throttle edits to avoid Telegram flood control
|
||||
_PROGRESS_EDIT_INTERVAL = 1.5 # Minimum seconds between edits
|
||||
|
||||
@@ -14687,6 +14789,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
_pdc = getattr(_status_adapter, "_post_delivery_callbacks", None)
|
||||
if _pdc is not None:
|
||||
_pdc[session_key] = _release_bg_review_messages
|
||||
# Memory update notifications in chat. Config: display.memory_notifications
|
||||
# off — no chat notification (still logged to stdout)
|
||||
# on — generic "💾 Memory updated" (default)
|
||||
# verbose — content preview: "💾 Memory ➕ Hermes Repo..."
|
||||
_mem_notif = user_config.get("display", {}).get("memory_notifications")
|
||||
if isinstance(_mem_notif, bool):
|
||||
_mem_notif = "on" if _mem_notif else "off"
|
||||
agent.memory_notifications = str(_mem_notif).lower() if _mem_notif else "on"
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Clarify callback: present a clarify prompt and block on a response.
|
||||
@@ -14763,6 +14873,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
|
||||
|
||||
agent.clarify_callback = _clarify_callback_sync
|
||||
|
||||
# Show assistant thinking between tool calls — independent of
|
||||
# tool_progress mode. Mattermost needs an explicit per-platform
|
||||
# opt-in so global scratch-text display does not leak into threads.
|
||||
agent.thinking_progress = _thinking_enabled
|
||||
# Store agent reference for interrupt support
|
||||
agent_holder[0] = agent
|
||||
# Capture the full tool definitions for transcript logging
|
||||
|
||||
+43
-10
@@ -197,6 +197,30 @@ class GatewayStreamConsumer:
|
||||
# this response and route through edit-based for graceful degradation.
|
||||
self._draft_failures = 0
|
||||
|
||||
def _metadata_for_send(
|
||||
self,
|
||||
*,
|
||||
final: bool = False,
|
||||
expect_edits: bool = False,
|
||||
) -> dict | None:
|
||||
"""Return per-send metadata for stream-created messages.
|
||||
|
||||
Mattermost treats notify-worthy sends as user-visible final content
|
||||
when deciding whether a broken thread root may fall back flat. Preview
|
||||
and progress sends keep their original metadata and remain thread-strict.
|
||||
|
||||
``expect_edits`` preserves the upstream Telegram streaming contract:
|
||||
preview messages that may be edited later must stay on the editable
|
||||
legacy send path, while fresh/fallback final sends can still use richer
|
||||
final-message delivery.
|
||||
"""
|
||||
meta = dict(self.metadata) if self.metadata else {}
|
||||
if expect_edits:
|
||||
meta["expect_edits"] = True
|
||||
if final:
|
||||
meta["notify"] = True
|
||||
return meta or None
|
||||
|
||||
@property
|
||||
def already_sent(self) -> bool:
|
||||
"""True if at least one message was sent or edited during the run."""
|
||||
@@ -513,7 +537,11 @@ class GatewayStreamConsumer:
|
||||
chunks_delivered = False
|
||||
reply_to = self._message_id or self._initial_reply_to_id
|
||||
for chunk in chunks:
|
||||
new_id = await self._send_new_chunk(chunk, reply_to)
|
||||
new_id = await self._send_new_chunk(
|
||||
chunk,
|
||||
reply_to,
|
||||
final=got_done,
|
||||
)
|
||||
if new_id is not None and new_id != reply_to:
|
||||
chunks_delivered = True
|
||||
self._accumulated = ""
|
||||
@@ -749,7 +777,13 @@ class GatewayStreamConsumer:
|
||||
# Strip trailing whitespace/newlines but preserve leading content
|
||||
return cleaned.rstrip()
|
||||
|
||||
async def _send_new_chunk(self, text: str, reply_to_id: Optional[str]) -> Optional[str]:
|
||||
async def _send_new_chunk(
|
||||
self,
|
||||
text: str,
|
||||
reply_to_id: Optional[str],
|
||||
*,
|
||||
final: bool = False,
|
||||
) -> Optional[str]:
|
||||
"""Send a new message chunk, optionally threaded to a previous message.
|
||||
|
||||
Returns the message_id so callers can thread subsequent chunks.
|
||||
@@ -758,15 +792,11 @@ class GatewayStreamConsumer:
|
||||
if not text.strip():
|
||||
return reply_to_id
|
||||
try:
|
||||
meta = dict(self.metadata) if self.metadata else {}
|
||||
# This chunk becomes the next edit target — adapters that support
|
||||
# rich final sends (Telegram) must keep it on the editable path.
|
||||
meta["expect_edits"] = True
|
||||
result = await self.adapter.send(
|
||||
chat_id=self.chat_id,
|
||||
content=text,
|
||||
reply_to=reply_to_id,
|
||||
metadata=meta,
|
||||
metadata=self._metadata_for_send(final=final, expect_edits=True),
|
||||
)
|
||||
if result.success and result.message_id:
|
||||
self._message_id = str(result.message_id)
|
||||
@@ -885,7 +915,7 @@ class GatewayStreamConsumer:
|
||||
result = await self.adapter.send(
|
||||
chat_id=self.chat_id,
|
||||
content=chunk,
|
||||
metadata=self.metadata,
|
||||
metadata=self._metadata_for_send(final=True),
|
||||
)
|
||||
if result.success:
|
||||
break
|
||||
@@ -1242,7 +1272,7 @@ class GatewayStreamConsumer:
|
||||
result = await self.adapter.send(
|
||||
chat_id=self.chat_id,
|
||||
content=text,
|
||||
metadata=self.metadata,
|
||||
metadata=self._metadata_for_send(final=True),
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug("Fresh-final send failed, falling back to edit: %s", e)
|
||||
@@ -1532,7 +1562,10 @@ class GatewayStreamConsumer:
|
||||
chat_id=self.chat_id,
|
||||
content=text,
|
||||
reply_to=self._initial_reply_to_id,
|
||||
metadata={**(self.metadata or {}), "expect_edits": True},
|
||||
metadata=self._metadata_for_send(
|
||||
final=finalize,
|
||||
expect_edits=True,
|
||||
),
|
||||
)
|
||||
if result.success:
|
||||
if result.message_id:
|
||||
|
||||
+16
-2
@@ -145,8 +145,16 @@ def build_top_level_parser():
|
||||
"--resume",
|
||||
"-r",
|
||||
metavar="SESSION",
|
||||
# nargs="?" + const=True: bare `--resume` parses to the sentinel True,
|
||||
# which `hermes --tui` turns into the session picker
|
||||
# (HERMES_TUI_RESUME=picker). `--resume <id|title>` is unchanged.
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=None,
|
||||
help="Resume a previous session by ID or title",
|
||||
help=(
|
||||
"Resume a previous session by ID or title. With --tui, bare "
|
||||
"--resume (no argument) opens the session picker."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--continue",
|
||||
@@ -301,8 +309,14 @@ def build_top_level_parser():
|
||||
"--resume",
|
||||
"-r",
|
||||
metavar="SESSION_ID",
|
||||
# Same bare-flag picker sentinel as the top-level --resume.
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=argparse.SUPPRESS,
|
||||
help="Resume a previous session by ID (shown on exit)",
|
||||
help=(
|
||||
"Resume a previous session by ID (shown on exit). With --tui, "
|
||||
"bare --resume opens the session picker."
|
||||
),
|
||||
)
|
||||
chat_parser.add_argument(
|
||||
"--continue",
|
||||
|
||||
+127
-11
@@ -3806,6 +3806,26 @@ def resolve_codex_runtime_credentials(
|
||||
"last_refresh": None,
|
||||
"auth_mode": "chatgpt",
|
||||
}
|
||||
pool_rate_limit = _codex_pool_rate_limit_status()
|
||||
if pool_rate_limit:
|
||||
reset_at = pool_rate_limit.get("reset_at")
|
||||
if isinstance(reset_at, (int, float)) and reset_at > time.time():
|
||||
remaining = int(reset_at - time.time())
|
||||
message = (
|
||||
f"Codex provider quota exhausted (429); retry after {remaining}s. "
|
||||
"Credentials are still valid."
|
||||
)
|
||||
else:
|
||||
message = (
|
||||
"Codex provider quota exhausted (429). Credentials are still valid; "
|
||||
"retry after the usage limit resets."
|
||||
)
|
||||
raise AuthError(
|
||||
message,
|
||||
provider="openai-codex",
|
||||
code=CODEX_RATE_LIMITED_CODE,
|
||||
relogin_required=False,
|
||||
)
|
||||
if read_error is not None:
|
||||
raise read_error
|
||||
raise AuthError(
|
||||
@@ -3852,6 +3872,79 @@ def resolve_codex_runtime_credentials(
|
||||
}
|
||||
|
||||
|
||||
def _codex_pool_rate_limit_status() -> Optional[Dict[str, Any]]:
|
||||
"""Return metadata for a pool-only Codex credential in quota cooldown."""
|
||||
def _parse_reset_at(value: Any) -> Optional[float]:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
if isinstance(value, (int, float)):
|
||||
numeric = float(value)
|
||||
if numeric <= 0:
|
||||
return None
|
||||
return numeric / 1000.0 if numeric > 1_000_000_000_000 else numeric
|
||||
if isinstance(value, str):
|
||||
raw = value.strip()
|
||||
if not raw:
|
||||
return None
|
||||
try:
|
||||
numeric = float(raw)
|
||||
except ValueError:
|
||||
numeric = None
|
||||
if numeric is not None:
|
||||
return numeric / 1000.0 if numeric > 1_000_000_000_000 else numeric
|
||||
try:
|
||||
return datetime.fromisoformat(raw.replace("Z", "+00:00")).timestamp()
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
try:
|
||||
with _auth_store_lock():
|
||||
auth_store = _load_auth_store()
|
||||
pool = auth_store.get("credential_pool")
|
||||
if not isinstance(pool, dict):
|
||||
return None
|
||||
entries = pool.get("openai-codex")
|
||||
if not isinstance(entries, list):
|
||||
return None
|
||||
now = time.time()
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
token = entry.get("access_token")
|
||||
if not isinstance(token, str) or not token.strip():
|
||||
continue
|
||||
if entry.get("last_status") != "exhausted":
|
||||
continue
|
||||
code = entry.get("last_error_code")
|
||||
reason = str(entry.get("last_error_reason") or "").lower()
|
||||
message = str(entry.get("last_error_message") or "").lower()
|
||||
is_rate_limited = (
|
||||
code == 429
|
||||
or "rate_limit" in reason
|
||||
or "usage_limit" in reason
|
||||
or "quota" in reason
|
||||
or "rate limit" in message
|
||||
or "usage limit" in message
|
||||
or "quota" in message
|
||||
)
|
||||
if not is_rate_limited:
|
||||
continue
|
||||
reset_at = _parse_reset_at(entry.get("last_error_reset_at"))
|
||||
if reset_at is not None and reset_at <= now:
|
||||
continue
|
||||
return {
|
||||
"label": entry.get("label"),
|
||||
"last_refresh": entry.get("last_refresh"),
|
||||
"reset_at": reset_at,
|
||||
"reason": entry.get("last_error_reason"),
|
||||
"message": entry.get("last_error_message"),
|
||||
}
|
||||
except Exception:
|
||||
logger.debug("Codex pool rate-limit lookup failed", exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def _pool_codex_access_token() -> str:
|
||||
"""Return the most-recent usable access_token from the openai-codex pool.
|
||||
|
||||
@@ -5763,18 +5856,24 @@ def _snapshot_nous_pool_status() -> Dict[str, Any]:
|
||||
# subscription-feature checks) call it many times per render — `hermes tools` → "All Platforms"
|
||||
# was firing the refresh ~31× during one menu paint, racking up >13s of HTTP and burning
|
||||
# single-use refresh tokens. Cache the snapshot for a few seconds, keyed on the auth.json
|
||||
# mtime so that `hermes auth login/logout/add/remove` invalidate naturally on the next call.
|
||||
# path + mtime so that profile switches do not share a process memo and
|
||||
# `hermes auth login/logout/add/remove` invalidate naturally on the next call.
|
||||
_NOUS_AUTH_STATUS_CACHE_TTL = 15.0 # seconds
|
||||
_nous_auth_status_cache: Optional[Tuple[float, Optional[float], Dict[str, Any]]] = None
|
||||
_nous_auth_status_cache: Optional[Tuple[float, str, Optional[float], Dict[str, Any]]] = None
|
||||
|
||||
|
||||
def _auth_file_mtime() -> Optional[float]:
|
||||
def _auth_file_cache_key() -> Tuple[str, Optional[float]]:
|
||||
auth_file = _auth_file_path()
|
||||
try:
|
||||
return _auth_file_path().stat().st_mtime
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
auth_file_key = str(auth_file.resolve(strict=False))
|
||||
except Exception:
|
||||
return None
|
||||
auth_file_key = str(auth_file)
|
||||
try:
|
||||
return auth_file_key, auth_file.stat().st_mtime
|
||||
except FileNotFoundError:
|
||||
return auth_file_key, None
|
||||
except Exception:
|
||||
return auth_file_key, None
|
||||
|
||||
|
||||
def invalidate_nous_auth_status_cache() -> None:
|
||||
@@ -5806,18 +5905,19 @@ def get_nous_auth_status() -> Dict[str, Any]:
|
||||
"""
|
||||
global _nous_auth_status_cache
|
||||
now = time.monotonic()
|
||||
mtime = _auth_file_mtime()
|
||||
auth_file_key, mtime = _auth_file_cache_key()
|
||||
cached = _nous_auth_status_cache
|
||||
if cached is not None:
|
||||
cached_at, cached_mtime, cached_status = cached
|
||||
cached_at, cached_auth_file_key, cached_mtime, cached_status = cached
|
||||
if (
|
||||
cached_mtime == mtime
|
||||
cached_auth_file_key == auth_file_key
|
||||
and cached_mtime == mtime
|
||||
and (now - cached_at) < _NOUS_AUTH_STATUS_CACHE_TTL
|
||||
):
|
||||
return dict(cached_status)
|
||||
|
||||
status = _compute_nous_auth_status()
|
||||
_nous_auth_status_cache = (now, mtime, dict(status))
|
||||
_nous_auth_status_cache = (now, auth_file_key, mtime, dict(status))
|
||||
return status
|
||||
|
||||
|
||||
@@ -5900,6 +6000,22 @@ def get_codex_auth_status() -> Dict[str, Any]:
|
||||
"source": f"pool:{getattr(entry, 'label', 'unknown')}",
|
||||
"api_key": api_key,
|
||||
}
|
||||
rate_limit = _codex_pool_rate_limit_status()
|
||||
if rate_limit:
|
||||
return {
|
||||
"logged_in": True,
|
||||
"auth_store": str(_auth_file_path()),
|
||||
"last_refresh": rate_limit.get("last_refresh"),
|
||||
"auth_mode": "chatgpt",
|
||||
"source": f"pool:{rate_limit.get('label') or 'unknown'}",
|
||||
"rate_limited": True,
|
||||
"error_code": CODEX_RATE_LIMITED_CODE,
|
||||
"error": (
|
||||
rate_limit.get("message")
|
||||
or "Codex provider quota exhausted; retry after the usage limit resets."
|
||||
),
|
||||
"reset_at": rate_limit.get("reset_at"),
|
||||
}
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
@@ -1053,7 +1053,8 @@ _SLACK_PRIORITY_ALIASES = ("btw", "bg")
|
||||
# the telegram-parity test reads it so an entry here is a deliberate
|
||||
# "Slack-via-/hermes" decision, not a silent clamp.
|
||||
# - credits: the billing/top-up surface; reached via /hermes credits on Slack.
|
||||
_SLACK_VIA_HERMES_ONLY = frozenset({"credits"})
|
||||
# - debug: the log/report upload surface; reached via /hermes debug on Slack.
|
||||
_SLACK_VIA_HERMES_ONLY = frozenset({"credits", "debug"})
|
||||
|
||||
|
||||
def _sanitize_slack_name(raw: str) -> str:
|
||||
|
||||
+13
-1
@@ -1428,6 +1428,12 @@ DEFAULT_CONFIG = {
|
||||
"tui_agents_nudge": True,
|
||||
"bell_on_complete": False,
|
||||
"show_reasoning": False,
|
||||
# Background self-improvement review notifications surfaced in chat.
|
||||
# "off" — no chat notification (the review still runs and writes)
|
||||
# "on" — generic "💾 Memory updated" line (default)
|
||||
# "verbose" — include a compact content preview of what changed
|
||||
# Per-platform overrides via display.platforms.<platform>.memory_notifications.
|
||||
"memory_notifications": "on",
|
||||
"streaming": False,
|
||||
"timestamps": False, # Show [HH:MM] on user and assistant labels
|
||||
"final_response_markdown": "strip", # render | strip | raw
|
||||
@@ -1479,6 +1485,12 @@ DEFAULT_CONFIG = {
|
||||
"tool_progress_command": False, # Enable /verbose command in messaging gateway
|
||||
"tool_progress_overrides": {}, # DEPRECATED — use display.platforms instead
|
||||
"tool_preview_length": 0, # Max chars for tool call previews (0 = no limit, show full paths/commands)
|
||||
# How gateway tool-progress is grouped on platforms that support message
|
||||
# editing: "accumulate" (default) edits one bubble in place; "separate"
|
||||
# sends one message per tool (the pre-v0.9 behavior, noisier). Only
|
||||
# applies where tool_progress is already enabled. Per-platform override
|
||||
# via display.platforms.<platform>.tool_progress_grouping.
|
||||
"tool_progress_grouping": "accumulate",
|
||||
# Auto-delete system-notice replies (e.g. "✨ New session started!",
|
||||
# "♻ Restarting gateway…", "⚡ Stopped…") after N seconds on platforms
|
||||
# that support message deletion (currently Telegram; other platforms
|
||||
@@ -1991,7 +2003,7 @@ DEFAULT_CONFIG = {
|
||||
"channel_prompts": {}, # Per-chat/topic ephemeral system prompts (topics inherit from parent group)
|
||||
"allowed_chats": "", # If set, bot ONLY responds in these group/supergroup chat IDs (whitelist)
|
||||
"extra": {
|
||||
"rich_messages": False, # Opt in to Bot API 10.1 rich messages; default uses legacy MarkdownV2
|
||||
"rich_messages": True, # Bot API 10.1 rich messages (tables/task lists/details/math) render natively; set False to force legacy MarkdownV2
|
||||
},
|
||||
},
|
||||
|
||||
|
||||
+394
-5
@@ -1640,8 +1640,286 @@ def _find_bundled_tui(hermes_cli_dir: Path | None = None) -> Path | None:
|
||||
return bundled if bundled.is_file() else None
|
||||
|
||||
|
||||
def _config_tui_engine_early() -> str | None:
|
||||
"""Read ``display.tui_engine`` from config via a minimal YAML read.
|
||||
|
||||
Returns the configured engine string, or ``None`` when unset/unreadable so the
|
||||
caller can apply the availability-gated default. Mirrors
|
||||
:func:`_config_default_interface_early`.
|
||||
"""
|
||||
try:
|
||||
home = os.environ.get("HERMES_HOME")
|
||||
cfg_path = (
|
||||
os.path.join(home, "config.yaml")
|
||||
if home
|
||||
else os.path.join(os.path.expanduser("~"), ".hermes", "config.yaml")
|
||||
)
|
||||
if os.path.exists(cfg_path):
|
||||
import yaml as _yaml_eng
|
||||
|
||||
with open(cfg_path, encoding="utf-8") as _f:
|
||||
raw = _yaml_eng.safe_load(_f) or {}
|
||||
disp = raw.get("display", {})
|
||||
if isinstance(disp, dict):
|
||||
eng = disp.get("tui_engine")
|
||||
if isinstance(eng, str) and eng.strip():
|
||||
return eng.strip().lower()
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _resolve_tui_engine() -> str:
|
||||
"""Which TUI engine to launch: "ink" (default) or "opentui".
|
||||
|
||||
Precedence: ``HERMES_TUI_ENGINE`` env > ``display.tui_engine`` config >
|
||||
(OpenTUI when this host can run it — Node >= 26.3 + the built package — else Ink).
|
||||
The OpenTUI engine runs on Node 26.3+ via the experimental ``node:ffi`` renderer,
|
||||
which is not validated on Windows or Termux — a request for "opentui" there falls
|
||||
back to "ink" with a notice so a stale flag never strands the user on an engine
|
||||
that can't start.
|
||||
"""
|
||||
env = (os.environ.get("HERMES_TUI_ENGINE") or "").strip().lower()
|
||||
# Explicit choice (env > config) wins; otherwise default to OpenTUI when this
|
||||
# host is genuinely set up for it (Node >= 26.3 + the built bundle), else Ink.
|
||||
engine = env or _config_tui_engine_early() or ("opentui" if _opentui_available() else "ink")
|
||||
if engine != "opentui":
|
||||
return "ink"
|
||||
|
||||
# opentui requested — gate on platform support.
|
||||
unsupported = sys.platform.startswith("win") or _is_termux_startup_environment()
|
||||
if unsupported:
|
||||
if not os.environ.get("HERMES_QUIET"):
|
||||
where = "Windows" if sys.platform.startswith("win") else "Termux"
|
||||
print(
|
||||
f"HERMES_TUI_ENGINE=opentui is not supported on {where} "
|
||||
f"(needs Node 26.3+ with experimental FFI) — falling back to the Ink engine.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return "ink"
|
||||
return "opentui"
|
||||
|
||||
|
||||
NODE26_MIN_VERSION = (26, 3, 0)
|
||||
|
||||
|
||||
def _node_version_tuple(node_bin: str) -> tuple[int, int, int] | None:
|
||||
"""Return (major, minor, patch) for a node binary, or ``None`` if unreadable."""
|
||||
try:
|
||||
out = subprocess.run([node_bin, "--version"], capture_output=True, text=True, timeout=5)
|
||||
except Exception:
|
||||
return None
|
||||
if out.returncode != 0:
|
||||
return None
|
||||
raw = (out.stdout or "").strip().lstrip("v").split("-", 1)[0]
|
||||
parts = raw.split(".")
|
||||
try:
|
||||
return (int(parts[0]), int(parts[1]), int(parts[2]))
|
||||
except (IndexError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _fnm_node26_candidates() -> list[str]:
|
||||
"""Node binaries from fnm's installed versions, newest first.
|
||||
|
||||
fnm keeps each version at ``<FNM_DIR>/node-versions/v<X.Y.Z>/installation/
|
||||
bin/node`` (default ``FNM_DIR``: ``$XDG_DATA_HOME/fnm`` or ``~/.local/share/
|
||||
fnm``; macOS Homebrew also uses ``~/Library/Application Support/fnm``). When
|
||||
the *active* node is older than 26.3 — e.g. the user's fnm default is on
|
||||
v25 — the right 26.x is still installed and usable; surface it so OpenTUI
|
||||
works without the user re-aliasing their global default. Version-sorted so
|
||||
the newest qualifying node wins.
|
||||
"""
|
||||
roots: list[Path] = []
|
||||
fnm_dir = os.environ.get("FNM_DIR")
|
||||
if fnm_dir:
|
||||
roots.append(Path(fnm_dir))
|
||||
xdg = os.environ.get("XDG_DATA_HOME")
|
||||
if xdg:
|
||||
roots.append(Path(xdg) / "fnm")
|
||||
roots.append(Path.home() / ".local" / "share" / "fnm")
|
||||
roots.append(Path.home() / "Library" / "Application Support" / "fnm")
|
||||
|
||||
seen: set[Path] = set()
|
||||
found: list[tuple[tuple[int, int, int], str]] = []
|
||||
for root in roots:
|
||||
versions_dir = root / "node-versions"
|
||||
if versions_dir in seen or not versions_dir.is_dir():
|
||||
continue
|
||||
seen.add(versions_dir)
|
||||
try:
|
||||
entries = list(versions_dir.iterdir())
|
||||
except OSError:
|
||||
continue
|
||||
for entry in entries:
|
||||
node_bin = entry / "installation" / "bin" / "node"
|
||||
if not (node_bin.is_file() and os.access(node_bin, os.X_OK)):
|
||||
continue
|
||||
# Trust the directory name for sorting; the real probe happens in
|
||||
# the caller (a renamed/symlinked dir still gets version-checked).
|
||||
name = entry.name.lstrip("v").split("-", 1)[0]
|
||||
parts = name.split(".")
|
||||
try:
|
||||
ver = (int(parts[0]), int(parts[1]), int(parts[2]))
|
||||
except (IndexError, ValueError):
|
||||
ver = (0, 0, 0)
|
||||
found.append((ver, str(node_bin)))
|
||||
found.sort(key=lambda pair: pair[0], reverse=True)
|
||||
return [path for _, path in found]
|
||||
|
||||
|
||||
def _node26_bin_or_none() -> str | None:
|
||||
"""Resolve a Node >= 26.3.0 binary (no exit — a probe), or ``None``.
|
||||
|
||||
Order: ``HERMES_NODE`` override > ``node`` on PATH > newest fnm-installed
|
||||
version. Each is gated on the real ``--version`` being >= 26.3.0. OpenTUI's
|
||||
native renderer loads via the experimental ``node:ffi`` API that only exists
|
||||
on Node 26.3+, so an older Node is treated as "not available" — but an
|
||||
installed-yet-inactive 26.x (common when fnm's default is on an older line)
|
||||
is discovered and used so the engine still launches.
|
||||
"""
|
||||
candidates: list[str] = []
|
||||
env_node = os.environ.get("HERMES_NODE")
|
||||
if env_node and os.path.isfile(env_node) and os.access(env_node, os.X_OK):
|
||||
candidates.append(env_node)
|
||||
path = shutil.which("node")
|
||||
if path:
|
||||
candidates.append(path)
|
||||
candidates.extend(_fnm_node26_candidates())
|
||||
for cand in candidates:
|
||||
ver = _node_version_tuple(cand)
|
||||
if ver is not None and ver >= NODE26_MIN_VERSION:
|
||||
return cand
|
||||
return None
|
||||
|
||||
|
||||
def _node26_bin() -> str:
|
||||
"""Resolve Node >= 26.3.0 for the OpenTUI engine, or exit with a clear message.
|
||||
|
||||
Use :func:`_node26_bin_or_none` for a non-fatal availability probe.
|
||||
"""
|
||||
node = _node26_bin_or_none()
|
||||
if node is not None:
|
||||
return node
|
||||
print(
|
||||
"Node.js >= 26.3.0 not found — the OpenTUI TUI engine needs it for the "
|
||||
"experimental node:ffi renderer.\n"
|
||||
"Install Node 26.3+ (e.g. via fnm/nvm) or set HERMES_NODE=/path/to/node, "
|
||||
"or unset HERMES_TUI_ENGINE to use the default Ink engine.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _opentui_npm() -> str:
|
||||
"""Resolve npm (ships with Node) to build the OpenTUI bundle, or exit."""
|
||||
npm = shutil.which("npm")
|
||||
if npm:
|
||||
return npm
|
||||
print(
|
||||
"npm not found — needed to build the OpenTUI engine bundle.\n"
|
||||
"Install Node 26.3+ (it ships npm), or unset HERMES_TUI_ENGINE for Ink.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _opentui_available() -> bool:
|
||||
"""Whether the OpenTUI engine can actually launch on this host.
|
||||
|
||||
True only when the platform is supported (not Windows/Termux), a Node >= 26.3
|
||||
binary resolves (the node:ffi floor), AND the v2 package is BUILT
|
||||
(``dist/main.js``) with its ``node_modules`` installed. This gates the DEFAULT
|
||||
engine: a host genuinely set up for OpenTUI defaults to it; everyone else stays
|
||||
on Ink. An explicit ``HERMES_TUI_ENGINE`` env or ``display.tui_engine`` config
|
||||
choice bypasses this probe (and triggers an on-demand build).
|
||||
"""
|
||||
if sys.platform.startswith("win") or _is_termux_startup_environment():
|
||||
return False
|
||||
if _node26_bin_or_none() is None:
|
||||
return False
|
||||
pkg = PROJECT_ROOT / "ui-opentui"
|
||||
built = pkg / "dist" / "main.js"
|
||||
return built.is_file() and (pkg / "node_modules" / "@opentui").is_dir()
|
||||
|
||||
|
||||
def _make_opentui_argv(tui_dev: bool) -> tuple[list[str], Path]:
|
||||
"""Argv for the native OpenTUI engine under Node 26 (no Bun).
|
||||
|
||||
Builds the Solid + Effect-at-boundary engine (``ui-opentui``) with esbuild
|
||||
(``npm run build`` → ``dist/main.js``) when the bundle is missing (or always, in
|
||||
``--dev``), then launches it on Node with the experimental FFI flag:
|
||||
|
||||
node --experimental-ffi --no-warnings dist/main.js
|
||||
|
||||
``--no-warnings`` keeps the ExperimentalWarning off the TUI's stderr. Returns the
|
||||
argv and the package cwd.
|
||||
|
||||
The spawned ``tui_gateway`` resolves its Python from ``HERMES_PYTHON_SRC_ROOT``
|
||||
(the caller sets it to ``PROJECT_ROOT``); the built bundle's own fallback also
|
||||
walks up to the checkout root, so the gateway resolves correctly either way.
|
||||
"""
|
||||
app_dir = PROJECT_ROOT / "ui-opentui"
|
||||
entry_src = app_dir / "src" / "entry" / "main.tsx"
|
||||
if not entry_src.is_file():
|
||||
print(
|
||||
f"OpenTUI v2 engine entry not found at {entry_src}.\n"
|
||||
f"Unset HERMES_TUI_ENGINE to use the default Ink engine.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
node = _node26_bin()
|
||||
|
||||
# The esbuild build needs the package's node_modules (esbuild + the @opentui
|
||||
# packages + the native blob). Without them the build/launch dies cryptically.
|
||||
if not (app_dir / "node_modules" / "@opentui").is_dir():
|
||||
print(
|
||||
f"OpenTUI engine dependencies are not installed in {app_dir}.\n"
|
||||
f"Run: (cd {app_dir} && npm install)\n"
|
||||
f"Or unset HERMES_TUI_ENGINE to use the default Ink engine.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
built = app_dir / "dist" / "main.js"
|
||||
if tui_dev or not built.is_file():
|
||||
npm = _opentui_npm()
|
||||
if not os.environ.get("HERMES_QUIET"):
|
||||
print("Building the OpenTUI engine…", file=sys.stderr)
|
||||
result = subprocess.run(
|
||||
[npm, "run", "build"],
|
||||
cwd=str(app_dir),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
combined = f"{result.stdout or ''}{result.stderr or ''}".strip()
|
||||
preview = "\n".join(combined.splitlines()[-30:])
|
||||
print("OpenTUI engine build failed.", file=sys.stderr)
|
||||
if preview:
|
||||
print(preview, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# --expose-gc (parity with Ink, main.py ~1909): makes `global.gc()` a real
|
||||
# callable so the OpenTUI engine's GC hooks (W2 proactive idle GC; /heapdump)
|
||||
# work instead of being silent no-ops. MUST be an argv flag — Node rejects
|
||||
# --expose-gc in NODE_OPTIONS (see the heap-cap injection below).
|
||||
return [node, "--experimental-ffi", "--no-warnings", "--expose-gc", str(built)], app_dir
|
||||
|
||||
|
||||
def _make_tui_argv(tui_dir: Path, tui_dev: bool) -> tuple[list[str], Path]:
|
||||
"""TUI: --dev → tsx src; else node dist (HERMES_TUI_DIR prebuilt or esbuild)."""
|
||||
"""TUI: --dev → tsx src; else node dist (HERMES_TUI_DIR prebuilt or esbuild).
|
||||
|
||||
Dual-engine: when ``HERMES_TUI_ENGINE``/``display.tui_engine`` selects the
|
||||
native OpenTUI engine, dispatch to ``_make_opentui_argv`` (Node 26 + its own
|
||||
esbuild build) BEFORE the Ink Node bootstrap — the OpenTUI engine resolves its
|
||||
own Node >= 26.3 and builds its own bundle, so it must not be routed through
|
||||
``_ensure_tui_node`` / the Ink prebuilt-dir logic.
|
||||
"""
|
||||
if _resolve_tui_engine() == "opentui":
|
||||
return _make_opentui_argv(tui_dev)
|
||||
|
||||
_ensure_tui_node()
|
||||
|
||||
def _node_bin(bin: str) -> str:
|
||||
@@ -1877,6 +2155,57 @@ def _read_cgroup_memory_limit() -> Optional[int]:
|
||||
return None
|
||||
|
||||
|
||||
def _config_tui_heap_mb_early() -> int | None:
|
||||
"""Read ``display.tui_heap_mb`` from config via a minimal YAML read.
|
||||
|
||||
Returns the configured V8 heap cap in MB, or ``None`` when unset/unreadable.
|
||||
Mirrors :func:`_config_tui_engine_early`. A non-secret behavioral setting, so
|
||||
it lives in ``config.yaml`` (NOT a ``HERMES_*`` env / the NODE_OPTIONS bridge,
|
||||
which is denylisted) — the ``HERMES_TUI_HEAP_MB`` env is only the per-launch
|
||||
override on top of this.
|
||||
"""
|
||||
try:
|
||||
home = os.environ.get("HERMES_HOME")
|
||||
cfg_path = (
|
||||
os.path.join(home, "config.yaml")
|
||||
if home
|
||||
else os.path.join(os.path.expanduser("~"), ".hermes", "config.yaml")
|
||||
)
|
||||
if os.path.exists(cfg_path):
|
||||
import yaml as _yaml_heap
|
||||
|
||||
with open(cfg_path, encoding="utf-8") as _f:
|
||||
raw = _yaml_heap.safe_load(_f) or {}
|
||||
disp = raw.get("display", {})
|
||||
if isinstance(disp, dict):
|
||||
val = disp.get("tui_heap_mb")
|
||||
if isinstance(val, bool): # guard: YAML true/false is an int subclass
|
||||
return None
|
||||
if isinstance(val, int) and val > 0:
|
||||
return val
|
||||
if isinstance(val, str) and val.strip().isdigit():
|
||||
n = int(val.strip())
|
||||
if n > 0:
|
||||
return n
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _resolve_tui_heap_override() -> int | None:
|
||||
"""The user's explicit V8 heap cap (MB), or ``None`` for the default path.
|
||||
|
||||
Precedence: ``HERMES_TUI_HEAP_MB`` env > ``display.tui_heap_mb`` config
|
||||
(matches the ``HERMES_TUI_ENGINE`` env-first pattern). Honored by BOTH engines
|
||||
via the shared ``NODE_OPTIONS`` injection. A positive integer wins; anything
|
||||
else (unset/garbage/non-positive) falls through to the cgroup-aware default.
|
||||
"""
|
||||
env_val = os.environ.get("HERMES_TUI_HEAP_MB", "").strip()
|
||||
if env_val.isdigit() and int(env_val) > 0:
|
||||
return int(env_val)
|
||||
return _config_tui_heap_mb_early()
|
||||
|
||||
|
||||
def _resolve_tui_heap_mb(default_mb: int = 8192) -> int:
|
||||
"""Pick a V8 ``--max-old-space-size`` (MB) that fits the container.
|
||||
|
||||
@@ -1885,7 +2214,16 @@ def _resolve_tui_heap_mb(default_mb: int = 8192) -> int:
|
||||
cgroup limit so the heap + non-heap RSS stays under the cgroup ceiling,
|
||||
clamped to a sane floor (1536MB — below this V8 GC-thrashes and the TUI
|
||||
is barely usable). Never exceeds ``default_mb``.
|
||||
|
||||
An explicit ``HERMES_TUI_HEAP_MB`` env / ``display.tui_heap_mb`` config
|
||||
override REPLACES the 8192 default (D3): setting it low is the low-mem opt-in,
|
||||
setting it high raises the ceiling. The cgroup-fit clamp still applies on top
|
||||
so an override never exceeds what the container can hold — a low override is
|
||||
honored as-is, a too-high one is still trimmed to ~75% of the cgroup limit.
|
||||
"""
|
||||
override = _resolve_tui_heap_override()
|
||||
if override is not None:
|
||||
default_mb = override
|
||||
limit = _read_cgroup_memory_limit()
|
||||
if not limit:
|
||||
return default_mb
|
||||
@@ -1902,7 +2240,8 @@ def _resolve_tui_heap_mb(default_mb: int = 8192) -> int:
|
||||
|
||||
|
||||
def _launch_tui(
|
||||
resume_session_id: Optional[str] = None,
|
||||
# str session id, the bare-`--resume` picker sentinel True, or None.
|
||||
resume_session_id: "Optional[str | bool]" = None,
|
||||
tui_dev: bool = False,
|
||||
model: Optional[str] = None,
|
||||
provider: Optional[str] = None,
|
||||
@@ -1921,6 +2260,14 @@ def _launch_tui(
|
||||
"""Replace current process with the TUI."""
|
||||
tui_dir = PROJECT_ROOT / "ui-tui"
|
||||
|
||||
# Bare `--resume` arrives as the argparse sentinel True: open the TUI
|
||||
# resume picker instead of resuming a specific session id. Normalize it
|
||||
# here so everything downstream (exit summary, env forwarding) keeps
|
||||
# seeing either a real session id string or None.
|
||||
resume_picker = resume_session_id is True
|
||||
if resume_picker:
|
||||
resume_session_id = None
|
||||
|
||||
import tempfile
|
||||
|
||||
env = os.environ.copy()
|
||||
@@ -1934,11 +2281,31 @@ def _launch_tui(
|
||||
)
|
||||
os.close(active_session_fd)
|
||||
env["HERMES_TUI_ACTIVE_SESSION_FILE"] = active_session_file
|
||||
# Tree-sitter grammar cache for the OpenTUI engine: grammars are fetched
|
||||
# from GitHub on first use and cached here (profile-aware). Unset → OpenTUI
|
||||
# falls back to its XDG default ($XDG_DATA_HOME/opentui). See
|
||||
# ui-opentui/src/boundary/parsers.ts.
|
||||
try:
|
||||
from hermes_cli.config import get_hermes_home
|
||||
|
||||
env["HERMES_TUI_PARSER_CACHE"] = str(
|
||||
get_hermes_home() / "cache" / "opentui-parsers"
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("Failed to resolve OpenTUI parser cache dir", exc_info=True)
|
||||
env["HERMES_PYTHON_SRC_ROOT"] = os.environ.get(
|
||||
"HERMES_PYTHON_SRC_ROOT", str(PROJECT_ROOT)
|
||||
)
|
||||
env.setdefault("HERMES_PYTHON", sys.executable)
|
||||
env.setdefault("HERMES_CWD", os.getcwd())
|
||||
# The TUI subprocess is launched with cwd=<engine package dir> (so its
|
||||
# build/resolution works), which means the gateway it spawns would otherwise
|
||||
# auto-detect THAT dir as the workspace (chrome bar showed "ui-opentui" no
|
||||
# matter where you ran hermes). TERMINAL_CWD is the gateway's canonical
|
||||
# launch-dir channel (_completion_cwd) — set it to the real cwd here so the
|
||||
# session, chrome bar, and terminal tool all anchor to where you actually
|
||||
# are. Worktree mode overrides it to the worktree path below.
|
||||
env.setdefault("TERMINAL_CWD", os.getcwd())
|
||||
env.setdefault("NODE_ENV", "development" if tui_dev else "production")
|
||||
|
||||
wt_info = None
|
||||
@@ -2015,6 +2382,11 @@ def _launch_tui(
|
||||
# --expose-gc is *not* added here: Node rejects it in NODE_OPTIONS
|
||||
# ("--expose-gc is not allowed in NODE_OPTIONS") and refuses to start.
|
||||
# It is passed as a direct argv flag in _make_tui_argv() instead.
|
||||
#
|
||||
# Both TUI engines run on Node/V8 now — Ink, and the native OpenTUI engine
|
||||
# (Node 26 + node:ffi). So --max-old-space-size (a V8/Node flag) applies to
|
||||
# both. (Pre-Node-26 the OpenTUI engine ran on Bun/JavaScriptCore, which has
|
||||
# no such flag; that gate is gone now that the engine is Node.)
|
||||
_tokens = env.get("NODE_OPTIONS", "").split()
|
||||
if not any(t.startswith("--max-old-space-size=") for t in _tokens):
|
||||
_tokens.append(f"--max-old-space-size={_resolve_tui_heap_mb()}")
|
||||
@@ -2027,7 +2399,11 @@ def _launch_tui(
|
||||
# resolved for this invocation; direct `node ui-tui/dist/entry.js` users can
|
||||
# still set HERMES_TUI_RESUME themselves.
|
||||
env.pop("HERMES_TUI_RESUME", None)
|
||||
if resume_session_id:
|
||||
if resume_picker:
|
||||
# Bare --resume: tell the TUI to open the resume picker before any
|
||||
# session.create (create is lazy, so nothing is wasted).
|
||||
env["HERMES_TUI_RESUME"] = "picker"
|
||||
elif resume_session_id:
|
||||
env["HERMES_TUI_RESUME"] = resume_session_id
|
||||
|
||||
argv, cwd = _make_tui_argv(tui_dir, tui_dev)
|
||||
@@ -2136,6 +2512,18 @@ def cmd_chat(args):
|
||||
"""Run interactive chat CLI."""
|
||||
use_tui = _resolve_use_tui(args)
|
||||
|
||||
# Bare `--resume` (argparse sentinel True) opens the TUI resume picker —
|
||||
# `_launch_tui` translates it to HERMES_TUI_RESUME=picker. The classic
|
||||
# REPL has no picker overlay, so point at the equivalents instead of
|
||||
# silently resuming something the user didn't choose.
|
||||
if getattr(args, "resume", None) is True and not use_tui:
|
||||
print("Bare --resume opens the session picker, which requires the TUI.")
|
||||
print(
|
||||
"Use 'hermes --tui --resume', 'hermes --resume <id|title>', "
|
||||
"'hermes -c', or 'hermes sessions browse'."
|
||||
)
|
||||
sys.exit(2)
|
||||
|
||||
# Resolve --continue into --resume with the latest session or by name
|
||||
continue_val = getattr(args, "continue_last", None)
|
||||
if continue_val and not getattr(args, "resume", None):
|
||||
@@ -2161,9 +2549,10 @@ def cmd_chat(args):
|
||||
print(f"No previous {kind} session found to continue.")
|
||||
sys.exit(1)
|
||||
|
||||
# Resolve --resume by title if it's not a direct session ID
|
||||
# Resolve --resume by title if it's not a direct session ID. The bare
|
||||
# picker sentinel (True) is not a name — leave it for _launch_tui.
|
||||
resume_val = getattr(args, "resume", None)
|
||||
if resume_val:
|
||||
if resume_val and resume_val is not True:
|
||||
resolved = _resolve_session_by_name_or_id(resume_val)
|
||||
if resolved:
|
||||
args.resume = resolved
|
||||
|
||||
+267
-107
@@ -247,6 +247,19 @@ def _has_valid_session_token(request: Request) -> bool:
|
||||
return hmac.compare_digest(auth.encode(), expected.encode())
|
||||
|
||||
|
||||
# Routes that may also authenticate via a ``?token=`` query param, for download
|
||||
# links opened by the OS shell or a new browser tab where the session header
|
||||
# can't be set. Kept narrow — same query-token tradeoff as the /api/pty WS.
|
||||
_QUERY_TOKEN_API_PATHS: frozenset[str] = frozenset({"/api/files/download"})
|
||||
|
||||
|
||||
def _has_valid_query_token(request: Request, path: str) -> bool:
|
||||
if path not in _QUERY_TOKEN_API_PATHS:
|
||||
return False
|
||||
token = request.query_params.get("token", "")
|
||||
return bool(token) and hmac.compare_digest(token.encode(), _SESSION_TOKEN.encode())
|
||||
|
||||
|
||||
def _require_token(request: Request) -> None:
|
||||
"""Authorize a sensitive endpoint, raising 401 if the caller isn't allowed.
|
||||
|
||||
@@ -403,7 +416,7 @@ async def auth_middleware(request: Request, call_next):
|
||||
return await call_next(request)
|
||||
path = request.url.path
|
||||
if path.startswith("/api/") and path not in _PUBLIC_API_PATHS:
|
||||
if not _has_valid_session_token(request):
|
||||
if not _has_valid_session_token(request) and not _has_valid_query_token(request, path):
|
||||
return JSONResponse(
|
||||
status_code=401,
|
||||
content={"detail": "Unauthorized"},
|
||||
@@ -1224,6 +1237,22 @@ def _default_hermes_root_is_opt_data() -> bool:
|
||||
return root == _HOSTED_MANAGED_FILES_ROOT
|
||||
|
||||
|
||||
def _dashboard_local_update_managed_externally() -> bool:
|
||||
"""Return true when the dashboard should not offer ``hermes update``.
|
||||
|
||||
Containerized dashboards are updated by the outer launcher/image, not by an
|
||||
in-browser local update action. Keep this dashboard capability separate
|
||||
from install-method detection: manual git/pip installs inside containers can
|
||||
still behave like their actual install method in the CLI.
|
||||
"""
|
||||
try:
|
||||
from hermes_constants import is_container
|
||||
|
||||
return is_container()
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _managed_files_policy(request: Request, *, create_root: bool = True) -> ManagedFilesPolicy:
|
||||
raw_forced_root = os.environ.get(_MANAGED_FILES_ROOT_ENV, "").strip()
|
||||
if raw_forced_root:
|
||||
@@ -1393,6 +1422,40 @@ async def read_managed_file(request: Request, path: str):
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/files/download")
|
||||
async def download_managed_file(request: Request, path: str):
|
||||
"""Stream a managed file as an attachment download.
|
||||
|
||||
Remote clients (desktop app, browser dashboard) open agent-written files
|
||||
that live on *this* gateway's disk, not theirs. Auth-gated like every other
|
||||
managed-files route — ``auth_middleware`` additionally accepts the session
|
||||
token as a ``?token=`` query param here so a shell/browser-opened download
|
||||
(which can't set the session header) still authenticates. See ``/api/pty``
|
||||
for the same query-token precedent.
|
||||
"""
|
||||
policy, target, _display_path = _resolve_managed_path(path, request)
|
||||
if not target.exists():
|
||||
raise HTTPException(status_code=404, detail="File not found")
|
||||
if not target.is_file():
|
||||
raise HTTPException(status_code=400, detail="Path is not a file")
|
||||
|
||||
try:
|
||||
size = target.stat().st_size
|
||||
except OSError as exc:
|
||||
raise HTTPException(status_code=500, detail=f"Could not stat file: {exc}")
|
||||
if size > _MANAGED_FILE_MAX_BYTES:
|
||||
raise HTTPException(status_code=413, detail="File is too large")
|
||||
|
||||
mime_type = mimetypes.guess_type(target.name)[0] or "application/octet-stream"
|
||||
|
||||
return FileResponse(
|
||||
path=str(target),
|
||||
media_type=mime_type,
|
||||
filename=target.name,
|
||||
content_disposition_type="attachment",
|
||||
)
|
||||
|
||||
|
||||
@app.post("/api/files/upload")
|
||||
async def upload_managed_file(payload: ManagedFileUpload, request: Request):
|
||||
policy, target, display_path = _resolve_managed_path(payload.path, request, for_write=True)
|
||||
@@ -1654,6 +1717,7 @@ async def get_status():
|
||||
"release_date": __release_date__,
|
||||
"config_version": current_ver,
|
||||
"latest_config_version": latest_ver,
|
||||
"can_update_hermes": not _dashboard_local_update_managed_externally(),
|
||||
"gateway_running": gateway_running,
|
||||
"gateway_state": gateway_state,
|
||||
"gateway_platforms": gateway_platforms,
|
||||
@@ -2165,6 +2229,22 @@ async def restart_gateway():
|
||||
@app.post("/api/hermes/update")
|
||||
async def update_hermes():
|
||||
"""Kick off ``hermes update`` in the background."""
|
||||
if _dashboard_local_update_managed_externally():
|
||||
message = (
|
||||
"Hermes updates are managed outside this dashboard in "
|
||||
"containerized environments. The built-in local updater is "
|
||||
"disabled here."
|
||||
)
|
||||
_record_completed_action("hermes-update", message, exit_code=1)
|
||||
return {
|
||||
"ok": False,
|
||||
"pid": None,
|
||||
"name": "hermes-update",
|
||||
"error": "dashboard_update_managed_externally",
|
||||
"message": message,
|
||||
"update_command": "managed outside dashboard",
|
||||
}
|
||||
|
||||
install_method = detect_install_method(PROJECT_ROOT)
|
||||
if install_method == "docker":
|
||||
message = format_docker_update_message()
|
||||
@@ -2264,6 +2344,20 @@ async def check_hermes_update(force: bool = False):
|
||||
desktop's remote update overlay renders this as "what's
|
||||
changed". Additive: existing consumers ignore it.
|
||||
"""
|
||||
if _dashboard_local_update_managed_externally():
|
||||
return {
|
||||
"install_method": "managed-runtime",
|
||||
"current_version": __version__,
|
||||
"behind": None,
|
||||
"update_available": False,
|
||||
"can_apply": False,
|
||||
"update_command": "managed outside dashboard",
|
||||
"message": (
|
||||
"Hermes updates are managed outside this dashboard in "
|
||||
"containerized environments."
|
||||
),
|
||||
}
|
||||
|
||||
install_method = detect_install_method(PROJECT_ROOT)
|
||||
update_command = recommended_update_command_for_method(install_method)
|
||||
|
||||
@@ -5144,7 +5238,7 @@ def _oauth_provider_disconnect_hint(provider: Dict[str, Any], status: Dict[str,
|
||||
|
||||
|
||||
@app.get("/api/providers/oauth")
|
||||
async def list_oauth_providers():
|
||||
async def list_oauth_providers(profile: Optional[str] = None):
|
||||
"""Enumerate every OAuth-capable LLM provider with current status.
|
||||
|
||||
Response shape (per provider):
|
||||
@@ -5161,83 +5255,89 @@ async def list_oauth_providers():
|
||||
expires_at ISO timestamp string or null
|
||||
has_refresh_token bool
|
||||
"""
|
||||
providers = []
|
||||
for p in _OAUTH_PROVIDER_CATALOG:
|
||||
status = _resolve_provider_status(p["id"], p.get("status_fn"))
|
||||
disconnect_hint = _oauth_provider_disconnect_hint(p, status)
|
||||
providers.append({
|
||||
"id": p["id"],
|
||||
"name": p["name"],
|
||||
"flow": p["flow"],
|
||||
"cli_command": p["cli_command"],
|
||||
"docs_url": p["docs_url"],
|
||||
"disconnect_hint": disconnect_hint,
|
||||
"disconnectable": disconnect_hint is None,
|
||||
"status": status,
|
||||
})
|
||||
return {"providers": providers}
|
||||
with _profile_scope(profile):
|
||||
providers = []
|
||||
for p in _OAUTH_PROVIDER_CATALOG:
|
||||
status = _resolve_provider_status(p["id"], p.get("status_fn"))
|
||||
disconnect_hint = _oauth_provider_disconnect_hint(p, status)
|
||||
providers.append({
|
||||
"id": p["id"],
|
||||
"name": p["name"],
|
||||
"flow": p["flow"],
|
||||
"cli_command": p["cli_command"],
|
||||
"docs_url": p["docs_url"],
|
||||
"disconnect_hint": disconnect_hint,
|
||||
"disconnectable": disconnect_hint is None,
|
||||
"status": status,
|
||||
})
|
||||
return {"providers": providers}
|
||||
|
||||
|
||||
@app.delete("/api/providers/oauth/{provider_id}")
|
||||
async def disconnect_oauth_provider(provider_id: str, request: Request):
|
||||
async def disconnect_oauth_provider(
|
||||
provider_id: str,
|
||||
request: Request,
|
||||
profile: Optional[str] = None,
|
||||
):
|
||||
"""Disconnect an OAuth provider. Token-protected (matches /env/reveal)."""
|
||||
_require_token(request)
|
||||
|
||||
catalog_by_id = {p["id"]: p for p in _OAUTH_PROVIDER_CATALOG}
|
||||
provider = catalog_by_id.get(provider_id)
|
||||
if provider is None:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Unknown provider: {provider_id}. "
|
||||
f"Available: {', '.join(sorted(catalog_by_id))}",
|
||||
)
|
||||
with _profile_scope(profile):
|
||||
catalog_by_id = {p["id"]: p for p in _OAUTH_PROVIDER_CATALOG}
|
||||
provider = catalog_by_id.get(provider_id)
|
||||
if provider is None:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Unknown provider: {provider_id}. "
|
||||
f"Available: {', '.join(sorted(catalog_by_id))}",
|
||||
)
|
||||
|
||||
disconnect_hint = _oauth_provider_disconnect_hint(provider, {})
|
||||
if disconnect_hint:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"{provider['name']} cannot be disconnected automatically. {disconnect_hint}",
|
||||
)
|
||||
disconnect_hint = _oauth_provider_disconnect_hint(provider, {})
|
||||
if disconnect_hint:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"{provider['name']} cannot be disconnected automatically. {disconnect_hint}",
|
||||
)
|
||||
|
||||
status = _resolve_provider_status(provider_id, provider.get("status_fn"))
|
||||
disconnect_hint = _oauth_provider_disconnect_hint(provider, status)
|
||||
if disconnect_hint:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"{provider['name']} cannot be disconnected automatically. {disconnect_hint}",
|
||||
)
|
||||
status = _resolve_provider_status(provider_id, provider.get("status_fn"))
|
||||
disconnect_hint = _oauth_provider_disconnect_hint(provider, status)
|
||||
if disconnect_hint:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"{provider['name']} cannot be disconnected automatically. {disconnect_hint}",
|
||||
)
|
||||
|
||||
# Anthropic clears only the Hermes-managed PKCE file and auth-store entry.
|
||||
# The separate claude-code catalog row is external/read-only and rejected
|
||||
# above so we never pretend to remove ~/.claude/* credentials owned by the CLI.
|
||||
if provider_id == "anthropic":
|
||||
cleared = False
|
||||
try:
|
||||
from agent.anthropic_adapter import _HERMES_OAUTH_FILE
|
||||
if _HERMES_OAUTH_FILE.exists():
|
||||
_HERMES_OAUTH_FILE.unlink()
|
||||
cleared = True
|
||||
except Exception:
|
||||
pass
|
||||
# Also clear the credential pool entry if present.
|
||||
try:
|
||||
from hermes_cli.auth import clear_provider_auth
|
||||
cleared = clear_provider_auth("anthropic") or cleared
|
||||
except Exception:
|
||||
pass
|
||||
_log.info("oauth/disconnect: %s", provider_id)
|
||||
return {"ok": bool(cleared), "provider": provider_id}
|
||||
|
||||
# Anthropic clears only the Hermes-managed PKCE file and auth-store entry.
|
||||
# The separate claude-code catalog row is external/read-only and rejected
|
||||
# above so we never pretend to remove ~/.claude/* credentials owned by the CLI.
|
||||
if provider_id == "anthropic":
|
||||
cleared = False
|
||||
try:
|
||||
from agent.anthropic_adapter import _HERMES_OAUTH_FILE
|
||||
if _HERMES_OAUTH_FILE.exists():
|
||||
_HERMES_OAUTH_FILE.unlink()
|
||||
cleared = True
|
||||
except Exception:
|
||||
pass
|
||||
# Also clear the credential pool entry if present.
|
||||
try:
|
||||
from hermes_cli.auth import clear_provider_auth
|
||||
cleared = clear_provider_auth("anthropic") or cleared
|
||||
except Exception:
|
||||
pass
|
||||
_log.info("oauth/disconnect: %s", provider_id)
|
||||
return {"ok": bool(cleared), "provider": provider_id}
|
||||
|
||||
try:
|
||||
from hermes_cli.auth import clear_provider_auth, invalidate_nous_auth_status_cache
|
||||
cleared = clear_provider_auth(provider_id)
|
||||
if provider_id == "nous":
|
||||
invalidate_nous_auth_status_cache()
|
||||
_log.info("oauth/disconnect: %s (cleared=%s)", provider_id, cleared)
|
||||
return {"ok": bool(cleared), "provider": provider_id}
|
||||
except Exception as e:
|
||||
_log.exception("disconnect %s failed", provider_id)
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
from hermes_cli.auth import clear_provider_auth, invalidate_nous_auth_status_cache
|
||||
cleared = clear_provider_auth(provider_id)
|
||||
if provider_id == "nous":
|
||||
invalidate_nous_auth_status_cache()
|
||||
_log.info("oauth/disconnect: %s (cleared=%s)", provider_id, cleared)
|
||||
return {"ok": bool(cleared), "provider": provider_id}
|
||||
except Exception as e:
|
||||
_log.exception("disconnect %s failed", provider_id)
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -5319,13 +5419,32 @@ def _gc_oauth_sessions() -> None:
|
||||
_oauth_sessions.pop(sid, None)
|
||||
|
||||
|
||||
def _new_oauth_session(provider_id: str, flow: str) -> tuple[str, Dict[str, Any]]:
|
||||
def _oauth_profile_name(profile: Optional[str]) -> Optional[str]:
|
||||
requested = (profile or "").strip()
|
||||
if not requested or requested.lower() == "current":
|
||||
return None
|
||||
return requested
|
||||
|
||||
|
||||
def _validate_oauth_profile(profile: Optional[str]) -> None:
|
||||
profile_name = _oauth_profile_name(profile)
|
||||
if profile_name:
|
||||
_resolve_profile_dir(profile_name)
|
||||
|
||||
|
||||
def _new_oauth_session(
|
||||
provider_id: str,
|
||||
flow: str,
|
||||
profile: Optional[str] = None,
|
||||
) -> tuple[str, Dict[str, Any]]:
|
||||
"""Create + register a new OAuth session, return (session_id, session_dict)."""
|
||||
sid = secrets.token_urlsafe(16)
|
||||
profile_name = _oauth_profile_name(profile)
|
||||
sess = {
|
||||
"session_id": sid,
|
||||
"provider": provider_id,
|
||||
"flow": flow,
|
||||
"profile": profile_name,
|
||||
"created_at": time.time(),
|
||||
"status": "pending", # pending | approved | denied | expired | error
|
||||
"error_message": None,
|
||||
@@ -5335,6 +5454,17 @@ def _new_oauth_session(provider_id: str, flow: str) -> tuple[str, Dict[str, Any]
|
||||
return sid, sess
|
||||
|
||||
|
||||
def _oauth_session_profile(
|
||||
session_id: str,
|
||||
fallback: Optional[str] = None,
|
||||
) -> Optional[str]:
|
||||
"""Return the profile that owns an OAuth session, if one was provided."""
|
||||
with _oauth_sessions_lock:
|
||||
sess = _oauth_sessions.get(session_id)
|
||||
profile = sess.get("profile") if sess else None
|
||||
return profile or _oauth_profile_name(fallback)
|
||||
|
||||
|
||||
def _save_anthropic_oauth_creds(access_token: str, refresh_token: str, expires_at_ms: int) -> None:
|
||||
"""Persist Anthropic PKCE creds to both Hermes file AND credential pool.
|
||||
|
||||
@@ -5402,12 +5532,12 @@ def _save_anthropic_oauth_creds(access_token: str, refresh_token: str, expires_a
|
||||
_log.warning("anthropic pool add (dashboard) failed: %s", e)
|
||||
|
||||
|
||||
def _start_anthropic_pkce() -> Dict[str, Any]:
|
||||
def _start_anthropic_pkce(profile: Optional[str] = None) -> Dict[str, Any]:
|
||||
"""Begin PKCE flow. Returns the auth URL the UI should open."""
|
||||
if not _ANTHROPIC_OAUTH_AVAILABLE:
|
||||
raise HTTPException(status_code=501, detail="Anthropic OAuth not available (missing adapter)")
|
||||
verifier, challenge = _generate_pkce_pair()
|
||||
sid, sess = _new_oauth_session("anthropic", "pkce")
|
||||
sid, sess = _new_oauth_session("anthropic", "pkce", profile=profile)
|
||||
sess["verifier"] = verifier
|
||||
sess["state"] = verifier # Anthropic round-trips verifier as state
|
||||
params = {
|
||||
@@ -5429,7 +5559,11 @@ def _start_anthropic_pkce() -> Dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def _submit_anthropic_pkce(session_id: str, code_input: str) -> Dict[str, Any]:
|
||||
def _submit_anthropic_pkce(
|
||||
session_id: str,
|
||||
code_input: str,
|
||||
profile: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Exchange authorization code for tokens. Persists on success."""
|
||||
with _oauth_sessions_lock:
|
||||
sess = _oauth_sessions.get(session_id)
|
||||
@@ -5483,7 +5617,8 @@ def _submit_anthropic_pkce(session_id: str, code_input: str) -> Dict[str, Any]:
|
||||
|
||||
expires_at_ms = int(time.time() * 1000) + (expires_in * 1000)
|
||||
try:
|
||||
_save_anthropic_oauth_creds(access_token, refresh_token, expires_at_ms)
|
||||
with _profile_scope(_oauth_session_profile(session_id, profile)):
|
||||
_save_anthropic_oauth_creds(access_token, refresh_token, expires_at_ms)
|
||||
except Exception as e:
|
||||
with _oauth_sessions_lock:
|
||||
sess["status"] = "error"
|
||||
@@ -5495,7 +5630,10 @@ def _submit_anthropic_pkce(session_id: str, code_input: str) -> Dict[str, Any]:
|
||||
return {"ok": True, "status": "approved"}
|
||||
|
||||
|
||||
async def _start_device_code_flow(provider_id: str) -> Dict[str, Any]:
|
||||
async def _start_device_code_flow(
|
||||
provider_id: str,
|
||||
profile: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Initiate a device-code flow (Nous, OpenAI Codex, or MiniMax).
|
||||
|
||||
Calls the provider's device-auth endpoint via the existing CLI helpers,
|
||||
@@ -5535,7 +5673,7 @@ async def _start_device_code_flow(provider_id: str) -> Dict[str, Any]:
|
||||
device_data, effective_scope = await asyncio.get_running_loop().run_in_executor(
|
||||
None, _do_nous_device_request
|
||||
)
|
||||
sid, sess = _new_oauth_session("nous", "device_code")
|
||||
sid, sess = _new_oauth_session("nous", "device_code", profile=profile)
|
||||
sess["device_code"] = str(device_data["device_code"])
|
||||
sess["interval"] = int(device_data["interval"])
|
||||
sess["expires_at"] = time.time() + int(device_data["expires_in"])
|
||||
@@ -5556,7 +5694,7 @@ async def _start_device_code_flow(provider_id: str) -> Dict[str, Any]:
|
||||
|
||||
if provider_id == "openai-codex":
|
||||
# Codex uses fixed OpenAI device-auth endpoints; reuse the helper.
|
||||
sid, _ = _new_oauth_session("openai-codex", "device_code")
|
||||
sid, _ = _new_oauth_session("openai-codex", "device_code", profile=profile)
|
||||
# Use the helper but in a thread because it polls inline.
|
||||
# We can't extract just the start step without refactoring auth.py,
|
||||
# so we run the full helper in a worker and proxy the user_code +
|
||||
@@ -5623,7 +5761,7 @@ async def _start_device_code_flow(provider_id: str) -> Dict[str, Any]:
|
||||
device_data = await asyncio.get_event_loop().run_in_executor(
|
||||
None, _do_minimax_request
|
||||
)
|
||||
sid, sess = _new_oauth_session("minimax-oauth", "device_code")
|
||||
sid, sess = _new_oauth_session("minimax-oauth", "device_code", profile=profile)
|
||||
# The CLI flow names this `interval_ms` because MiniMax's
|
||||
# `interval` field is in milliseconds (defensive default 2000ms
|
||||
# in _minimax_poll_token).
|
||||
@@ -5677,7 +5815,7 @@ async def _start_device_code_flow(provider_id: str) -> Dict[str, Any]:
|
||||
_XAI_LOOPBACK_TIMEOUT_SECONDS = 300.0
|
||||
|
||||
|
||||
def _start_xai_loopback_flow() -> Dict[str, Any]:
|
||||
def _start_xai_loopback_flow(profile: Optional[str] = None) -> Dict[str, Any]:
|
||||
"""Begin the xAI loopback PKCE flow.
|
||||
|
||||
Binds the local callback server, builds the authorize URL, and spawns a
|
||||
@@ -5716,7 +5854,7 @@ def _start_xai_loopback_flow() -> Dict[str, Any]:
|
||||
pass
|
||||
raise
|
||||
|
||||
sid, sess = _new_oauth_session("xai-oauth", "loopback")
|
||||
sid, sess = _new_oauth_session("xai-oauth", "loopback", profile=profile)
|
||||
sess["server"] = server
|
||||
sess["thread"] = thread
|
||||
sess["callback_result"] = callback_result
|
||||
@@ -5819,13 +5957,14 @@ def _xai_loopback_worker(session_id: str) -> None:
|
||||
}
|
||||
if _cancelled():
|
||||
return
|
||||
hauth._save_xai_oauth_tokens(
|
||||
tokens,
|
||||
discovery=sess.get("discovery"),
|
||||
redirect_uri=sess["redirect_uri"],
|
||||
last_refresh=last_refresh,
|
||||
)
|
||||
_add_xai_oauth_pool_entry(access_token, refresh_token, base_url, last_refresh)
|
||||
with _profile_scope(_oauth_session_profile(session_id)):
|
||||
hauth._save_xai_oauth_tokens(
|
||||
tokens,
|
||||
discovery=sess.get("discovery"),
|
||||
redirect_uri=sess["redirect_uri"],
|
||||
last_refresh=last_refresh,
|
||||
)
|
||||
_add_xai_oauth_pool_entry(access_token, refresh_token, base_url, last_refresh)
|
||||
except Exception as exc:
|
||||
_fail(f"xAI token exchange failed: {exc}")
|
||||
return
|
||||
@@ -5928,13 +6067,14 @@ def _nous_poller(session_id: str) -> None:
|
||||
),
|
||||
"expires_in": token_ttl,
|
||||
}
|
||||
full_state = refresh_nous_oauth_from_state(
|
||||
auth_state,
|
||||
timeout_seconds=15.0,
|
||||
force_refresh=False,
|
||||
)
|
||||
from hermes_cli.auth import persist_nous_credentials
|
||||
persist_nous_credentials(full_state)
|
||||
with _profile_scope(_oauth_session_profile(session_id)):
|
||||
full_state = refresh_nous_oauth_from_state(
|
||||
auth_state,
|
||||
timeout_seconds=15.0,
|
||||
force_refresh=False,
|
||||
)
|
||||
from hermes_cli.auth import persist_nous_credentials
|
||||
persist_nous_credentials(full_state)
|
||||
with _oauth_sessions_lock:
|
||||
sess["status"] = "approved"
|
||||
_log.info("oauth/device: nous login completed (session=%s)", session_id)
|
||||
@@ -6017,7 +6157,8 @@ def _minimax_poller(session_id: str) -> None:
|
||||
).isoformat(),
|
||||
"expires_in": expires_in_s,
|
||||
}
|
||||
_minimax_save_auth_state(auth_state)
|
||||
with _profile_scope(_oauth_session_profile(session_id)):
|
||||
_minimax_save_auth_state(auth_state)
|
||||
with _oauth_sessions_lock:
|
||||
sess["status"] = "approved"
|
||||
_log.info("oauth/device: minimax login completed (session=%s)", session_id)
|
||||
@@ -6130,10 +6271,11 @@ def _codex_full_login_worker(session_id: str) -> None:
|
||||
|
||||
from hermes_cli.auth import _save_codex_tokens
|
||||
|
||||
_save_codex_tokens({
|
||||
"access_token": access_token,
|
||||
"refresh_token": refresh_token,
|
||||
})
|
||||
with _profile_scope(_oauth_session_profile(session_id)):
|
||||
_save_codex_tokens({
|
||||
"access_token": access_token,
|
||||
"refresh_token": refresh_token,
|
||||
})
|
||||
with _oauth_sessions_lock:
|
||||
sess["status"] = "approved"
|
||||
_log.info("oauth/device: openai-codex login completed (session=%s)", session_id)
|
||||
@@ -6147,10 +6289,15 @@ def _codex_full_login_worker(session_id: str) -> None:
|
||||
|
||||
|
||||
@app.post("/api/providers/oauth/{provider_id}/start")
|
||||
async def start_oauth_login(provider_id: str, request: Request):
|
||||
async def start_oauth_login(
|
||||
provider_id: str,
|
||||
request: Request,
|
||||
profile: Optional[str] = None,
|
||||
):
|
||||
"""Initiate an OAuth login flow. Token-protected."""
|
||||
_require_token(request)
|
||||
_gc_oauth_sessions()
|
||||
_validate_oauth_profile(profile)
|
||||
valid = {p["id"] for p in _OAUTH_PROVIDER_CATALOG}
|
||||
if provider_id not in valid:
|
||||
raise HTTPException(status_code=400, detail=f"Unknown provider {provider_id}")
|
||||
@@ -6168,12 +6315,12 @@ async def start_oauth_login(provider_id: str, request: Request):
|
||||
# change for MiniMax). New PKCE providers must add their own
|
||||
# start function and an explicit branch here.
|
||||
if catalog_entry["flow"] == "pkce" and provider_id == "anthropic":
|
||||
return _start_anthropic_pkce()
|
||||
return _start_anthropic_pkce(profile=profile)
|
||||
if catalog_entry["flow"] == "device_code":
|
||||
return await _start_device_code_flow(provider_id)
|
||||
return await _start_device_code_flow(provider_id, profile=profile)
|
||||
if catalog_entry["flow"] == "loopback" and provider_id == "xai-oauth":
|
||||
return await asyncio.get_running_loop().run_in_executor(
|
||||
None, _start_xai_loopback_flow
|
||||
None, _start_xai_loopback_flow, profile,
|
||||
)
|
||||
except HTTPException:
|
||||
raise
|
||||
@@ -6189,18 +6336,27 @@ class OAuthSubmitBody(BaseModel):
|
||||
|
||||
|
||||
@app.post("/api/providers/oauth/{provider_id}/submit")
|
||||
async def submit_oauth_code(provider_id: str, body: OAuthSubmitBody, request: Request):
|
||||
async def submit_oauth_code(
|
||||
provider_id: str,
|
||||
body: OAuthSubmitBody,
|
||||
request: Request,
|
||||
profile: Optional[str] = None,
|
||||
):
|
||||
"""Submit the auth code for PKCE flows. Token-protected."""
|
||||
_require_token(request)
|
||||
if provider_id == "anthropic":
|
||||
return await asyncio.get_running_loop().run_in_executor(
|
||||
None, _submit_anthropic_pkce, body.session_id, body.code,
|
||||
None, _submit_anthropic_pkce, body.session_id, body.code, profile,
|
||||
)
|
||||
raise HTTPException(status_code=400, detail=f"submit not supported for {provider_id}")
|
||||
|
||||
|
||||
@app.get("/api/providers/oauth/{provider_id}/poll/{session_id}")
|
||||
async def poll_oauth_session(provider_id: str, session_id: str):
|
||||
async def poll_oauth_session(
|
||||
provider_id: str,
|
||||
session_id: str,
|
||||
profile: Optional[str] = None,
|
||||
):
|
||||
"""Poll a session's status (no auth — read-only state).
|
||||
|
||||
Shared by the device-code flows (Nous, OpenAI Codex, MiniMax) and the
|
||||
@@ -6223,7 +6379,11 @@ async def poll_oauth_session(provider_id: str, session_id: str):
|
||||
|
||||
|
||||
@app.delete("/api/providers/oauth/sessions/{session_id}")
|
||||
async def cancel_oauth_session(session_id: str, request: Request):
|
||||
async def cancel_oauth_session(
|
||||
session_id: str,
|
||||
request: Request,
|
||||
profile: Optional[str] = None,
|
||||
):
|
||||
"""Cancel a pending OAuth session. Token-protected."""
|
||||
_require_token(request)
|
||||
with _oauth_sessions_lock:
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
---
|
||||
name: mpp-agent
|
||||
description: Pay HTTP 402 APIs via Machine Payments Protocol (MPP).
|
||||
version: 0.1.0
|
||||
author: Teknium (teknium1), Hermes Agent
|
||||
license: MIT
|
||||
platforms: [linux, macos]
|
||||
metadata:
|
||||
hermes:
|
||||
tags: [Payments, MPP, HTTP-402, Tempo, Stripe]
|
||||
related_skills: [stripe-link-cli, stripe-projects]
|
||||
---
|
||||
|
||||
# MPP Agent Skill
|
||||
|
||||
Wraps the Machine Payments Protocol (MPP, https://mpp.dev) clients so Hermes can pay for per-request API access against servers that respond with `HTTP 402 Payment Required`.
|
||||
|
||||
Three client options, all distributed via npm. Pick the lightest one that solves the user's need. Gated `[linux, macos]` while the broader payments tooling matures on Windows.
|
||||
|
||||
## When to Use
|
||||
|
||||
- A merchant API returns `HTTP 402` with a `www-authenticate` header — and the user wants to actually pay it, not just log the response.
|
||||
- The user asks to "pay per request", "set up an agent wallet", "use Tempo / Privy / AgentCash", or wants to discover MPP-priced services.
|
||||
- A Stripe Link spend has produced a Shared Payment Token (SPT) and the agent needs to attach it to the 402 challenge — in that flow, prefer `link-cli mpp pay` (see the `stripe-link-cli` skill).
|
||||
|
||||
## Choosing a client
|
||||
|
||||
| Tool | When | Setup |
|
||||
|---|---|---|
|
||||
| `link-cli` | User already has Stripe Link set up, or the 402 challenge advertises `method="stripe"` | see the `stripe-link-cli` skill |
|
||||
| Tempo Wallet | MPP services with spend controls, service discovery | `tempo wallet login` |
|
||||
| Privy Agent CLI | Multi-chain wallets, browser-based funding | `privy-agent-wallets login` |
|
||||
| AgentCash | 300+ pre-priced APIs via one USDC.e balance | `npx agentcash onboard` |
|
||||
| `mppx` | Dev + debugging, smallest dep surface | `npm install -g mppx` then `mppx account create` |
|
||||
|
||||
Default: if the user already has Stripe Link configured or the 402 challenge specifies `method="stripe"`, use `link-cli mpp pay` (the `stripe-link-cli` skill). Otherwise `mppx` for one-off paid calls and debugging, and Tempo Wallet when the user wants persistent spend controls.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Node.js 20+ on `PATH`
|
||||
- A funded wallet (Tempo / Privy / AgentCash) OR an `mppx` account
|
||||
- For Tempo / Privy / AgentCash: follow their respective onboarding skills:
|
||||
- `https://tempo.xyz/SKILL.md`
|
||||
- `https://agents.privy.io/skill.md`
|
||||
- `https://agentcash.dev/skill.md`
|
||||
|
||||
Use `web_extract` to fetch any of those SKILL.md files if the user picks one.
|
||||
|
||||
## Procedure (mppx, fastest path)
|
||||
|
||||
Run all commands through the `terminal` tool.
|
||||
|
||||
### 1. Install + create an account
|
||||
|
||||
```
|
||||
npm install -g mppx
|
||||
mppx account create
|
||||
```
|
||||
|
||||
Store the resulting account credentials wherever the CLI tells you (the CLI writes them under its own config — do not paste them into the agent transcript).
|
||||
|
||||
### 2. Inspect the merchant's 402 challenge
|
||||
|
||||
If the user gives you a URL, probe it first to confirm it actually speaks MPP:
|
||||
|
||||
```
|
||||
curl -i <url>
|
||||
```
|
||||
|
||||
A real MPP 402 looks like:
|
||||
|
||||
```
|
||||
HTTP/1.1 402 Payment Required
|
||||
www-authenticate: tempo amount=0.1 currency=...
|
||||
```
|
||||
|
||||
### 3. Pay the request
|
||||
|
||||
```
|
||||
mppx <url>
|
||||
```
|
||||
|
||||
For non-GET methods or request bodies:
|
||||
|
||||
```
|
||||
mppx <url> --method POST --data '<json>'
|
||||
```
|
||||
|
||||
`mppx` handles the 402 challenge/credential dance automatically and prints the merchant's actual response on success.
|
||||
|
||||
### 4. Verify the receipt
|
||||
|
||||
`mppx` attaches the receipt header automatically. To inspect:
|
||||
|
||||
```
|
||||
mppx <url> -v
|
||||
```
|
||||
|
||||
## Procedure (Tempo Wallet)
|
||||
|
||||
The Tempo Wallet skill at https://tempo.xyz/SKILL.md is the canonical reference; fetch it with `web_extract` and follow it. Headline:
|
||||
|
||||
```
|
||||
tempo wallet login
|
||||
tempo wallet pay <url>
|
||||
```
|
||||
|
||||
Spend controls and service discovery live in the wallet UI at https://wallet.tempo.xyz.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- **`HTTP 402` without `method="stripe"` cannot be paid by Stripe Link.** If the challenge advertises only Tempo / other methods, use `mppx` (or whichever wallet matches) — Link will reject it. Conversely, if it advertises `method="stripe"`, prefer Link via the `stripe-link-cli` skill so the spend goes through the user's approved card.
|
||||
- **Multiple challenges in one header.** `www-authenticate` may list several methods (e.g. `tempo, stripe`). The Link CLI's `mpp decode` will pick the Stripe one; `mppx` will pick Tempo. There's no single "right" client — pick by which wallet the user has funded.
|
||||
- **Zero-amount challenges.** Some MPP endpoints charge `$0.00` and just want a proof credential. These work without a funded wallet. Don't refuse them as "broken."
|
||||
- **Wallet keys never enter agent context.** All four clients store keys under their own config dirs (or generate per-session ephemeral keypairs, in Privy's case). Do not `cat`/`read_file` them.
|
||||
- **Server-side MPP is a different skill.** If the user wants to ADD 402 to their own API, this skill is wrong — point them at https://mpp.dev/quickstart/server and the `mppx/nextjs` / `mppx/hono` / `mppx/express` / `mppx/elysia` middlewares. A dedicated `mpp-server` skill may land later.
|
||||
|
||||
## Verification
|
||||
|
||||
```
|
||||
mppx --version && mppx account list
|
||||
```
|
||||
|
||||
Exit code 0 means installed and an account exists.
|
||||
@@ -0,0 +1,184 @@
|
||||
---
|
||||
name: stripe-link-cli
|
||||
description: Agent payments via Stripe Link — cards, SPT, approvals.
|
||||
version: 0.1.0
|
||||
author: Teknium (teknium1), Hermes Agent
|
||||
license: MIT
|
||||
platforms: [linux, macos]
|
||||
metadata:
|
||||
hermes:
|
||||
tags: [Payments, Stripe, Link, Checkout, MPP]
|
||||
related_skills: [mpp-agent, stripe-projects]
|
||||
---
|
||||
|
||||
# Stripe Link CLI Skill
|
||||
|
||||
Wraps [@stripe/link-cli](https://github.com/stripe/link-cli) so Hermes can complete purchases on the user's behalf using one-time-use virtual cards or Shared Payment Tokens (SPT). Every spend is gated by an in-app approval in the Link mobile/web app — Hermes cannot self-approve.
|
||||
|
||||
US-only at the moment (Link account requirement). Windows is not supported by the upstream CLI — this skill is gated `[linux, macos]`.
|
||||
|
||||
## When to Use
|
||||
|
||||
Trigger phrases:
|
||||
|
||||
- "buy X", "pay for X", "make a purchase", "complete checkout"
|
||||
- "get me a card", "I need a payment method"
|
||||
- "log in to Link", "connect my Link wallet"
|
||||
- HTTP 402 response from a merchant API with `www-authenticate: ... method="stripe"`
|
||||
|
||||
If the user wants a paid API call (HTTP 402, no checkout form), the `card` path is wrong — use SPT via this same skill, or hand off to the `mpp-agent` skill.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Node.js 20+ available on `PATH` (`node --version`)
|
||||
- US-based (Link account requirement)
|
||||
|
||||
The Link account, payment method, and spend-approval app do NOT need to be set up before Hermes attempts to pay — the CLI walks the user through them on first run:
|
||||
|
||||
- A Link account at https://app.link.com — created/linked during first `link-cli` auth
|
||||
- At least one payment method — added during first run at https://app.link.com/wallet
|
||||
- The Link mobile/web app — opened to approve the first spend request when it's made
|
||||
|
||||
No env vars required — auth state is stored locally by the CLI under its own config directory.
|
||||
|
||||
## Install
|
||||
|
||||
Install once, globally:
|
||||
|
||||
```
|
||||
npm install -g @stripe/link-cli
|
||||
```
|
||||
|
||||
Or invoke ad-hoc via `npx @stripe/link-cli`. The skill below uses the installed `link-cli` form.
|
||||
|
||||
## How to Run
|
||||
|
||||
All commands run through the `terminal` tool. The CLI auto-detects non-TTY callers and emits compact `toon` output by default — fine for the model. Pass `--format json` if a step needs structured fields.
|
||||
|
||||
Discover commands: `link-cli --llms-full`.
|
||||
Get a command's schema before invoking: `link-cli <command> --schema`.
|
||||
|
||||
## Procedure
|
||||
|
||||
### 1. Check / establish auth
|
||||
|
||||
```
|
||||
link-cli auth status
|
||||
```
|
||||
|
||||
If not authenticated, log in with a clear client name (this label shows in the user's Link app):
|
||||
|
||||
```
|
||||
link-cli auth login --client-name "Hermes" --interval 5 --timeout 300
|
||||
```
|
||||
|
||||
The `--interval`/`--timeout` form polls inline so the agent doesn't need to manage a `_next` step. Print the verification URL + phrase to the user and wait for the CLI to return.
|
||||
|
||||
**Do not proceed past this step until `auth status` confirms login.**
|
||||
|
||||
### 2. Evaluate the merchant before creating a spend request
|
||||
|
||||
Decide the credential type:
|
||||
|
||||
| Merchant surface | `--credential-type` |
|
||||
|---|---|
|
||||
| Standard web checkout form / Stripe Elements | `card` (default) |
|
||||
| Returns HTTP 402 with `method="stripe"` in `www-authenticate` | `shared_payment_token` |
|
||||
| Returns HTTP 402 without `method="stripe"` | unsupported — stop |
|
||||
|
||||
For 402 responses, do NOT decode the challenge manually. Pass the raw header:
|
||||
|
||||
```
|
||||
link-cli mpp decode --challenge '<full WWW-Authenticate header>'
|
||||
```
|
||||
|
||||
This validates the challenge and extracts the network ID + decoded request body.
|
||||
|
||||
### 3. List payment methods + shipping
|
||||
|
||||
```
|
||||
link-cli payment-methods list
|
||||
link-cli shipping-address list
|
||||
```
|
||||
|
||||
Use the first entry unless the user specifies otherwise. The `id` from `payment-methods list` is the `--payment-method-id` in the next step.
|
||||
|
||||
### 4. Create the spend request
|
||||
|
||||
Confirm the final total with the user before issuing this command. Amounts are in cents.
|
||||
|
||||
```
|
||||
link-cli spend-request create \
|
||||
--payment-method-id <pm_id> \
|
||||
--merchant-name "<name>" \
|
||||
--merchant-url "<url>" \
|
||||
--context "<one sentence: what is being purchased and why>" \
|
||||
--amount <cents> \
|
||||
--line-item "name:<item>,unit_amount:<cents>,quantity:1" \
|
||||
--total "type:total,display_text:Total,amount:<cents>" \
|
||||
--request-approval
|
||||
```
|
||||
|
||||
For MPP merchants add `--credential-type shared_payment_token`.
|
||||
|
||||
`--request-approval` pings the user's Link app and polls until they approve or deny. The CLI exits non-zero on deny / timeout.
|
||||
|
||||
### 5. Retrieve the credential — SECURELY
|
||||
|
||||
**Do not print card details to stdout.** Use `--output-file` so the PAN never enters the agent's transcript or logs:
|
||||
|
||||
```
|
||||
link-cli spend-request retrieve <lsrq_id> \
|
||||
--include card \
|
||||
--output-file /tmp/link-card.json \
|
||||
--format json
|
||||
```
|
||||
|
||||
The file is written with `0600` perms; stdout shows only redacted fields (brand, last4, expiry) plus a `card_output_file` path.
|
||||
|
||||
### 6. Use the credential
|
||||
|
||||
- For web checkout: hand the file path to the user, OR pass it to a browser-driving tool that fills the form directly from disk. Never `read_file` or `cat` the card file into the agent's reasoning context.
|
||||
- For MPP merchants:
|
||||
|
||||
```
|
||||
link-cli mpp pay <merchant-url> \
|
||||
--spend-request-id <lsrq_id> \
|
||||
--method POST \
|
||||
--data '<json body>'
|
||||
```
|
||||
|
||||
### 7. Clean up
|
||||
|
||||
Delete the card file as soon as the purchase is done:
|
||||
|
||||
```
|
||||
rm -f /tmp/link-card.json
|
||||
```
|
||||
|
||||
## Optional: run as an MCP server instead
|
||||
|
||||
`@stripe/link-cli --mcp` exposes the same commands as MCP tools over stdio. To register it with Hermes' native MCP:
|
||||
|
||||
```
|
||||
hermes mcp add stripe-link --command "npx" --args "@stripe/link-cli --mcp"
|
||||
```
|
||||
|
||||
Then `hermes mcp list` should show `stripe-link`. The same approval rules apply — MCP doesn't bypass the Link app approval step.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- **US-only.** Outside the US, `auth login` will fail. Tell the user, don't keep retrying.
|
||||
- **Card PAN must never enter agent context.** Use `--output-file` every time. If you've already retrieved without it, immediately `link-cli auth logout` is not enough — the card is one-time-use but rotate hygiene matters.
|
||||
- **`--request-approval` blocks until the user acts.** If the user is asleep, the CLI will hit its timeout. Set expectations.
|
||||
- **Multi-step `_next` commands.** Some commands return `_next.command` that must be executed to continue. When in doubt, prefer the inline-polling flags (`--interval`/`--timeout`).
|
||||
- **Output format defaults to `toon`** in non-TTY mode. Fine for prose, but if a downstream step needs to parse a specific field, pass `--format json`.
|
||||
- **Don't default to `card`.** The merchant-evaluation step (Section 2) exists because picking the wrong credential type fails the purchase silently or leaks more data than needed.
|
||||
|
||||
## Verification
|
||||
|
||||
```
|
||||
link-cli --version && link-cli auth status
|
||||
```
|
||||
|
||||
Exit code 0 means installed and logged in.
|
||||
@@ -0,0 +1,120 @@
|
||||
---
|
||||
name: stripe-projects
|
||||
description: Provision SaaS services + sync creds via Stripe Projects.
|
||||
version: 0.1.0
|
||||
author: Teknium (teknium1), Hermes Agent
|
||||
license: MIT
|
||||
platforms: [linux, macos]
|
||||
metadata:
|
||||
hermes:
|
||||
tags: [Payments, Stripe, Projects, Provisioning, Infrastructure]
|
||||
related_skills: [stripe-link-cli, mpp-agent]
|
||||
---
|
||||
|
||||
# Stripe Projects Skill
|
||||
|
||||
Wraps the [Stripe Projects](https://projects.dev) CLI plugin so Hermes can provision SaaS services (Neon, Twilio, Vercel, etc.), generate and sync credentials into the user's `.env`, and manage billing across providers from one place.
|
||||
|
||||
Gated `[linux, macos]` while the broader payments cluster matures on Windows. The Stripe CLI itself is cross-platform; this gate is a posture for the cluster, not a hard limit.
|
||||
|
||||
## When to Use
|
||||
|
||||
Trigger phrases:
|
||||
|
||||
- "set up <provider>", "provision <Neon|Twilio|Vercel|...>", "create a database"
|
||||
- "give me a <Postgres|Redis|Twilio number|...> for this project"
|
||||
- "manage my stack credentials", "rotate this key", "upgrade my plan"
|
||||
- "what providers can I add?"
|
||||
|
||||
If the user already has the service set up manually and just wants to use it, this skill is not the right entry point.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Stripe CLI installed (Homebrew on macOS, package manager on Linux, or download from https://docs.stripe.com/stripe-cli/install)
|
||||
- Stripe Projects plugin installed
|
||||
- A Stripe account, logged in via `stripe login`
|
||||
|
||||
## Install
|
||||
|
||||
macOS:
|
||||
|
||||
```
|
||||
brew install stripe/stripe-cli/stripe
|
||||
stripe plugin install projects
|
||||
```
|
||||
|
||||
Linux: follow the platform-specific install at https://docs.stripe.com/stripe-cli/install, then:
|
||||
|
||||
```
|
||||
stripe plugin install projects
|
||||
```
|
||||
|
||||
## How to Run
|
||||
|
||||
All commands run through the `terminal` tool from inside the user's project directory (the CLI writes `.env` and `.projects/vault/vault.json` into the CWD).
|
||||
|
||||
## Procedure
|
||||
|
||||
### 1. Initialize the project
|
||||
|
||||
```
|
||||
cd <project-root>
|
||||
stripe projects init
|
||||
```
|
||||
|
||||
This creates `.projects/vault/vault.json` (encrypted credential store) and prepares the project to receive providers.
|
||||
|
||||
### 2. Discover available providers
|
||||
|
||||
```
|
||||
stripe projects catalog
|
||||
```
|
||||
|
||||
Lists every provider Stripe Projects supports — databases, hosting, auth, AI, analytics, messaging, etc.
|
||||
|
||||
### 3. Add a service
|
||||
|
||||
```
|
||||
stripe projects add <provider>/<service>
|
||||
```
|
||||
|
||||
Examples:
|
||||
|
||||
- `stripe projects add neon/postgres`
|
||||
- `stripe projects add twilio/sms`
|
||||
- `stripe projects add runloop/sandbox`
|
||||
|
||||
The CLI provisions the service in the user's own account with the provider, generates credentials, syncs them into `.env`, and records the resource in the vault. The user may need to confirm a tier selection or pricing prompt.
|
||||
|
||||
### 4. Verify
|
||||
|
||||
```
|
||||
stripe projects list
|
||||
```
|
||||
|
||||
Should show the newly added provider and its `.env` keys.
|
||||
|
||||
### 5. Manage / upgrade / remove
|
||||
|
||||
```
|
||||
stripe projects upgrade <provider> # tier change
|
||||
stripe projects remove <provider> # deprovision
|
||||
stripe projects rotate <provider> # rotate credentials
|
||||
```
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- **`.env` writes are real writes.** The CLI appends to whatever `.env` is in the project root. If the user's `.env` is gitignored (normal), the keys land safely; if not, this skill could be a credential-leak vector. Always check `.gitignore` first.
|
||||
- **Per-project state.** `.projects/vault/vault.json` is per-project. Provisioning the same service in two different projects creates two separate resources — and two bills.
|
||||
- **Billing happens on Stripe's side.** Tier prompts during `add`/`upgrade` are real charges; surface them to the user before confirming.
|
||||
- **Provider availability changes.** The catalog grows; if a provider the user names isn't listed, `stripe projects catalog | grep <name>` first instead of failing the `add` call.
|
||||
- **Credentials in vault are encrypted but `.env` is plaintext.** Standard `.env` hygiene applies — never commit it.
|
||||
- **Removing a service does NOT always destroy the underlying resource.** Some providers leave a paused/dormant resource behind. Check the provider's own dashboard after `remove` for high-cost services (managed databases especially).
|
||||
|
||||
## Verification
|
||||
|
||||
```
|
||||
stripe projects --version && stripe projects list
|
||||
```
|
||||
|
||||
Exit code 0 inside an initialized project means the plugin is healthy.
|
||||
@@ -137,10 +137,11 @@ In gateway deployments (Telegram, Discord, Slack, etc.) each user arrives with a
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `pinUserPeer` | bool | `false` | When `true`, every gateway runtime user collapses to `peerName`. Single-operator deployments where you want all your platforms (and any other users) to share one peer. Also accepted as `pinPeerName` |
|
||||
| `pinPeerName` | bool | `false` | Alias for `pinUserPeer`; same effect |
|
||||
| `userPeerAliases` | object | `{}` | Map of runtime IDs to peer IDs (`{"86701400": "eri"}`). Many-to-one is the intended pattern — alias all your runtime IDs to one peer name. One-to-many is not supported; one runtime ID resolves to exactly one peer |
|
||||
| `runtimePeerPrefix` | string | `""` | Prepended to unknown runtime IDs to namespace them (e.g. `"telegram_"` → `telegram_86701400`). Used only when no alias matches. Prevents collisions between platforms whose runtime IDs share the same shape |
|
||||
| `pinUserPeer` | bool | `false` | When `true`, every gateway runtime user collapses to `peerName`. Single-operator deployments where you want all your platforms (and any other users) to share one peer |
|
||||
| `userPeerAliases` | object | `{}` | Map of runtime IDs to peer IDs (`{"7654321": "alice"}`). Many-to-one is the intended pattern — alias all your runtime IDs to one peer name. One-to-many is not supported; one runtime ID resolves to exactly one peer |
|
||||
| `runtimePeerPrefix` | string | `""` | Prepended to unknown runtime IDs to namespace them (e.g. `"telegram_"` → `telegram_7654321`). Used only when no alias matches. Prevents collisions between platforms whose runtime IDs share the same shape |
|
||||
|
||||
> **Deprecated:** `pinPeerName` is a legacy alias for `pinUserPeer`, still read for back-compat (`pinUserPeer` wins where both are set). `hermes honcho setup` migrates it onto `pinUserPeer` on touch and never writes it.
|
||||
|
||||
**Resolver ladder** (first match wins):
|
||||
|
||||
@@ -158,13 +159,15 @@ In gateway deployments (Telegram, Discord, Slack, etc.) each user arrives with a
|
||||
|
||||
**Host vs root semantics.** All three keys are accepted at both root and `hosts.<host>` levels. Host-level wins. For maps and prefixes, host-level *replaces* the root value as a whole (not merge), so a host can intentionally own its identity universe or wipe it with `userPeerAliases: {}` / `runtimePeerPrefix: ""`.
|
||||
|
||||
**Deployment shapes** (`hermes memory setup honcho` asks one prompt to set these):
|
||||
**Setup — gateway identity tree.** `hermes honcho setup` only asks about identity mapping when it detects a connected gateway platform (it inspects the gateway config; off-gateway the step is skipped because these keys do nothing without a runtime user ID). When it runs, it asks *who talks to this gateway?* and derives the keys:
|
||||
|
||||
- **Single-operator** — `pinUserPeer: true`. All gateway users → `peerName`. Recommended for personal use where you connect Hermes to your own Telegram/Discord/etc.
|
||||
- **Multi-user gateway** — `pinUserPeer: false`, optional `runtimePeerPrefix`. Each runtime user → own peer. Recommended for bots serving many humans.
|
||||
- **Hybrid** — `pinUserPeer: false`, `userPeerAliases` mapping the operator's runtime IDs to `peerName`. Multi-user gateway where YOU are routed but others stay distinct.
|
||||
- **just me** → `pinUserPeer: true`. Every non-agent gateway user collapses to `peerName`; the pin overrides all aliases, so pick this only when no user-side identity needs its own peer. Personal use where you connect Hermes to your own Telegram/Discord/etc. If separate agents reach the gateway and each needs a distinct peer, do **not** pin — leave `pinUserPeer: false` and map them via `userPeerAliases` (the `[e]` editor).
|
||||
- **me + other people, pooled** → `pinUserPeer: false` + `userPeerAliases` mapping your runtime IDs to `peerName`. You stay on the shared history; everyone else gets their own peer.
|
||||
- **me + other people / only other people** → `pinUserPeer: false`, optional `runtimePeerPrefix`. Each runtime user → own peer. For bots serving many humans.
|
||||
|
||||
**Migrating single → multi.** Flipping `pinUserPeer` from `true` to `false` does not migrate data. Memory accumulated under `peerName` while pinned stays there; runtime users now resolve to fresh, empty peers. To preserve your own continuity, use the **hybrid** shape — alias your runtime IDs back to `peerName` so your turns keep landing on the pooled history while other users get their own peers. The setup wizard offers this path automatically when it detects a single → multi transition.
|
||||
Pick **[e]** at the prompt to set the three keys directly instead of going through the tree.
|
||||
|
||||
**Un-pinning (single → per-user).** Flipping `pinUserPeer` from `true` to `false` does not migrate data. Memory accumulated under `peerName` while pinned stays there; runtime users now resolve to fresh, empty peers. To preserve your own continuity, choose the **pooled** path — alias your runtime IDs back to `peerName` so your turns keep landing on the pooled history while other users get their own peers. The wizard offers this steer automatically when it detects you're un-pinning a previously pinned profile.
|
||||
|
||||
### Memory & Recall
|
||||
|
||||
@@ -205,7 +208,7 @@ The Honcho session name determines which conversation bucket memory lands in. Re
|
||||
|
||||
Gateway platforms always resolve via priority 3 (per-chat isolation) regardless of `sessionStrategy`. The strategy setting only affects CLI sessions.
|
||||
|
||||
If `sessionPeerPrefix` is `true`, the peer name is prepended: `eri-hermes-agent`.
|
||||
If `sessionPeerPrefix` is `true`, the peer name is prepended: `alice-hermes-agent`.
|
||||
|
||||
#### What each strategy produces
|
||||
|
||||
|
||||
+231
-119
@@ -41,22 +41,20 @@ def clone_honcho_for_profile(profile_name: str) -> bool:
|
||||
return False # already exists
|
||||
|
||||
# Clone settings from default block, override identity fields.
|
||||
# Identity-mapping keys (pinPeerName/pinUserPeer, userPeerAliases,
|
||||
# runtimePeerPrefix) carry the operator's runtime-to-peer routing
|
||||
# intent from #27371. Both pin keys are inherited because
|
||||
# HonchoClientConfig prefers pinUserPeer over pinPeerName — leaving
|
||||
# the canonical key off this allowlist silently drops the pin on
|
||||
# cloned profiles when the default uses the newer name.
|
||||
# Identity-mapping keys (pinUserPeer, userPeerAliases, runtimePeerPrefix)
|
||||
# carry the operator's runtime-to-peer routing intent from #27371.
|
||||
new_block = {}
|
||||
for key in ("recallMode", "writeFrequency", "sessionStrategy",
|
||||
"sessionPeerPrefix", "contextTokens", "dialecticReasoningLevel",
|
||||
"dialecticDynamic", "dialecticMaxChars", "messageMaxChars",
|
||||
"dialecticMaxInputChars", "saveMessages", "observation",
|
||||
"pinPeerName", "pinUserPeer", "userPeerAliases",
|
||||
"runtimePeerPrefix"):
|
||||
"pinUserPeer", "userPeerAliases", "runtimePeerPrefix"):
|
||||
val = default_block.get(key)
|
||||
if val is not None:
|
||||
new_block[key] = val
|
||||
# Carry a legacy default-block pinPeerName forward under the canonical key.
|
||||
if "pinUserPeer" not in new_block and default_block.get("pinPeerName") is not None:
|
||||
new_block["pinUserPeer"] = default_block["pinPeerName"]
|
||||
|
||||
# Inherit peer name from default
|
||||
peer_name = default_block.get("peerName") or cfg.get("peerName")
|
||||
@@ -371,15 +369,122 @@ def _resolve_effective_identity_mapping(
|
||||
def _scrub_identity_mapping(hermes_host: dict) -> None:
|
||||
"""Drop every peer-mapping key from the host block.
|
||||
|
||||
Called before the wizard writes a chosen shape so latent precedence
|
||||
conflicts can't survive — e.g. a stray host ``pinUserPeer: false``
|
||||
that would silently outrank a freshly written ``pinPeerName: true``
|
||||
(host ``pinUserPeer`` is first in the resolver ladder).
|
||||
Called before the wizard writes a chosen shape so a stale alias, prefix,
|
||||
or pin from an earlier run can't bleed into the new mapping.
|
||||
"""
|
||||
for key in _IDENTITY_MAPPING_KEYS:
|
||||
hermes_host.pop(key, None)
|
||||
|
||||
|
||||
def _migrate_pin_key(block: dict) -> bool:
|
||||
"""Rewrite a legacy ``pinPeerName`` to canonical ``pinUserPeer`` in place.
|
||||
|
||||
``pinUserPeer`` wins over ``pinPeerName`` in the resolver, so setup writes
|
||||
only the canonical form and migrates on touch to stop configs carrying
|
||||
both. Returns True if the block changed.
|
||||
"""
|
||||
if "pinPeerName" not in block:
|
||||
return False
|
||||
legacy = block.pop("pinPeerName")
|
||||
if "pinUserPeer" not in block:
|
||||
block["pinUserPeer"] = legacy
|
||||
return True
|
||||
|
||||
|
||||
def _gateway_platforms() -> list[str] | None:
|
||||
"""Connected gateway platforms, or None if undetectable.
|
||||
|
||||
Identity mapping only affects gateway runtime users, so setup gates the
|
||||
whole step on this. Best-effort and dependency-free: the memory plugin
|
||||
must not hard-depend on the gateway package, so the import is lazy and
|
||||
guarded (matching the idiom hermes_cli already uses for gateway refs).
|
||||
"""
|
||||
try:
|
||||
from gateway.config import load_gateway_config
|
||||
return [p.value for p in load_gateway_config().get_connected_platforms()]
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _collect_operator_aliases(existing: dict, peer_target: str) -> dict:
|
||||
"""Prompt for the operator's per-platform runtime IDs, aliasing each to
|
||||
``peer_target``. Existing entries are preserved."""
|
||||
aliases = dict(existing)
|
||||
print(f"\n Add runtime IDs that should alias to peer '{peer_target}'.")
|
||||
print(" Leave blank to skip a platform. Existing aliases are preserved.")
|
||||
for platform_label, alias_hint in (
|
||||
("Telegram UID", "e.g. 7654321"),
|
||||
("Discord snowflake", "e.g. 491827364"),
|
||||
("Slack user ID", "e.g. U04ABCDEF"),
|
||||
("Matrix MXID", "e.g. @you:matrix.org"),
|
||||
):
|
||||
entered = _prompt(f" {platform_label} ({alias_hint})", default="").strip()
|
||||
if entered:
|
||||
aliases[entered] = peer_target
|
||||
return aliases
|
||||
|
||||
|
||||
def _apply_runtime_prefix(
|
||||
hermes_host: dict, current_prefix: str, prefix_from_root: bool, label: str
|
||||
) -> None:
|
||||
"""Write a host-level runtimePeerPrefix only when it diverges from an
|
||||
inherited root value; otherwise let the root cascade stand."""
|
||||
new_prefix = _prompt(label, default=current_prefix or "").strip()
|
||||
if new_prefix and not (prefix_from_root and new_prefix == current_prefix):
|
||||
hermes_host["runtimePeerPrefix"] = new_prefix
|
||||
|
||||
|
||||
def _echo_identity_mapping(hermes_host: dict) -> None:
|
||||
"""Show the resulting keys so the operator can verify what was written."""
|
||||
aliases = hermes_host.get("userPeerAliases")
|
||||
prefix = hermes_host.get("runtimePeerPrefix")
|
||||
print(" resolved →")
|
||||
print(f" pinUserPeer = {bool(hermes_host.get('pinUserPeer'))}")
|
||||
print(f" userPeerAliases = {aliases if aliases else '{}'}")
|
||||
print(f" runtimePeerPrefix = {prefix if prefix else '(none)'}")
|
||||
|
||||
|
||||
def _configure_raw_identity_mapping(
|
||||
hermes_host: dict,
|
||||
current_pin: bool,
|
||||
current_aliases: dict,
|
||||
current_prefix: str,
|
||||
aliases_from_root: bool,
|
||||
prefix_from_root: bool,
|
||||
) -> None:
|
||||
"""Power-user escape hatch: set the three resolver knobs directly."""
|
||||
print("\n Raw identity-mapping keys (resolver tries them top-down):")
|
||||
pin_in = _prompt(
|
||||
"pinUserPeer — pin all gateway users to your peer? (true/false)",
|
||||
default=str(bool(current_pin)).lower(),
|
||||
).strip().lower()
|
||||
pin = pin_in in {"true", "t", "yes", "y", "1"}
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinUserPeer"] = pin
|
||||
if pin:
|
||||
return
|
||||
aliases = (
|
||||
dict(current_aliases)
|
||||
if isinstance(current_aliases, dict) and not aliases_from_root
|
||||
else {}
|
||||
)
|
||||
print(" userPeerAliases — 'runtime_id=peer' pairs (blank line to finish):")
|
||||
while True:
|
||||
entry = _prompt(" alias", default="").strip()
|
||||
if not entry:
|
||||
break
|
||||
if "=" in entry:
|
||||
rid, peer = (p.strip() for p in entry.split("=", 1))
|
||||
if rid and peer:
|
||||
aliases[rid] = peer
|
||||
if aliases:
|
||||
hermes_host["userPeerAliases"] = aliases
|
||||
_apply_runtime_prefix(
|
||||
hermes_host, current_prefix, prefix_from_root,
|
||||
"runtimePeerPrefix — namespace for unknown IDs (blank for none)",
|
||||
)
|
||||
|
||||
|
||||
def _prompt(label: str, default: str | None = None, secret: bool = False) -> str:
|
||||
suffix = f" [{default}]" if default else ""
|
||||
sys.stdout.write(f" {label}{suffix}: ")
|
||||
@@ -446,6 +551,10 @@ def cmd_setup(args) -> None:
|
||||
hosts = cfg.setdefault("hosts", {})
|
||||
hermes_host = hosts.setdefault(_host_key(), {})
|
||||
|
||||
# Canonicalize any legacy pinPeerName before detection/writes.
|
||||
_migrate_pin_key(cfg)
|
||||
_migrate_pin_key(hermes_host)
|
||||
|
||||
# --- 1. Cloud or local? ---
|
||||
print(" Deployment:")
|
||||
print(" cloud -- Honcho cloud (api.honcho.dev)")
|
||||
@@ -545,18 +654,15 @@ def cmd_setup(args) -> None:
|
||||
if new_workspace:
|
||||
hermes_host["workspace"] = new_workspace
|
||||
|
||||
# --- 3b. Deployment shape ---
|
||||
# Determines how runtime user identities (Telegram UIDs, Discord
|
||||
# snowflakes, etc.) map to Honcho peers in gateway sessions. Three
|
||||
# shapes cover the realistic deployments; each writes a different
|
||||
# combination of pinPeerName / userPeerAliases / runtimePeerPrefix.
|
||||
# See plugins/memory/honcho/README.md for the resolver ladder.
|
||||
# --- 3b. Gateway identity mapping ---
|
||||
# These keys only affect the Hermes GATEWAY (Telegram/Discord/Slack/...),
|
||||
# the one entrypoint that supplies a runtime user ID. CLI/TUI/desktop/ACP
|
||||
# sessions have no runtime ID and fall through to peerName, so the step is
|
||||
# moot off-gateway — gate it behind detection.
|
||||
#
|
||||
# Detection must mirror the gateway resolver: root-level config and
|
||||
# ``pinUserPeer`` (which outranks ``pinPeerName`` at the same level)
|
||||
# both affect effective routing, so reading host-only fields would
|
||||
# mis-classify a profile that inherits its mapping from root or uses
|
||||
# the newer canonical key.
|
||||
# Detection mirrors the gateway resolver: root-level config and the
|
||||
# canonical ``pinUserPeer`` both affect routing, so host-only reads would
|
||||
# mis-classify a profile that inherits its mapping from root.
|
||||
(
|
||||
current_pin,
|
||||
current_aliases,
|
||||
@@ -572,103 +678,109 @@ def cmd_setup(args) -> None:
|
||||
else:
|
||||
current_shape = "multi"
|
||||
|
||||
print("\n Deployment shape (how gateway users map to peers):")
|
||||
print(" single -- all platforms route to your peer (recommended for personal use)")
|
||||
print(" multi -- each platform user gets their own peer (multi-user bots)")
|
||||
print(" hybrid -- multi-user, but YOUR runtime IDs alias to your peer")
|
||||
print(" skip -- don't touch identity-mapping config")
|
||||
new_shape = _prompt("Deployment shape", default=current_shape).strip().lower()
|
||||
|
||||
# Transitioning single → multi orphans the peerName pool for runtime users
|
||||
# (their resolved peers go from peerName to runtime-derived IDs with empty
|
||||
# history). Steer the operator toward hybrid so their own continuity is
|
||||
# preserved via alias mappings.
|
||||
if current_shape == "single" and new_shape == "multi":
|
||||
peer_target = hermes_host.get("peerName") or current_peer or "user"
|
||||
print(
|
||||
f"\n ⚠ Switching from single to multi will orphan memory accumulated\n"
|
||||
f" under peer '{peer_target}'. Existing runtime users (Telegram,\n"
|
||||
f" Discord, etc.) will resolve to fresh, empty peers."
|
||||
)
|
||||
print(" To keep your own continuity, choose 'hybrid' and alias your\n"
|
||||
" runtime IDs back to peerName.")
|
||||
confirm = _prompt("Continue with multi anyway? (yes/hybrid/no)", default="hybrid").strip().lower()
|
||||
if confirm in {"hybrid", "h"}:
|
||||
new_shape = "hybrid"
|
||||
elif confirm not in {"yes", "y"}:
|
||||
new_shape = "skip"
|
||||
|
||||
# Each shape branch scrubs every peer-mapping key before writing its own,
|
||||
# so a stale ``pinUserPeer`` left behind by an earlier setup run can't
|
||||
# outrank the freshly written ``pinPeerName`` via host-level precedence.
|
||||
if new_shape == "single":
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinPeerName"] = True
|
||||
print(f" pinPeerName=true → all gateway users route to '{hermes_host.get('peerName', '?')}'.")
|
||||
elif new_shape == "multi":
|
||||
# Preserve operator-curated, host-level aliases so multi → multi
|
||||
# re-runs don't drop them. Root-sourced aliases are left to
|
||||
# cascade naturally and are NOT copied down into the host.
|
||||
prior_aliases = (
|
||||
dict(current_aliases)
|
||||
if isinstance(current_aliases, dict) and not aliases_from_root
|
||||
else {}
|
||||
)
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinPeerName"] = False
|
||||
# Do NOT auto-write ``userPeerAliases: {}``: an empty host map
|
||||
# would override any root-level ``userPeerAliases`` the operator
|
||||
# set as a cross-host baseline, silently disabling those aliases.
|
||||
# Absence is the right "no host opinion" signal.
|
||||
if prior_aliases:
|
||||
hermes_host["userPeerAliases"] = prior_aliases
|
||||
_prefix_default = current_prefix or ""
|
||||
_new_prefix = _prompt(
|
||||
"Runtime peer prefix (e.g. 'telegram_', blank for none)",
|
||||
default=_prefix_default,
|
||||
).strip()
|
||||
# Only write a host-level prefix when the operator typed one that
|
||||
# diverges from the inherited root value; otherwise let the root
|
||||
# cascade continue unmodified.
|
||||
if _new_prefix and not (prefix_from_root and _new_prefix == current_prefix):
|
||||
hermes_host["runtimePeerPrefix"] = _new_prefix
|
||||
print(" Multi-user mode: each runtime ID → own peer. Use 'hermes honcho status' to inspect.")
|
||||
elif new_shape == "hybrid":
|
||||
# Hybrid encodes operator intent at the host level: collect existing
|
||||
# entries (host or root) so the wizard never silently drops a known
|
||||
# alias, then write the combined map. Materialising root entries
|
||||
# into the host is the right move here — once the operator answers
|
||||
# the alias prompts for a host, they're declaring "this host owns
|
||||
# the mapping".
|
||||
existing_aliases = dict(current_aliases) if isinstance(current_aliases, dict) else {}
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinPeerName"] = False
|
||||
peer_target = hermes_host.get("peerName") or current_peer or "user"
|
||||
print(f"\n Add runtime IDs that should alias to peer '{peer_target}'.")
|
||||
print(" Leave blank to skip a platform. Existing aliases are preserved.")
|
||||
for platform_label, alias_hint in (
|
||||
("Telegram UID", "e.g. 86701400"),
|
||||
("Discord snowflake", "e.g. 491827364"),
|
||||
("Slack user ID", "e.g. U04ABCDEF"),
|
||||
("Matrix MXID", "e.g. @you:matrix.org"),
|
||||
):
|
||||
entered = _prompt(f" {platform_label} ({alias_hint})", default="").strip()
|
||||
if entered:
|
||||
existing_aliases[entered] = peer_target
|
||||
if existing_aliases:
|
||||
hermes_host["userPeerAliases"] = existing_aliases
|
||||
_prefix_default = current_prefix or ""
|
||||
_new_prefix = _prompt(
|
||||
"Runtime peer prefix for unknown users (e.g. 'telegram_', blank for none)",
|
||||
default=_prefix_default,
|
||||
).strip()
|
||||
if _new_prefix and not (prefix_from_root and _new_prefix == current_prefix):
|
||||
hermes_host["runtimePeerPrefix"] = _new_prefix
|
||||
print(f" Hybrid mode: your runtime IDs → '{peer_target}', others → own peer.")
|
||||
elif new_shape == "skip":
|
||||
pass # leave config untouched
|
||||
gw_platforms = _gateway_platforms()
|
||||
if gw_platforms is None:
|
||||
print("\n Gateway identity mapping routes platform users to memory peers.")
|
||||
run_mapping = _prompt(
|
||||
"Running the Hermes gateway (Telegram/Discord/etc.)? (y/N)",
|
||||
default="n",
|
||||
).strip().lower() in {"y", "yes"}
|
||||
elif not gw_platforms:
|
||||
print("\n No gateway platforms connected — identity mapping only affects")
|
||||
print(" gateway users, so this step doesn't apply here.")
|
||||
run_mapping = _prompt(
|
||||
"Configure gateway mapping anyway? (y/N)", default="n",
|
||||
).strip().lower() in {"y", "yes"}
|
||||
else:
|
||||
print(f" Unknown shape '{new_shape}' — leaving identity-mapping config untouched.")
|
||||
print(f"\n Gateway platforms detected: {', '.join(gw_platforms)}")
|
||||
run_mapping = True
|
||||
|
||||
if run_mapping:
|
||||
peer_target = hermes_host.get("peerName") or current_peer or "user"
|
||||
default_choice = {"single": "1", "hybrid": "2", "multi": "3"}.get(current_shape, "3")
|
||||
print("\n How should gateway users map to memory peers?")
|
||||
print(" [1] just me — every non-agent user collapses to your peer")
|
||||
print(" [2] me + other people — keep mine pooled, others separate")
|
||||
print(" [3] only other people — everyone gets their own peer")
|
||||
print(" [s] skip (leave untouched) [e] edit raw keys")
|
||||
choice = _prompt("Choice", default=default_choice).strip().lower()
|
||||
|
||||
if choice in {"2", "me+others", "both"}:
|
||||
pooled = _prompt(
|
||||
" Keep my own memory pooled across platforms? (Y/n)", default="y",
|
||||
).strip().lower()
|
||||
shape = "hybrid" if pooled in {"y", "yes", ""} else "multi"
|
||||
elif choice in {"1", "me", "just-me"}:
|
||||
shape = "single"
|
||||
elif choice in {"3", "others"}:
|
||||
shape = "multi"
|
||||
elif choice in {"e", "edit", "raw"}:
|
||||
shape = "raw"
|
||||
else:
|
||||
shape = "skip"
|
||||
|
||||
# Un-pinning a currently-pinned profile without aliasing strands the
|
||||
# pooled peerName history; steer the operator toward pooling instead.
|
||||
if current_pin and shape == "multi":
|
||||
print(
|
||||
f"\n ⚠ Un-pinning will orphan memory accumulated under peer\n"
|
||||
f" '{peer_target}'. Existing gateway users resolve to fresh,\n"
|
||||
f" empty peers."
|
||||
)
|
||||
confirm = _prompt(
|
||||
" Pool my own memory instead (alias my IDs to peerName)? (Y/n)",
|
||||
default="y",
|
||||
).strip().lower()
|
||||
if confirm in {"y", "yes", ""}:
|
||||
shape = "hybrid"
|
||||
|
||||
# Each branch scrubs every peer-mapping key first so a stale alias,
|
||||
# prefix, or pin from an earlier run starts clean.
|
||||
if shape == "single":
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinUserPeer"] = True
|
||||
print(f" All non-agent gateway users route to '{peer_target}' (pin overrides aliases).")
|
||||
_echo_identity_mapping(hermes_host)
|
||||
elif shape == "multi":
|
||||
# Preserve operator-curated host-level aliases across multi → multi
|
||||
# re-runs. Root-sourced aliases cascade naturally and are NOT
|
||||
# copied down — an empty host map would mask a root baseline.
|
||||
prior_aliases = (
|
||||
dict(current_aliases)
|
||||
if isinstance(current_aliases, dict) and not aliases_from_root
|
||||
else {}
|
||||
)
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinUserPeer"] = False
|
||||
if prior_aliases:
|
||||
hermes_host["userPeerAliases"] = prior_aliases
|
||||
_apply_runtime_prefix(
|
||||
hermes_host, current_prefix, prefix_from_root,
|
||||
"Runtime peer prefix (e.g. 'telegram_', blank for none)",
|
||||
)
|
||||
print(" Each gateway user → own peer.")
|
||||
_echo_identity_mapping(hermes_host)
|
||||
elif shape == "hybrid":
|
||||
existing_aliases = dict(current_aliases) if isinstance(current_aliases, dict) else {}
|
||||
_scrub_identity_mapping(hermes_host)
|
||||
hermes_host["pinUserPeer"] = False
|
||||
merged = _collect_operator_aliases(existing_aliases, peer_target)
|
||||
if merged:
|
||||
hermes_host["userPeerAliases"] = merged
|
||||
_apply_runtime_prefix(
|
||||
hermes_host, current_prefix, prefix_from_root,
|
||||
"Runtime peer prefix for unknown users (e.g. 'telegram_', blank for none)",
|
||||
)
|
||||
print(f" Your runtime IDs → '{peer_target}', others → own peer.")
|
||||
_echo_identity_mapping(hermes_host)
|
||||
elif shape == "raw":
|
||||
_configure_raw_identity_mapping(
|
||||
hermes_host, current_pin, current_aliases, current_prefix,
|
||||
aliases_from_root, prefix_from_root,
|
||||
)
|
||||
_echo_identity_mapping(hermes_host)
|
||||
else: # skip
|
||||
print(" Identity mapping left untouched.")
|
||||
|
||||
# --- 4. Observation mode ---
|
||||
current_obs = hermes_host.get("observationMode") or cfg.get("observationMode", "directional")
|
||||
|
||||
@@ -96,6 +96,9 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
or os.getenv("MATTERMOST_REPLY_MODE", "off")
|
||||
).lower()
|
||||
|
||||
self._last_post_status: Optional[int] = None
|
||||
self._last_post_error: str = ""
|
||||
|
||||
# Dedup cache (prevent reprocessing)
|
||||
self._dedup = MessageDeduplicator()
|
||||
|
||||
@@ -130,20 +133,79 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
"""POST /api/v4/{path} with JSON body."""
|
||||
import aiohttp
|
||||
url = f"{self._base_url}/api/v4/{path.lstrip('/')}"
|
||||
self._last_post_status = None
|
||||
self._last_post_error = ""
|
||||
try:
|
||||
async with self._session.post(
|
||||
url, headers=self._headers(), json=payload,
|
||||
timeout=aiohttp.ClientTimeout(total=30)
|
||||
) as resp:
|
||||
self._last_post_status = resp.status
|
||||
if resp.status >= 400:
|
||||
body = await resp.text()
|
||||
self._last_post_error = body or ""
|
||||
logger.error("MM API POST %s → %s: %s", path, resp.status, body[:200])
|
||||
return {}
|
||||
return await resp.json()
|
||||
except aiohttp.ClientError as exc:
|
||||
self._last_post_error = str(exc)
|
||||
logger.error("MM API POST %s network error: %s", path, exc)
|
||||
return {}
|
||||
|
||||
async def _thread_root_for_send(
|
||||
self,
|
||||
reply_to: Optional[str],
|
||||
metadata: Optional[Dict[str, Any]],
|
||||
) -> Optional[str]:
|
||||
"""Resolve the Mattermost root_id from reply_to or metadata."""
|
||||
if self._reply_mode != "thread":
|
||||
return None
|
||||
candidate = reply_to
|
||||
if not candidate and isinstance(metadata, dict):
|
||||
candidate = metadata.get("thread_id") or metadata.get("root_id")
|
||||
if not candidate:
|
||||
return None
|
||||
return await self._resolve_root_id(str(candidate))
|
||||
|
||||
def _last_post_failure_is_broken_thread_root(self) -> bool:
|
||||
"""Return True only for clear invalid/missing Mattermost thread roots."""
|
||||
if self._last_post_status not in {400, 404}:
|
||||
return False
|
||||
body = (self._last_post_error or "").lower()
|
||||
if not body:
|
||||
return False
|
||||
rootish = any(marker in body for marker in ("root_id", "rootid", "root id", "thread", "post"))
|
||||
broken = any(marker in body for marker in ("invalid", "not found", "does not exist", "missing"))
|
||||
return rootish and broken
|
||||
|
||||
async def _post_preserving_thread(
|
||||
self,
|
||||
chat_id: str,
|
||||
payload: Dict[str, Any],
|
||||
metadata: Optional[Dict[str, Any]],
|
||||
) -> Dict[str, Any]:
|
||||
"""Post once, optionally falling back flat for final notify content."""
|
||||
data = await self._api_post("posts", payload)
|
||||
if data or "root_id" not in payload:
|
||||
return data
|
||||
if not (isinstance(metadata, dict) and metadata.get("notify")):
|
||||
return data
|
||||
if not self._last_post_failure_is_broken_thread_root():
|
||||
return data
|
||||
|
||||
flat_payload = dict(payload)
|
||||
flat_payload.pop("root_id", None)
|
||||
original = str(flat_payload.get("message") or "")
|
||||
flat_payload["message"] = (
|
||||
"⚠️ Mattermost thread delivery failed; posting final reply in channel.\n\n"
|
||||
+ original
|
||||
).strip()
|
||||
logger.warning(
|
||||
"Mattermost: falling back to flat channel delivery for notify-worthy post in %s",
|
||||
chat_id,
|
||||
)
|
||||
return await self._api_post("posts", flat_payload)
|
||||
|
||||
async def _api_put(
|
||||
self, path: str, payload: Dict[str, Any]
|
||||
) -> Dict[str, Any]:
|
||||
@@ -286,14 +348,12 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
"channel_id": chat_id,
|
||||
"message": chunk,
|
||||
}
|
||||
# Thread support: reply_to is the root post ID.
|
||||
if reply_to and self._reply_mode == "thread":
|
||||
# Ensure root_id points to the thread root, not a reply.
|
||||
# Mattermost rejects non-root post IDs as root_id.
|
||||
resolved_root = await self._resolve_root_id(reply_to)
|
||||
# Thread support: reply_to or metadata["thread_id"] is the root post ID.
|
||||
resolved_root = await self._thread_root_for_send(reply_to, metadata)
|
||||
if resolved_root:
|
||||
payload["root_id"] = resolved_root
|
||||
|
||||
data = await self._api_post("posts", payload)
|
||||
data = await self._post_preserving_thread(chat_id, payload, metadata)
|
||||
if not data or "id" not in data:
|
||||
return SendResult(success=False, error="Failed to create post")
|
||||
last_id = data["id"]
|
||||
@@ -346,7 +406,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
) -> SendResult:
|
||||
"""Download an image and upload it as a file attachment."""
|
||||
return await self._send_url_as_file(
|
||||
chat_id, image_url, caption, reply_to, "image"
|
||||
chat_id, image_url, caption, reply_to, "image", metadata
|
||||
)
|
||||
|
||||
async def send_image_file(
|
||||
@@ -359,7 +419,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
) -> SendResult:
|
||||
"""Upload a local image file."""
|
||||
return await self._send_local_file(
|
||||
chat_id, image_path, caption, reply_to
|
||||
chat_id, image_path, caption, reply_to, metadata=metadata
|
||||
)
|
||||
|
||||
async def send_document(
|
||||
@@ -373,7 +433,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
) -> SendResult:
|
||||
"""Upload a local file as a document."""
|
||||
return await self._send_local_file(
|
||||
chat_id, file_path, caption, reply_to, file_name
|
||||
chat_id, file_path, caption, reply_to, file_name, metadata
|
||||
)
|
||||
|
||||
async def send_voice(
|
||||
@@ -386,7 +446,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
) -> SendResult:
|
||||
"""Upload an audio file."""
|
||||
return await self._send_local_file(
|
||||
chat_id, audio_path, caption, reply_to
|
||||
chat_id, audio_path, caption, reply_to, metadata=metadata
|
||||
)
|
||||
|
||||
async def send_video(
|
||||
@@ -399,7 +459,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
) -> SendResult:
|
||||
"""Upload a video file."""
|
||||
return await self._send_local_file(
|
||||
chat_id, video_path, caption, reply_to
|
||||
chat_id, video_path, caption, reply_to, metadata=metadata
|
||||
)
|
||||
|
||||
def format_message(self, content: str) -> str:
|
||||
@@ -423,12 +483,13 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
caption: Optional[str],
|
||||
reply_to: Optional[str],
|
||||
kind: str = "file",
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> SendResult:
|
||||
"""Download a URL and upload it as a file attachment."""
|
||||
from tools.url_safety import is_safe_url
|
||||
if not is_safe_url(url):
|
||||
logger.warning("Mattermost: blocked unsafe URL (SSRF protection)")
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to, metadata=metadata)
|
||||
|
||||
import aiohttp
|
||||
|
||||
@@ -446,7 +507,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
await asyncio.sleep(1.5 * (attempt + 1))
|
||||
continue
|
||||
if resp.status >= 400:
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to, metadata=metadata)
|
||||
file_data = await resp.read()
|
||||
ct = resp.content_type or "application/octet-stream"
|
||||
break
|
||||
@@ -455,25 +516,26 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
await asyncio.sleep(1.5 * (attempt + 1))
|
||||
continue
|
||||
logger.warning("Mattermost: failed to download %s after %d attempts: %s", url, attempt + 1, exc)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to, metadata=metadata)
|
||||
|
||||
if file_data is None:
|
||||
logger.warning("Mattermost: download returned no data for %s", url)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to, metadata=metadata)
|
||||
|
||||
file_id = await self._upload_file(chat_id, file_data, fname, ct)
|
||||
if not file_id:
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to)
|
||||
return await self.send(chat_id, f"{caption or ''}\n{url}".strip(), reply_to, metadata=metadata)
|
||||
|
||||
payload: Dict[str, Any] = {
|
||||
"channel_id": chat_id,
|
||||
"message": caption or "",
|
||||
"file_ids": [file_id],
|
||||
}
|
||||
if reply_to and self._reply_mode == "thread":
|
||||
payload["root_id"] = await self._resolve_root_id(reply_to)
|
||||
resolved_root = await self._thread_root_for_send(reply_to, metadata)
|
||||
if resolved_root:
|
||||
payload["root_id"] = resolved_root
|
||||
|
||||
data = await self._api_post("posts", payload)
|
||||
data = await self._post_preserving_thread(chat_id, payload, metadata)
|
||||
if not data or "id" not in data:
|
||||
return SendResult(success=False, error="Failed to post with file")
|
||||
return SendResult(success=True, message_id=data["id"])
|
||||
@@ -485,6 +547,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
caption: Optional[str],
|
||||
reply_to: Optional[str],
|
||||
file_name: Optional[str] = None,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> SendResult:
|
||||
"""Upload a local file and attach it to a post."""
|
||||
import mimetypes
|
||||
@@ -509,10 +572,11 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
"message": caption or "",
|
||||
"file_ids": [file_id],
|
||||
}
|
||||
if reply_to and self._reply_mode == "thread":
|
||||
payload["root_id"] = await self._resolve_root_id(reply_to)
|
||||
resolved_root = await self._thread_root_for_send(reply_to, metadata)
|
||||
if resolved_root:
|
||||
payload["root_id"] = resolved_root
|
||||
|
||||
data = await self._api_post("posts", payload)
|
||||
data = await self._post_preserving_thread(chat_id, payload, metadata)
|
||||
if not data or "id" not in data:
|
||||
return SendResult(success=False, error="Failed to post with file")
|
||||
return SendResult(success=True, message_id=data["id"])
|
||||
@@ -596,11 +660,14 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
"message": "\n".join(caption_parts),
|
||||
"file_ids": file_ids,
|
||||
}
|
||||
resolved_root = await self._thread_root_for_send(None, metadata)
|
||||
if resolved_root:
|
||||
payload["root_id"] = resolved_root
|
||||
logger.info(
|
||||
"Mattermost: sending %d image(s) as single post (chunk %d/%d)",
|
||||
len(file_ids), chunk_idx + 1, len(chunks),
|
||||
)
|
||||
data = await self._api_post("posts", payload)
|
||||
data = await self._post_preserving_thread(chat_id, payload, metadata)
|
||||
if not data or "id" not in data:
|
||||
logger.warning("Mattermost: multi-image post failed, falling back")
|
||||
await super().send_multiple_images(chat_id, chunk, metadata, human_delay=human_delay)
|
||||
@@ -786,8 +853,16 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
sender_id = post.get("user_id", "")
|
||||
sender_name = data.get("sender_name", "").lstrip("@") or sender_id
|
||||
|
||||
# Thread support: if the post is in a thread, use root_id.
|
||||
# Thread support: if the post is in a thread, use root_id. In
|
||||
# thread mode, top-level channel posts are valid roots for progress.
|
||||
thread_id = post.get("root_id") or None
|
||||
if (
|
||||
not thread_id
|
||||
and self._reply_mode == "thread"
|
||||
and channel_type_raw != "D"
|
||||
and post_id
|
||||
):
|
||||
thread_id = post_id
|
||||
|
||||
# Determine message type.
|
||||
file_ids = post.get("file_ids") or []
|
||||
@@ -849,6 +924,7 @@ class MattermostAdapter(BasePlatformAdapter):
|
||||
user_id=sender_id,
|
||||
user_name=sender_name,
|
||||
thread_id=thread_id,
|
||||
message_id=post_id,
|
||||
)
|
||||
|
||||
# Per-channel ephemeral prompt
|
||||
|
||||
@@ -54,8 +54,10 @@ hermes gateway start
|
||||
1. **Device login** (RFC 8628, `client_id=photon-cli`) — opens
|
||||
`https://app.photon.codes/` for approval and stores the bearer token.
|
||||
2. **Find or create** the `Hermes Agent` project on the Photon dashboard.
|
||||
3. **Enable Spectrum**, read the project's `spectrumProjectId`, rotate the
|
||||
project secret, and persist both.
|
||||
3. **Provision the project secret** — mint a fresh project secret (the
|
||||
dashboard reveals it only once) and persist it to `~/.hermes/.env` so the
|
||||
sidecar can authenticate `spectrum-ts`. Spectrum is always on, so there's no
|
||||
separate enable step.
|
||||
4. **Register your phone number** as a Spectrum user (idempotent — skipped if
|
||||
a user with that number already exists).
|
||||
5. **Print the assigned iMessage line** — the number you text to reach your
|
||||
@@ -75,7 +77,7 @@ Runtime SDK credentials live in `~/.hermes/.env` (the same place every other
|
||||
channel keeps its token), and the adapter reads them from the environment:
|
||||
|
||||
```bash
|
||||
PHOTON_PROJECT_ID=<spectrumProjectId> # the SDK's projectId
|
||||
PHOTON_PROJECT_ID=<projectId> # the SDK's projectId (same as the dashboard project id)
|
||||
PHOTON_PROJECT_SECRET=<projectSecret>
|
||||
```
|
||||
|
||||
@@ -89,8 +91,8 @@ Management metadata lives in `~/.hermes/auth.json` under `credential_pool`:
|
||||
],
|
||||
"photon_project": [
|
||||
{
|
||||
"dashboard_project_id": "<dashboard id>",
|
||||
"spectrum_project_id": "<spectrumProjectId>",
|
||||
"dashboard_project_id": "<project id>",
|
||||
"spectrum_project_id": "<project id>",
|
||||
"project_secret": "<projectSecret>",
|
||||
"name": "Hermes Agent"
|
||||
}
|
||||
@@ -99,9 +101,9 @@ Management metadata lives in `~/.hermes/auth.json` under `credential_pool`:
|
||||
}
|
||||
```
|
||||
|
||||
> **Note on ids.** A Photon project has two identifiers: the dashboard `id`
|
||||
> (used for management API calls) and the `spectrumProjectId` (what the SDK
|
||||
> authenticates with). `PHOTON_PROJECT_ID` is the **spectrum** id.
|
||||
> **Note on ids.** A Photon project's dashboard id and its Spectrum project id
|
||||
> are the same value, exposed as `PHOTON_PROJECT_ID`. The `dashboard_project_id`
|
||||
> and `spectrum_project_id` keys in `auth.json` both hold that id.
|
||||
|
||||
## Configuration knobs
|
||||
|
||||
|
||||
@@ -3,29 +3,29 @@ Photon Dashboard API client + device-code login flow.
|
||||
|
||||
This module is pure Python — it intentionally does not depend on
|
||||
``spectrum-ts``. Every management-plane operation (login, find/create
|
||||
project, enable Spectrum, rotate the project secret, register a user,
|
||||
list the assigned iMessage line) talks to Photon's **Dashboard API** on a
|
||||
single host, exactly like the official Photon CLI (``photon-hq/cli``):
|
||||
project, rotate the project secret, register a user, list the assigned
|
||||
iMessage line) talks to Photon's **Dashboard API** on a single host,
|
||||
exactly like the official Photon CLI (``photon-hq/cli``):
|
||||
|
||||
Dashboard API https://app.photon.codes/api/...
|
||||
OAuth 2.0 device flow, Bearer access token
|
||||
|
||||
A Photon project carries two distinct identifiers:
|
||||
|
||||
* ``id`` — the Dashboard project id (used in API paths)
|
||||
* ``spectrumProjectId`` — the Spectrum Cloud project id, populated when
|
||||
Spectrum is enabled on the project
|
||||
A Photon project has a single identifier: the dashboard ``id`` *is* the
|
||||
Spectrum Cloud project id. They used to diverge (a separate
|
||||
``spectrumProjectId`` field), but the dashboard unified them — every
|
||||
project is created with matching ids and the pre-existing diverged rows
|
||||
were backfilled so ``project.id == spectrumProjectId`` everywhere
|
||||
(dashboard ENG-1582). Spectrum is always enabled and provisioned at
|
||||
create-time, so there is no enable/toggle step anymore.
|
||||
|
||||
The ``spectrum-ts`` SDK (run by the Node sidecar) authenticates to Spectrum
|
||||
Cloud with ``(spectrumProjectId, projectSecret)`` — so the value we persist
|
||||
as ``PHOTON_PROJECT_ID`` for the runtime is the **spectrumProjectId**, not
|
||||
the Dashboard ``id``. The Dashboard ``id`` is kept only for management
|
||||
calls.
|
||||
Cloud with ``(id, projectSecret)`` — the same ``id`` used in Dashboard API
|
||||
paths — which we persist as ``PHOTON_PROJECT_ID`` for the runtime.
|
||||
|
||||
Credential storage mirrors every other Hermes channel:
|
||||
|
||||
* runtime SDK creds -> ``~/.hermes/.env`` (``PHOTON_PROJECT_ID`` =
|
||||
spectrumProjectId, ``PHOTON_PROJECT_SECRET``) via ``save_env_value``
|
||||
project id, ``PHOTON_PROJECT_SECRET``) via ``save_env_value``
|
||||
* management metadata -> ``~/.hermes/auth.json`` under
|
||||
``credential_pool.photon`` (device token),
|
||||
``credential_pool.photon_project`` (dashboard id, spectrum id, name), and
|
||||
@@ -148,8 +148,8 @@ def load_project_credentials() -> Tuple[Optional[str], Optional[str]]:
|
||||
|
||||
Precedence: process env (``~/.hermes/.env`` is loaded into the gateway's
|
||||
environment at startup) wins, then ``auth.json`` for offline / status
|
||||
use. This is the pair the Node sidecar feeds to ``spectrum-ts`` — the id
|
||||
is the **spectrumProjectId**, not the Dashboard id.
|
||||
use. This is the pair the Node sidecar feeds to ``spectrum-ts``; the id
|
||||
is the unified project id (dashboard id == spectrumProjectId).
|
||||
"""
|
||||
env_id = os.getenv("PHOTON_PROJECT_ID")
|
||||
env_sec = os.getenv("PHOTON_PROJECT_SECRET")
|
||||
@@ -166,14 +166,26 @@ def load_project_credentials() -> Tuple[Optional[str], Optional[str]]:
|
||||
|
||||
|
||||
def load_dashboard_project_id() -> Optional[str]:
|
||||
"""Return the Dashboard project id (for management API calls)."""
|
||||
"""Return the project id used for management API calls.
|
||||
|
||||
Post-unification the dashboard id and the Spectrum id are the same value,
|
||||
so we prefer the stored ``spectrum_project_id``: for pre-backfill installs
|
||||
the old ``dashboard_project_id`` is the diverged id that the unification
|
||||
rewrote (it now 404s), while the Spectrum id always matches the live row.
|
||||
Falls back to the legacy keys for older records.
|
||||
"""
|
||||
env_id = os.getenv("PHOTON_DASHBOARD_PROJECT_ID")
|
||||
if env_id:
|
||||
return env_id
|
||||
auth = _load_auth()
|
||||
proj = auth.get("credential_pool", {}).get("photon_project") or []
|
||||
if isinstance(proj, list) and proj:
|
||||
return proj[0].get("dashboard_project_id") or proj[0].get("project_id")
|
||||
entry = proj[0]
|
||||
return (
|
||||
entry.get("spectrum_project_id")
|
||||
or entry.get("dashboard_project_id")
|
||||
or entry.get("project_id")
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
@@ -646,30 +658,23 @@ def find_project_by_name(token: str, name: str) -> Optional[Dict[str, Any]]:
|
||||
return None
|
||||
|
||||
|
||||
def get_project(token: str, project_id: str) -> Dict[str, Any]:
|
||||
"""GET ``/api/projects/{id}`` — includes ``spectrum`` + ``spectrumProjectId``."""
|
||||
if httpx is None:
|
||||
raise RuntimeError("httpx is required for Photon")
|
||||
url = f"{_dashboard_host()}/api/projects/{project_id}"
|
||||
resp = httpx.get(url, headers=_bearer(token), timeout=30.0)
|
||||
resp.raise_for_status()
|
||||
return resp.json() or {}
|
||||
|
||||
|
||||
def create_project(
|
||||
token: str,
|
||||
*,
|
||||
name: str = DEFAULT_PROJECT_NAME,
|
||||
location: str = "United States",
|
||||
) -> Dict[str, Any]:
|
||||
"""POST ``/api/projects`` with ``spectrum: true`` and return ``{success, id}``."""
|
||||
"""POST ``/api/projects`` and return ``{success, id}``.
|
||||
|
||||
Spectrum is always provisioned at create-time, so the request body no
|
||||
longer carries a ``spectrum`` flag (the field was dropped from the API).
|
||||
"""
|
||||
if httpx is None:
|
||||
raise RuntimeError("httpx is required for Photon project creation")
|
||||
url = f"{_dashboard_host()}/api/projects"
|
||||
body: Dict[str, Any] = {
|
||||
"name": name,
|
||||
"location": location,
|
||||
"spectrum": True,
|
||||
"template": False,
|
||||
"observability": False,
|
||||
}
|
||||
@@ -683,29 +688,6 @@ def create_project(
|
||||
return data
|
||||
|
||||
|
||||
def ensure_spectrum_enabled(token: str, project_id: str) -> Dict[str, Any]:
|
||||
"""Enable Spectrum on the project if needed; return the project dict.
|
||||
|
||||
The dashboard exposes Spectrum as a toggle, so we only flip it when
|
||||
``spectrum`` is currently false, then re-fetch to pick up the freshly
|
||||
populated ``spectrumProjectId``.
|
||||
"""
|
||||
if httpx is None:
|
||||
raise RuntimeError("httpx is required for Photon")
|
||||
proj = get_project(token, project_id)
|
||||
if not proj.get("spectrum"):
|
||||
url = f"{_dashboard_host()}/api/projects/{project_id}/spectrum/toggle"
|
||||
resp = httpx.post(url, json={}, headers=_bearer(token), timeout=30.0)
|
||||
resp.raise_for_status()
|
||||
proj = get_project(token, project_id)
|
||||
if not proj.get("spectrumProjectId"):
|
||||
raise RuntimeError(
|
||||
"Spectrum is enabled but the project has no spectrumProjectId yet — "
|
||||
"retry in a moment, or enable Spectrum from the dashboard."
|
||||
)
|
||||
return proj
|
||||
|
||||
|
||||
def regenerate_project_secret(token: str, project_id: str) -> str:
|
||||
"""POST ``/api/projects/{id}/regenerate-secret`` → the new project secret.
|
||||
|
||||
@@ -1007,8 +989,9 @@ def print_credential_summary(emit: Any = print) -> None:
|
||||
else "✗ missing (run `hermes photon setup`)"
|
||||
)
|
||||
sid, sec = load_project_credentials()
|
||||
labels["spectrum_project_id"] = sid if sid else "✗ missing"
|
||||
labels["dashboard_project_id"] = load_dashboard_project_id() or "—"
|
||||
# Dashboard id and Spectrum id are the same value now (ids unified), so
|
||||
# there's a single project id to show.
|
||||
labels["project_id"] = sid if sid else "✗ missing"
|
||||
labels["project_key"] = "✓ stored" if sec else "✗ missing"
|
||||
phone, assigned = load_user_numbers()
|
||||
labels["phone_number"] = (
|
||||
@@ -1022,8 +1005,7 @@ def print_credential_summary(emit: Any = print) -> None:
|
||||
"Photon iMessage status",
|
||||
"──────────────────────",
|
||||
" device token : " + labels["device_token"],
|
||||
" dashboard project : " + labels["dashboard_project_id"],
|
||||
" spectrum project id : " + labels["spectrum_project_id"],
|
||||
" project id : " + labels["project_id"],
|
||||
" project secret : " + labels["project_key"],
|
||||
" my number : " + labels["phone_number"],
|
||||
" assigned number : " + labels["assigned_phone_number"],
|
||||
@@ -1039,7 +1021,7 @@ def credential_summary() -> Dict[str, str]:
|
||||
else "✗ missing (run `hermes photon setup`)"
|
||||
)
|
||||
|
||||
def _present_spectrum_id() -> str:
|
||||
def _present_project_id() -> str:
|
||||
sid, _sec = load_project_credentials()
|
||||
return sid or "✗ missing"
|
||||
|
||||
@@ -1057,8 +1039,7 @@ def credential_summary() -> Dict[str, str]:
|
||||
|
||||
return {
|
||||
"device_token": _present_token(),
|
||||
"dashboard_project_id": load_dashboard_project_id() or "—",
|
||||
"spectrum_project_id": _present_spectrum_id(),
|
||||
"project_id": _present_project_id(),
|
||||
"project_key": _present_secret(),
|
||||
"phone_number": _present_phone(),
|
||||
"assigned_phone_number": _present_assigned_phone(),
|
||||
|
||||
@@ -164,16 +164,14 @@ def _cmd_setup(args: argparse.Namespace) -> int:
|
||||
print("could not resolve a Photon project id", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
# 3. Enable Spectrum, fetch the spectrum project id, rotate the secret,
|
||||
# and persist both (runtime creds -> ~/.hermes/.env, ids -> auth.json).
|
||||
# 3. Rotate the project secret and persist creds (runtime -> ~/.hermes/.env,
|
||||
# ids -> auth.json). Spectrum is always enabled and provisioned at
|
||||
# create-time, and the dashboard project id *is* the Spectrum project id
|
||||
# (ids unified), so there's nothing to enable — the id we already have is
|
||||
# the Spectrum id.
|
||||
try:
|
||||
print("[3/5] Enabling Spectrum and provisioning credentials...")
|
||||
proj = photon_auth.ensure_spectrum_enabled(token, dashboard_id)
|
||||
spectrum_id = proj.get("spectrumProjectId")
|
||||
if not spectrum_id:
|
||||
print("spectrum provisioning failed: no spectrum project id", file=sys.stderr)
|
||||
return 1
|
||||
spectrum_id = str(spectrum_id)
|
||||
print("[3/5] Provisioning Spectrum credentials...")
|
||||
spectrum_id = dashboard_id
|
||||
secret = photon_auth.regenerate_project_secret(token, dashboard_id)
|
||||
photon_auth.store_project_credentials(
|
||||
spectrum_project_id=spectrum_id,
|
||||
@@ -182,7 +180,7 @@ def _cmd_setup(args: argparse.Namespace) -> int:
|
||||
name=name,
|
||||
)
|
||||
# spectrum_id is an opaque non-secret id; safe to show.
|
||||
print(f" ✓ Spectrum enabled (project id {spectrum_id}) — secret saved")
|
||||
print(f" ✓ Spectrum ready (project id {spectrum_id}) — secret saved")
|
||||
except Exception as e:
|
||||
print(f"spectrum provisioning failed: {e}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
+7
-1
@@ -1411,10 +1411,15 @@ class AIAgent:
|
||||
def _summarize_background_review_actions(
|
||||
review_messages: List[Dict],
|
||||
prior_snapshot: List[Dict],
|
||||
notification_mode: str = "on",
|
||||
) -> List[str]:
|
||||
"""Forwarder — see ``agent.background_review.summarize_background_review_actions``."""
|
||||
from agent.background_review import summarize_background_review_actions
|
||||
return summarize_background_review_actions(review_messages, prior_snapshot)
|
||||
return summarize_background_review_actions(
|
||||
review_messages,
|
||||
prior_snapshot,
|
||||
notification_mode=notification_mode,
|
||||
)
|
||||
|
||||
def _spawn_background_review(
|
||||
self,
|
||||
@@ -5140,6 +5145,7 @@ class AIAgent:
|
||||
acp_command=function_args.get("acp_command"),
|
||||
acp_args=function_args.get("acp_args"),
|
||||
role=function_args.get("role"),
|
||||
background=function_args.get("background"),
|
||||
parent_agent=self,
|
||||
)
|
||||
|
||||
|
||||
+78
-1
@@ -268,7 +268,7 @@ emit_manifest() {
|
||||
if [ "$INCLUDE_DESKTOP" = true ]; then
|
||||
desktop_stage='{"name":"desktop","title":"Build desktop app","category":"runtime","needs_user_input":false},'
|
||||
fi
|
||||
printf '%s' '{"protocol_version":1,"stages":[{"name":"prerequisites","title":"System prerequisites","category":"runtime","needs_user_input":false},{"name":"repository","title":"Download Hermes Agent","category":"runtime","needs_user_input":false},{"name":"venv","title":"Create Python virtual environment","category":"runtime","needs_user_input":false},{"name":"python-deps","title":"Install Python dependencies","category":"runtime","needs_user_input":false},{"name":"node-deps","title":"Install browser-tool dependencies","category":"runtime","needs_user_input":false},{"name":"path","title":"Install hermes command","category":"runtime","needs_user_input":false},{"name":"config","title":"Prepare config and skills","category":"configuration","needs_user_input":false},{"name":"setup","title":"Configure API keys and settings","category":"configuration","needs_user_input":true},{"name":"gateway","title":"Configure gateway service","category":"configuration","needs_user_input":true},'"$desktop_stage"'{"name":"complete","title":"Finish install","category":"runtime","needs_user_input":false}]}'
|
||||
printf '%s' '{"protocol_version":1,"stages":[{"name":"prerequisites","title":"System prerequisites","category":"runtime","needs_user_input":false},{"name":"repository","title":"Download Hermes Agent","category":"runtime","needs_user_input":false},{"name":"venv","title":"Create Python virtual environment","category":"runtime","needs_user_input":false},{"name":"python-deps","title":"Install Python dependencies","category":"runtime","needs_user_input":false},{"name":"node-deps","title":"Install browser-tool dependencies","category":"runtime","needs_user_input":false},{"name":"opentui-engine","title":"Set up OpenTUI engine","category":"runtime","needs_user_input":false},{"name":"path","title":"Install hermes command","category":"runtime","needs_user_input":false},{"name":"config","title":"Prepare config and skills","category":"configuration","needs_user_input":false},{"name":"setup","title":"Configure API keys and settings","category":"configuration","needs_user_input":true},{"name":"gateway","title":"Configure gateway service","category":"configuration","needs_user_input":true},'"$desktop_stage"'{"name":"complete","title":"Finish install","category":"runtime","needs_user_input":false}]}'
|
||||
printf '\n'
|
||||
}
|
||||
|
||||
@@ -1980,6 +1980,76 @@ install_node_deps() {
|
||||
restore_dirty_lockfiles "$INSTALL_DIR"
|
||||
}
|
||||
|
||||
# Provision the native OpenTUI engine on NODE 26.3+ (no Bun): `npm install` +
|
||||
# `npm run build` (esbuild → dist/main.js) in ui-opentui. The engine's
|
||||
# renderer loads via the experimental `node:ffi` API that only exists on Node
|
||||
# 26.3+. The launcher (hermes_cli/main.py:_opentui_available) only uses OpenTUI
|
||||
# when a Node >= 26.3 resolves AND the v2 package is built; otherwise it falls
|
||||
# back to the Ink engine. So this stage is STRICTLY best-effort: any failure
|
||||
# (unsupported platform, Node < 26.3, no network, install/build fails) logs a
|
||||
# warning and returns 0. A skipped OpenTUI setup just means the user gets Ink —
|
||||
# breaking the install would be far worse than skipping OpenTUI. Every sub-step
|
||||
# is guarded; this function never `exit`s and never returns non-zero.
|
||||
install_opentui() {
|
||||
# node:ffi isn't validated on Windows/Termux — keep those hosts on Ink.
|
||||
if [ "$OS" = "windows" ] || [ "$DISTRO" = "termux" ] || [ "$OS" = "android" ]; then
|
||||
log_info "Skipping OpenTUI engine (unsupported platform) — using Ink."
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Only meaningful if the v2 package is present in this checkout.
|
||||
if [ ! -f "$INSTALL_DIR/ui-opentui/package.json" ]; then
|
||||
log_info "Skipping OpenTUI engine (ui-opentui not present) — using Ink."
|
||||
return 0
|
||||
fi
|
||||
|
||||
log_info "Setting up OpenTUI engine (native TUI, Node 26.3+ / node:ffi)..."
|
||||
|
||||
# Resolve a Node >= 26.3.0 (the node:ffi floor): HERMES_NODE > node on PATH,
|
||||
# version-checked. We do NOT install Node here — if one new enough isn't
|
||||
# available the launcher cleanly falls back to Ink.
|
||||
local node_bin=""
|
||||
for cand in "${HERMES_NODE:-}" "$(command -v node 2>/dev/null || true)"; do
|
||||
[ -n "$cand" ] && [ -x "$cand" ] || continue
|
||||
if "$cand" -e 'const p=process.versions.node.split(".").map(Number); process.exit(p[0]>26||(p[0]===26&&p[1]>=3)?0:1)' 2>/dev/null; then
|
||||
node_bin="$cand"
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ -z "$node_bin" ]; then
|
||||
log_warn "OpenTUI engine setup skipped (needs Node >= 26.3.0; none found) — using the Ink engine. Install Node 26.3+ or set HERMES_NODE."
|
||||
return 0
|
||||
fi
|
||||
log_success "Node found ($("$node_bin" --version 2>/dev/null || echo "unknown"))"
|
||||
|
||||
# npm ships with Node; the build (`node scripts/build.mjs`) runs fine on any
|
||||
# recent Node — only the runtime needs 26.3, which the launcher re-checks.
|
||||
local npm_bin
|
||||
npm_bin="$(command -v npm 2>/dev/null || true)"
|
||||
if [ -z "$npm_bin" ]; then
|
||||
log_warn "OpenTUI engine setup skipped (npm not found) — using the Ink engine."
|
||||
return 0
|
||||
fi
|
||||
|
||||
cd "$INSTALL_DIR/ui-opentui" || { log_warn "OpenTUI engine setup skipped (cd failed) — using Ink."; return 0; }
|
||||
|
||||
# Pull deps (fetches the per-arch @opentui/core-<arch> native lib) then build
|
||||
# the Node bundle (dist/main.js). Both idempotent.
|
||||
log_info "Installing OpenTUI dependencies (npm install)..."
|
||||
if ! "$npm_bin" install --no-audit --no-fund >/dev/null 2>&1; then
|
||||
log_warn "OpenTUI engine setup skipped (npm install failed) — the Ink engine will be used."
|
||||
return 0
|
||||
fi
|
||||
log_info "Building OpenTUI engine (npm run build)..."
|
||||
if ! "$npm_bin" run build >/dev/null 2>&1; then
|
||||
log_warn "OpenTUI engine setup skipped (build failed) — the Ink engine will be used."
|
||||
return 0
|
||||
fi
|
||||
|
||||
log_success "OpenTUI engine ready (opt-in: HERMES_TUI_ENGINE=opentui; default is Ink)."
|
||||
return 0
|
||||
}
|
||||
|
||||
run_setup_wizard() {
|
||||
if [ "$RUN_SETUP" = false ]; then
|
||||
log_info "Skipping setup wizard (--skip-setup)"
|
||||
@@ -2636,6 +2706,12 @@ run_stage_body() {
|
||||
check_node
|
||||
install_node_deps
|
||||
;;
|
||||
opentui-engine)
|
||||
detect_os
|
||||
resolve_install_layout
|
||||
require_install_dir
|
||||
install_opentui
|
||||
;;
|
||||
path)
|
||||
detect_os
|
||||
resolve_install_layout
|
||||
@@ -2743,6 +2819,7 @@ main() {
|
||||
setup_venv
|
||||
install_deps
|
||||
install_node_deps
|
||||
install_opentui
|
||||
setup_path
|
||||
copy_config_templates
|
||||
run_setup_wizard
|
||||
|
||||
@@ -49,6 +49,7 @@ AUTHOR_MAP = {
|
||||
"rio.jeong@thebytesize.ai": "rio-jeong",
|
||||
"yehaotian@xuanshudeMac-mini.local": "ArcanePivot",
|
||||
"dbeyer7@gmail.com": "benegessarit",
|
||||
"264773240+MrDiamondBallz@users.noreply.github.com": "MrDiamondBallz",
|
||||
"kenmege@yahoo.com": "Kenmege",
|
||||
"tianying.x@eukarya.io": "xtymac",
|
||||
"dkobi16@gmail.com": "Diyoncrz18",
|
||||
@@ -89,6 +90,7 @@ AUTHOR_MAP = {
|
||||
"290859878+synapsesx@users.noreply.github.com": "synapsesx",
|
||||
"157689911+itsflownium@users.noreply.github.com": "itsflownium",
|
||||
"dirtyren@users.noreply.github.com": "dirtyren",
|
||||
"evansrory@gmail.com": "zimigit2020",
|
||||
"237263164+ft-ioxcs@users.noreply.github.com": "ft-ioxcs",
|
||||
"tharushkadinujaya05@gmail.com": "0xneobyte",
|
||||
"138671361+Veritas-7@users.noreply.github.com": "Veritas-7",
|
||||
|
||||
@@ -1653,6 +1653,37 @@ class TestAuxiliaryFallbackLayering:
|
||||
exc.status_code = 402
|
||||
return exc
|
||||
|
||||
def test_auto_provider_uses_task_then_main_chain_before_builtin_chain(self, monkeypatch):
|
||||
"""Auto aux call failures try per-task then top-level fallback before built-ins."""
|
||||
primary_client = MagicMock()
|
||||
primary_client.chat.completions.create.side_effect = self._make_payment_err()
|
||||
|
||||
main_chain_client = MagicMock()
|
||||
main_chain_client.chat.completions.create.return_value = MagicMock(choices=[
|
||||
MagicMock(message=MagicMock(content="from main fallback chain"))
|
||||
])
|
||||
|
||||
with patch("agent.auxiliary_client._get_cached_client",
|
||||
return_value=(primary_client, "qwen/qwen3.5-122b-a10b")), \
|
||||
patch("agent.auxiliary_client._resolve_task_provider_model",
|
||||
return_value=("auto", None, None, None, None)), \
|
||||
patch("agent.auxiliary_client._try_configured_fallback_chain",
|
||||
return_value=(None, None, "")) as mock_task_chain, \
|
||||
patch("agent.auxiliary_client._try_main_fallback_chain",
|
||||
return_value=(main_chain_client, "inclusionai/ring-2.6-1t:free", "openrouter")) as mock_main_chain, \
|
||||
patch("agent.auxiliary_client._try_payment_fallback") as mock_builtin_chain:
|
||||
result = call_llm(
|
||||
task="title_generation",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
)
|
||||
|
||||
assert main_chain_client.chat.completions.create.called
|
||||
mock_task_chain.assert_called_once_with(
|
||||
"title_generation", "auto", reason="payment error")
|
||||
mock_main_chain.assert_called_once_with(
|
||||
"title_generation", "auto", reason="payment error")
|
||||
mock_builtin_chain.assert_not_called()
|
||||
|
||||
def test_explicit_provider_uses_configured_chain_first(self, monkeypatch, caplog):
|
||||
"""When a user has fallback_chain configured, it's tried BEFORE the main agent model."""
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
|
||||
|
||||
@@ -118,6 +118,64 @@ class TestResolveAutoMainFirst:
|
||||
assert client is chain_client
|
||||
assert model == "google/gemini-3-flash-preview"
|
||||
|
||||
def test_main_unavailable_uses_task_fallback_chain_before_builtin_chain(self):
|
||||
"""Auto aux resolution honors auxiliary.<task>.fallback_chain before built-ins."""
|
||||
task_client = MagicMock()
|
||||
with patch(
|
||||
"agent.auxiliary_client._read_main_provider", return_value="nvidia",
|
||||
), patch(
|
||||
"agent.auxiliary_client._read_main_model", return_value="qwen/qwen3.5-122b-a10b",
|
||||
), patch(
|
||||
"agent.auxiliary_client.resolve_provider_client",
|
||||
return_value=(None, None), # main provider has no client
|
||||
), patch(
|
||||
"agent.auxiliary_client._try_configured_fallback_chain",
|
||||
return_value=(task_client, "task-free-model", "fallback_chain[0](openrouter)"),
|
||||
) as mock_task_chain, patch(
|
||||
"agent.auxiliary_client._try_main_fallback_chain",
|
||||
) as mock_main_chain, patch(
|
||||
"agent.auxiliary_client._try_openrouter",
|
||||
) as mock_openrouter:
|
||||
from agent.auxiliary_client import _resolve_auto
|
||||
|
||||
client, model = _resolve_auto(task="title_generation")
|
||||
|
||||
assert client is task_client
|
||||
assert model == "task-free-model"
|
||||
mock_task_chain.assert_called_once_with(
|
||||
"title_generation", "nvidia", reason="main provider unavailable")
|
||||
mock_main_chain.assert_not_called()
|
||||
mock_openrouter.assert_not_called()
|
||||
|
||||
def test_main_unavailable_uses_main_fallback_chain_before_builtin_chain(self):
|
||||
"""Auto aux resolution honors top-level fallback_providers before built-ins."""
|
||||
main_fallback_client = MagicMock()
|
||||
with patch(
|
||||
"agent.auxiliary_client._read_main_provider", return_value="nvidia",
|
||||
), patch(
|
||||
"agent.auxiliary_client._read_main_model", return_value="qwen/qwen3.5-122b-a10b",
|
||||
), patch(
|
||||
"agent.auxiliary_client.resolve_provider_client",
|
||||
return_value=(None, None), # main provider has no client
|
||||
), patch(
|
||||
"agent.auxiliary_client._try_configured_fallback_chain",
|
||||
return_value=(None, None, ""),
|
||||
), patch(
|
||||
"agent.auxiliary_client._try_main_fallback_chain",
|
||||
return_value=(main_fallback_client, "inclusionai/ring-2.6-1t:free", "openrouter"),
|
||||
) as mock_main_chain, patch(
|
||||
"agent.auxiliary_client._try_openrouter",
|
||||
) as mock_openrouter:
|
||||
from agent.auxiliary_client import _resolve_auto
|
||||
|
||||
client, model = _resolve_auto(task="title_generation")
|
||||
|
||||
assert client is main_fallback_client
|
||||
assert model == "inclusionai/ring-2.6-1t:free"
|
||||
mock_main_chain.assert_called_once_with(
|
||||
"title_generation", "nvidia", reason="main provider unavailable")
|
||||
mock_openrouter.assert_not_called()
|
||||
|
||||
def test_no_main_config_uses_chain_directly(self):
|
||||
"""No main provider configured → skip step 1, use chain (no regression)."""
|
||||
chain_client = MagicMock()
|
||||
|
||||
@@ -1498,7 +1498,7 @@ class TestAgentConfigSignatureUserId:
|
||||
from gateway.run import GatewayRunner
|
||||
runtime = {"provider": "anthropic", "api_key": "k", "base_url": "", "api_mode": "chat_completions"}
|
||||
sig_a = GatewayRunner._agent_config_signature(
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="86701400"
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="7654321"
|
||||
)
|
||||
sig_b = GatewayRunner._agent_config_signature(
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="491827364"
|
||||
@@ -1509,10 +1509,10 @@ class TestAgentConfigSignatureUserId:
|
||||
from gateway.run import GatewayRunner
|
||||
runtime = {"provider": "anthropic", "api_key": "k", "base_url": "", "api_mode": "chat_completions"}
|
||||
sig_1 = GatewayRunner._agent_config_signature(
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="86701400"
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="7654321"
|
||||
)
|
||||
sig_2 = GatewayRunner._agent_config_signature(
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="86701400"
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "", user_id="7654321"
|
||||
)
|
||||
assert sig_1 == sig_2
|
||||
|
||||
@@ -1521,11 +1521,11 @@ class TestAgentConfigSignatureUserId:
|
||||
runtime = {"provider": "anthropic", "api_key": "k", "base_url": "", "api_mode": "chat_completions"}
|
||||
sig_a = GatewayRunner._agent_config_signature(
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "",
|
||||
user_id="86701400", user_id_alt="@igor_tg",
|
||||
user_id="7654321", user_id_alt="@igor_tg",
|
||||
)
|
||||
sig_b = GatewayRunner._agent_config_signature(
|
||||
"claude-sonnet-4", runtime, ["hermes-telegram"], "",
|
||||
user_id="86701400", user_id_alt="@erosika_tg",
|
||||
user_id="7654321", user_id_alt="@erosika_tg",
|
||||
)
|
||||
assert sig_a != sig_b
|
||||
|
||||
|
||||
@@ -832,7 +832,7 @@ class TestLoadGatewayConfig:
|
||||
|
||||
assert config.platforms[Platform.TELEGRAM].extra["rich_messages"] is False
|
||||
|
||||
def test_load_config_default_disables_telegram_rich_messages(self, tmp_path, monkeypatch):
|
||||
def test_load_config_default_enables_telegram_rich_messages(self, tmp_path, monkeypatch):
|
||||
hermes_home = tmp_path / ".hermes"
|
||||
hermes_home.mkdir()
|
||||
|
||||
@@ -842,7 +842,7 @@ class TestLoadGatewayConfig:
|
||||
|
||||
config = load_config()
|
||||
|
||||
assert config["telegram"]["extra"]["rich_messages"] is False
|
||||
assert config["telegram"]["extra"]["rich_messages"] is True
|
||||
|
||||
def test_bridges_telegram_extra_base_url_from_config_yaml(self, tmp_path, monkeypatch):
|
||||
hermes_home = tmp_path / ".hermes"
|
||||
|
||||
@@ -450,3 +450,63 @@ class TestCleanupProgress:
|
||||
}
|
||||
}
|
||||
assert resolve_display_setting(config, "telegram", "cleanup_progress") is True, val
|
||||
|
||||
|
||||
class TestToolProgressGrouping:
|
||||
"""resolve_display_setting() for the tool_progress_grouping knob."""
|
||||
|
||||
def test_default_is_accumulate(self):
|
||||
"""No config anywhere → global default 'accumulate'."""
|
||||
from gateway.display_config import resolve_display_setting
|
||||
|
||||
assert (
|
||||
resolve_display_setting({}, "telegram", "tool_progress_grouping")
|
||||
== "accumulate"
|
||||
)
|
||||
|
||||
def test_global_separate(self):
|
||||
from gateway.display_config import resolve_display_setting
|
||||
|
||||
config = {"display": {"tool_progress_grouping": "separate"}}
|
||||
assert (
|
||||
resolve_display_setting(config, "discord", "tool_progress_grouping")
|
||||
== "separate"
|
||||
)
|
||||
|
||||
def test_platform_override_wins(self):
|
||||
from gateway.display_config import resolve_display_setting
|
||||
|
||||
config = {
|
||||
"display": {
|
||||
"tool_progress_grouping": "accumulate",
|
||||
"platforms": {"discord": {"tool_progress_grouping": "separate"}},
|
||||
}
|
||||
}
|
||||
assert (
|
||||
resolve_display_setting(config, "discord", "tool_progress_grouping")
|
||||
== "separate"
|
||||
)
|
||||
# Other platforms still get the global value.
|
||||
assert (
|
||||
resolve_display_setting(config, "telegram", "tool_progress_grouping")
|
||||
== "accumulate"
|
||||
)
|
||||
|
||||
def test_invalid_value_falls_back_to_accumulate(self):
|
||||
"""_normalise rejects anything outside accumulate|separate."""
|
||||
from gateway.display_config import resolve_display_setting
|
||||
|
||||
config = {"display": {"tool_progress_grouping": "bogus"}}
|
||||
assert (
|
||||
resolve_display_setting(config, "telegram", "tool_progress_grouping")
|
||||
== "accumulate"
|
||||
)
|
||||
|
||||
def test_case_insensitive(self):
|
||||
from gateway.display_config import resolve_display_setting
|
||||
|
||||
config = {"display": {"tool_progress_grouping": "SEPARATE"}}
|
||||
assert (
|
||||
resolve_display_setting(config, "telegram", "tool_progress_grouping")
|
||||
== "separate"
|
||||
)
|
||||
|
||||
@@ -6,6 +6,124 @@ import pytest
|
||||
from unittest.mock import MagicMock, patch, AsyncMock
|
||||
|
||||
from gateway.config import Platform, PlatformConfig
|
||||
from gateway.run import (
|
||||
_resolve_gateway_display_bool,
|
||||
_resolve_progress_thread_id,
|
||||
)
|
||||
|
||||
|
||||
class TestMattermostProgressThreadRouting:
|
||||
def test_top_level_mattermost_progress_uses_event_message_id(self):
|
||||
assert _resolve_progress_thread_id(
|
||||
Platform.MATTERMOST,
|
||||
source_thread_id=None,
|
||||
event_message_id="top_post_123",
|
||||
) == "top_post_123"
|
||||
|
||||
def test_threaded_mattermost_progress_prefers_existing_thread_root(self):
|
||||
assert _resolve_progress_thread_id(
|
||||
Platform.MATTERMOST,
|
||||
source_thread_id="root_post_123",
|
||||
event_message_id="reply_post_456",
|
||||
) == "root_post_123"
|
||||
|
||||
def test_telegram_progress_does_not_use_message_id_as_thread_id(self):
|
||||
assert _resolve_progress_thread_id(
|
||||
Platform.TELEGRAM,
|
||||
source_thread_id=None,
|
||||
event_message_id="12345",
|
||||
) is None
|
||||
|
||||
|
||||
class TestMattermostDisplayHygiene:
|
||||
def test_mattermost_requires_platform_opt_in_for_interim_assistant_messages(self):
|
||||
"""Global interim commentary must not make Mattermost leak scratch notes."""
|
||||
user_config = {"display": {"interim_assistant_messages": True}}
|
||||
|
||||
assert _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
"mattermost",
|
||||
"interim_assistant_messages",
|
||||
default=True,
|
||||
platform=Platform.MATTERMOST,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
) is False
|
||||
|
||||
def test_mattermost_platform_opt_in_can_enable_interim_assistant_messages(self):
|
||||
"""Mattermost can still opt into commentary explicitly per platform."""
|
||||
user_config = {
|
||||
"display": {
|
||||
"interim_assistant_messages": False,
|
||||
"platforms": {
|
||||
"mattermost": {"interim_assistant_messages": True},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
assert _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
"mattermost",
|
||||
"interim_assistant_messages",
|
||||
default=True,
|
||||
platform=Platform.MATTERMOST,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
) is True
|
||||
|
||||
def test_mattermost_requires_platform_opt_in_for_thinking_progress(self):
|
||||
"""Global thinking_progress must not surface internal analysis in Mattermost."""
|
||||
user_config = {"display": {"thinking_progress": True}}
|
||||
|
||||
assert _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
"mattermost",
|
||||
"thinking_progress",
|
||||
default=False,
|
||||
platform=Platform.MATTERMOST,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
) is False
|
||||
|
||||
def test_mattermost_requires_platform_opt_in_for_show_reasoning(self):
|
||||
"""Global show_reasoning must not prepend scratch reasoning in Mattermost."""
|
||||
user_config = {"display": {"show_reasoning": True}}
|
||||
|
||||
assert _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
"mattermost",
|
||||
"show_reasoning",
|
||||
default=False,
|
||||
platform=Platform.MATTERMOST,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
) is False
|
||||
|
||||
def test_mattermost_platform_opt_in_can_enable_show_reasoning(self):
|
||||
user_config = {
|
||||
"display": {
|
||||
"show_reasoning": False,
|
||||
"platforms": {"mattermost": {"show_reasoning": True}},
|
||||
}
|
||||
}
|
||||
|
||||
assert _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
"mattermost",
|
||||
"show_reasoning",
|
||||
default=False,
|
||||
platform=Platform.MATTERMOST,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
) is True
|
||||
|
||||
def test_global_thinking_progress_still_applies_to_other_platforms(self):
|
||||
"""The Mattermost guard must not silently neuter Telegram/other chats."""
|
||||
user_config = {"display": {"thinking_progress": True}}
|
||||
|
||||
assert _resolve_gateway_display_bool(
|
||||
user_config,
|
||||
"telegram",
|
||||
"thinking_progress",
|
||||
default=False,
|
||||
platform=Platform.TELEGRAM,
|
||||
require_platform_override_for={Platform.MATTERMOST},
|
||||
) is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -237,6 +355,110 @@ class TestMattermostSend:
|
||||
payload = self.adapter._session.post.call_args[1]["json"]
|
||||
assert "root_id" not in payload
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_uses_metadata_thread_id_for_progress_messages(self):
|
||||
"""Progress/status messages pass Mattermost thread context via metadata."""
|
||||
self.adapter._reply_mode = "thread"
|
||||
self.adapter._api_get = AsyncMock(return_value={"id": "root_post_123", "root_id": ""})
|
||||
self.adapter._api_post = AsyncMock(return_value={"id": "progress_post"})
|
||||
|
||||
result = await self.adapter.send(
|
||||
"channel_1",
|
||||
"⚡ terminal...",
|
||||
metadata={"thread_id": "root_post_123"},
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
payload = self.adapter._api_post.call_args_list[0][0][1]
|
||||
assert payload["root_id"] == "root_post_123"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_progress_send_with_invalid_thread_root_never_falls_back_flat(self):
|
||||
"""Tool/status/progress bubbles must stay quiet when the thread is broken."""
|
||||
self.adapter._reply_mode = "thread"
|
||||
self.adapter._api_get = AsyncMock(return_value={"id": "bad_root", "root_id": ""})
|
||||
self.adapter._last_post_status = 400
|
||||
self.adapter._last_post_error = "api.context.invalid_param.app_error: invalid root_id"
|
||||
self.adapter._api_post = AsyncMock(return_value={})
|
||||
|
||||
result = await self.adapter.send(
|
||||
"channel_1",
|
||||
"⚙️ terminal...",
|
||||
metadata={"thread_id": "bad_root"},
|
||||
)
|
||||
|
||||
assert result.success is False
|
||||
assert self.adapter._api_post.call_count == 1
|
||||
payload = self.adapter._api_post.call_args_list[0][0][1]
|
||||
assert payload["root_id"] == "bad_root"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_notify_send_with_invalid_thread_root_falls_back_flat_with_warning(self):
|
||||
"""Notify-worthy replies may fall back flat so the answer is not lost."""
|
||||
self.adapter._reply_mode = "thread"
|
||||
self.adapter._api_get = AsyncMock(return_value={"id": "bad_root", "root_id": ""})
|
||||
self.adapter._last_post_status = 400
|
||||
self.adapter._last_post_error = "api.context.invalid_param.app_error: invalid root_id"
|
||||
self.adapter._api_post = AsyncMock(side_effect=[{}, {"id": "flat_final"}])
|
||||
|
||||
result = await self.adapter.send(
|
||||
"channel_1",
|
||||
"Final answer body",
|
||||
reply_to="bad_root",
|
||||
metadata={"notify": True},
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.message_id == "flat_final"
|
||||
assert self.adapter._api_post.call_count == 2
|
||||
threaded_payload = self.adapter._api_post.call_args_list[0][0][1]
|
||||
flat_payload = self.adapter._api_post.call_args_list[1][0][1]
|
||||
assert threaded_payload["root_id"] == "bad_root"
|
||||
assert "root_id" not in flat_payload
|
||||
assert flat_payload["channel_id"] == "channel_1"
|
||||
assert "Mattermost thread delivery failed" in flat_payload["message"]
|
||||
assert "Final answer body" in flat_payload["message"]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_notify_send_with_server_error_does_not_fall_back_flat(self):
|
||||
"""Notify fallback is only for broken thread roots, not generic API failures."""
|
||||
self.adapter._reply_mode = "thread"
|
||||
self.adapter._api_get = AsyncMock(return_value={"id": "root_post", "root_id": ""})
|
||||
self.adapter._last_post_status = 500
|
||||
self.adapter._last_post_error = "Internal Server Error"
|
||||
self.adapter._api_post = AsyncMock(return_value={})
|
||||
|
||||
result = await self.adapter.send(
|
||||
"channel_1",
|
||||
"Final answer body",
|
||||
reply_to="root_post",
|
||||
metadata={"notify": True},
|
||||
)
|
||||
|
||||
assert result.success is False
|
||||
assert self.adapter._api_post.call_count == 1
|
||||
payload = self.adapter._api_post.call_args_list[0][0][1]
|
||||
assert payload["root_id"] == "root_post"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_progress_send_with_invalid_thread_root_never_falls_back_flat(self):
|
||||
"""Tool/status/progress bubbles must stay quiet when the thread is broken."""
|
||||
self.adapter._reply_mode = "thread"
|
||||
self.adapter._api_get = AsyncMock(return_value={"id": "bad_root", "root_id": ""})
|
||||
self.adapter._api_post = AsyncMock(return_value={})
|
||||
|
||||
result = await self.adapter.send(
|
||||
"channel_1",
|
||||
"⚙️ terminal...",
|
||||
metadata={"thread_id": "bad_root"},
|
||||
)
|
||||
|
||||
assert result.success is False
|
||||
assert self.adapter._api_post.call_count == 1
|
||||
payload = self.adapter._api_post.call_args_list[0][0][1]
|
||||
assert payload["root_id"] == "bad_root"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_api_failure(self):
|
||||
"""When API returns error, send should return failure."""
|
||||
@@ -750,3 +972,65 @@ class TestMattermostMediaTypes:
|
||||
assert msg.media_types == ["application/pdf"]
|
||||
assert not msg.media_types[0].startswith("image/")
|
||||
assert not msg.media_types[0].startswith("audio/")
|
||||
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mattermost_top_level_channel_post_is_thread_root():
|
||||
adapter = _make_adapter()
|
||||
adapter._reply_mode = "thread"
|
||||
adapter._bot_user_id = "bot_user_id"
|
||||
adapter._bot_username = "hermes-bot"
|
||||
adapter.handle_message = AsyncMock()
|
||||
post_data = {
|
||||
"id": "top_post_123",
|
||||
"user_id": "user_123",
|
||||
"channel_id": "chan_456",
|
||||
"message": "@hermes-bot start work",
|
||||
"root_id": "",
|
||||
}
|
||||
event = {
|
||||
"event": "posted",
|
||||
"data": {
|
||||
"post": json.dumps(post_data),
|
||||
"channel_type": "O",
|
||||
"sender_name": "@alice",
|
||||
},
|
||||
}
|
||||
|
||||
await adapter._handle_ws_event(event)
|
||||
|
||||
msg_event = adapter.handle_message.call_args[0][0]
|
||||
assert msg_event.source.thread_id == "top_post_123"
|
||||
assert msg_event.source.message_id == "top_post_123"
|
||||
assert msg_event.message_id == "top_post_123"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mattermost_dm_post_does_not_seed_thread_root():
|
||||
adapter = _make_adapter()
|
||||
adapter._reply_mode = "thread"
|
||||
adapter._bot_user_id = "bot_user_id"
|
||||
adapter._bot_username = "hermes-bot"
|
||||
adapter.handle_message = AsyncMock()
|
||||
post_data = {
|
||||
"id": "dm_post_123",
|
||||
"user_id": "user_123",
|
||||
"channel_id": "dm_chan",
|
||||
"message": "hello",
|
||||
"root_id": "",
|
||||
}
|
||||
event = {
|
||||
"event": "posted",
|
||||
"data": {
|
||||
"post": json.dumps(post_data),
|
||||
"channel_type": "D",
|
||||
"sender_name": "@alice",
|
||||
},
|
||||
}
|
||||
|
||||
await adapter._handle_ws_event(event)
|
||||
|
||||
msg_event = adapter.handle_message.call_args[0][0]
|
||||
assert msg_event.source.thread_id is None
|
||||
assert msg_event.source.message_id == "dm_post_123"
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Contract: media-send overrides must accept the ``metadata`` kwarg.
|
||||
|
||||
``BasePlatformAdapter.send_multiple_images`` passes ``metadata=metadata``
|
||||
to ``send_image`` / ``send_image_file`` / ``send_animation`` on every send.
|
||||
An override whose signature stops at ``reply_to`` raises ``TypeError:
|
||||
send_image() got an unexpected keyword argument 'metadata'`` at runtime —
|
||||
which is exactly how image delivery broke on WhatsApp and email.
|
||||
|
||||
This mirrors ``test_discord_media_metadata.py`` but covers the two
|
||||
adapters that previously slipped, plus a best-effort sweep over every
|
||||
adapter that imports cleanly so the next slip is caught at test time.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import inspect
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _accepts_metadata(method) -> bool:
|
||||
params = inspect.signature(method).parameters
|
||||
if "metadata" in params:
|
||||
return True
|
||||
# A ``**kwargs`` catch-all also absorbs metadata (the convention used by
|
||||
# WhatsApp's send_video / send_voice / send_document overrides).
|
||||
return any(p.kind is inspect.Parameter.VAR_KEYWORD for p in params.values())
|
||||
|
||||
|
||||
# (module, class) for the two adapters this fix targeted. These must import
|
||||
# in CI, so assert directly rather than skipping.
|
||||
@pytest.mark.parametrize(
|
||||
"module_name, class_name",
|
||||
[
|
||||
("gateway.platforms.whatsapp", "WhatsAppAdapter"),
|
||||
("gateway.platforms.email", "EmailAdapter"),
|
||||
],
|
||||
)
|
||||
def test_send_image_accepts_metadata(module_name, class_name):
|
||||
cls = getattr(importlib.import_module(module_name), class_name)
|
||||
assert _accepts_metadata(cls.send_image), (
|
||||
f"{class_name}.send_image must accept 'metadata' (or **kwargs) — "
|
||||
f"send_multiple_images passes it on every send"
|
||||
)
|
||||
|
||||
|
||||
# Best-effort sweep across all shipped adapters. Modules whose optional
|
||||
# platform SDK isn't installed are skipped; an adapter that imports but
|
||||
# whose override drops metadata is a hard failure.
|
||||
_ALL_ADAPTERS = [
|
||||
("gateway.platforms.bluebubbles", "BlueBubblesAdapter"),
|
||||
("gateway.platforms.dingtalk", "DingTalkAdapter"),
|
||||
("gateway.platforms.discord", "DiscordAdapter"),
|
||||
("gateway.platforms.email", "EmailAdapter"),
|
||||
("gateway.platforms.feishu", "FeishuAdapter"),
|
||||
("gateway.platforms.matrix", "MatrixAdapter"),
|
||||
("gateway.platforms.mattermost", "MattermostAdapter"),
|
||||
("gateway.platforms.signal", "SignalAdapter"),
|
||||
("gateway.platforms.slack", "SlackAdapter"),
|
||||
("gateway.platforms.telegram", "TelegramAdapter"),
|
||||
("gateway.platforms.wecom", "WeComAdapter"),
|
||||
("gateway.platforms.weixin", "WeixinAdapter"),
|
||||
("gateway.platforms.whatsapp", "WhatsAppAdapter"),
|
||||
("gateway.platforms.yuanbao", "YuanbaoAdapter"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("module_name, class_name", _ALL_ADAPTERS)
|
||||
def test_all_adapters_send_image_metadata_sweep(module_name, class_name):
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
except Exception as exc: # optional platform dep not installed
|
||||
pytest.skip(f"{module_name} not importable: {exc}")
|
||||
cls = getattr(module, class_name, None)
|
||||
if cls is None or "send_image" not in cls.__dict__:
|
||||
pytest.skip(f"{class_name} has no send_image override")
|
||||
assert _accepts_metadata(cls.send_image), (
|
||||
f"{class_name}.send_image drops the 'metadata' kwarg"
|
||||
)
|
||||
@@ -106,6 +106,42 @@ class TestInitialReplyToId:
|
||||
assert call_kwargs["metadata"] == {**metadata, "expect_edits": True}
|
||||
assert metadata == {"thread_id": "omt_topic789"}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_final_first_send_marks_metadata_notify_true(self):
|
||||
"""Final streaming sends should use the existing notify=True marker."""
|
||||
adapter = _make_adapter()
|
||||
consumer = GatewayStreamConsumer(
|
||||
adapter,
|
||||
"chat_123",
|
||||
metadata={"thread_id": "root_post_123"},
|
||||
initial_reply_to_id="reply_post_456",
|
||||
)
|
||||
|
||||
await consumer._send_or_edit("Final answer", finalize=True)
|
||||
|
||||
call_kwargs = adapter.send.call_args[1]
|
||||
metadata = call_kwargs["metadata"]
|
||||
assert metadata["thread_id"] == "root_post_123"
|
||||
assert metadata["notify"] is True
|
||||
assert "delivery_kind" not in metadata
|
||||
assert "allow_flat_fallback" not in metadata
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_nonfinal_first_send_does_not_mark_notify(self):
|
||||
"""Preview/interim streaming sends must not be notify-worthy."""
|
||||
adapter = _make_adapter()
|
||||
consumer = GatewayStreamConsumer(
|
||||
adapter,
|
||||
"chat_123",
|
||||
metadata={"thread_id": "root_post_123"},
|
||||
initial_reply_to_id="reply_post_456",
|
||||
)
|
||||
|
||||
await consumer._send_or_edit("Preview", finalize=False)
|
||||
|
||||
metadata = adapter.send.call_args[1]["metadata"]
|
||||
assert metadata == {"thread_id": "root_post_123", "expect_edits": True}
|
||||
|
||||
|
||||
class TestOverflowFirstMessage:
|
||||
"""Verify thread routing is preserved when the first message overflows."""
|
||||
|
||||
@@ -134,7 +134,7 @@ async def test_stream_consumer_fallback_sends_tail_after_partial_overflow():
|
||||
|
||||
adapter.send.assert_awaited_once()
|
||||
assert adapter.send.await_args.kwargs["content"] == "world"
|
||||
assert adapter.send.await_args.kwargs["metadata"] == {"thread_id": "77"}
|
||||
assert adapter.send.await_args.kwargs["metadata"] == {"thread_id": "77", "notify": True}
|
||||
adapter.delete_message.assert_not_awaited()
|
||||
assert consumer.final_response_sent is True
|
||||
assert consumer.final_content_delivered is True
|
||||
|
||||
@@ -61,6 +61,8 @@ def _make_adapter(extra=None):
|
||||
bot.send_message = AsyncMock(return_value=MagicMock(message_id=1))
|
||||
bot.send_chat_action = AsyncMock() # keeps the post-send typing re-trigger quiet
|
||||
bot.send_message_draft = AsyncMock(return_value=True) # legacy draft fallback
|
||||
bot.edit_message_text = AsyncMock(return_value=MagicMock(message_id=1)) # legacy edit path
|
||||
bot.delete_message = AsyncMock(return_value=True)
|
||||
adapter._bot = bot
|
||||
return adapter
|
||||
|
||||
@@ -184,7 +186,10 @@ async def test_rich_messages_opt_out_accepts_string_false():
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_rich_messages_default_is_disabled():
|
||||
async def test_rich_messages_default_is_enabled():
|
||||
"""Rich messages are on by default (Bot API 10.1); rich-eligible content
|
||||
(tables/task lists/details/math) goes through sendRichMessage without the
|
||||
user having to opt in."""
|
||||
config = PlatformConfig(enabled=True, token="fake-token")
|
||||
adapter = TelegramAdapter(config)
|
||||
bot = MagicMock()
|
||||
@@ -195,6 +200,42 @@ async def test_rich_messages_default_is_disabled():
|
||||
|
||||
result = await adapter.send("12345", RICH_CONTENT)
|
||||
|
||||
assert result.success is True
|
||||
bot = adapter._bot
|
||||
assert bot is not None
|
||||
bot.do_api_request.assert_awaited_once()
|
||||
bot.send_message.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_rich_messages_can_be_opted_out():
|
||||
"""Setting platforms.telegram.extra.rich_messages: false keeps every reply
|
||||
on the legacy MarkdownV2 path even for rich-eligible content."""
|
||||
config = PlatformConfig(
|
||||
enabled=True, token="fake-token", extra={"rich_messages": False}
|
||||
)
|
||||
adapter = TelegramAdapter(config)
|
||||
bot = MagicMock()
|
||||
bot.do_api_request = AsyncMock(return_value=SimpleNamespace(message_id=123))
|
||||
bot.send_message = AsyncMock(return_value=MagicMock(message_id=1))
|
||||
bot.send_chat_action = AsyncMock()
|
||||
adapter._bot = bot
|
||||
|
||||
result = await adapter.send("12345", RICH_CONTENT)
|
||||
|
||||
assert result.success is True
|
||||
bot.do_api_request.assert_not_called()
|
||||
bot.send_message.assert_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_plain_markdown_stays_on_legacy_path():
|
||||
"""Ordinary replies (no table/task-list/details/math) stay on the legacy
|
||||
MarkdownV2 path for consistent client rendering, even with rich enabled."""
|
||||
adapter = _make_adapter()
|
||||
|
||||
result = await adapter.send("12345", "Hello **there**\n\nA normal reply.")
|
||||
|
||||
assert result.success is True
|
||||
bot = adapter._bot
|
||||
assert bot is not None
|
||||
@@ -240,7 +281,9 @@ async def test_oversized_content_skips_rich_and_chunks():
|
||||
async def test_rich_limit_is_characters_not_bytes():
|
||||
"""Telegram's rich limit is UTF-8 characters, not encoded bytes."""
|
||||
adapter = _make_adapter()
|
||||
cjk = "测" * 20000 # 20k chars, 60k UTF-8 bytes
|
||||
# Rich-eligible (table) so the content takes the rich path; the CJK body
|
||||
# is 20k chars / 60k UTF-8 bytes — over the byte count, under the char cap.
|
||||
cjk = "| a | b |\n|---|---|\n" + "测" * 20000 # 20k chars, ~60k UTF-8 bytes
|
||||
assert len(cjk.encode("utf-8")) > TelegramAdapter.RICH_MESSAGE_MAX_BYTES
|
||||
assert len(cjk) <= TelegramAdapter.RICH_MESSAGE_MAX_CHARS
|
||||
|
||||
@@ -324,7 +367,9 @@ async def test_real_ptb_endpoint_missing_falls_back_and_latches_off(exc):
|
||||
async def test_rich_payload_preserves_link_preview_disable():
|
||||
adapter = _make_adapter(extra={"disable_link_previews": True})
|
||||
|
||||
result = await adapter.send("12345", "See https://example.com")
|
||||
result = await adapter.send(
|
||||
"12345", "| Link | Note |\n|---|---|\n| See https://example.com | x |"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
api_kwargs = _rich_api_kwargs(adapter)
|
||||
@@ -575,3 +620,139 @@ async def test_rich_draft_opt_out_uses_legacy():
|
||||
assert bot is not None
|
||||
bot.do_api_request.assert_not_called()
|
||||
bot.send_message_draft.assert_awaited_once()
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Rich finalize via editMessageText (Bot API 10.1 rich_message edit param).
|
||||
# Streamed previews finalize by editing the existing message IN PLACE as rich,
|
||||
# so tables/task lists survive without a fresh send + delete (no duplicate).
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _rich_edit_kwargs(adapter):
|
||||
"""Return the api_kwargs dict from the single editMessageText rich call."""
|
||||
call = adapter._bot.do_api_request.call_args
|
||||
assert call.args[0] == "editMessageText"
|
||||
return call.kwargs["api_kwargs"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_finalize_edit_uses_rich_for_table_content():
|
||||
"""Finalizing a streamed preview whose content is a table edits the
|
||||
existing message IN PLACE via editMessageText's rich_message param —
|
||||
no fresh send, no delete, no duplicate."""
|
||||
adapter = _make_adapter()
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", RICH_CONTENT, finalize=True,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.message_id == "555" # same message, edited in place
|
||||
api_kwargs = _rich_edit_kwargs(adapter)
|
||||
assert api_kwargs["message_id"] == 555
|
||||
# RAW markdown is passed through so table pipes survive.
|
||||
assert api_kwargs["rich_message"]["markdown"] == RICH_CONTENT
|
||||
# No fresh send / delete — the whole point of the in-place rich edit.
|
||||
adapter._bot.edit_message_text.assert_not_called()
|
||||
adapter._bot.delete_message.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_finalize_edit_plain_content_stays_legacy():
|
||||
"""Finalizing plain content (no table/task-list/details/math) uses the
|
||||
legacy MarkdownV2 edit_message_text path, not the rich edit endpoint."""
|
||||
adapter = _make_adapter()
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", "Just a normal answer, no rich constructs.", finalize=True,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
adapter._bot.do_api_request.assert_not_called()
|
||||
adapter._bot.edit_message_text.assert_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_finalize_edit_rich_capability_error_falls_back_to_legacy():
|
||||
"""A capability error on the rich edit latches rich off and falls back to
|
||||
the legacy MarkdownV2 edit so the user still gets the final answer."""
|
||||
adapter = _make_adapter()
|
||||
adapter._bot.do_api_request = AsyncMock(side_effect=PTB_ENDPOINT_NOT_FOUND)
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", RICH_CONTENT, finalize=True,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert adapter._rich_send_disabled is True
|
||||
adapter._bot.edit_message_text.assert_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_finalize_edit_rich_not_modified_is_success_noop():
|
||||
"""'Message is not modified' on a rich edit is a no-op success — must NOT
|
||||
fall through to a redundant legacy edit."""
|
||||
adapter = _make_adapter()
|
||||
adapter._bot.do_api_request = AsyncMock(
|
||||
side_effect=BadRequest("Message is not modified")
|
||||
)
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", RICH_CONTENT, finalize=True,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
adapter._bot.edit_message_text.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_non_finalize_edit_never_uses_rich():
|
||||
"""Intermediate (non-finalize) stream edits stay on the plain edit path;
|
||||
rich is only applied on the final edit."""
|
||||
adapter = _make_adapter()
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", RICH_CONTENT, finalize=False,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
adapter._bot.do_api_request.assert_not_called()
|
||||
adapter._bot.edit_message_text.assert_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_finalize_edit_opt_out_uses_legacy():
|
||||
"""With rich_messages: false, even a table finalizes via the legacy
|
||||
MarkdownV2 edit path."""
|
||||
adapter = _make_adapter(extra={"rich_messages": False})
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", RICH_CONTENT, finalize=True,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
adapter._bot.do_api_request.assert_not_called()
|
||||
adapter._bot.edit_message_text.assert_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_finalize_edit_rich_over_markdownv2_limit_not_split():
|
||||
"""A rich table that exceeds the 4,096 MarkdownV2 limit but fits the 32,768
|
||||
rich cap is edited in place as one rich message, NOT split into legacy
|
||||
chunks."""
|
||||
adapter = _make_adapter()
|
||||
big_table = "| a | b |\n|---|---|\n" + "\n".join(
|
||||
f"| {'x' * 50} | {'y' * 50} |" for _ in range(40)
|
||||
)
|
||||
assert len(big_table) > TelegramAdapter.MAX_MESSAGE_LENGTH
|
||||
assert len(big_table) <= TelegramAdapter.RICH_MESSAGE_MAX_CHARS
|
||||
|
||||
result = await adapter.edit_message(
|
||||
"12345", "555", big_table, finalize=True,
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
api_kwargs = _rich_edit_kwargs(adapter)
|
||||
assert api_kwargs["rich_message"]["markdown"] == big_table
|
||||
adapter._bot.edit_message_text.assert_not_called()
|
||||
|
||||
@@ -76,7 +76,7 @@ async def test_base_adapter_routes_telegram_flac_media_tag_to_document_sender(tm
|
||||
adapter.send_document.assert_awaited_once_with(
|
||||
chat_id="chat-1",
|
||||
file_path=str(media_file),
|
||||
metadata=None,
|
||||
metadata={"notify": True},
|
||||
)
|
||||
adapter.send_voice.assert_not_awaited()
|
||||
|
||||
@@ -95,7 +95,7 @@ async def test_base_adapter_routes_non_voice_telegram_ogg_media_tag_to_document_
|
||||
adapter.send_document.assert_awaited_once_with(
|
||||
chat_id="chat-1",
|
||||
file_path=str(media_file),
|
||||
metadata=None,
|
||||
metadata={"notify": True},
|
||||
)
|
||||
adapter.send_voice.assert_not_awaited()
|
||||
|
||||
@@ -116,7 +116,7 @@ async def test_base_adapter_routes_voice_tagged_telegram_ogg_media_tag_to_voice_
|
||||
adapter.send_voice.assert_awaited_once_with(
|
||||
chat_id="chat-1",
|
||||
audio_path=str(media_file),
|
||||
metadata=None,
|
||||
metadata={"notify": True},
|
||||
)
|
||||
adapter.send_document.assert_not_awaited()
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -25,6 +26,37 @@ def _jwt_with_email(email: str) -> str:
|
||||
return f"{header}.{payload}.signature"
|
||||
|
||||
|
||||
def _codex_pool_only_store(*, exhausted: bool = False) -> dict:
|
||||
entry = {
|
||||
"id": "codex-1",
|
||||
"label": "codex@example.com",
|
||||
"auth_type": "oauth",
|
||||
"priority": 0,
|
||||
"source": "manual:device_code",
|
||||
"access_token": _jwt_with_email("codex@example.com"),
|
||||
"refresh_token": "refresh-token",
|
||||
"base_url": "https://chatgpt.com/backend-api/codex",
|
||||
"last_refresh": "2026-06-15T10:00:00Z",
|
||||
}
|
||||
if exhausted:
|
||||
entry.update(
|
||||
{
|
||||
"last_status": "exhausted",
|
||||
"last_status_at": time.time(),
|
||||
"last_error_code": 429,
|
||||
"last_error_reason": "usage_limit_reached",
|
||||
"last_error_message": "The usage limit has been reached",
|
||||
"last_error_reset_at": time.time() + 3600,
|
||||
}
|
||||
)
|
||||
return {
|
||||
"version": 1,
|
||||
"active_provider": "openai-codex",
|
||||
"providers": {},
|
||||
"credential_pool": {"openai-codex": [entry]},
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_provider_env(monkeypatch):
|
||||
for key in (
|
||||
@@ -483,6 +515,44 @@ def test_auth_add_codex_oauth_keeps_distinct_pool_accounts(tmp_path, monkeypatch
|
||||
assert payload["active_provider"] == "openai-codex"
|
||||
|
||||
|
||||
def test_codex_auth_status_reports_pool_only_credential(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hermes"))
|
||||
_write_auth_store(tmp_path, _codex_pool_only_store())
|
||||
|
||||
from hermes_cli.auth import get_codex_auth_status
|
||||
|
||||
status = get_codex_auth_status()
|
||||
|
||||
assert status["logged_in"] is True
|
||||
assert status["source"] == "pool:codex@example.com"
|
||||
|
||||
|
||||
def test_codex_auth_status_reports_pool_only_rate_limit(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hermes"))
|
||||
_write_auth_store(tmp_path, _codex_pool_only_store(exhausted=True))
|
||||
|
||||
from hermes_cli.auth import get_codex_auth_status
|
||||
|
||||
status = get_codex_auth_status()
|
||||
|
||||
assert status["logged_in"] is True
|
||||
assert status["rate_limited"] is True
|
||||
assert status["error_code"] == "codex_rate_limited"
|
||||
|
||||
|
||||
def test_codex_runtime_pool_only_rate_limit_is_not_missing_auth(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hermes"))
|
||||
_write_auth_store(tmp_path, _codex_pool_only_store(exhausted=True))
|
||||
|
||||
from hermes_cli.auth import AuthError, CODEX_RATE_LIMITED_CODE, resolve_codex_runtime_credentials
|
||||
|
||||
with pytest.raises(AuthError) as exc_info:
|
||||
resolve_codex_runtime_credentials()
|
||||
|
||||
assert exc_info.value.code == CODEX_RATE_LIMITED_CODE
|
||||
assert exc_info.value.relogin_required is False
|
||||
|
||||
|
||||
def test_auth_add_xai_oauth_sets_active_provider(tmp_path, monkeypatch):
|
||||
"""hermes auth add xai-oauth must write providers singleton and set active_provider.
|
||||
|
||||
|
||||
@@ -772,6 +772,25 @@ class TestUpdateCheckEndpoint:
|
||||
assert body["message"]
|
||||
assert body["behind"] is None
|
||||
|
||||
def test_managed_runtime_dashboard_is_not_applyable(self, monkeypatch):
|
||||
import hermes_cli.web_server as ws
|
||||
|
||||
monkeypatch.setattr(ws, "_dashboard_local_update_managed_externally", lambda: True)
|
||||
monkeypatch.setattr(
|
||||
ws,
|
||||
"detect_install_method",
|
||||
lambda *a, **k: pytest.fail(
|
||||
"managed runtime update check should not probe install method"
|
||||
),
|
||||
)
|
||||
|
||||
body = self.client.get("/api/hermes/update/check").json()
|
||||
assert body["install_method"] == "managed-runtime"
|
||||
assert body["can_apply"] is False
|
||||
assert body["update_available"] is False
|
||||
assert body["behind"] is None
|
||||
assert "managed outside this dashboard" in body["message"]
|
||||
|
||||
def test_check_failure_is_soft(self, monkeypatch):
|
||||
import hermes_cli.web_server as ws
|
||||
import hermes_cli.banner as banner
|
||||
|
||||
@@ -3,8 +3,9 @@
|
||||
The cache avoids re-validating Nous credentials on every menu paint —
|
||||
`hermes tools` → "All Platforms" used to fire ~31 OAuth refresh POSTs
|
||||
against portal.nousresearch.com during one render. The cache is keyed
|
||||
on auth.json mtime so login/logout flows invalidate naturally; tests
|
||||
and other writers can also call invalidate_nous_auth_status_cache().
|
||||
on auth.json path + mtime so profile switches stay isolated while
|
||||
login/logout flows invalidate naturally; tests and other writers can
|
||||
also call invalidate_nous_auth_status_cache().
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -88,6 +89,42 @@ def test_get_nous_auth_status_invalidates_on_auth_file_mtime(tmp_path, monkeypat
|
||||
auth_mod.invalidate_nous_auth_status_cache()
|
||||
|
||||
|
||||
def test_get_nous_auth_status_cache_is_scoped_by_auth_file_path(tmp_path, monkeypatch):
|
||||
"""Two profile homes with missing auth.json must not share cached status."""
|
||||
profile_a = tmp_path / "profiles" / "a"
|
||||
profile_b = tmp_path / "profiles" / "b"
|
||||
profile_a.mkdir(parents=True)
|
||||
profile_b.mkdir(parents=True)
|
||||
|
||||
from hermes_cli import auth as auth_mod
|
||||
|
||||
auth_mod.invalidate_nous_auth_status_cache()
|
||||
|
||||
call_count = {"n": 0}
|
||||
seen_auth_files = []
|
||||
|
||||
def fake_compute():
|
||||
call_count["n"] += 1
|
||||
seen_auth_files.append(auth_mod._auth_file_path())
|
||||
return {"logged_in": False, "call": call_count["n"]}
|
||||
|
||||
with patch.object(auth_mod, "_compute_nous_auth_status", side_effect=fake_compute):
|
||||
monkeypatch.setenv("HERMES_HOME", str(profile_a))
|
||||
first = auth_mod.get_nous_auth_status()
|
||||
monkeypatch.setenv("HERMES_HOME", str(profile_b))
|
||||
second = auth_mod.get_nous_auth_status()
|
||||
|
||||
assert call_count["n"] == 2
|
||||
assert first["call"] == 1
|
||||
assert second["call"] == 2
|
||||
assert seen_auth_files == [
|
||||
profile_a / "auth.json",
|
||||
profile_b / "auth.json",
|
||||
]
|
||||
|
||||
auth_mod.invalidate_nous_auth_status_cache()
|
||||
|
||||
|
||||
def test_invalidate_nous_auth_status_cache_forces_recompute(tmp_path, monkeypatch):
|
||||
"""Explicit invalidate forces the next call to re-compute."""
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
"""Node-26 resolution for the OpenTUI engine + the launch-cwd channel.
|
||||
|
||||
Regression coverage for two ways the local TUI silently fell back to Ink /
|
||||
showed the wrong directory:
|
||||
|
||||
1. fnm's active/default node was on an older line (v25) while a usable v26.3
|
||||
sat installed-but-inactive — ``_node26_bin_or_none`` only checked
|
||||
``HERMES_NODE`` + ``which node`` and so reported "no node 26" → OpenTUI
|
||||
unavailable → Ink fallback.
|
||||
2. ``TERMINAL_CWD`` (the gateway's launch-dir channel) was only exported in
|
||||
worktree mode, so a normal launch let the gateway auto-detect the engine's
|
||||
own package dir as the workspace.
|
||||
"""
|
||||
|
||||
import os
|
||||
import stat
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.main as main_mod
|
||||
|
||||
|
||||
def _fake_node(path, version: str) -> None:
|
||||
"""Write a stub `node` that prints `version` for `--version`."""
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(f'#!/bin/sh\necho "{version}"\n')
|
||||
path.chmod(path.stat().st_mode | stat.S_IEXEC | stat.S_IXGRP | stat.S_IXOTH)
|
||||
|
||||
|
||||
class TestFnmNode26Discovery:
|
||||
def test_discovers_inactive_v26_when_default_is_older(self, tmp_path, monkeypatch):
|
||||
"""A v26.3 installed under fnm is found even when PATH `node` is v25."""
|
||||
fnm_dir = tmp_path / "fnm"
|
||||
for ver in ("24.11.0", "25.9.0", "26.3.0"):
|
||||
_fake_node(fnm_dir / "node-versions" / f"v{ver}" / "installation" / "bin" / "node", f"v{ver}")
|
||||
monkeypatch.setenv("FNM_DIR", str(fnm_dir))
|
||||
monkeypatch.delenv("HERMES_NODE", raising=False)
|
||||
# PATH node is the too-old default (v25).
|
||||
monkeypatch.setattr(main_mod.shutil, "which", lambda _b: str(
|
||||
fnm_dir / "node-versions" / "v25.9.0" / "installation" / "bin" / "node"
|
||||
))
|
||||
|
||||
resolved = main_mod._node26_bin_or_none()
|
||||
assert resolved is not None
|
||||
assert "v26.3.0" in resolved # newest qualifying, not the v25 default
|
||||
|
||||
def test_candidates_sorted_newest_first(self, tmp_path, monkeypatch):
|
||||
fnm_dir = tmp_path / "fnm"
|
||||
for ver in ("26.1.0", "26.4.0", "25.0.0"):
|
||||
_fake_node(fnm_dir / "node-versions" / f"v{ver}" / "installation" / "bin" / "node", f"v{ver}")
|
||||
monkeypatch.setenv("FNM_DIR", str(fnm_dir))
|
||||
cands = main_mod._fnm_node26_candidates()
|
||||
# Directory-name order: 26.4 before 26.1 before 25.0.
|
||||
idx = [next(i for i, c in enumerate(cands) if f"v{v}" in c) for v in ("26.4.0", "26.1.0", "25.0.0")]
|
||||
assert idx == sorted(idx)
|
||||
|
||||
def test_no_fnm_dir_is_empty_not_error(self, tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("FNM_DIR", str(tmp_path / "does-not-exist"))
|
||||
monkeypatch.setenv("XDG_DATA_HOME", str(tmp_path / "xdg-none"))
|
||||
monkeypatch.setattr(main_mod.Path, "home", classmethod(lambda cls: tmp_path / "home-none"))
|
||||
assert main_mod._fnm_node26_candidates() == []
|
||||
|
||||
def test_hermes_node_still_wins(self, tmp_path, monkeypatch):
|
||||
"""An explicit HERMES_NODE >= 26.3 takes precedence over fnm discovery."""
|
||||
explicit = tmp_path / "explicit" / "node"
|
||||
_fake_node(explicit, "v26.5.0")
|
||||
monkeypatch.setenv("HERMES_NODE", str(explicit))
|
||||
monkeypatch.setattr(main_mod.shutil, "which", lambda _b: None)
|
||||
assert main_mod._node26_bin_or_none() == str(explicit)
|
||||
@@ -99,6 +99,70 @@ class TestResolveTuiHeapMb:
|
||||
assert self._resolve(64 * GB) == 8192
|
||||
|
||||
|
||||
class TestHeapOverride:
|
||||
"""HERMES_TUI_HEAP_MB env / display.tui_heap_mb config override (W1/D3).
|
||||
|
||||
The override REPLACES the 8192 default; the cgroup-fit clamp still applies on
|
||||
top so a too-high override can't exceed the container. Precedence: env > config.
|
||||
"""
|
||||
|
||||
def _resolve(self, limit_bytes, env=None, config_mb=None):
|
||||
with mock.patch.object(m, "_read_cgroup_memory_limit", return_value=limit_bytes), \
|
||||
mock.patch.object(m, "_config_tui_heap_mb_early", return_value=config_mb), \
|
||||
mock.patch.dict(m.os.environ, env or {}, clear=False):
|
||||
if env is None:
|
||||
m.os.environ.pop("HERMES_TUI_HEAP_MB", None)
|
||||
return m._resolve_tui_heap_mb()
|
||||
|
||||
def test_env_override_unconstrained(self):
|
||||
# explicit low cap, no cgroup limit -> used as-is (the low-mem opt-in).
|
||||
assert self._resolve(None, env={"HERMES_TUI_HEAP_MB": "256"}) == 256
|
||||
|
||||
def test_env_override_raises_ceiling(self):
|
||||
# a higher-than-default cap is honored when unconstrained.
|
||||
assert self._resolve(None, env={"HERMES_TUI_HEAP_MB": "16384"}) == 16384
|
||||
|
||||
def test_env_wins_over_config(self):
|
||||
assert self._resolve(None, env={"HERMES_TUI_HEAP_MB": "512"}, config_mb=4096) == 512
|
||||
|
||||
def test_config_used_when_no_env(self):
|
||||
assert self._resolve(None, config_mb=2048) == 2048
|
||||
|
||||
def test_override_still_cgroup_clamped(self):
|
||||
# user asks for 16GB but the container is 4GB -> trimmed to 75% = 3072.
|
||||
assert self._resolve(4 * GB, env={"HERMES_TUI_HEAP_MB": "16384"}) == 3072
|
||||
|
||||
def test_low_override_honored_under_big_container(self):
|
||||
# a deliberately low cap is NOT raised by a roomy container.
|
||||
assert self._resolve(16 * GB, env={"HERMES_TUI_HEAP_MB": "256"}) == 256
|
||||
|
||||
def test_garbage_env_falls_through_to_default(self):
|
||||
assert self._resolve(None, env={"HERMES_TUI_HEAP_MB": "nope"}) == 8192
|
||||
|
||||
def test_nonpositive_env_falls_through(self):
|
||||
assert self._resolve(None, env={"HERMES_TUI_HEAP_MB": "0"}) == 8192
|
||||
|
||||
|
||||
class TestExposeGcOnOpenTuiArgv:
|
||||
"""W1/D4: the OpenTUI engine argv must carry --expose-gc (parity with Ink) so
|
||||
global.gc() is a real call, not a no-op."""
|
||||
|
||||
def test_opentui_argv_has_expose_gc(self, tmp_path):
|
||||
app_dir = tmp_path / "ui-opentui"
|
||||
(app_dir / "src" / "entry").mkdir(parents=True)
|
||||
(app_dir / "src" / "entry" / "main.tsx").write_text("// entry")
|
||||
(app_dir / "node_modules" / "@opentui").mkdir(parents=True)
|
||||
(app_dir / "dist").mkdir()
|
||||
(app_dir / "dist" / "main.js").write_text("// built")
|
||||
with mock.patch.object(m, "PROJECT_ROOT", tmp_path), \
|
||||
mock.patch.object(m, "_node26_bin", return_value="/usr/bin/node"):
|
||||
argv, cwd = m._make_opentui_argv(tui_dev=False)
|
||||
assert "--expose-gc" in argv
|
||||
assert argv[0] == "/usr/bin/node"
|
||||
assert argv[-1].endswith("dist/main.js")
|
||||
assert cwd == app_dir
|
||||
|
||||
|
||||
class TestNodeOptionsTokenMerge:
|
||||
"""The _launch_tui token-merge block must add the sized cap unless the user
|
||||
already supplied one, and must preserve unrelated NODE_OPTIONS flags."""
|
||||
|
||||
@@ -14,6 +14,20 @@ def main_mod():
|
||||
return m
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _pin_ink_engine(monkeypatch):
|
||||
"""These tests exercise the Ink/npm bootstrap inside ``_make_tui_argv``.
|
||||
|
||||
The dual-engine dispatch (``_resolve_tui_engine``) auto-selects the native
|
||||
OpenTUI engine whenever ``ui-opentui/dist`` is built in the repo, which
|
||||
would route ``_make_tui_argv`` away from the npm path under test. Pin the
|
||||
engine to ink, mirroring test_tui_resume_flow.py.
|
||||
"""
|
||||
import hermes_cli.main as m
|
||||
|
||||
monkeypatch.setattr(m, "_resolve_tui_engine", lambda: "ink")
|
||||
|
||||
|
||||
def _touch_ink(root: Path) -> None:
|
||||
ink = root / "node_modules" / "@hermes" / "ink" / "package.json"
|
||||
ink.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
@@ -118,6 +118,82 @@ def test_cmd_chat_tui_resume_resolves_title_before_launch(monkeypatch, main_mod)
|
||||
assert captured["resume"] == "20260409_000000_aa11bb"
|
||||
|
||||
|
||||
def test_bare_resume_parses_to_picker_sentinel():
|
||||
from hermes_cli._parser import build_top_level_parser
|
||||
|
||||
parser, _subparsers, _chat_parser = build_top_level_parser()
|
||||
|
||||
args = parser.parse_args(["--tui", "--resume"])
|
||||
assert args.resume is True
|
||||
|
||||
args = parser.parse_args(["--resume", "abc123"])
|
||||
assert args.resume == "abc123"
|
||||
|
||||
args = parser.parse_args(["chat", "--tui", "--resume"])
|
||||
assert args.resume is True
|
||||
|
||||
|
||||
def test_cmd_chat_tui_bare_resume_skips_resolution_and_launches_picker(
|
||||
monkeypatch, main_mod
|
||||
):
|
||||
captured = {}
|
||||
|
||||
def fake_launch(resume_session_id=None, **kwargs):
|
||||
captured["resume"] = resume_session_id
|
||||
raise SystemExit(0)
|
||||
|
||||
def boom(_val):
|
||||
raise AssertionError("bare --resume must not hit name/id resolution")
|
||||
|
||||
monkeypatch.setattr(main_mod, "_resolve_session_by_name_or_id", boom)
|
||||
monkeypatch.setattr(main_mod, "_launch_tui", fake_launch)
|
||||
|
||||
with pytest.raises(SystemExit):
|
||||
main_mod.cmd_chat(_args(resume=True))
|
||||
|
||||
assert captured["resume"] is True
|
||||
|
||||
|
||||
def test_cmd_chat_bare_resume_without_tui_exits_with_guidance(
|
||||
monkeypatch, capsys, main_mod
|
||||
):
|
||||
monkeypatch.setattr(main_mod, "_resolve_use_tui", lambda args: False)
|
||||
monkeypatch.setattr(
|
||||
main_mod,
|
||||
"_launch_tui",
|
||||
lambda *a, **kw: pytest.fail("must not launch the TUI"),
|
||||
)
|
||||
|
||||
with pytest.raises(SystemExit) as exc:
|
||||
main_mod.cmd_chat(_args(tui=False, resume=True))
|
||||
|
||||
assert exc.value.code == 2
|
||||
out = capsys.readouterr().out
|
||||
assert "requires the TUI" in out
|
||||
assert "hermes --tui --resume" in out
|
||||
|
||||
|
||||
def test_launch_tui_sets_picker_env_for_bare_resume(monkeypatch, main_mod):
|
||||
captured = {}
|
||||
|
||||
monkeypatch.setenv("HERMES_TUI_RESUME", "stale-missing-session")
|
||||
monkeypatch.setattr(
|
||||
main_mod,
|
||||
"_make_tui_argv",
|
||||
lambda tui_dir, tui_dev: (["node", "dist/entry.js"], Path(".")),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
main_mod.subprocess,
|
||||
"call",
|
||||
lambda argv, cwd=None, env=None: captured.update({"env": env}) or 1,
|
||||
)
|
||||
|
||||
with pytest.raises(SystemExit):
|
||||
main_mod._launch_tui(resume_session_id=True)
|
||||
|
||||
assert captured["env"]["HERMES_TUI_RESUME"] == "picker"
|
||||
|
||||
|
||||
def test_cmd_chat_tui_passes_model_and_provider(monkeypatch, main_mod):
|
||||
captured = {}
|
||||
|
||||
@@ -1008,6 +1084,10 @@ def test_make_tui_argv_dev_prebuilds_hermes_ink(monkeypatch, main_mod, tmp_path)
|
||||
monkeypatch.setattr(main_mod, "_tui_need_npm_install", lambda _tui_dir: False)
|
||||
monkeypatch.delenv("HERMES_TUI_DIR", raising=False)
|
||||
monkeypatch.setattr(main_mod.shutil, "which", lambda bin_name: f"/usr/bin/{bin_name}")
|
||||
# _make_tui_argv now dispatches on the TUI engine first; resolving "opentui"
|
||||
# availability probes `node --version` (a subprocess.run this test would
|
||||
# otherwise record). Pin the Ink engine — this test covers the Ink dev path.
|
||||
monkeypatch.setattr(main_mod, "_resolve_tui_engine", lambda: "ink")
|
||||
|
||||
calls = []
|
||||
|
||||
|
||||
@@ -34,6 +34,13 @@ client = TestClient(app)
|
||||
HEADERS = {"X-Hermes-Session-Token": _SESSION_TOKEN}
|
||||
|
||||
|
||||
def _make_profile_home(tmp_path, monkeypatch, profile="coder"):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
profile_home = tmp_path / "profiles" / profile
|
||||
profile_home.mkdir(parents=True)
|
||||
return profile_home
|
||||
|
||||
|
||||
def _fake_nous_device_data():
|
||||
return {
|
||||
"device_code": "device-code",
|
||||
@@ -127,6 +134,67 @@ def test_nous_dashboard_device_flow_ignores_legacy_scope_override(monkeypatch):
|
||||
ws._oauth_sessions.pop(result["session_id"], None)
|
||||
|
||||
|
||||
def test_oauth_provider_status_uses_profile_query(tmp_path, monkeypatch):
|
||||
from hermes_cli import web_server as ws
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
profile_home = _make_profile_home(tmp_path, monkeypatch)
|
||||
observed_homes = []
|
||||
|
||||
def fake_status():
|
||||
observed_homes.append(get_hermes_home())
|
||||
return {"logged_in": False, "source": None}
|
||||
|
||||
fake_catalog = ({
|
||||
"id": "fake-oauth",
|
||||
"name": "Fake OAuth",
|
||||
"flow": "pkce",
|
||||
"cli_command": "hermes auth add fake-oauth",
|
||||
"docs_url": "https://example.com",
|
||||
"status_fn": fake_status,
|
||||
},)
|
||||
monkeypatch.setattr(ws, "_OAUTH_PROVIDER_CATALOG", fake_catalog)
|
||||
|
||||
resp = client.get("/api/providers/oauth?profile=coder", headers=HEADERS)
|
||||
|
||||
assert resp.status_code == 200, resp.text
|
||||
assert observed_homes == [profile_home]
|
||||
|
||||
|
||||
def test_oauth_start_stores_profile_for_background_completion(tmp_path, monkeypatch):
|
||||
from hermes_cli import web_server as ws
|
||||
|
||||
_make_profile_home(tmp_path, monkeypatch)
|
||||
fake_user_code_resp = {
|
||||
"user_code": "ABCD-1234",
|
||||
"verification_uri": "https://api.minimax.io/oauth/verify",
|
||||
"expired_in": 600,
|
||||
"interval": 2000,
|
||||
"state": "stub-state",
|
||||
}
|
||||
with patch(
|
||||
"hermes_cli.auth._minimax_request_user_code",
|
||||
return_value=fake_user_code_resp,
|
||||
), patch(
|
||||
"hermes_cli.auth._minimax_pkce_pair",
|
||||
return_value=("verifier-stub", "challenge-stub", "stub-state"),
|
||||
), patch(
|
||||
"hermes_cli.web_server._minimax_poller",
|
||||
return_value=None,
|
||||
):
|
||||
resp = client.post(
|
||||
"/api/providers/oauth/minimax-oauth/start?profile=coder",
|
||||
headers=HEADERS,
|
||||
)
|
||||
|
||||
assert resp.status_code == 200, resp.text
|
||||
session_id = resp.json()["session_id"]
|
||||
try:
|
||||
assert ws._oauth_sessions[session_id]["profile"] == "coder"
|
||||
finally:
|
||||
ws._oauth_sessions.pop(session_id, None)
|
||||
|
||||
|
||||
def test_nous_dashboard_device_flow_does_not_retry_legacy_scope_on_invoke_refusal(monkeypatch):
|
||||
from hermes_cli import auth as auth_mod
|
||||
from hermes_cli import web_server as ws
|
||||
@@ -207,6 +275,71 @@ def test_codex_dashboard_worker_persists_runtime_provider(tmp_path, monkeypatch)
|
||||
ws._oauth_sessions.pop(sid, None)
|
||||
|
||||
|
||||
def test_codex_dashboard_worker_persists_inside_session_profile(tmp_path, monkeypatch):
|
||||
from hermes_cli import auth as auth_mod
|
||||
from hermes_cli import web_server as ws
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
profile_home = _make_profile_home(tmp_path, monkeypatch)
|
||||
|
||||
class _Resp:
|
||||
def __init__(self, status_code, payload):
|
||||
self.status_code = status_code
|
||||
self._payload = payload
|
||||
|
||||
def json(self):
|
||||
return self._payload
|
||||
|
||||
class _Client:
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *args):
|
||||
return False
|
||||
|
||||
def post(self, url, **kwargs):
|
||||
if url.endswith("/deviceauth/usercode"):
|
||||
return _Resp(200, {
|
||||
"device_auth_id": "device-auth-id",
|
||||
"interval": 3,
|
||||
"user_code": "CODEX-1234",
|
||||
})
|
||||
if url.endswith("/deviceauth/token"):
|
||||
return _Resp(200, {
|
||||
"authorization_code": "authorization-code",
|
||||
"code_verifier": "code-verifier",
|
||||
})
|
||||
return _Resp(200, {
|
||||
"access_token": "codex-access",
|
||||
"refresh_token": "codex-refresh",
|
||||
})
|
||||
|
||||
saved_homes = []
|
||||
monkeypatch.setattr(httpx, "Client", _Client)
|
||||
monkeypatch.setattr(ws.time, "sleep", lambda _: None)
|
||||
monkeypatch.setattr(
|
||||
auth_mod,
|
||||
"_save_codex_tokens",
|
||||
lambda tokens: saved_homes.append(get_hermes_home()),
|
||||
)
|
||||
|
||||
sid, _ = ws._new_oauth_session(
|
||||
"openai-codex",
|
||||
"device_code",
|
||||
profile="coder",
|
||||
)
|
||||
try:
|
||||
ws._codex_full_login_worker(sid)
|
||||
|
||||
assert ws._oauth_sessions[sid]["status"] == "approved"
|
||||
assert saved_homes == [profile_home]
|
||||
finally:
|
||||
ws._oauth_sessions.pop(sid, None)
|
||||
|
||||
|
||||
def test_nous_dashboard_poller_preserves_effective_scope_when_token_omits_scope(monkeypatch):
|
||||
from hermes_cli import auth as auth_mod
|
||||
from hermes_cli import web_server as ws
|
||||
|
||||
@@ -245,6 +245,24 @@ class TestWebServerEndpoints:
|
||||
assert "version" in data
|
||||
assert "hermes_home" in data
|
||||
assert "active_sessions" in data
|
||||
assert data["can_update_hermes"] is True
|
||||
|
||||
def test_get_status_hides_update_capability_in_managed_runtime(self, monkeypatch):
|
||||
import hermes_cli.web_server as web_server
|
||||
|
||||
monkeypatch.setattr(web_server, "_dashboard_local_update_managed_externally", lambda: True)
|
||||
|
||||
resp = self.client.get("/api/status")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["can_update_hermes"] is False
|
||||
|
||||
def test_dashboard_update_capability_detects_generic_container(self, monkeypatch):
|
||||
import hermes_constants
|
||||
import hermes_cli.web_server as web_server
|
||||
|
||||
monkeypatch.setattr(hermes_constants, "is_container", lambda: True)
|
||||
|
||||
assert web_server._dashboard_local_update_managed_externally() is True
|
||||
|
||||
# ── GET /api/media (remote image display) ───────────────────────────
|
||||
|
||||
@@ -912,6 +930,48 @@ class TestWebServerEndpoints:
|
||||
assert status_data["pid"] is None
|
||||
assert any("docker pull nousresearch/hermes-agent:latest" in line for line in status_data["lines"])
|
||||
|
||||
def test_update_hermes_returns_managed_runtime_guidance_without_spawning(self, monkeypatch):
|
||||
import hermes_cli.web_server as web_server
|
||||
|
||||
spawned = False
|
||||
detected = False
|
||||
|
||||
def fail_spawn(*_args, **_kwargs):
|
||||
nonlocal spawned
|
||||
spawned = True
|
||||
raise AssertionError("managed runtime update guard should not spawn hermes update")
|
||||
|
||||
def fail_detect(*_args, **_kwargs):
|
||||
nonlocal detected
|
||||
detected = True
|
||||
raise AssertionError("managed runtime update guard should not detect install method")
|
||||
|
||||
monkeypatch.setattr(web_server, "_dashboard_local_update_managed_externally", lambda: True)
|
||||
monkeypatch.setattr(web_server, "detect_install_method", fail_detect)
|
||||
monkeypatch.setattr(web_server, "_spawn_hermes_action", fail_spawn)
|
||||
web_server._ACTION_PROCS.pop("hermes-update", None)
|
||||
web_server._ACTION_RESULTS.pop("hermes-update", None)
|
||||
|
||||
resp = self.client.post("/api/hermes/update")
|
||||
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert data["ok"] is False
|
||||
assert data["name"] == "hermes-update"
|
||||
assert data["pid"] is None
|
||||
assert data["error"] == "dashboard_update_managed_externally"
|
||||
assert "managed outside this dashboard" in data["message"]
|
||||
assert spawned is False
|
||||
assert detected is False
|
||||
|
||||
status = self.client.get("/api/actions/hermes-update/status")
|
||||
assert status.status_code == 200
|
||||
status_data = status.json()
|
||||
assert status_data["running"] is False
|
||||
assert status_data["exit_code"] == 1
|
||||
assert status_data["pid"] is None
|
||||
assert any("managed outside this dashboard" in line for line in status_data["lines"])
|
||||
|
||||
def test_update_hermes_spawns_on_non_docker_install(self, monkeypatch):
|
||||
import hermes_cli.web_server as web_server
|
||||
|
||||
|
||||
@@ -254,6 +254,66 @@ def test_local_mode_upload_read_mkdir_delete_roundtrip(local_files_client):
|
||||
assert not folder.exists()
|
||||
|
||||
|
||||
def _seed_file(client, root, name="out/hello.txt"):
|
||||
file_path = root / name
|
||||
created = client.post(
|
||||
"/api/files/upload",
|
||||
json={"path": str(file_path), "data_url": "data:text/plain;base64,aGVsbG8="},
|
||||
)
|
||||
assert created.status_code == 200
|
||||
return file_path
|
||||
|
||||
|
||||
def test_download_returns_file_as_attachment(forced_files_client):
|
||||
client, root = forced_files_client
|
||||
file_path = _seed_file(client, root)
|
||||
|
||||
resp = client.get("/api/files/download", params={"path": str(file_path)})
|
||||
assert resp.status_code == 200
|
||||
assert resp.content == b"hello"
|
||||
disposition = resp.headers["content-disposition"]
|
||||
assert "attachment" in disposition
|
||||
assert "hello.txt" in disposition
|
||||
|
||||
|
||||
def test_download_authenticates_via_query_token(forced_files_client):
|
||||
client, root = forced_files_client
|
||||
file_path = _seed_file(client, root)
|
||||
|
||||
# Drop the session header so only the ?token= query param authenticates —
|
||||
# mirrors a browser/shell-opened download that can't set the session header.
|
||||
del client.headers[web_server._SESSION_HEADER_NAME]
|
||||
|
||||
ok = client.get(
|
||||
"/api/files/download",
|
||||
params={"path": str(file_path), "token": web_server._SESSION_TOKEN},
|
||||
)
|
||||
assert ok.status_code == 200
|
||||
assert ok.content == b"hello"
|
||||
|
||||
assert client.get(
|
||||
"/api/files/download", params={"path": str(file_path), "token": "nope"}
|
||||
).status_code == 401
|
||||
assert client.get(
|
||||
"/api/files/download", params={"path": str(file_path)}
|
||||
).status_code == 401
|
||||
|
||||
|
||||
def test_query_token_does_not_authenticate_other_endpoints(forced_files_client):
|
||||
client, root = forced_files_client
|
||||
file_path = _seed_file(client, root)
|
||||
|
||||
del client.headers[web_server._SESSION_HEADER_NAME]
|
||||
|
||||
# The query-token escape hatch is scoped to /api/files/download only; it must
|
||||
# not unlock the rest of the API surface.
|
||||
leaked = client.get(
|
||||
"/api/files/read",
|
||||
params={"path": str(file_path), "token": web_server._SESSION_TOKEN},
|
||||
)
|
||||
assert leaked.status_code == 401
|
||||
|
||||
|
||||
def test_hosted_policy_locks_to_opt_data(monkeypatch):
|
||||
monkeypatch.delenv("HERMES_DASHBOARD_FILES_ROOT", raising=False)
|
||||
monkeypatch.setenv("HERMES_HOME", "/opt/data")
|
||||
|
||||
+153
-78
@@ -239,7 +239,7 @@ class TestCloneHonchoForProfile:
|
||||
"""Identity-key carryover during profile cloning.
|
||||
|
||||
The host-scoped identity-mapping keys (``userPeerAliases``,
|
||||
``runtimePeerPrefix``, ``pinPeerName``) must survive a clone; otherwise
|
||||
``runtimePeerPrefix``, ``pinUserPeer``) must survive a clone; otherwise
|
||||
the new profile silently fragments memory by resolving gateway users to
|
||||
raw runtime IDs instead of operator-declared peers.
|
||||
"""
|
||||
@@ -263,7 +263,7 @@ class TestCloneHonchoForProfile:
|
||||
"apiKey": "***",
|
||||
"hosts": {
|
||||
"hermes": {
|
||||
"userPeerAliases": {"86701400": "eri", "discord-491827364": "eri"},
|
||||
"userPeerAliases": {"7654321": "eri", "discord-491827364": "eri"},
|
||||
"peerName": "eri",
|
||||
},
|
||||
},
|
||||
@@ -272,7 +272,7 @@ class TestCloneHonchoForProfile:
|
||||
ok = honcho_cli.clone_honcho_for_profile("coder")
|
||||
assert ok is True
|
||||
new_block = written["cfg"]["hosts"]["hermes_coder"]
|
||||
assert new_block["userPeerAliases"] == {"86701400": "eri", "discord-491827364": "eri"}
|
||||
assert new_block["userPeerAliases"] == {"7654321": "eri", "discord-491827364": "eri"}
|
||||
|
||||
def test_runtime_peer_prefix_carries_into_cloned_profile(self, monkeypatch, tmp_path):
|
||||
cfg = {
|
||||
@@ -290,7 +290,7 @@ class TestCloneHonchoForProfile:
|
||||
new_block = written["cfg"]["hosts"]["hermes_coder"]
|
||||
assert new_block["runtimePeerPrefix"] == "telegram_"
|
||||
|
||||
def test_pin_peer_name_carries_into_cloned_profile(self, monkeypatch, tmp_path):
|
||||
def test_legacy_pin_peer_name_migrates_to_canonical_on_clone(self, monkeypatch, tmp_path):
|
||||
cfg = {
|
||||
"apiKey": "***",
|
||||
"hosts": {
|
||||
@@ -304,7 +304,8 @@ class TestCloneHonchoForProfile:
|
||||
ok = honcho_cli.clone_honcho_for_profile("coder")
|
||||
assert ok is True
|
||||
new_block = written["cfg"]["hosts"]["hermes_coder"]
|
||||
assert new_block["pinPeerName"] is True
|
||||
assert new_block["pinUserPeer"] is True
|
||||
assert "pinPeerName" not in new_block
|
||||
|
||||
def test_unset_identity_keys_do_not_appear_in_cloned_profile(self, monkeypatch, tmp_path):
|
||||
cfg = {
|
||||
@@ -317,23 +318,25 @@ class TestCloneHonchoForProfile:
|
||||
new_block = written["cfg"]["hosts"]["hermes_coder"]
|
||||
assert "userPeerAliases" not in new_block
|
||||
assert "runtimePeerPrefix" not in new_block
|
||||
assert "pinUserPeer" not in new_block
|
||||
assert "pinPeerName" not in new_block
|
||||
|
||||
|
||||
class TestSetupWizardDeploymentShape:
|
||||
"""The deployment-shape step writes pinPeerName / userPeerAliases /
|
||||
runtimePeerPrefix based on the operator's chosen shape.
|
||||
"""The gateway identity-mapping tree writes pinUserPeer / userPeerAliases /
|
||||
runtimePeerPrefix based on the operator's intent.
|
||||
|
||||
Single-operator deployments collapse all platforms to peerName.
|
||||
Multi-user gateways leave the resolver to route per-runtime.
|
||||
Hybrid deployments alias the operator's own runtime IDs only.
|
||||
Choice [1] (just me) collapses all platforms to peerName.
|
||||
Choice [3] (only other people) leaves the resolver to route per-runtime.
|
||||
Choice [2] (me + others, pooled) aliases the operator's own runtime IDs.
|
||||
|
||||
These tests script the interactive _prompt calls and assert the
|
||||
resulting hermes_host block, so the wizard's deployment-shape
|
||||
These tests mock gateway detection and script the interactive _prompt
|
||||
calls, asserting the resulting hermes_host block so the tree's routing
|
||||
semantics stay locked even as adjacent prompts are added.
|
||||
"""
|
||||
|
||||
def _run_setup(self, monkeypatch, tmp_path, *, answers, initial_cfg=None):
|
||||
def _run_setup(self, monkeypatch, tmp_path, *, answers, initial_cfg=None,
|
||||
gateway_platforms=("telegram",)):
|
||||
import plugins.memory.honcho.cli as honcho_cli
|
||||
|
||||
cfg_path = tmp_path / "config.json"
|
||||
@@ -346,6 +349,10 @@ class TestSetupWizardDeploymentShape:
|
||||
monkeypatch.setattr(honcho_cli, "_host_key", lambda: "hermes")
|
||||
monkeypatch.setattr(honcho_cli, "_ensure_sdk_installed", lambda: True)
|
||||
monkeypatch.setattr(honcho_cli, "_write_config", lambda *a, **k: None)
|
||||
# Gate detection is mocked so tests control whether the tree runs.
|
||||
# None → undetectable; list (possibly empty) → connected platforms.
|
||||
gw = None if gateway_platforms is None else list(gateway_platforms)
|
||||
monkeypatch.setattr(honcho_cli, "_gateway_platforms", lambda: gw)
|
||||
|
||||
# Bypass config.yaml + connection test side effects.
|
||||
monkeypatch.setattr(
|
||||
@@ -391,14 +398,14 @@ class TestSetupWizardDeploymentShape:
|
||||
honcho_cli.cmd_setup(SimpleNamespace())
|
||||
return cfg["hosts"]["hermes"]
|
||||
|
||||
def test_single_shape_sets_pin_peer_name_and_clears_aliases(self, monkeypatch, tmp_path):
|
||||
def test_just_me_pins_and_clears_aliases(self, monkeypatch, tmp_path):
|
||||
answers = [
|
||||
"cloud", # deployment
|
||||
"", # api key (keep)
|
||||
"eri", # peer name
|
||||
"hermetika", # ai peer
|
||||
"hermes", # workspace
|
||||
"single", # deployment shape ← key answer
|
||||
"1", # tree: just me ← key answer
|
||||
# remaining prompts fall through to defaults
|
||||
]
|
||||
initial_cfg = {
|
||||
@@ -409,51 +416,54 @@ class TestSetupWizardDeploymentShape:
|
||||
}},
|
||||
}
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is True
|
||||
assert host["pinUserPeer"] is True
|
||||
assert "userPeerAliases" not in host
|
||||
assert "runtimePeerPrefix" not in host
|
||||
|
||||
def test_multi_shape_leaves_pin_false_and_accepts_prefix(self, monkeypatch, tmp_path):
|
||||
def test_only_others_leaves_pin_false_and_accepts_prefix(self, monkeypatch, tmp_path):
|
||||
answers = [
|
||||
"cloud", # deployment
|
||||
"", # api key (keep)
|
||||
"eri", # peer name
|
||||
"hermetika", # ai peer
|
||||
"hermes", # workspace
|
||||
"multi", # deployment shape
|
||||
"3", # tree: only other people
|
||||
"telegram_", # runtime peer prefix
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers)
|
||||
assert host["pinPeerName"] is False
|
||||
assert host["pinUserPeer"] is False
|
||||
# Multi must NOT auto-write ``userPeerAliases: {}``: an empty host
|
||||
# map would silently override a root-level baseline. Absence is
|
||||
# the correct "no host opinion" signal.
|
||||
assert "userPeerAliases" not in host
|
||||
assert host["runtimePeerPrefix"] == "telegram_"
|
||||
|
||||
def test_hybrid_shape_aliases_operator_runtime_ids_to_peer_name(self, monkeypatch, tmp_path):
|
||||
def test_pooled_aliases_operator_runtime_ids_to_peer_name(self, monkeypatch, tmp_path):
|
||||
answers = [
|
||||
"cloud", # deployment
|
||||
"", # api key (keep)
|
||||
"eri", # peer name
|
||||
"hermetika", # ai peer
|
||||
"hermes", # workspace
|
||||
"hybrid", # deployment shape
|
||||
"86701400", # telegram uid
|
||||
"2", # tree: me + other people
|
||||
"y", # keep my memory pooled? → hybrid
|
||||
"7654321", # telegram uid
|
||||
"491827364", # discord snowflake
|
||||
"", # slack (skip)
|
||||
"", # matrix (skip)
|
||||
"", # runtime peer prefix (skip)
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers)
|
||||
assert host["pinPeerName"] is False
|
||||
assert host["pinUserPeer"] is False
|
||||
assert host["userPeerAliases"] == {
|
||||
"86701400": "eri",
|
||||
"7654321": "eri",
|
||||
"491827364": "eri",
|
||||
}
|
||||
assert "runtimePeerPrefix" not in host
|
||||
|
||||
def test_skip_shape_preserves_existing_identity_config(self, monkeypatch, tmp_path):
|
||||
# Seeds the legacy ``pinPeerName``: skip must leave the mapping intact
|
||||
# except for the on-load migration onto the canonical key.
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
"hosts": {"hermes": {
|
||||
@@ -463,17 +473,18 @@ class TestSetupWizardDeploymentShape:
|
||||
}},
|
||||
}
|
||||
answers = [
|
||||
"cloud", "", "eri", "hermetika", "hermes", "skip",
|
||||
"cloud", "", "eri", "hermetika", "hermes", "s",
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is True
|
||||
assert host["pinUserPeer"] is True
|
||||
assert "pinPeerName" not in host
|
||||
assert host["userPeerAliases"] == {"keep": "me"}
|
||||
assert host["runtimePeerPrefix"] == "keep_"
|
||||
|
||||
def test_single_to_multi_steers_to_hybrid_by_default(self, monkeypatch, tmp_path):
|
||||
"""Flipping single → multi triggers a warning that auto-steers the
|
||||
operator to ``hybrid`` (default), so their own runtime IDs keep
|
||||
landing on peerName instead of orphaning the pinned-pool history.
|
||||
def test_unpin_steers_to_pooled_by_default(self, monkeypatch, tmp_path):
|
||||
"""Choosing 'only other people' on a currently-pinned profile triggers
|
||||
the orphan warning, which auto-steers to pooled (hybrid) so the
|
||||
operator's own runtime IDs keep landing on peerName.
|
||||
"""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
@@ -485,60 +496,57 @@ class TestSetupWizardDeploymentShape:
|
||||
"eri", # peer name
|
||||
"hermetika", # ai peer
|
||||
"hermes", # workspace
|
||||
"multi", # deployment shape — triggers the guard
|
||||
"hybrid", # guard response: accept the steer
|
||||
"86701400", # telegram uid
|
||||
"3", # tree: only others — triggers the orphan guard
|
||||
"y", # pool my own memory instead? → hybrid
|
||||
"7654321", # telegram uid
|
||||
"", # discord (skip)
|
||||
"", # slack (skip)
|
||||
"", # matrix (skip)
|
||||
"", # runtime prefix (skip)
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is False
|
||||
assert host["userPeerAliases"] == {"86701400": "eri"}
|
||||
assert host["pinUserPeer"] is False
|
||||
assert host["userPeerAliases"] == {"7654321": "eri"}
|
||||
|
||||
def test_single_to_multi_yes_override_keeps_multi(self, monkeypatch, tmp_path):
|
||||
"""Operator can override the steer by answering ``yes`` and accept
|
||||
the orphaning consequences. This is the explicit undo-the-pin path.
|
||||
"""
|
||||
def test_unpin_decline_steer_keeps_per_user(self, monkeypatch, tmp_path):
|
||||
"""Operator can decline the steer ('n') and accept orphaning, ending
|
||||
up with per-user peers (no aliases)."""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
"hosts": {"hermes": {"pinPeerName": True, "peerName": "eri"}},
|
||||
}
|
||||
answers = [
|
||||
"cloud", "", "eri", "hermetika", "hermes",
|
||||
"multi", # deployment shape — triggers the guard
|
||||
"yes", # guard response: confirm multi
|
||||
"3", # tree: only others — triggers the orphan guard
|
||||
"n", # decline pooling, accept orphaning
|
||||
"telegram_", # runtime peer prefix
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is False
|
||||
# See test_multi_shape_leaves_pin_false_and_accepts_prefix.
|
||||
assert host["pinUserPeer"] is False
|
||||
assert "userPeerAliases" not in host
|
||||
assert host["runtimePeerPrefix"] == "telegram_"
|
||||
|
||||
def test_host_pin_user_peer_true_is_detected_as_single(self, monkeypatch, tmp_path):
|
||||
"""Host-level ``pinUserPeer: true`` must classify as ``single``.
|
||||
|
||||
Pressing Enter at the shape prompt then preserves the pin instead
|
||||
of falling through to ``multi`` and orphaning the user's memory
|
||||
pool — the bug the wizard regressed when ``pinUserPeer`` landed
|
||||
as a higher-precedence alias.
|
||||
Pressing Enter at the choice prompt then preserves the pin instead
|
||||
of falling through to per-user routing and orphaning the user's
|
||||
memory pool — the bug the wizard regressed when ``pinUserPeer``
|
||||
landed as a higher-precedence alias.
|
||||
"""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
"hosts": {"hermes": {"pinUserPeer": True, "peerName": "eri"}},
|
||||
}
|
||||
# Exhaust the iterator before the shape prompt so the scripted
|
||||
# mock falls through to the prompt's default (which is the
|
||||
# wizard-detected shape). Scripting an explicit "" would NOT
|
||||
# exercise that fallthrough — the mock returns it literally.
|
||||
# Exhaust the iterator before the choice prompt so the scripted
|
||||
# mock falls through to the prompt's default (the detected shape →
|
||||
# choice "1"). Scripting an explicit "" would NOT exercise that
|
||||
# fallthrough — the mock returns it literally.
|
||||
answers = ["cloud", "", "eri", "hermetika", "hermes"]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
# Scrub-then-write normalises onto pinPeerName and drops the alias
|
||||
# so resolver precedence can't reintroduce ambiguity.
|
||||
assert host["pinPeerName"] is True
|
||||
assert "pinUserPeer" not in host
|
||||
# Scrub-then-write normalises onto the canonical pinUserPeer.
|
||||
assert host["pinUserPeer"] is True
|
||||
assert "pinPeerName" not in host
|
||||
|
||||
def test_host_pin_user_peer_false_overrides_root_pin_peer_name(
|
||||
self, monkeypatch, tmp_path
|
||||
@@ -558,8 +566,8 @@ class TestSetupWizardDeploymentShape:
|
||||
}
|
||||
answers = ["cloud", "", "eri", "hermetika", "hermes"]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is False
|
||||
assert "pinUserPeer" not in host
|
||||
assert host["pinUserPeer"] is False
|
||||
assert "pinPeerName" not in host
|
||||
|
||||
def test_root_user_peer_aliases_detected_as_hybrid(self, monkeypatch, tmp_path):
|
||||
"""Root-level ``userPeerAliases`` must classify as ``hybrid`` even
|
||||
@@ -567,26 +575,26 @@ class TestSetupWizardDeploymentShape:
|
||||
"""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
"userPeerAliases": {"86701400": "eri"},
|
||||
"userPeerAliases": {"7654321": "eri"},
|
||||
"hosts": {"hermes": {"peerName": "eri"}},
|
||||
}
|
||||
answers = ["cloud", "", "eri", "hermetika", "hermes"]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is False
|
||||
assert host["pinUserPeer"] is False
|
||||
# Hybrid materialises the root aliases into the host so subsequent
|
||||
# operator edits live on the host block they're inspecting.
|
||||
assert host["userPeerAliases"] == {"86701400": "eri"}
|
||||
assert host["userPeerAliases"] == {"7654321": "eri"}
|
||||
|
||||
def test_multi_does_not_override_root_user_peer_aliases(self, monkeypatch, tmp_path):
|
||||
"""Explicit ``multi`` must leave the host ``userPeerAliases`` key
|
||||
absent, preserving any root-level aliases as a cross-host baseline.
|
||||
def test_only_others_does_not_override_root_user_peer_aliases(self, monkeypatch, tmp_path):
|
||||
"""Explicitly choosing 'only other people' must leave the host
|
||||
``userPeerAliases`` key absent, preserving any root-level aliases as a
|
||||
cross-host baseline.
|
||||
|
||||
Picking ``multi`` here is an active choice — detection would have
|
||||
defaulted to ``hybrid`` because root aliases exist — so the
|
||||
operator's intent is to drop the alias mapping for this host.
|
||||
We honor that by writing ``pinPeerName: false`` only, and rely
|
||||
on the host's absence of ``userPeerAliases`` to inherit root.
|
||||
That inheritance is intentional: a true wipe would require the
|
||||
Picking [3] here is an active choice — detection would have defaulted
|
||||
to [2]/hybrid because root aliases exist — so the operator's intent is
|
||||
to drop the alias mapping for this host. We honor that by writing
|
||||
``pinUserPeer: false`` only, relying on the host's absence of
|
||||
``userPeerAliases`` to inherit root. A true wipe would require the
|
||||
operator to delete the root key explicitly.
|
||||
"""
|
||||
initial_cfg = {
|
||||
@@ -596,17 +604,15 @@ class TestSetupWizardDeploymentShape:
|
||||
}
|
||||
answers = [
|
||||
"cloud", "", "eri", "hermetika", "hermes",
|
||||
"multi", # explicit multi override of detected hybrid
|
||||
"3", # explicit per-user override of detected hybrid
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is False
|
||||
assert host["pinUserPeer"] is False
|
||||
assert "userPeerAliases" not in host
|
||||
|
||||
def test_single_scrubs_stale_pin_user_peer_false(self, monkeypatch, tmp_path):
|
||||
"""Choosing ``single`` must drop any host-level ``pinUserPeer``,
|
||||
otherwise an existing ``pinUserPeer: false`` would outrank the
|
||||
freshly written ``pinPeerName: true`` and leave the profile
|
||||
effectively unpinned (the P1 latent-precedence regression).
|
||||
def test_just_me_scrubs_stale_pin_user_peer_false(self, monkeypatch, tmp_path):
|
||||
"""Choosing 'just me' must overwrite a stale ``pinUserPeer: false``
|
||||
with ``pinUserPeer: true`` so the profile ends up genuinely pinned.
|
||||
"""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
@@ -617,11 +623,56 @@ class TestSetupWizardDeploymentShape:
|
||||
}
|
||||
answers = [
|
||||
"cloud", "", "eri", "hermetika", "hermes",
|
||||
"single",
|
||||
"1",
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg)
|
||||
assert host["pinPeerName"] is True
|
||||
assert host["pinUserPeer"] is True
|
||||
|
||||
def test_no_gateway_connected_skips_mapping_when_declined(self, monkeypatch, tmp_path):
|
||||
"""With no gateway platforms connected, the tree is gated off; declining
|
||||
the 'configure anyway?' prompt leaves identity mapping untouched."""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
"hosts": {"hermes": {"peerName": "eri"}},
|
||||
}
|
||||
answers = ["cloud", "", "eri", "hermetika", "hermes", "n"]
|
||||
host = self._run_setup(
|
||||
monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg,
|
||||
gateway_platforms=[],
|
||||
)
|
||||
assert "pinUserPeer" not in host
|
||||
assert "userPeerAliases" not in host
|
||||
assert "runtimePeerPrefix" not in host
|
||||
|
||||
def test_undetectable_gateway_skips_mapping_when_declined(self, monkeypatch, tmp_path):
|
||||
"""When the gateway package can't be inspected (None), the wizard asks
|
||||
whether the gateway is running; 'no' skips the mapping step."""
|
||||
initial_cfg = {
|
||||
"apiKey": "***",
|
||||
"hosts": {"hermes": {"peerName": "eri"}},
|
||||
}
|
||||
answers = ["cloud", "", "eri", "hermetika", "hermes", "n"]
|
||||
host = self._run_setup(
|
||||
monkeypatch, tmp_path, answers=answers, initial_cfg=initial_cfg,
|
||||
gateway_platforms=None,
|
||||
)
|
||||
assert "pinUserPeer" not in host
|
||||
|
||||
def test_raw_edit_sets_resolver_knobs_directly(self, monkeypatch, tmp_path):
|
||||
"""The [e] escape hatch lets a power user set pinUserPeer + an alias +
|
||||
prefix directly, bypassing the intent tree."""
|
||||
answers = [
|
||||
"cloud", "", "eri", "hermetika", "hermes",
|
||||
"e", # tree: edit raw keys
|
||||
"false", # pinUserPeer
|
||||
"99887766=eri", # one alias pair
|
||||
"", # finish aliases
|
||||
"discord_", # runtimePeerPrefix
|
||||
]
|
||||
host = self._run_setup(monkeypatch, tmp_path, answers=answers)
|
||||
assert host["pinUserPeer"] is False
|
||||
assert host["userPeerAliases"] == {"99887766": "eri"}
|
||||
assert host["runtimePeerPrefix"] == "discord_"
|
||||
|
||||
|
||||
class TestCloneCarriesPinUserPeer:
|
||||
@@ -653,3 +704,27 @@ class TestCloneCarriesPinUserPeer:
|
||||
assert ok is True
|
||||
new_block = written["cfg"]["hosts"]["hermes_partner"]
|
||||
assert new_block["pinUserPeer"] is True
|
||||
|
||||
|
||||
class TestMigratePinKey:
|
||||
"""``_migrate_pin_key`` rewrites the legacy ``pinPeerName`` onto the
|
||||
canonical ``pinUserPeer`` in place, without clobbering an existing
|
||||
canonical value."""
|
||||
|
||||
def test_legacy_key_renamed_to_canonical(self):
|
||||
import plugins.memory.honcho.cli as honcho_cli
|
||||
block = {"pinPeerName": True}
|
||||
assert honcho_cli._migrate_pin_key(block) is True
|
||||
assert block == {"pinUserPeer": True}
|
||||
|
||||
def test_canonical_key_wins_when_both_present(self):
|
||||
import plugins.memory.honcho.cli as honcho_cli
|
||||
block = {"pinPeerName": True, "pinUserPeer": False}
|
||||
assert honcho_cli._migrate_pin_key(block) is True
|
||||
assert block == {"pinUserPeer": False}
|
||||
|
||||
def test_noop_when_no_legacy_key(self):
|
||||
import plugins.memory.honcho.cli as honcho_cli
|
||||
block = {"pinUserPeer": True}
|
||||
assert honcho_cli._migrate_pin_key(block) is False
|
||||
assert block == {"pinUserPeer": True}
|
||||
|
||||
@@ -105,7 +105,7 @@ class TestRuntimePeerMappingConfigParsing:
|
||||
config_file.write_text(json.dumps({
|
||||
"apiKey": "k",
|
||||
"userPeerAliases": {
|
||||
" 86701400 ": " Igor ",
|
||||
" 7654321 ": " Igor ",
|
||||
"": "ignored",
|
||||
"empty-value": " ",
|
||||
"null-value": None,
|
||||
@@ -115,7 +115,7 @@ class TestRuntimePeerMappingConfigParsing:
|
||||
|
||||
config = HonchoClientConfig.from_global_config(config_path=config_file)
|
||||
|
||||
assert config.user_peer_aliases == {"86701400": "Igor"}
|
||||
assert config.user_peer_aliases == {"7654321": "Igor"}
|
||||
assert config.runtime_peer_prefix == "telegram_"
|
||||
|
||||
def test_host_aliases_override_root_aliases_as_whole_map(self, tmp_path):
|
||||
@@ -226,12 +226,12 @@ class TestPeerResolutionOrder:
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._config(peer_name="Igor", pin_peer_name=False),
|
||||
runtime_user_peer_name="86701400", # e.g. Telegram UID
|
||||
runtime_user_peer_name="7654321", # e.g. Telegram UID
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == "86701400", (
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "7654321", (
|
||||
"pin_peer_name=False is the multi-user default — the gateway's "
|
||||
"platform-native user ID must win so each user gets their own "
|
||||
"peer scope. If this regresses, every Telegram/Discord/Slack "
|
||||
@@ -245,14 +245,14 @@ class TestPeerResolutionOrder:
|
||||
config=self._config(
|
||||
peer_name="Igor",
|
||||
pin_peer_name=False,
|
||||
user_peer_aliases={"86701400": "Igor"},
|
||||
user_peer_aliases={"7654321": "Igor"},
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "Igor"
|
||||
|
||||
def test_unknown_runtime_id_uses_prefix(self):
|
||||
@@ -264,12 +264,12 @@ class TestPeerResolutionOrder:
|
||||
pin_peer_name=False,
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == "telegram_86701400"
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "telegram_7654321"
|
||||
|
||||
def test_prefixed_runtime_id_hashes_when_sanitization_is_lossy(self):
|
||||
"""Generated prefixed IDs avoid merges caused by lossy sanitization."""
|
||||
@@ -291,43 +291,43 @@ class TestPeerResolutionOrder:
|
||||
|
||||
def test_prefixed_runtime_id_hashes_when_it_collides_with_peer_name(self):
|
||||
"""Unknown generated peers should not silently merge into peerName."""
|
||||
raw_peer_id = "telegram_86701400"
|
||||
raw_peer_id = "telegram_7654321"
|
||||
expected_hash = hashlib.sha256(raw_peer_id.encode("utf-8")).hexdigest()[:8]
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._config(
|
||||
peer_name="telegram_86701400",
|
||||
peer_name="telegram_7654321",
|
||||
pin_peer_name=False,
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == f"telegram_86701400-{expected_hash}"
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == f"telegram_7654321-{expected_hash}"
|
||||
|
||||
def test_prefixed_runtime_id_hashes_when_it_collides_with_alias_target(self):
|
||||
"""Unknown generated peers should not silently merge into alias targets."""
|
||||
raw_peer_id = "telegram_86701400"
|
||||
raw_peer_id = "telegram_7654321"
|
||||
expected_hash = hashlib.sha256(raw_peer_id.encode("utf-8")).hexdigest()[:8]
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._config(
|
||||
peer_name=None,
|
||||
pin_peer_name=False,
|
||||
user_peer_aliases={"known-user": "telegram_86701400"},
|
||||
user_peer_aliases={"known-user": "telegram_7654321"},
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == f"telegram_86701400-{expected_hash}"
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == f"telegram_7654321-{expected_hash}"
|
||||
|
||||
def test_prefixed_runtime_id_extends_hash_when_short_hash_collides(self):
|
||||
raw_peer_id = "telegram_86701400"
|
||||
raw_peer_id = "telegram_7654321"
|
||||
digest = hashlib.sha256(raw_peer_id.encode("utf-8")).hexdigest()
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
@@ -335,17 +335,17 @@ class TestPeerResolutionOrder:
|
||||
peer_name=None,
|
||||
pin_peer_name=False,
|
||||
user_peer_aliases={
|
||||
"known-user": "telegram_86701400",
|
||||
"reserved-user": f"telegram_86701400-{digest[:8]}",
|
||||
"known-user": "telegram_7654321",
|
||||
"reserved-user": f"telegram_7654321-{digest[:8]}",
|
||||
},
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == f"telegram_86701400-{digest[:12]}"
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == f"telegram_7654321-{digest[:12]}"
|
||||
|
||||
def test_alias_value_is_sanitized_after_selection(self):
|
||||
mgr = HonchoSessionManager(
|
||||
@@ -353,13 +353,13 @@ class TestPeerResolutionOrder:
|
||||
config=self._config(
|
||||
peer_name=None,
|
||||
pin_peer_name=False,
|
||||
user_peer_aliases={"86701400": "Alice Smith!"},
|
||||
user_peer_aliases={"7654321": "Alice Smith!"},
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "Alice-Smith-"
|
||||
|
||||
def test_alias_keys_match_raw_runtime_id_before_sanitization(self):
|
||||
@@ -391,13 +391,13 @@ class TestPeerResolutionOrder:
|
||||
runtime_peer_prefix="telegram_",
|
||||
session_peer_prefix=True,
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == "telegram_86701400"
|
||||
assert session.honcho_session_id == "telegram-86701400"
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "telegram_7654321"
|
||||
assert session.honcho_session_id == "telegram-7654321"
|
||||
|
||||
def test_config_wins_when_pin_is_true(self):
|
||||
"""With pin enabled, configured peer_name beats runtime ID."""
|
||||
@@ -406,14 +406,14 @@ class TestPeerResolutionOrder:
|
||||
config=self._config(
|
||||
peer_name="Igor",
|
||||
pin_peer_name=True,
|
||||
user_peer_aliases={"86701400": "Alias"},
|
||||
user_peer_aliases={"7654321": "Alias"},
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400", # Telegram pushes this in
|
||||
runtime_user_peer_name="7654321", # Telegram pushes this in
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "Igor", (
|
||||
"With pinPeerName=true the user's configured peer_name must "
|
||||
"beat the platform-native runtime ID so memory stays unified "
|
||||
@@ -429,26 +429,26 @@ class TestPeerResolutionOrder:
|
||||
config=self._config(
|
||||
peer_name=None,
|
||||
pin_peer_name=True,
|
||||
user_peer_aliases={"86701400": "Igor"},
|
||||
user_peer_aliases={"7654321": "Igor"},
|
||||
runtime_peer_prefix="telegram_",
|
||||
),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "Igor"
|
||||
|
||||
def test_pin_noop_without_peer_name_or_mapping_preserves_runtime(self):
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._config(peer_name=None, pin_peer_name=True),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
assert session.user_peer_id == "86701400"
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "7654321"
|
||||
|
||||
def test_alt_runtime_id_can_match_alias_without_changing_raw_fallback(self):
|
||||
"""Stable alternate IDs can map known users while primary ID fallback stays unchanged."""
|
||||
@@ -526,11 +526,11 @@ class TestPeerResolutionOrder:
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=cfg,
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
|
||||
session = mgr.get_or_create("telegram:86701400")
|
||||
session = mgr.get_or_create("telegram:7654321")
|
||||
assert session.user_peer_id == "Igor"
|
||||
assert session.assistant_peer_id == "hermes-assistant"
|
||||
|
||||
@@ -556,10 +556,10 @@ class TestCrossPlatformMemoryUnification:
|
||||
mgr_telegram = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._config_pinned(),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr_telegram)
|
||||
telegram_session = mgr_telegram.get_or_create("telegram:86701400")
|
||||
telegram_session = mgr_telegram.get_or_create("telegram:7654321")
|
||||
|
||||
# Discord turn (separate manager instance — simulates a fresh
|
||||
# platform-adapter invocation)
|
||||
@@ -701,20 +701,20 @@ class TestPinTransition:
|
||||
pinned_mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._pinned(),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(pinned_mgr)
|
||||
before = pinned_mgr.get_or_create("telegram:86701400")
|
||||
before = pinned_mgr.get_or_create("telegram:7654321")
|
||||
assert before.user_peer_id == "Igor"
|
||||
|
||||
unpinned_mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._unpinned(),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(unpinned_mgr)
|
||||
after = unpinned_mgr.get_or_create("telegram:86701400")
|
||||
assert after.user_peer_id == "86701400", (
|
||||
after = unpinned_mgr.get_or_create("telegram:7654321")
|
||||
assert after.user_peer_id == "7654321", (
|
||||
"After flipping pinPeerName off, the same runtime ID must resolve "
|
||||
"to its own peer — otherwise multi-user mode silently merges users."
|
||||
)
|
||||
@@ -723,14 +723,14 @@ class TestPinTransition:
|
||||
mgr = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._pinned(),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr)
|
||||
first = mgr.get_or_create("telegram:86701400")
|
||||
first = mgr.get_or_create("telegram:7654321")
|
||||
assert first.user_peer_id == "Igor"
|
||||
|
||||
mgr._config = self._unpinned()
|
||||
second = mgr.get_or_create("telegram:86701400")
|
||||
second = mgr.get_or_create("telegram:7654321")
|
||||
assert second.user_peer_id == "Igor", (
|
||||
"The per-key session cache is keyed by session-key, not by "
|
||||
"resolved peer. In-process flips don't invalidate it — the "
|
||||
@@ -764,7 +764,7 @@ class TestPinTransition:
|
||||
cfg_path.write_text(json.dumps({
|
||||
"apiKey": "k",
|
||||
"peerName": "Igor",
|
||||
"userPeerAliases": {"86701400": "Igor"},
|
||||
"userPeerAliases": {"7654321": "Igor"},
|
||||
}))
|
||||
sig_with_aliases = GatewayRunner._extract_cache_busting_config({"memory": {"provider": "honcho"}})
|
||||
|
||||
@@ -839,18 +839,18 @@ class TestProfilePeerUniqueness:
|
||||
mgr_a = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._pinned_to("alice"),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr_a)
|
||||
sess_a = mgr_a.get_or_create("telegram:86701400")
|
||||
sess_a = mgr_a.get_or_create("telegram:7654321")
|
||||
|
||||
mgr_b = HonchoSessionManager(
|
||||
honcho=MagicMock(),
|
||||
config=self._pinned_to("bob"),
|
||||
runtime_user_peer_name="86701400",
|
||||
runtime_user_peer_name="7654321",
|
||||
)
|
||||
_patch_manager_for_resolution_test(mgr_b)
|
||||
sess_b = mgr_b.get_or_create("telegram:86701400")
|
||||
sess_b = mgr_b.get_or_create("telegram:7654321")
|
||||
|
||||
assert sess_a.user_peer_id == "alice"
|
||||
assert sess_b.user_peer_id == "bob"
|
||||
|
||||
@@ -88,7 +88,10 @@ def test_store_project_credentials_round_trip(
|
||||
sid, secret = photon_auth.load_project_credentials()
|
||||
assert sid == "sp-123"
|
||||
assert secret == "secret-key"
|
||||
assert photon_auth.load_dashboard_project_id() == "dash-456"
|
||||
# Post-unification the management id resolves to the Spectrum id, not the
|
||||
# stored dashboard id — so a pre-backfill diverged install (whose old
|
||||
# dashboard id was rewritten and now 404s) still reaches the live row.
|
||||
assert photon_auth.load_dashboard_project_id() == "sp-123"
|
||||
|
||||
|
||||
def test_store_project_credentials_writes_env(tmp_hermes_home: Path) -> None:
|
||||
@@ -284,7 +287,7 @@ def test_find_project_by_name_case_insensitive(monkeypatch: pytest.MonkeyPatch)
|
||||
assert proj is not None and proj["id"] == "p2"
|
||||
|
||||
|
||||
def test_create_project_sends_spectrum_true(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def test_create_project_omits_spectrum_flag(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
captured: Dict[str, Any] = {}
|
||||
|
||||
def fake_post(url: str, **kwargs: Any) -> _FakeResponse:
|
||||
@@ -296,7 +299,9 @@ def test_create_project_sends_spectrum_true(monkeypatch: pytest.MonkeyPatch) ->
|
||||
monkeypatch.setattr(photon_auth.httpx, "post", fake_post)
|
||||
data = photon_auth.create_project("tok", name="Hermes Agent")
|
||||
assert data["id"] == "new-proj"
|
||||
assert captured["body"]["spectrum"] is True
|
||||
# Spectrum is always provisioned at create-time; the field was dropped
|
||||
# from the API schema, so we must not send it.
|
||||
assert "spectrum" not in captured["body"]
|
||||
assert captured["body"]["name"] == "Hermes Agent"
|
||||
assert captured["headers"]["Authorization"] == "Bearer tok"
|
||||
assert captured["url"].endswith("/api/projects")
|
||||
@@ -311,46 +316,6 @@ def test_create_project_raises_without_id(monkeypatch: pytest.MonkeyPatch) -> No
|
||||
photon_auth.create_project("tok")
|
||||
|
||||
|
||||
def test_ensure_spectrum_enabled_toggles_when_off(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
get_calls = {"n": 0}
|
||||
posted = {"toggle": False}
|
||||
|
||||
def fake_get(url: str, **kwargs: Any) -> _FakeResponse:
|
||||
get_calls["n"] += 1
|
||||
if get_calls["n"] == 1:
|
||||
return _FakeResponse(json_body={"id": "p", "spectrum": False, "spectrumProjectId": None})
|
||||
return _FakeResponse(json_body={"id": "p", "spectrum": True, "spectrumProjectId": "sp-1"})
|
||||
|
||||
def fake_post(url: str, **kwargs: Any) -> _FakeResponse:
|
||||
if url.endswith("/spectrum/toggle"):
|
||||
posted["toggle"] = True
|
||||
return _FakeResponse(json_body={"success": True})
|
||||
|
||||
monkeypatch.setattr(photon_auth.httpx, "get", fake_get)
|
||||
monkeypatch.setattr(photon_auth.httpx, "post", fake_post)
|
||||
proj = photon_auth.ensure_spectrum_enabled("tok", "p")
|
||||
assert posted["toggle"] is True
|
||||
assert proj["spectrumProjectId"] == "sp-1"
|
||||
|
||||
|
||||
def test_ensure_spectrum_enabled_skips_toggle_when_on(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
posted = {"toggle": False}
|
||||
|
||||
def fake_get(url: str, **kwargs: Any) -> _FakeResponse:
|
||||
return _FakeResponse(json_body={"id": "p", "spectrum": True, "spectrumProjectId": "sp-1"})
|
||||
|
||||
def fake_post(url: str, **kwargs: Any) -> _FakeResponse:
|
||||
if url.endswith("/spectrum/toggle"):
|
||||
posted["toggle"] = True
|
||||
return _FakeResponse(json_body={"success": True})
|
||||
|
||||
monkeypatch.setattr(photon_auth.httpx, "get", fake_get)
|
||||
monkeypatch.setattr(photon_auth.httpx, "post", fake_post)
|
||||
proj = photon_auth.ensure_spectrum_enabled("tok", "p")
|
||||
assert posted["toggle"] is False
|
||||
assert proj["spectrumProjectId"] == "sp-1"
|
||||
|
||||
|
||||
def test_regenerate_project_secret(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def fake_post(url: str, **kwargs: Any) -> _FakeResponse:
|
||||
assert url.endswith("/regenerate-secret")
|
||||
@@ -498,8 +463,8 @@ def test_credential_summary_no_secret_leak(
|
||||
assert "secret-bbbb" not in blob
|
||||
assert summary["device_token"].startswith("✓")
|
||||
assert summary["project_key"].startswith("✓")
|
||||
assert summary["spectrum_project_id"] == "sp-uuid"
|
||||
assert summary["dashboard_project_id"] == "dash-uuid"
|
||||
# Unified id: dashboard id == Spectrum id, surfaced as one project id.
|
||||
assert summary["project_id"] == "sp-uuid"
|
||||
assert summary["phone_number"].startswith("✗ missing")
|
||||
assert summary["assigned_phone_number"].startswith("✗ missing")
|
||||
|
||||
|
||||
@@ -73,12 +73,10 @@ def test_env_enablement_home_channel_defaults_name(monkeypatch: pytest.MonkeyPat
|
||||
|
||||
def test_setup_hint_uses_gateway_service_command(monkeypatch: pytest.MonkeyPatch, capsys) -> None:
|
||||
monkeypatch.setattr(cli.photon_auth, "load_photon_token", lambda: "token")
|
||||
# The dashboard id *is* the Spectrum project id (ids unified), so setup no
|
||||
# longer enables Spectrum or fetches a separate spectrumProjectId — it
|
||||
# reuses this id directly.
|
||||
monkeypatch.setattr(cli.photon_auth, "load_dashboard_project_id", lambda: "dashboard")
|
||||
monkeypatch.setattr(
|
||||
cli.photon_auth,
|
||||
"ensure_spectrum_enabled",
|
||||
lambda token, dashboard_id: {"spectrumProjectId": "project_123"},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
cli.photon_auth,
|
||||
"regenerate_project_secret",
|
||||
|
||||
@@ -176,11 +176,16 @@ class TestClientCacheBoundedGrowth:
|
||||
"""When the loop changes, the old entry should be replaced, not duplicated."""
|
||||
from agent.auxiliary_client import (
|
||||
_client_cache,
|
||||
_client_cache_key,
|
||||
_client_cache_lock,
|
||||
_get_cached_client,
|
||||
)
|
||||
|
||||
key = ("test_replace", True, "", "", "", (), False, "")
|
||||
key = _client_cache_key(
|
||||
"test_replace",
|
||||
async_mode=True,
|
||||
task="",
|
||||
)
|
||||
|
||||
# Simulate a stale entry from a closed loop
|
||||
old_loop = asyncio.new_event_loop()
|
||||
|
||||
@@ -115,10 +115,11 @@ def test_background_review_summarizer_receives_captured_messages_after_close(mon
|
||||
# must have snapshot them before this runs.
|
||||
self._session_messages = []
|
||||
|
||||
def fake_summarize(review_messages, prior_snapshot):
|
||||
def fake_summarize(review_messages, prior_snapshot, notification_mode="on"):
|
||||
events.append("summarize")
|
||||
captured["review_messages"] = list(review_messages)
|
||||
captured["prior_snapshot"] = list(prior_snapshot)
|
||||
captured["notification_mode"] = notification_mode
|
||||
return []
|
||||
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", FakeReviewAgent)
|
||||
@@ -146,6 +147,7 @@ def test_background_review_summarizer_receives_captured_messages_after_close(mon
|
||||
]
|
||||
assert captured["review_messages"] == [review_tool_message]
|
||||
assert captured["prior_snapshot"] == messages_snapshot
|
||||
assert captured["notification_mode"] == "on"
|
||||
|
||||
|
||||
def test_background_review_installs_auto_deny_approval_callback(monkeypatch):
|
||||
@@ -313,3 +315,117 @@ def test_background_review_fork_skips_external_memory_plugins(monkeypatch):
|
||||
"the fork leaks harness prompts into the user's real memory "
|
||||
"namespace via on_turn_start / prefetch_all / sync_all."
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# memory_notifications mode: off | on | verbose
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
import json as _json
|
||||
|
||||
from agent.background_review import summarize_background_review_actions
|
||||
|
||||
|
||||
def _memory_add_review():
|
||||
"""A minimal review transcript: one memory add (assistant call + tool result)."""
|
||||
return [
|
||||
{
|
||||
"role": "assistant",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_mem1",
|
||||
"function": {
|
||||
"name": "memory",
|
||||
"arguments": _json.dumps(
|
||||
{
|
||||
"action": "add",
|
||||
"target": "memory",
|
||||
"content": "User prefers terse replies",
|
||||
}
|
||||
),
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_call_id": "call_mem1",
|
||||
"content": _json.dumps(
|
||||
{"success": True, "message": "Entry added.", "target": "memory"}
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _skill_patch_review():
|
||||
return [
|
||||
{
|
||||
"role": "assistant",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_skill1",
|
||||
"function": {
|
||||
"name": "skill_manage",
|
||||
"arguments": _json.dumps(
|
||||
{"action": "patch", "name": "demo", "old_string": "a", "new_string": "b"}
|
||||
),
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_call_id": "call_skill1",
|
||||
"content": _json.dumps(
|
||||
{
|
||||
"success": True,
|
||||
"message": "Patched SKILL.md in skill 'demo' (1 replacement).",
|
||||
"_change": {"old": "a", "new": "b"},
|
||||
}
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def test_memory_notifications_off_returns_nothing():
|
||||
actions = summarize_background_review_actions(
|
||||
_memory_add_review(), [], notification_mode="off"
|
||||
)
|
||||
assert actions == []
|
||||
|
||||
|
||||
def test_memory_notifications_on_returns_generic_line():
|
||||
actions = summarize_background_review_actions(
|
||||
_memory_add_review(), [], notification_mode="on"
|
||||
)
|
||||
assert actions == ["Memory updated"]
|
||||
|
||||
|
||||
def test_memory_notifications_verbose_includes_content_preview():
|
||||
actions = summarize_background_review_actions(
|
||||
_memory_add_review(), [], notification_mode="verbose"
|
||||
)
|
||||
assert len(actions) == 1
|
||||
# Verbose surfaces the actual content that was saved.
|
||||
assert "User prefers terse replies" in actions[0]
|
||||
assert actions[0] != "Memory updated"
|
||||
|
||||
|
||||
def test_memory_notifications_default_is_on():
|
||||
"""No mode passed → behaves like 'on' (generic line, not empty/verbose)."""
|
||||
actions = summarize_background_review_actions(_memory_add_review(), [])
|
||||
assert actions == ["Memory updated"]
|
||||
|
||||
|
||||
def test_skill_patch_off_silent_verbose_shows_diff():
|
||||
assert (
|
||||
summarize_background_review_actions(
|
||||
_skill_patch_review(), [], notification_mode="off"
|
||||
)
|
||||
== []
|
||||
)
|
||||
verbose = summarize_background_review_actions(
|
||||
_skill_patch_review(), [], notification_mode="verbose"
|
||||
)
|
||||
assert len(verbose) == 1
|
||||
assert "demo" in verbose[0] and "→" in verbose[0]
|
||||
|
||||
@@ -7,6 +7,7 @@ import time
|
||||
import types
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from unittest.mock import patch
|
||||
|
||||
from hermes_constants import reset_hermes_home_override, set_hermes_home_override
|
||||
@@ -380,6 +381,78 @@ def test_tui_verbose_tool_events_omit_details_when_redaction_fails(monkeypatch):
|
||||
assert "result_text" not in events[1][2]
|
||||
|
||||
|
||||
def test_tool_complete_emits_full_unified_diff(monkeypatch):
|
||||
events: list[tuple[str, str, dict]] = []
|
||||
monkeypatch.setattr(
|
||||
server, "_emit", lambda event_type, sid, payload: events.append((event_type, sid, payload))
|
||||
)
|
||||
monkeypatch.setitem(
|
||||
server._sessions,
|
||||
"diff-test",
|
||||
{"tool_progress_mode": "concise", "tool_started_at": {}, "edit_snapshots": {}},
|
||||
)
|
||||
|
||||
diff = "--- a/x.py\n+++ b/x.py\n@@ -1 +1 @@\n-a = 1\n+a = 2\n"
|
||||
result = json.dumps({"success": True, "diff": diff})
|
||||
server._on_tool_complete("diff-test", "tool-1", "patch", {"mode": "replace", "path": "x.py"}, result)
|
||||
|
||||
assert events and events[0][0] == "tool.complete"
|
||||
payload = events[0][2]
|
||||
# the raw unified diff rides alongside the pretty/capped inline_diff
|
||||
assert payload["diff_unified"] == diff
|
||||
assert "inline_diff" in payload
|
||||
|
||||
|
||||
def test_verbose_result_text_drops_diff_echo_when_diff_unified_ships(monkeypatch):
|
||||
# A tall edit's result JSON embeds the WHOLE diff; tail-capping that echo
|
||||
# yields an unparseable JSON-looking fragment the TUI can't suppress
|
||||
# reliably. When diff_unified ships, result_text must carry only the
|
||||
# non-diff signal — small, parseable, never the diff echo.
|
||||
events: list[tuple[str, str, dict]] = []
|
||||
monkeypatch.setattr(
|
||||
server, "_emit", lambda event_type, sid, payload: events.append((event_type, sid, payload))
|
||||
)
|
||||
monkeypatch.setitem(
|
||||
server._sessions,
|
||||
"diff-echo-test",
|
||||
{"tool_progress_mode": "verbose", "tool_started_at": {}, "edit_snapshots": {}},
|
||||
)
|
||||
|
||||
lines = "\n".join(f"+def fn_{i}() -> int: return {i}" for i in range(60))
|
||||
diff = f"--- a/x.py\n+++ b/x.py\n@@ -1,0 +1,60 @@\n{lines}\n"
|
||||
result = json.dumps(
|
||||
{"success": True, "diff": diff, "files_modified": ["x.py"], "_warning": "stale read"}
|
||||
)
|
||||
server._on_tool_complete("diff-echo-test", "tool-1", "patch", {"mode": "patch"}, result)
|
||||
|
||||
payload = events[0][2]
|
||||
assert payload["diff_unified"] == diff
|
||||
text = payload["result_text"]
|
||||
assert "[showing verbose tail" not in text # small enough to dodge the cap
|
||||
parsed = json.loads(text) # parseable …
|
||||
assert "diff" not in parsed # … with the echo gone
|
||||
assert parsed["_warning"] == "stale read" # non-diff signal survives
|
||||
# without diff_unified (non-edit tools) the result_text is untouched
|
||||
assert server._result_sans_diff_echo("plain text result") == "plain text result"
|
||||
assert server._result_sans_diff_echo('{"output": "x"}') == '{"output": "x"}'
|
||||
|
||||
|
||||
def test_cap_diff_unified_truncates_at_line_boundary():
|
||||
line = "+" + "x" * 63 # 64 bytes per line incl. newline
|
||||
diff = "\n".join([line] * 100)
|
||||
|
||||
capped = server._cap_diff_unified(diff, max_bytes=1000)
|
||||
|
||||
body, _, marker = capped.rpartition("\n")
|
||||
assert marker.startswith("# … diff truncated (")
|
||||
assert marker.endswith(" more bytes)")
|
||||
assert len(body.encode("utf-8")) <= 1000
|
||||
# cut on a line boundary: every surviving line is intact
|
||||
assert all(l == line for l in body.split("\n"))
|
||||
# under the cap → untouched
|
||||
assert server._cap_diff_unified("small", max_bytes=1000) == "small"
|
||||
|
||||
|
||||
def test_dispatch_rejects_non_object_request():
|
||||
resp = server.dispatch([])
|
||||
|
||||
@@ -4162,6 +4235,51 @@ def test_session_info_includes_mcp_servers(monkeypatch):
|
||||
assert info["mcp_servers"] == fake_status
|
||||
|
||||
|
||||
def test_session_info_includes_session_title(monkeypatch):
|
||||
"""session.info carries the live session title (window-title chrome).
|
||||
|
||||
Resolution order mirrors _session_live_title: DB row wins, a queued
|
||||
pending_title fills in before the row exists, "" until either lands.
|
||||
"""
|
||||
agent = types.SimpleNamespace(tools=[], model="m", provider="p")
|
||||
|
||||
# No session at all -> "" (and never a crash).
|
||||
assert server._session_info(agent, None)["title"] == ""
|
||||
|
||||
# pending_title before the DB row exists.
|
||||
session = {"session_key": "k1", "pending_title": "rename the moon"}
|
||||
monkeypatch.setattr(server, "_get_db", lambda: None)
|
||||
assert server._session_info(agent, session)["title"] == "rename the moon"
|
||||
|
||||
# DB row wins over pending.
|
||||
fake_db = types.SimpleNamespace(get_session_title=lambda key: "db title")
|
||||
monkeypatch.setattr(server, "_get_db", lambda: fake_db)
|
||||
assert server._session_info(agent, session)["title"] == "db title"
|
||||
|
||||
|
||||
def test_emit_title_refresh_pushes_session_info(monkeypatch):
|
||||
"""_emit_title_refresh emits a session.info for a live session and is a
|
||||
silent no-op for unknown/agent-less sessions (it runs on the auto-title
|
||||
worker thread -- it must never raise)."""
|
||||
events = []
|
||||
monkeypatch.setattr(
|
||||
server, "_emit", lambda event_type, sid, payload: events.append((event_type, sid, payload))
|
||||
)
|
||||
monkeypatch.setattr(server, "_session_info", lambda agent, session: {"title": "t"})
|
||||
|
||||
agent = types.SimpleNamespace(tools=[], model="m", provider="p")
|
||||
monkeypatch.setitem(server._sessions, "title-test", {"agent": agent, "session_key": "k"})
|
||||
|
||||
server._emit_title_refresh("title-test")
|
||||
assert events == [("session.info", "title-test", {"title": "t"})]
|
||||
|
||||
# Unknown sid / no agent -> no emission, no exception.
|
||||
server._emit_title_refresh("missing-sid")
|
||||
monkeypatch.setitem(server._sessions, "agentless", {"session_key": "k2"})
|
||||
server._emit_title_refresh("agentless")
|
||||
assert len(events) == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# History-mutating commands must reject while session.running is True.
|
||||
# Without these guards, prompt.submit's post-run history write either
|
||||
@@ -4562,6 +4680,90 @@ def test_respond_unpacks_sid_tuple_correctly():
|
||||
server._answers.pop("rid-x", None)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Blocking prompts wait for the human (v6 north-star #5): _block with
|
||||
# timeout=None must never expire — interrupt/shutdown (_clear_pending)
|
||||
# are the only releases.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _run_block_in_thread(monkeypatch, sid):
|
||||
"""Start _block(timeout=None) on a background thread; return
|
||||
(thread, results, get_rid) where get_rid polls for the pending rid."""
|
||||
monkeypatch.setattr(server, "_emit", lambda *a, **kw: None)
|
||||
results: list[str] = []
|
||||
|
||||
def runner():
|
||||
results.append(server._block("clarify.request", sid, {"question": "q"}))
|
||||
|
||||
t = threading.Thread(target=runner, daemon=True)
|
||||
t.start()
|
||||
|
||||
def get_rid():
|
||||
deadline = time.time() + 5
|
||||
while time.time() < deadline:
|
||||
with server._prompt_lock:
|
||||
for rid, (owner, _ev) in server._pending.items():
|
||||
if owner == sid:
|
||||
return rid
|
||||
time.sleep(0.01)
|
||||
raise AssertionError("pending rid never appeared for sid=%s" % sid)
|
||||
|
||||
return t, results, get_rid
|
||||
|
||||
|
||||
def test_block_no_timeout_waits_for_delayed_answer(monkeypatch):
|
||||
"""_block(timeout=None) must keep blocking until the answer arrives —
|
||||
no premature empty return."""
|
||||
t, results, get_rid = _run_block_in_thread(monkeypatch, "sid_block_wait")
|
||||
rid = get_rid()
|
||||
|
||||
# Answer after a short delay; _block must still be waiting.
|
||||
time.sleep(0.3)
|
||||
assert t.is_alive(), "_block returned before any answer was provided"
|
||||
with server._prompt_lock:
|
||||
server._answers[rid] = "green"
|
||||
server._pending[rid][1].set()
|
||||
|
||||
t.join(timeout=5)
|
||||
assert not t.is_alive()
|
||||
assert results == ["green"]
|
||||
|
||||
|
||||
def test_clear_pending_releases_no_timeout_block(monkeypatch):
|
||||
"""_clear_pending(sid) must release a timeout=None _block with ''."""
|
||||
t, results, get_rid = _run_block_in_thread(monkeypatch, "sid_block_clear")
|
||||
get_rid()
|
||||
|
||||
server._clear_pending("sid_block_clear")
|
||||
t.join(timeout=5)
|
||||
assert not t.is_alive()
|
||||
assert results == [""]
|
||||
|
||||
|
||||
def test_clear_pending_other_sid_does_not_release_block(monkeypatch):
|
||||
"""_clear_pending on an unrelated session must NOT release a pending
|
||||
timeout=None _block (session scoping)."""
|
||||
t, results, get_rid = _run_block_in_thread(monkeypatch, "sid_block_scoped")
|
||||
rid = get_rid()
|
||||
|
||||
server._clear_pending("sid_some_other_session")
|
||||
time.sleep(0.2)
|
||||
assert t.is_alive(), (
|
||||
"_clear_pending on another sid released a prompt owned by "
|
||||
"sid_block_scoped — session scoping is broken"
|
||||
)
|
||||
assert not results
|
||||
|
||||
# Clean up: release properly so the thread joins.
|
||||
with server._prompt_lock:
|
||||
server._answers[rid] = "done"
|
||||
server._pending[rid][1].set()
|
||||
t.join(timeout=5)
|
||||
assert not t.is_alive()
|
||||
assert results == ["done"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# /model switch and other agent-mutating commands must reject while the
|
||||
# session is running. agent.switch_model() mutates self.model, self.provider,
|
||||
@@ -4901,33 +5103,45 @@ def test_session_create_no_race_keeps_worker_alive(monkeypatch):
|
||||
)
|
||||
monkeypatch.setattr(_approval, "load_permanent_allowlist", lambda: None)
|
||||
|
||||
resp = server.handle_request(
|
||||
{
|
||||
"id": "1",
|
||||
"method": "session.create",
|
||||
"params": {"cols": 80},
|
||||
}
|
||||
)
|
||||
sid = resp["result"]["session_id"]
|
||||
# Isolate from sibling-test leakage: daemon build threads from prior
|
||||
# session.create tests in the same shard process mutate the shared
|
||||
# ``server._sessions`` dict under ``_sessions_lock`` and can replace/pop
|
||||
# entries mid-run, which would flip this build thread's ``replaced`` check
|
||||
# to True and trigger a spurious unregister. Snapshot, clear, and restore
|
||||
# so this test sees only its own session regardless of shard composition.
|
||||
_saved_sessions = dict(server._sessions)
|
||||
server._sessions.clear()
|
||||
|
||||
# Wait for the build to finish (ready event inside session dict).
|
||||
session = server._sessions[sid]
|
||||
session["agent_ready"].wait(timeout=2.0)
|
||||
try:
|
||||
resp = server.handle_request(
|
||||
{
|
||||
"id": "1",
|
||||
"method": "session.create",
|
||||
"params": {"cols": 80},
|
||||
}
|
||||
)
|
||||
sid = resp["result"]["session_id"]
|
||||
|
||||
# Build finished without a close race — nothing should have been
|
||||
# cleaned up by the orphan check.
|
||||
assert (
|
||||
closed_workers == []
|
||||
), f"build thread closed its own worker despite no race: {closed_workers}"
|
||||
assert (
|
||||
unregistered_keys == []
|
||||
), f"build thread unregistered its own notify despite no race: {unregistered_keys}"
|
||||
# Wait for the build to finish (ready event inside session dict).
|
||||
session = server._sessions[sid]
|
||||
built = session["agent_ready"].wait(timeout=10.0)
|
||||
assert built, "agent build did not complete within timeout"
|
||||
|
||||
# Session should have the live worker installed.
|
||||
assert session.get("slash_worker") is not None
|
||||
# Build finished without a close race — nothing should have been
|
||||
# cleaned up by the orphan check.
|
||||
assert (
|
||||
closed_workers == []
|
||||
), f"build thread closed its own worker despite no race: {closed_workers}"
|
||||
assert (
|
||||
unregistered_keys == []
|
||||
), f"build thread unregistered its own notify despite no race: {unregistered_keys}"
|
||||
|
||||
# Cleanup
|
||||
server._sessions.pop(sid, None)
|
||||
# Session should have the live worker installed.
|
||||
assert session.get("slash_worker") is not None
|
||||
finally:
|
||||
# Cleanup + restore sibling sessions we snapshotted.
|
||||
server._sessions.clear()
|
||||
server._sessions.update(_saved_sessions)
|
||||
|
||||
|
||||
def test_get_db_degrades_cleanly_when_sessiondb_init_fails(monkeypatch):
|
||||
@@ -5764,6 +5978,351 @@ def test_session_most_recent_handles_db_unavailable(monkeypatch):
|
||||
assert resp["result"]["session_id"] is None
|
||||
|
||||
|
||||
# ── session.list (resume-picker filters + widened projection) ───────
|
||||
|
||||
|
||||
def _picker_row(sid, source, started, **extra):
|
||||
row = {
|
||||
"id": sid,
|
||||
"source": source,
|
||||
"title": f"title-{sid}",
|
||||
"preview": f"preview-{sid}",
|
||||
"started_at": started,
|
||||
"last_active": started,
|
||||
"message_count": 3,
|
||||
"ended_at": None,
|
||||
"cwd": None,
|
||||
"model": None,
|
||||
}
|
||||
row.update(extra)
|
||||
return row
|
||||
|
||||
|
||||
class _PickerDB:
|
||||
"""list_sessions_rich stand-in honouring the kwargs session.list maps."""
|
||||
|
||||
def __init__(self, rows):
|
||||
self.rows = rows
|
||||
self.calls = []
|
||||
|
||||
def list_sessions_rich(
|
||||
self,
|
||||
*,
|
||||
source=None,
|
||||
limit=20,
|
||||
offset=0,
|
||||
order_by_last_active=False,
|
||||
id_query=None,
|
||||
):
|
||||
self.calls.append(
|
||||
{
|
||||
"source": source,
|
||||
"limit": limit,
|
||||
"offset": offset,
|
||||
"order_by_last_active": order_by_last_active,
|
||||
"id_query": id_query,
|
||||
}
|
||||
)
|
||||
rows = [dict(r) for r in self.rows]
|
||||
if source:
|
||||
rows = [r for r in rows if r.get("source") == source]
|
||||
if id_query:
|
||||
needle = id_query.strip().lower()
|
||||
rows = [r for r in rows if needle in r["id"].lower()]
|
||||
if order_by_last_active:
|
||||
rows.sort(key=lambda r: r.get("last_active") or 0, reverse=True)
|
||||
else:
|
||||
rows.sort(key=lambda r: r.get("started_at") or 0, reverse=True)
|
||||
return rows[offset : offset + limit]
|
||||
|
||||
|
||||
def _picker_db():
|
||||
return _PickerDB(
|
||||
[
|
||||
_picker_row(
|
||||
"tui-1",
|
||||
"tui",
|
||||
100,
|
||||
cwd="/home/u/proj",
|
||||
model="nous/hermes-4",
|
||||
ended_at=150.0,
|
||||
),
|
||||
_picker_row("cli-1", "cli", 90),
|
||||
_picker_row("cron-1", "cron", 80),
|
||||
_picker_row("cron-2", "cron", 70),
|
||||
_picker_row("tg-1", "telegram", 60),
|
||||
_picker_row("tool-1", "tool", 50),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def test_session_list_no_params_keeps_legacy_behavior_with_widened_rows(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
resp = server.handle_request({"id": "1", "method": "session.list", "params": {}})
|
||||
|
||||
rows = resp["result"]["sessions"]
|
||||
# Legacy semantics: started_at DESC, `tool` denied, one DB fetch.
|
||||
assert [r["id"] for r in rows] == ["tui-1", "cli-1", "cron-1", "cron-2", "tg-1"]
|
||||
assert db.calls == [
|
||||
{
|
||||
"source": None,
|
||||
"limit": 400,
|
||||
"offset": 0,
|
||||
"order_by_last_active": False,
|
||||
"id_query": None,
|
||||
}
|
||||
]
|
||||
# Widened projection, None-safe for rows missing the new columns.
|
||||
first, second = rows[0], rows[1]
|
||||
assert first["cwd"] == "/home/u/proj"
|
||||
assert first["model"] == "nous/hermes-4"
|
||||
assert first["ended_at"] == 150.0
|
||||
assert first["last_active"] == 100
|
||||
assert second["cwd"] is None
|
||||
assert second["model"] is None
|
||||
assert second["ended_at"] is None
|
||||
assert second["last_active"] == 90
|
||||
# Legacy keys are still present and unchanged.
|
||||
assert second["title"] == "title-cli-1"
|
||||
assert second["preview"] == "preview-cli-1"
|
||||
assert second["message_count"] == 3
|
||||
assert second["source"] == "cli"
|
||||
|
||||
|
||||
def test_session_list_single_source_passes_through_to_db(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.list", "params": {"sources": ["tui"]}}
|
||||
)
|
||||
|
||||
assert [r["id"] for r in resp["result"]["sessions"]] == ["tui-1"]
|
||||
assert db.calls[0]["source"] == "tui" # pushed into SQL, not Python-filtered
|
||||
|
||||
|
||||
def test_session_list_multi_source_filters_gateway_side(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{
|
||||
"id": "1",
|
||||
"method": "session.list",
|
||||
"params": {"sources": ["cron", "telegram"]},
|
||||
}
|
||||
)
|
||||
|
||||
assert [r["id"] for r in resp["result"]["sessions"]] == [
|
||||
"cron-1",
|
||||
"cron-2",
|
||||
"tg-1",
|
||||
]
|
||||
assert db.calls[0]["source"] is None # multi-source: DB scan + gateway filter
|
||||
|
||||
|
||||
def test_session_list_sources_tool_stays_denied(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.list", "params": {"sources": ["tool"]}}
|
||||
)
|
||||
|
||||
assert resp["result"]["sessions"] == []
|
||||
|
||||
|
||||
def test_session_list_query_maps_to_id_query_on_last_active_path(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.list", "params": {"query": "cron"}}
|
||||
)
|
||||
|
||||
assert [r["id"] for r in resp["result"]["sessions"]] == ["cron-1", "cron-2"]
|
||||
assert db.calls[0]["id_query"] == "cron"
|
||||
assert db.calls[0]["order_by_last_active"] is True
|
||||
|
||||
|
||||
def test_session_list_offset_limit_paginate(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
def page(offset, limit):
|
||||
resp = server.handle_request(
|
||||
{
|
||||
"id": "1",
|
||||
"method": "session.list",
|
||||
"params": {
|
||||
"sources": ["tui", "cli", "cron", "telegram"],
|
||||
"offset": offset,
|
||||
"limit": limit,
|
||||
},
|
||||
}
|
||||
)
|
||||
return [r["id"] for r in resp["result"]["sessions"]]
|
||||
|
||||
assert page(0, 2) == ["tui-1", "cli-1"]
|
||||
assert page(2, 2) == ["cron-1", "cron-2"]
|
||||
assert page(4, 2) == ["tg-1"]
|
||||
assert page(6, 2) == []
|
||||
|
||||
|
||||
# ── session.peek (resume-picker Space preview) ───────────────────────
|
||||
|
||||
|
||||
class _PeekDB:
|
||||
def __init__(self, session, messages):
|
||||
self.session = session
|
||||
self.messages = messages
|
||||
|
||||
def get_session(self, session_id):
|
||||
if self.session and session_id == self.session["id"]:
|
||||
return dict(self.session)
|
||||
return None
|
||||
|
||||
def get_messages(self, session_id):
|
||||
return [dict(m) for m in self.messages]
|
||||
|
||||
|
||||
def _peek_db():
|
||||
session = {
|
||||
"id": "sess-1",
|
||||
"title": "picker demo",
|
||||
"source": "tui",
|
||||
"model": "nous/hermes-4",
|
||||
"cwd": "/home/u/proj",
|
||||
"started_at": 100.0,
|
||||
"ended_at": 400.0,
|
||||
"end_reason": "tui_shutdown",
|
||||
"message_count": 6,
|
||||
"actual_cost_usd": None,
|
||||
"estimated_cost_usd": 0.42,
|
||||
}
|
||||
messages = [
|
||||
{"id": 1, "role": "system", "content": "sys prompt", "timestamp": 100.0},
|
||||
{"id": 2, "role": "user", "content": "first prompt", "timestamp": 110.0},
|
||||
{"id": 3, "role": "assistant", "content": "first answer", "timestamp": 120.0},
|
||||
{"id": 4, "role": "tool", "content": "tool output", "timestamp": 130.0},
|
||||
{"id": 5, "role": "user", "content": "second prompt", "timestamp": 140.0},
|
||||
{"id": 6, "role": "assistant", "content": "final answer", "timestamp": 150.0},
|
||||
]
|
||||
return _PeekDB(session, messages)
|
||||
|
||||
|
||||
def test_session_peek_returns_metadata_and_head_tail(monkeypatch):
|
||||
monkeypatch.setattr(server, "_get_db", _peek_db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{
|
||||
"id": "1",
|
||||
"method": "session.peek",
|
||||
"params": {"session_id": "sess-1", "head": 1, "tail": 2},
|
||||
}
|
||||
)
|
||||
|
||||
result = resp["result"]
|
||||
meta = result["session"]
|
||||
assert meta == {
|
||||
"id": "sess-1",
|
||||
"title": "picker demo",
|
||||
"source": "tui",
|
||||
"model": "nous/hermes-4",
|
||||
"cwd": "/home/u/proj",
|
||||
"started_at": 100.0,
|
||||
"ended_at": 400.0,
|
||||
"end_reason": "tui_shutdown",
|
||||
"message_count": 6,
|
||||
"last_active": 150.0, # last message timestamp, not ended_at
|
||||
"cost_usd": 0.42, # estimated fallback when actual is None
|
||||
}
|
||||
# Only displayable (user/assistant) messages, system/tool rows skipped.
|
||||
assert [(m["role"], m["content"]) for m in result["head"]] == [
|
||||
("user", "first prompt")
|
||||
]
|
||||
assert [(m["role"], m["content"]) for m in result["tail"]] == [
|
||||
("user", "second prompt"),
|
||||
("assistant", "final answer"),
|
||||
]
|
||||
assert result["total_messages"] == 4
|
||||
assert all(m["truncated"] is False for m in result["head"] + result["tail"])
|
||||
|
||||
|
||||
def test_session_peek_head_tail_never_overlap(monkeypatch):
|
||||
db = _peek_db()
|
||||
db.messages = db.messages[:3] # system + user + assistant
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{
|
||||
"id": "1",
|
||||
"method": "session.peek",
|
||||
"params": {"session_id": "sess-1", "head": 2, "tail": 2},
|
||||
}
|
||||
)
|
||||
|
||||
result = resp["result"]
|
||||
head_ids = [m["id"] for m in result["head"]]
|
||||
tail_ids = [m["id"] for m in result["tail"]]
|
||||
assert head_ids == [2, 3]
|
||||
assert tail_ids == [] # both displayable rows already consumed by head
|
||||
assert result["total_messages"] == 2
|
||||
|
||||
|
||||
def test_session_peek_does_not_build_an_agent(monkeypatch):
|
||||
spawned = []
|
||||
monkeypatch.setattr(server, "_get_db", _peek_db)
|
||||
monkeypatch.setattr(
|
||||
server, "_make_agent", lambda *a, **kw: spawned.append("make") or None
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
server, "_start_agent_build", lambda *a, **kw: spawned.append("build")
|
||||
)
|
||||
before_sessions = dict(server._sessions)
|
||||
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.peek", "params": {"session_id": "sess-1"}}
|
||||
)
|
||||
|
||||
assert "result" in resp
|
||||
assert spawned == []
|
||||
assert server._sessions == before_sessions # no live session registered
|
||||
|
||||
|
||||
def test_session_peek_unknown_id_is_clean_error(monkeypatch):
|
||||
monkeypatch.setattr(server, "_get_db", _peek_db)
|
||||
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.peek", "params": {"session_id": "nope"}}
|
||||
)
|
||||
|
||||
assert resp["error"]["code"] == 4007
|
||||
assert resp["error"]["message"] == "session not found"
|
||||
|
||||
|
||||
def test_session_peek_requires_session_id(monkeypatch):
|
||||
monkeypatch.setattr(server, "_get_db", _peek_db)
|
||||
|
||||
resp = server.handle_request({"id": "1", "method": "session.peek", "params": {}})
|
||||
|
||||
assert resp["error"]["code"] == 4006
|
||||
|
||||
|
||||
def test_session_peek_db_unavailable(monkeypatch):
|
||||
monkeypatch.setattr(server, "_get_db", lambda: None)
|
||||
monkeypatch.setattr(server, "_db_error", "locked")
|
||||
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.peek", "params": {"session_id": "sess-1"}}
|
||||
)
|
||||
|
||||
assert resp["error"]["code"] == 5046
|
||||
assert "state.db unavailable" in resp["error"]["message"]
|
||||
|
||||
|
||||
# ── browser.manage ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -6928,6 +7487,8 @@ def test_notification_event_dedup_key_preserves_distinct_watch_matches():
|
||||
|
||||
def test_notification_poller_emits_distinct_watch_matches_once(monkeypatch):
|
||||
"""Distinct watch matches from one process emit; exact replay is deduped."""
|
||||
import queue as _queue_mod
|
||||
|
||||
from tools.process_registry import process_registry
|
||||
|
||||
turns = []
|
||||
@@ -6943,8 +7504,8 @@ def test_notification_poller_emits_distinct_watch_matches_once(monkeypatch):
|
||||
monkeypatch.setattr(server, "_emit", lambda *a, **kw: emitted.append(a))
|
||||
monkeypatch.setattr(server, "_run_prompt_submit", _fake_run_prompt_submit)
|
||||
|
||||
while not process_registry.completion_queue.empty():
|
||||
process_registry.completion_queue.get_nowait()
|
||||
isolated_queue: _queue_mod.Queue = _queue_mod.Queue()
|
||||
monkeypatch.setattr(process_registry, "completion_queue", isolated_queue)
|
||||
|
||||
base = {
|
||||
"type": "watch_match",
|
||||
@@ -6954,9 +7515,9 @@ def test_notification_poller_emits_distinct_watch_matches_once(monkeypatch):
|
||||
"output": "READY on port 8000",
|
||||
"suppressed": 0,
|
||||
}
|
||||
process_registry.completion_queue.put(base)
|
||||
process_registry.completion_queue.put({**base, "output": "READY on port 9000"})
|
||||
process_registry.completion_queue.put(dict(base))
|
||||
isolated_queue.put(base)
|
||||
isolated_queue.put({**base, "output": "READY on port 9000"})
|
||||
isolated_queue.put(dict(base))
|
||||
|
||||
stop = threading.Event()
|
||||
stop.set()
|
||||
@@ -7209,6 +7770,84 @@ def test_sniff_image_ext_magic_and_filename():
|
||||
assert server._sniff_image_ext(b"\x89PNG", "photo.jpeg") == ".jpeg"
|
||||
|
||||
|
||||
def test_tool_complete_derives_error_from_result_convention(monkeypatch):
|
||||
# The repo convention (agent.display._result_succeeded): a JSON-object
|
||||
# result with a truthy string "error" — or success:false — IS a failure.
|
||||
# The gateway surfaces it as payload["error"] so clients (OpenTUI ✗ state,
|
||||
# Ink trail ✗) don't sniff the convention themselves.
|
||||
events: list[tuple[str, str, dict]] = []
|
||||
monkeypatch.setattr(
|
||||
server, "_emit", lambda event_type, sid, payload: events.append((event_type, sid, payload))
|
||||
)
|
||||
monkeypatch.setitem(
|
||||
server._sessions,
|
||||
"err-test",
|
||||
{"tool_progress_mode": "verbose", "tool_started_at": {}, "edit_snapshots": {}},
|
||||
)
|
||||
|
||||
# 1) error-string result → flattened, capped error on the payload
|
||||
server._on_tool_complete(
|
||||
"err-test", "t1", "read_file", {"path": "/nope"},
|
||||
json.dumps({"error": "File not found:\n /nope"}),
|
||||
)
|
||||
assert events[-1][2]["error"] == "File not found: /nope"
|
||||
|
||||
# 2) success:false without error string → generic failure marker
|
||||
server._on_tool_complete(
|
||||
"err-test", "t2", "patch", {"path": "x"}, json.dumps({"success": False})
|
||||
)
|
||||
assert events[-1][2]["error"] == "tool reported failure"
|
||||
|
||||
# 3) plain-text result → NEVER a failure
|
||||
server._on_tool_complete("err-test", "t3", "terminal", {"command": "ls"}, "file-a\nfile-b")
|
||||
assert "error" not in events[-1][2]
|
||||
|
||||
# 4) JSON result with error: null / no error key → success
|
||||
server._on_tool_complete(
|
||||
"err-test", "t4", "web_search", {"q": "x"},
|
||||
json.dumps({"results": [1, 2], "error": None}),
|
||||
)
|
||||
assert "error" not in events[-1][2]
|
||||
|
||||
|
||||
def test_session_list_reports_scan_cap_truncation(monkeypatch):
|
||||
# The bounded multi-source scan must say so when the 10k safety cap stops
|
||||
# it with the requested window unfilled — an empty page past the cap is
|
||||
# "truncated", not "no more sessions".
|
||||
class _DenyAllDB:
|
||||
def __init__(self):
|
||||
self.offsets = []
|
||||
|
||||
def list_sessions_rich(self, source=None, limit=20, offset=0, **kw):
|
||||
self.offsets.append(offset)
|
||||
return [
|
||||
{"id": f"s{offset}-{i}", "source": "tool", "title": "", "preview": "",
|
||||
"started_at": 1, "message_count": 1}
|
||||
for i in range(limit)
|
||||
]
|
||||
|
||||
db = _DenyAllDB()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
resp = server.handle_request(
|
||||
{"id": "1", "method": "session.list", "params": {"sources": ["tui", "cli"], "limit": 5}}
|
||||
)
|
||||
result = resp["result"]
|
||||
assert result["sessions"] == []
|
||||
assert result["truncated"] is True
|
||||
assert max(db.offsets) <= 10_000
|
||||
|
||||
|
||||
def test_session_list_truncated_false_on_normal_paths(monkeypatch):
|
||||
db = _picker_db()
|
||||
monkeypatch.setattr(server, "_get_db", lambda: db)
|
||||
legacy = server.handle_request({"id": "1", "method": "session.list", "params": {}})
|
||||
assert legacy["result"]["truncated"] is False
|
||||
filtered = server.handle_request(
|
||||
{"id": "2", "method": "session.list", "params": {"sources": ["tui"], "limit": 5}}
|
||||
)
|
||||
assert filtered["result"]["truncated"] is False
|
||||
|
||||
|
||||
def test_slash_worker_close_reaps_zombie_and_closes_fds():
|
||||
"""A hung worker is SIGKILLed, the zombie reaped, all pipes closed — once."""
|
||||
calls = {k: 0 for k in ("terminate", "kill", "wait", "stdin", "stdout", "stderr")}
|
||||
@@ -7483,3 +8122,51 @@ def test_reap_idle_sessions_closes_only_evictable(monkeypatch):
|
||||
assert closed == [("stale", "idle_timeout")]
|
||||
finally:
|
||||
server._sessions.clear()
|
||||
|
||||
|
||||
class TestProcessCompletionCard:
|
||||
"""_emit_process_completion_card surfaces a background-process completion to
|
||||
the TUI as a notification.show card (Option B, glitch 2026-06-14) — in
|
||||
addition to the agent turn the completion triggers."""
|
||||
|
||||
@staticmethod
|
||||
def _capture(monkeypatch):
|
||||
emitted: list = []
|
||||
monkeypatch.setattr(server, "_emit", lambda event, sid, payload=None: emitted.append((event, sid, payload)))
|
||||
return emitted
|
||||
|
||||
def test_completion_exit_zero_is_an_info_card(self, monkeypatch):
|
||||
emitted = self._capture(monkeypatch)
|
||||
server._emit_process_completion_card(
|
||||
"s1", {"type": "completion", "session_id": "proc_1", "command": "sleep 20 && echo hi", "exit_code": 0}
|
||||
)
|
||||
assert len(emitted) == 1
|
||||
event, sid, payload = emitted[0]
|
||||
assert event == "notification.show"
|
||||
assert sid == "s1"
|
||||
assert payload["text"] == "sleep 20 && echo hi exited 0"
|
||||
assert payload["level"] == "info"
|
||||
assert payload["kind"] == "process.complete"
|
||||
assert payload["key"] == "proc:proc_1"
|
||||
|
||||
def test_nonzero_exit_is_a_warn_card(self, monkeypatch):
|
||||
emitted = self._capture(monkeypatch)
|
||||
server._emit_process_completion_card("s1", {"type": "completion", "command": "build", "exit_code": 1, "session_id": "p2"})
|
||||
assert emitted[0][2]["level"] == "warn"
|
||||
assert emitted[0][2]["text"] == "build exited 1"
|
||||
|
||||
def test_watch_match_is_not_carded(self, monkeypatch):
|
||||
emitted = self._capture(monkeypatch)
|
||||
server._emit_process_completion_card("s1", {"type": "watch_match", "command": "tail -f log"})
|
||||
assert emitted == []
|
||||
|
||||
def test_long_command_is_truncated(self, monkeypatch):
|
||||
emitted = self._capture(monkeypatch)
|
||||
server._emit_process_completion_card("s1", {"type": "completion", "command": "x" * 100, "exit_code": 0, "session_id": "p3"})
|
||||
assert "…" in emitted[0][2]["text"]
|
||||
assert len(emitted[0][2]["text"]) < 80
|
||||
|
||||
def test_missing_exit_code_says_finished(self, monkeypatch):
|
||||
emitted = self._capture(monkeypatch)
|
||||
server._emit_process_completion_card("s1", {"type": "completion", "command": "daemon", "session_id": "p4"})
|
||||
assert emitted[0][2]["text"] == "daemon finished"
|
||||
|
||||
@@ -957,3 +957,74 @@ class TestPinnedGuard:
|
||||
side_effect=RuntimeError("sidecar broken")):
|
||||
result = _delete_skill("my-skill")
|
||||
assert result["success"] is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _delete_skill — recursive-delete safety (port of Kilo Code #11240)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestDeleteSkillRmtreeGuard:
|
||||
"""Defense-in-depth before ``shutil.rmtree`` in ``_delete_skill``.
|
||||
|
||||
Mirrors the Kilo Code #11227 fix: never let a recursive skill delete
|
||||
escape the skills tree, target a skills root, or follow a symlink.
|
||||
"""
|
||||
|
||||
def test_normal_delete_still_works(self, tmp_path):
|
||||
with _skill_dir(tmp_path):
|
||||
_create_skill("good-skill", VALID_SKILL_CONTENT)
|
||||
result = _delete_skill("good-skill", absorbed_into="")
|
||||
assert result["success"] is True, result
|
||||
assert not (tmp_path / "good-skill").exists()
|
||||
|
||||
def test_symlinked_skill_dir_refused(self, tmp_path):
|
||||
"""A skill dir that is a symlink must not be rmtree'd — rmtree would
|
||||
otherwise follow it and delete the link target's contents."""
|
||||
victim = tmp_path.parent / "precious_victim"
|
||||
victim.mkdir()
|
||||
(victim / "important.txt").write_text("DO NOT DELETE")
|
||||
skills = tmp_path / "skills"
|
||||
skills.mkdir()
|
||||
evil = skills / "evil-skill"
|
||||
evil.symlink_to(victim, target_is_directory=True)
|
||||
try:
|
||||
with patch("tools.skill_manager_tool.SKILLS_DIR", skills), \
|
||||
patch("agent.skill_utils.get_all_skills_dirs", return_value=[skills]), \
|
||||
patch("tools.skill_manager_tool._find_skill",
|
||||
return_value={"path": evil}):
|
||||
result = _delete_skill("evil-skill", absorbed_into="")
|
||||
assert result["success"] is False
|
||||
assert "symlink" in result["error"].lower()
|
||||
assert (victim / "important.txt").exists()
|
||||
finally:
|
||||
import shutil as _sh
|
||||
_sh.rmtree(victim, ignore_errors=True)
|
||||
|
||||
def test_skills_root_itself_refused(self, tmp_path):
|
||||
"""If discovery ever hands back the skills root, refuse — rmtree would
|
||||
wipe every installed skill."""
|
||||
with patch("tools.skill_manager_tool.SKILLS_DIR", tmp_path), \
|
||||
patch("agent.skill_utils.get_all_skills_dirs", return_value=[tmp_path]), \
|
||||
patch("tools.skill_manager_tool._find_skill",
|
||||
return_value={"path": tmp_path}):
|
||||
result = _delete_skill("root-attack", absorbed_into="")
|
||||
assert result["success"] is False
|
||||
assert "skills root" in result["error"].lower()
|
||||
assert tmp_path.exists()
|
||||
|
||||
def test_out_of_tree_path_refused(self, tmp_path):
|
||||
"""A path that resolves outside every known skills root is refused."""
|
||||
skills = tmp_path / "skills"
|
||||
skills.mkdir()
|
||||
outside = tmp_path / "outside_skill"
|
||||
outside.mkdir()
|
||||
(outside / "SKILL.md").write_text("x")
|
||||
with patch("tools.skill_manager_tool.SKILLS_DIR", skills), \
|
||||
patch("agent.skill_utils.get_all_skills_dirs", return_value=[skills]), \
|
||||
patch("tools.skill_manager_tool._find_skill",
|
||||
return_value={"path": outside}):
|
||||
result = _delete_skill("outside", absorbed_into="")
|
||||
assert result["success"] is False
|
||||
assert "skills root" in result["error"].lower()
|
||||
assert outside.exists()
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user