opentui(phase3): launcher integration — HERMES_TUI_ENGINE dual-engine
hermes --tui launches the native OpenTUI engine (Bun) when HERMES_TUI_ENGINE=opentui (env) or display.tui_engine=opentui (config); Ink stays the default and the shipping path is untouched. - _resolve_tui_engine() (env > config > ink); refuses opentui on Windows/Termux (no Bun) -> falls back to ink with a notice. - _make_opentui_argv() -> [bun, src/entry.real.tsx] (no build step). - _bun_bin() with HERMES_BUN override. - Branch at top of _make_tui_argv BEFORE _ensure_tui_node (Bun-only host must not bootstrap Node). - Gate _launch_tui NODE_OPTIONS/--max-old-space-size on engine==ink (Bun is JSC; the V8 flag errors/ignores). Verified end-to-end via tmux: real hermes --tui -> Bun -> OpenTUI -> real Python gateway streamed a real reply. No-flag default still ink.
This commit is contained in:
+25
-205
@@ -190,8 +190,6 @@ DEFAULT_XAI_BASE_URL = "https://api.x.ai/v1"
|
||||
DEFAULT_GEMINI_TTS_MODEL = "gemini-2.5-flash-preview-tts"
|
||||
DEFAULT_GEMINI_TTS_VOICE = "Kore"
|
||||
DEFAULT_GEMINI_TTS_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"
|
||||
DEFAULT_GEMINI_AUDIO_TAGS = False
|
||||
GEMINI_AUDIO_TAG_REWRITE_TASK = "tts_audio_tags"
|
||||
# PCM output specs for Gemini TTS (fixed by the API)
|
||||
GEMINI_TTS_SAMPLE_RATE = 24000
|
||||
GEMINI_TTS_CHANNELS = 1
|
||||
@@ -206,8 +204,8 @@ DEFAULT_OUTPUT_DIR = _get_default_output_dir()
|
||||
# ---------------------------------------------------------------------------
|
||||
# Per-provider input-character limits (from official provider docs).
|
||||
# A single global cap was wrong: OpenAI is 4096, xAI is 15k, MiniMax is 10k,
|
||||
# ElevenLabs is model-dependent (5k / 10k / 30k / 40k), Gemini has a 32k-token
|
||||
# context window. Users can override any of these via
|
||||
# ElevenLabs is model-dependent (5k / 10k / 30k / 40k), Gemini caps at ~8k
|
||||
# input tokens. Users can override any of these via
|
||||
# ``tts.<provider>.max_text_length`` in config.yaml.
|
||||
# ---------------------------------------------------------------------------
|
||||
PROVIDER_MAX_TEXT_LENGTH: Dict[str, int] = {
|
||||
@@ -216,7 +214,7 @@ PROVIDER_MAX_TEXT_LENGTH: Dict[str, int] = {
|
||||
"xai": 15000, # https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
||||
"minimax": 10000, # https://platform.minimax.io/docs/api-reference/speech-t2a-http (sync)
|
||||
"mistral": 4000, # conservative; no published per-request cap
|
||||
"gemini": 32000, # Gemini TTS has a 32k-token context window; char cap is conservative
|
||||
"gemini": 5000, # Gemini TTS caps at ~8k input tokens / ~655s audio
|
||||
"elevenlabs": 10000, # fallback when model-aware lookup can't resolve (multilingual_v2)
|
||||
"neutts": 2000, # local model, quality falls off on long text
|
||||
"kittentts": 2000, # local 25MB model
|
||||
@@ -235,23 +233,6 @@ ELEVENLABS_MODEL_MAX_TEXT_LENGTH: Dict[str, int] = {
|
||||
"eleven_flash_v2_5": 40000,
|
||||
}
|
||||
|
||||
|
||||
def _config_bool(value: Any, default: bool = False) -> bool:
|
||||
"""Coerce common YAML/env bool spellings without treating random strings as true."""
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
if value is None:
|
||||
return default
|
||||
if isinstance(value, (int, float)):
|
||||
return bool(value)
|
||||
if isinstance(value, str):
|
||||
normalized = value.strip().lower()
|
||||
if normalized in {"1", "true", "yes", "on", "enabled"}:
|
||||
return True
|
||||
if normalized in {"0", "false", "no", "off", "disabled"}:
|
||||
return False
|
||||
return default
|
||||
|
||||
# Final fallback when provider isn't recognised at all.
|
||||
FALLBACK_MAX_TEXT_LENGTH = 4000
|
||||
|
||||
@@ -712,7 +693,6 @@ def _terminate_command_tts_process_tree(proc: subprocess.Popen) -> None:
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
timeout=5,
|
||||
stdin=subprocess.DEVNULL,
|
||||
)
|
||||
except Exception:
|
||||
proc.kill()
|
||||
@@ -765,7 +745,7 @@ def _run_command_tts(command: str, timeout: float) -> subprocess.CompletedProces
|
||||
else:
|
||||
popen_kwargs["start_new_session"] = True
|
||||
|
||||
proc = subprocess.Popen(command, **popen_kwargs, stdin=subprocess.DEVNULL)
|
||||
proc = subprocess.Popen(command, **popen_kwargs)
|
||||
try:
|
||||
stdout, stderr = proc.communicate(timeout=timeout)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
@@ -902,7 +882,6 @@ def _convert_to_opus(mp3_path: str) -> Optional[str]:
|
||||
["ffmpeg", "-i", mp3_path, "-acodec", "libopus",
|
||||
"-ac", "1", "-b:a", "64k", "-vbr", "off", ogg_path, "-y"],
|
||||
capture_output=True, timeout=30,
|
||||
stdin=subprocess.DEVNULL,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
logger.warning("ffmpeg conversion failed with return code %d: %s",
|
||||
@@ -1088,7 +1067,20 @@ _XAI_FIRST_SENTENCE_RE = re.compile(r"^(.{12,120}?[.!?…])\s+(?=\S)", flags=re.
|
||||
|
||||
|
||||
def _xai_bool_config(value: Any, default: bool = False) -> bool:
|
||||
return _config_bool(value, default=default)
|
||||
"""Coerce common YAML/env bool spellings without treating random strings as true."""
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
if value is None:
|
||||
return default
|
||||
if isinstance(value, (int, float)):
|
||||
return bool(value)
|
||||
if isinstance(value, str):
|
||||
normalized = value.strip().lower()
|
||||
if normalized in {"1", "true", "yes", "on", "enabled"}:
|
||||
return True
|
||||
if normalized in {"0", "false", "no", "off", "disabled"}:
|
||||
return False
|
||||
return default
|
||||
|
||||
|
||||
def _apply_xai_auto_speech_tags(text: str) -> str:
|
||||
@@ -1400,160 +1392,6 @@ def _wrap_pcm_as_wav(
|
||||
return riff_header + fmt_chunk + data_chunk_header + pcm_bytes
|
||||
|
||||
|
||||
def _resolve_gemini_persona_prompt_path(gemini_config: Dict[str, Any]) -> Optional[Path]:
|
||||
"""Return the configured persona prompt file path, if any."""
|
||||
raw = gemini_config.get("persona_prompt_file")
|
||||
if not isinstance(raw, str) or not raw.strip():
|
||||
return None
|
||||
|
||||
expanded = os.path.expandvars(raw.strip())
|
||||
path = Path(expanded).expanduser()
|
||||
if not path.is_absolute():
|
||||
try:
|
||||
from hermes_constants import get_hermes_home
|
||||
path = get_hermes_home() / path
|
||||
except Exception:
|
||||
path = Path.cwd() / path
|
||||
return path
|
||||
|
||||
|
||||
def _read_gemini_persona_prompt(gemini_config: Dict[str, Any]) -> str:
|
||||
"""Read the Gemini persona prompt file, failing soft on config mistakes."""
|
||||
path = _resolve_gemini_persona_prompt_path(gemini_config)
|
||||
if path is None:
|
||||
return ""
|
||||
try:
|
||||
return path.read_text(encoding="utf-8").strip()
|
||||
except (OSError, UnicodeDecodeError) as exc:
|
||||
logger.warning(
|
||||
"Gemini TTS persona prompt file unavailable at %s: %s",
|
||||
path,
|
||||
exc,
|
||||
)
|
||||
return ""
|
||||
|
||||
|
||||
def _gemini_model_supports_audio_tags(model: str) -> bool:
|
||||
"""Return True for Gemini TTS models known to support expressive audio tags."""
|
||||
normalized = (model or "").strip().lower().rsplit("/", 1)[-1]
|
||||
return "gemini-3.1" in normalized and "tts" in normalized
|
||||
|
||||
|
||||
def _gemini_audio_tags_enabled(gemini_config: Dict[str, Any], model: str) -> bool:
|
||||
raw = gemini_config.get("audio_tags")
|
||||
if isinstance(raw, dict):
|
||||
raw = raw.get("enabled")
|
||||
enabled = _config_bool(raw, default=DEFAULT_GEMINI_AUDIO_TAGS)
|
||||
if not enabled:
|
||||
return False
|
||||
if not _gemini_model_supports_audio_tags(model):
|
||||
logger.warning(
|
||||
"Gemini TTS audio_tags enabled, but model %s is not known to support "
|
||||
"Gemini audio tags; skipping hidden tag rewrite",
|
||||
model,
|
||||
)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _clean_gemini_audio_tag_rewrite(content: str) -> str:
|
||||
clean = (content or "").strip()
|
||||
fence = re.fullmatch(r"```(?:[A-Za-z0-9_-]+)?\s*(.*?)\s*```", clean, flags=re.DOTALL)
|
||||
if fence:
|
||||
clean = fence.group(1).strip()
|
||||
return clean
|
||||
|
||||
|
||||
def _extract_auxiliary_message_content(response: Any) -> str:
|
||||
try:
|
||||
choice = response.choices[0]
|
||||
message = getattr(choice, "message", None)
|
||||
if isinstance(message, dict):
|
||||
return str(message.get("content") or "")
|
||||
return str(getattr(message, "content", "") or "")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _rewrite_gemini_tts_audio_tags(text: str, persona_prompt: str = "") -> str:
|
||||
"""Use the configured auxiliary model to insert Gemini audio tags."""
|
||||
transcript = text.strip()
|
||||
if not transcript:
|
||||
return text
|
||||
|
||||
system_prompt = (
|
||||
"You rewrite transcripts for Gemini 3.1 Flash TTS by inserting expressive "
|
||||
"audio tags.\n\n"
|
||||
"Audio tags are inline square-bracket modifiers such as [whispers], "
|
||||
"[excitedly], [very slow], [sarcastically], [laughs], [sighs], or [gasp]. "
|
||||
"There is no fixed allowlist. Use creative freeform tags generously but "
|
||||
"naturally to control tone, pace, emotional vibe, emphasis, section-level "
|
||||
"delivery, and non-verbal sounds. Use English audio tags even when the "
|
||||
"spoken transcript is not English.\n\n"
|
||||
"Rules:\n"
|
||||
"- Preserve the spoken words, order, and meaning.\n"
|
||||
"- Do not add new spoken sentences or remove existing spoken words.\n"
|
||||
"- Use square brackets for every audio tag.\n"
|
||||
"- Do not use SSML or XML tags.\n"
|
||||
"- Do not explain or comment.\n"
|
||||
"- Return only the tagged TTS script."
|
||||
)
|
||||
context = persona_prompt.strip() or "(none)"
|
||||
user_prompt = (
|
||||
"PERSONA AND DIRECTOR CONTEXT:\n"
|
||||
f"{context}\n\n"
|
||||
"TRANSCRIPT TO TAG:\n"
|
||||
f"{transcript}"
|
||||
)
|
||||
try:
|
||||
from agent.auxiliary_client import call_llm
|
||||
|
||||
response = call_llm(
|
||||
task=GEMINI_AUDIO_TAG_REWRITE_TASK,
|
||||
messages=[
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": user_prompt},
|
||||
],
|
||||
temperature=0.7,
|
||||
)
|
||||
tagged = _clean_gemini_audio_tag_rewrite(_extract_auxiliary_message_content(response))
|
||||
return tagged or text
|
||||
except Exception as exc:
|
||||
logger.warning("Gemini TTS audio tag rewrite failed; using untagged text: %s", exc)
|
||||
return text
|
||||
|
||||
|
||||
def _compose_gemini_tts_prompt(
|
||||
text: str,
|
||||
gemini_config: Dict[str, Any],
|
||||
persona_prompt: Optional[str] = None,
|
||||
) -> str:
|
||||
"""Build the Gemini prompt from persona direction plus the live transcript."""
|
||||
transcript = text.strip()
|
||||
if persona_prompt is None:
|
||||
persona_prompt = _read_gemini_persona_prompt(gemini_config)
|
||||
if not persona_prompt:
|
||||
return transcript
|
||||
|
||||
preamble = (
|
||||
"Synthesize speech from the TRANSCRIPT only. Treat AUDIO PROFILE, "
|
||||
"SCENE, DIRECTOR'S NOTES, and SAMPLE CONTEXT as performance direction; "
|
||||
"do not speak those sections aloud."
|
||||
)
|
||||
|
||||
placeholder_patterns = (
|
||||
re.compile(r"\{\{\s*transcript\s*\}\}", flags=re.IGNORECASE),
|
||||
re.compile(r"\{\s*transcript\s*\}", flags=re.IGNORECASE),
|
||||
)
|
||||
prompt = persona_prompt
|
||||
for pattern in placeholder_patterns:
|
||||
if pattern.search(prompt):
|
||||
prompt = pattern.sub(transcript, prompt)
|
||||
return f"{preamble}\n\n{prompt}".strip()
|
||||
|
||||
return f"{preamble}\n\n{persona_prompt}\n\n#### TRANSCRIPT\n{transcript}".strip()
|
||||
|
||||
|
||||
def _generate_gemini_tts(text: str, output_path: str, tts_config: Dict[str, Any]) -> str:
|
||||
"""Generate audio using Google Gemini TTS.
|
||||
|
||||
@@ -1579,8 +1417,7 @@ def _generate_gemini_tts(text: str, output_path: str, tts_config: Dict[str, Any]
|
||||
"GEMINI_API_KEY not set. Get one at https://aistudio.google.com/app/apikey"
|
||||
)
|
||||
|
||||
raw_gemini_config = tts_config.get("gemini", {})
|
||||
gemini_config = raw_gemini_config if isinstance(raw_gemini_config, dict) else {}
|
||||
gemini_config = tts_config.get("gemini", {})
|
||||
model = str(gemini_config.get("model", DEFAULT_GEMINI_TTS_MODEL)).strip() or DEFAULT_GEMINI_TTS_MODEL
|
||||
voice = str(gemini_config.get("voice", DEFAULT_GEMINI_TTS_VOICE)).strip() or DEFAULT_GEMINI_TTS_VOICE
|
||||
base_url = str(
|
||||
@@ -1588,25 +1425,9 @@ def _generate_gemini_tts(text: str, output_path: str, tts_config: Dict[str, Any]
|
||||
or get_env_value("GEMINI_BASE_URL")
|
||||
or DEFAULT_GEMINI_TTS_BASE_URL
|
||||
).strip().rstrip("/")
|
||||
persona_prompt = _read_gemini_persona_prompt(gemini_config)
|
||||
tts_script = text
|
||||
if _gemini_audio_tags_enabled(gemini_config, model):
|
||||
tts_script = _rewrite_gemini_tts_audio_tags(text, persona_prompt=persona_prompt)
|
||||
prompt_text = _compose_gemini_tts_prompt(
|
||||
tts_script,
|
||||
gemini_config,
|
||||
persona_prompt=persona_prompt,
|
||||
)
|
||||
max_len = _resolve_max_text_length("gemini", tts_config)
|
||||
if len(prompt_text) > max_len:
|
||||
logger.warning(
|
||||
"Gemini TTS composed prompt too long (%d chars), truncating to %d",
|
||||
len(prompt_text), max_len,
|
||||
)
|
||||
prompt_text = prompt_text[:max_len]
|
||||
|
||||
payload: Dict[str, Any] = {
|
||||
"contents": [{"parts": [{"text": prompt_text}]}],
|
||||
"contents": [{"parts": [{"text": text}]}],
|
||||
"generationConfig": {
|
||||
"responseModalities": ["AUDIO"],
|
||||
"speechConfig": {
|
||||
@@ -1683,7 +1504,7 @@ def _generate_gemini_tts(text: str, output_path: str, tts_config: Dict[str, Any]
|
||||
]
|
||||
else:
|
||||
cmd = [ffmpeg, "-i", wav_path, "-y", "-loglevel", "error", output_path]
|
||||
result = subprocess.run(cmd, capture_output=True, timeout=30, stdin=subprocess.DEVNULL)
|
||||
result = subprocess.run(cmd, capture_output=True, timeout=30)
|
||||
if result.returncode != 0:
|
||||
stderr = result.stderr.decode("utf-8", errors="ignore")[:300]
|
||||
raise RuntimeError(f"ffmpeg conversion failed: {stderr}")
|
||||
@@ -1766,7 +1587,7 @@ def _generate_neutts(text: str, output_path: str, tts_config: Dict[str, Any]) ->
|
||||
"--device", device,
|
||||
]
|
||||
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, timeout=120, stdin=subprocess.DEVNULL)
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
|
||||
if result.returncode != 0:
|
||||
stderr = result.stderr.strip()
|
||||
# Filter out the "OK:" line from stderr
|
||||
@@ -1778,7 +1599,7 @@ def _generate_neutts(text: str, output_path: str, tts_config: Dict[str, Any]) ->
|
||||
ffmpeg = shutil.which("ffmpeg")
|
||||
if ffmpeg:
|
||||
conv_cmd = [ffmpeg, "-i", wav_path, "-y", "-loglevel", "error", output_path]
|
||||
subprocess.run(conv_cmd, check=True, timeout=30, stdin=subprocess.DEVNULL)
|
||||
subprocess.run(conv_cmd, check=True, timeout=30)
|
||||
os.remove(wav_path)
|
||||
else:
|
||||
# No ffmpeg — just rename the WAV to the expected path
|
||||
@@ -1849,7 +1670,6 @@ def _resolve_piper_voice_path(voice: str, download_dir: Path) -> str:
|
||||
[_sys.executable, "-m", "piper.download_voices", voice,
|
||||
"--download-dir", str(download_dir)],
|
||||
capture_output=True, text=True, timeout=300,
|
||||
stdin=subprocess.DEVNULL,
|
||||
)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
raise RuntimeError(
|
||||
@@ -1937,7 +1757,7 @@ def _generate_piper_tts(text: str, output_path: str, tts_config: Dict[str, Any])
|
||||
ffmpeg = shutil.which("ffmpeg")
|
||||
if ffmpeg:
|
||||
conv_cmd = [ffmpeg, "-i", wav_path, "-y", "-loglevel", "error", output_path]
|
||||
subprocess.run(conv_cmd, check=True, timeout=30, stdin=subprocess.DEVNULL)
|
||||
subprocess.run(conv_cmd, check=True, timeout=30)
|
||||
try:
|
||||
os.remove(wav_path)
|
||||
except OSError:
|
||||
@@ -2003,7 +1823,7 @@ def _generate_kittentts(text: str, output_path: str, tts_config: Dict[str, Any])
|
||||
ffmpeg = shutil.which("ffmpeg")
|
||||
if ffmpeg:
|
||||
conv_cmd = [ffmpeg, "-i", wav_path, "-y", "-loglevel", "error", output_path]
|
||||
subprocess.run(conv_cmd, check=True, timeout=30, stdin=subprocess.DEVNULL)
|
||||
subprocess.run(conv_cmd, check=True, timeout=30)
|
||||
os.remove(wav_path)
|
||||
else:
|
||||
# No ffmpeg — rename the WAV to the expected path
|
||||
|
||||
Reference in New Issue
Block a user