Merge commit '6110aed9b' into feat/whatsapp-cloud-api
This commit is contained in:
+63
-10
@@ -45,6 +45,28 @@ _jobs_file_lock = threading.Lock()
|
||||
OUTPUT_DIR = CRON_DIR / "output"
|
||||
ONESHOT_GRACE_SECONDS = 120
|
||||
|
||||
# Fields on a cron job that must never change after creation. ``id`` is used
|
||||
# as a filesystem path component under ``OUTPUT_DIR``; allowing it to be
|
||||
# updated lets an unsafe value (``../escape``, absolute path, nested) leak
|
||||
# into output writes/deletes.
|
||||
_IMMUTABLE_JOB_FIELDS = frozenset({"id"})
|
||||
|
||||
|
||||
def _job_output_dir(job_id: str) -> Path:
|
||||
"""Resolve a job's output directory, rejecting any path-escape attempt.
|
||||
|
||||
Job IDs are filesystem path components under ``OUTPUT_DIR``. A legacy or
|
||||
crafted ID containing ``..``, absolute paths, or nested separators would
|
||||
allow output writes/deletes to escape the cron output sandbox. Reject
|
||||
anything that isn't a single safe path component.
|
||||
"""
|
||||
text = str(job_id or "").strip()
|
||||
if not text or text in {".", ".."} or "/" in text or "\\" in text:
|
||||
raise ValueError(f"Invalid cron job id for output path: {job_id!r}")
|
||||
if Path(text).is_absolute() or Path(text).drive:
|
||||
raise ValueError(f"Invalid cron job id for output path: {job_id!r}")
|
||||
return OUTPUT_DIR / text
|
||||
|
||||
|
||||
def _normalize_skill_list(skill: Optional[str] = None, skills: Optional[Any] = None) -> List[str]:
|
||||
"""Normalize legacy/single-skill and multi-skill inputs into a unique ordered list."""
|
||||
@@ -406,22 +428,18 @@ def load_jobs() -> List[Dict[str, Any]]:
|
||||
ensure_dirs()
|
||||
if not JOBS_FILE.exists():
|
||||
return []
|
||||
|
||||
|
||||
_strict_retry = False # track whether we used the strict=False fallback
|
||||
|
||||
try:
|
||||
with open(JOBS_FILE, 'r', encoding='utf-8') as f:
|
||||
data = json.load(f)
|
||||
return data.get("jobs", [])
|
||||
except json.JSONDecodeError:
|
||||
# Retry with strict=False to handle bare control chars in string values
|
||||
_strict_retry = True
|
||||
try:
|
||||
with open(JOBS_FILE, 'r', encoding='utf-8') as f:
|
||||
data = json.loads(f.read(), strict=False)
|
||||
jobs = data.get("jobs", [])
|
||||
if jobs:
|
||||
# Auto-repair: rewrite with proper escaping
|
||||
save_jobs(jobs)
|
||||
logger.warning("Auto-repaired jobs.json (had invalid control characters)")
|
||||
return jobs
|
||||
except Exception as e:
|
||||
logger.error("Failed to auto-repair jobs.json: %s", e)
|
||||
raise RuntimeError(f"Cron database corrupted and unrepairable: {e}") from e
|
||||
@@ -429,6 +447,29 @@ def load_jobs() -> List[Dict[str, Any]]:
|
||||
logger.error("IOError reading jobs.json: %s", e)
|
||||
raise RuntimeError(f"Failed to read cron database: {e}") from e
|
||||
|
||||
# Validate the top-level JSON shape: accept a dict (expected) or a bare
|
||||
# list (auto-repair). Anything else (str/number/null) is corruption that
|
||||
# would otherwise raise an uncaught AttributeError on ``.get()`` and take
|
||||
# down the whole cron subsystem.
|
||||
if isinstance(data, dict):
|
||||
jobs = data.get("jobs", [])
|
||||
if _strict_retry and jobs:
|
||||
# Hit control-character corruption — rewrite with proper escaping.
|
||||
save_jobs(jobs)
|
||||
logger.warning("Auto-repaired jobs.json (had invalid control characters)")
|
||||
return jobs
|
||||
if isinstance(data, list):
|
||||
# Bare array — likely saved/edited outside save_jobs(). Wrap it back
|
||||
# into the expected {"jobs": [...]} structure.
|
||||
if data:
|
||||
save_jobs(data)
|
||||
logger.warning("Auto-repaired jobs.json (bare list wrapped as dict)")
|
||||
return data
|
||||
|
||||
raise RuntimeError(
|
||||
f"Cron database corrupted: expected {{'jobs': [...]}}, got {type(data).__name__}"
|
||||
)
|
||||
|
||||
|
||||
def save_jobs(jobs: List[Dict[str, Any]]):
|
||||
"""Save all jobs to storage."""
|
||||
@@ -728,6 +769,15 @@ def list_jobs(include_disabled: bool = False) -> List[Dict[str, Any]]:
|
||||
|
||||
def update_job(job_id: str, updates: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
"""Update a job by ID, refreshing derived schedule fields when needed."""
|
||||
# Block mutation of immutable fields. ``id`` in particular is a filesystem
|
||||
# path component under OUTPUT_DIR — letting an update change it leaks
|
||||
# path-escape values into output writes/deletes.
|
||||
bad_fields = _IMMUTABLE_JOB_FIELDS.intersection(updates or {})
|
||||
if bad_fields:
|
||||
raise ValueError(
|
||||
f"Cron job field(s) cannot be updated: {', '.join(sorted(bad_fields))}"
|
||||
)
|
||||
|
||||
jobs = load_jobs()
|
||||
for i, job in enumerate(jobs):
|
||||
if job["id"] != job_id:
|
||||
@@ -845,9 +895,12 @@ def remove_job(job_id: str) -> bool:
|
||||
original_len = len(jobs)
|
||||
jobs = [j for j in jobs if j["id"] != canonical_id]
|
||||
if len(jobs) < original_len:
|
||||
# Resolve the output dir BEFORE saving so a legacy unsafe ID (e.g.
|
||||
# left over from before the create-time guard) fails closed without
|
||||
# half-applying the removal.
|
||||
job_output_dir = _job_output_dir(canonical_id)
|
||||
save_jobs(jobs)
|
||||
# Clean up output directory to prevent orphaned dirs accumulating
|
||||
job_output_dir = OUTPUT_DIR / canonical_id
|
||||
if job_output_dir.exists():
|
||||
shutil.rmtree(job_output_dir)
|
||||
return True
|
||||
@@ -1061,7 +1114,7 @@ def _get_due_jobs_locked() -> List[Dict[str, Any]]:
|
||||
def save_job_output(job_id: str, output: str):
|
||||
"""Save job output to file."""
|
||||
ensure_dirs()
|
||||
job_output_dir = OUTPUT_DIR / job_id
|
||||
job_output_dir = _job_output_dir(job_id)
|
||||
job_output_dir.mkdir(parents=True, exist_ok=True)
|
||||
_secure_dir(job_output_dir)
|
||||
|
||||
|
||||
+378
-40
@@ -9,6 +9,7 @@ runs at a time if multiple processes overlap.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import atexit
|
||||
import concurrent.futures
|
||||
import contextvars
|
||||
import json
|
||||
@@ -17,6 +18,7 @@ import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
from contextlib import contextmanager
|
||||
|
||||
# fcntl is Unix-only; on Windows use msvcrt for file locking
|
||||
@@ -57,6 +59,29 @@ class CronPromptInjectionBlocked(Exception):
|
||||
"""
|
||||
|
||||
|
||||
def _resolve_cron_disabled_toolsets(cfg: dict) -> list[str]:
|
||||
"""Toolsets a cron-spawned agent must never receive.
|
||||
|
||||
Three protected toolsets are always disabled in cron context:
|
||||
- ``cronjob`` — would let a cron-spawned agent schedule more cron jobs
|
||||
- ``messaging`` — interactive, needs a live gateway session
|
||||
- ``clarify`` — interactive, blocks waiting for user input
|
||||
|
||||
User-level ``agent.disabled_toolsets`` from config.yaml is layered on top
|
||||
so per-job ``enabled_toolsets`` cannot bypass policy that applies to
|
||||
ordinary agent runs (#25752 — LLM-supplied enabled_toolsets was widening
|
||||
past config.yaml's denylist).
|
||||
"""
|
||||
disabled = ["cronjob", "messaging", "clarify"]
|
||||
agent_cfg = (cfg or {}).get("agent") or {}
|
||||
user_disabled = agent_cfg.get("disabled_toolsets") or []
|
||||
for name in user_disabled:
|
||||
name = str(name).strip()
|
||||
if name and name not in disabled:
|
||||
disabled.append(name)
|
||||
return disabled
|
||||
|
||||
|
||||
def _resolve_cron_enabled_toolsets(job: dict, cfg: dict) -> list[str] | None:
|
||||
"""Resolve the toolset list for a cron job.
|
||||
|
||||
@@ -132,6 +157,69 @@ from cron.jobs import get_due_jobs, mark_job_run, save_job_output, advance_next_
|
||||
# locally for audit.
|
||||
SILENT_MARKER = "[SILENT]"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Persistent thread pool for parallel cron jobs.
|
||||
# The tick function submits jobs here and returns immediately so the ticker
|
||||
# thread is never blocked by long-running jobs (e.g. the fixer running 15+ min).
|
||||
# ---------------------------------------------------------------------------
|
||||
_parallel_pool: Optional[concurrent.futures.ThreadPoolExecutor] = None
|
||||
_parallel_pool_max_workers: Optional[int] = None
|
||||
_running_job_ids: set = set()
|
||||
_running_lock = threading.Lock()
|
||||
|
||||
# Sequential (env/context-mutating) cron jobs — workdir/profile jobs that touch
|
||||
# process-global runtime state — must run one at a time, but must NOT block the
|
||||
# ticker thread. A persistent single-thread executor preserves ordering across
|
||||
# ticks while keeping dispatch fire-and-forget, the same as the parallel pool.
|
||||
_sequential_pool: Optional[concurrent.futures.ThreadPoolExecutor] = None
|
||||
|
||||
|
||||
def _get_parallel_pool(max_workers: Optional[int]) -> concurrent.futures.ThreadPoolExecutor:
|
||||
"""Return (or create) the persistent parallel pool."""
|
||||
global _parallel_pool, _parallel_pool_max_workers
|
||||
if _parallel_pool is None or _parallel_pool_max_workers != max_workers:
|
||||
if _parallel_pool is not None:
|
||||
_parallel_pool.shutdown(wait=False, cancel_futures=False)
|
||||
_parallel_pool = concurrent.futures.ThreadPoolExecutor(
|
||||
max_workers=max_workers,
|
||||
thread_name_prefix="cron-parallel",
|
||||
)
|
||||
_parallel_pool_max_workers = max_workers
|
||||
return _parallel_pool
|
||||
|
||||
|
||||
def _get_sequential_pool() -> concurrent.futures.ThreadPoolExecutor:
|
||||
"""Return (or create) the persistent single-thread sequential pool.
|
||||
|
||||
A single worker guarantees env/context-mutating jobs never overlap, even
|
||||
across ticks: a job queued by a newer tick waits for the previous tick's
|
||||
sequential jobs to finish rather than corrupting their os.environ /
|
||||
profile state.
|
||||
"""
|
||||
global _sequential_pool
|
||||
if _sequential_pool is None:
|
||||
_sequential_pool = concurrent.futures.ThreadPoolExecutor(
|
||||
max_workers=1,
|
||||
thread_name_prefix="cron-seq",
|
||||
)
|
||||
return _sequential_pool
|
||||
|
||||
|
||||
def _shutdown_parallel_pool() -> None:
|
||||
"""Shut down the persistent pools on process exit."""
|
||||
global _parallel_pool, _parallel_pool_max_workers, _sequential_pool
|
||||
if _parallel_pool is not None:
|
||||
_parallel_pool.shutdown(wait=True, cancel_futures=False)
|
||||
_parallel_pool = None
|
||||
_parallel_pool_max_workers = None
|
||||
if _sequential_pool is not None:
|
||||
_sequential_pool.shutdown(wait=True, cancel_futures=False)
|
||||
_sequential_pool = None
|
||||
|
||||
|
||||
atexit.register(_shutdown_parallel_pool)
|
||||
|
||||
|
||||
# Backward-compatible module override used by tests and emergency monkeypatches.
|
||||
_hermes_home: Path | None = None
|
||||
|
||||
@@ -235,6 +323,30 @@ def _resolve_origin(job: dict) -> Optional[dict]:
|
||||
return None
|
||||
|
||||
|
||||
def _cron_job_origin_log_suffix(job: dict) -> str:
|
||||
"""Return safe provenance details for security warnings about a cron job.
|
||||
|
||||
The scheduler normally has no live HTTP request object when it detects a
|
||||
bad stored ``context_from`` reference. Including the job's saved origin
|
||||
makes future probe logs actionable without exposing secrets: platform/chat
|
||||
metadata for gateway-created jobs, and optional source-IP fields for API
|
||||
surfaces that persist them in origin metadata.
|
||||
"""
|
||||
origin = job.get("origin")
|
||||
if not isinstance(origin, dict):
|
||||
return ""
|
||||
|
||||
fields = []
|
||||
for key in ("platform", "chat_id", "thread_id", "source_ip", "remote", "forwarded_for"):
|
||||
value = origin.get(key)
|
||||
if value is None:
|
||||
continue
|
||||
text = str(value).replace("\r", " ").replace("\n", " ").strip()
|
||||
if text:
|
||||
fields.append(f"origin_{key}={text[:200]!r}")
|
||||
return " " + " ".join(fields) if fields else ""
|
||||
|
||||
|
||||
def _plugin_cron_env_var(platform_name: str) -> str:
|
||||
"""Return the cron home-channel env var registered by a plugin platform.
|
||||
|
||||
@@ -337,6 +449,47 @@ def _iter_home_target_platforms():
|
||||
pass
|
||||
|
||||
|
||||
def cron_delivery_targets() -> list[dict]:
|
||||
"""Return the platforms a cron job can auto-deliver to.
|
||||
|
||||
Single source of truth for any UI (dashboard dropdown, etc.) that lets a
|
||||
user pick a cron delivery target. A platform is included when it is a valid
|
||||
cron delivery platform AND its gateway is configured (enabled + credentials
|
||||
present). Each entry reports whether the platform's home target (the
|
||||
room/channel cron posts to) is set — a platform can be configured for
|
||||
interactive use but still lack the home target an unattended cron job needs.
|
||||
|
||||
Returns a list of dicts: ``{"id", "name", "home_target_set", "home_env_var"}``
|
||||
ordered by the gateway's canonical platform order. Callers should always
|
||||
prepend the implicit ``local`` option themselves — it needs no config.
|
||||
"""
|
||||
targets: list[dict] = []
|
||||
try:
|
||||
from gateway.config import load_gateway_config
|
||||
|
||||
gateway_config = load_gateway_config()
|
||||
connected = {p.value for p in gateway_config.get_connected_platforms()}
|
||||
except Exception:
|
||||
logger.debug("cron_delivery_targets: gateway config unavailable", exc_info=True)
|
||||
connected = set()
|
||||
|
||||
for name in _iter_home_target_platforms():
|
||||
if name not in connected:
|
||||
continue
|
||||
if not _is_known_delivery_platform(name):
|
||||
continue
|
||||
env_var = _resolve_home_env_var(name)
|
||||
targets.append(
|
||||
{
|
||||
"id": name,
|
||||
"name": name.replace("_", " ").title(),
|
||||
"home_target_set": bool(_get_home_target_chat_id(name)),
|
||||
"home_env_var": env_var or None,
|
||||
}
|
||||
)
|
||||
return targets
|
||||
|
||||
|
||||
def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[dict]:
|
||||
"""Resolve one concrete auto-delivery target for a cron job."""
|
||||
|
||||
@@ -530,7 +683,9 @@ def _send_media_via_adapter(
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
from gateway.platforms.base import should_send_media_as_audio
|
||||
from gateway.platforms.base import BasePlatformAdapter, should_send_media_as_audio
|
||||
|
||||
media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files)
|
||||
|
||||
for media_path, _is_voice in media_files:
|
||||
try:
|
||||
@@ -615,6 +770,7 @@ def _deliver_result(job: dict, content: str, adapters=None, loop=None) -> Option
|
||||
# Extract MEDIA: tags so attachments are forwarded as files, not raw text
|
||||
from gateway.platforms.base import BasePlatformAdapter
|
||||
media_files, cleaned_delivery_content = BasePlatformAdapter.extract_media(delivery_content)
|
||||
media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files)
|
||||
|
||||
try:
|
||||
config = load_gateway_config()
|
||||
@@ -963,8 +1119,15 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
result is used for prompt injection. When omitted, the script
|
||||
(if any) runs inline as before.
|
||||
"""
|
||||
prompt = str(job.get("prompt") or "")
|
||||
user_prompt = str(job.get("prompt") or "")
|
||||
prompt = user_prompt
|
||||
skills = job.get("skills")
|
||||
# True when runtime-collected DATA (script stdout, upstream-job output)
|
||||
# has been injected into the prompt. Data content legitimately quotes
|
||||
# command-shape strings (a triage feed ingesting a bug report that
|
||||
# pastes `rm -rf /`), so it must not be scanned with the strict
|
||||
# user-prompt pattern set — see _scan_assembled_cron_prompt.
|
||||
has_injected_data = False
|
||||
|
||||
# Run data-collection script if configured, inject output as context.
|
||||
script_path = job.get("script")
|
||||
@@ -982,6 +1145,7 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
f"```\n{script_output}\n```\n\n"
|
||||
f"{prompt}"
|
||||
)
|
||||
has_injected_data = True
|
||||
else:
|
||||
# Script produced no output — nothing to report, skip AI call.
|
||||
return None
|
||||
@@ -992,6 +1156,7 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
f"```\n{script_output}\n```\n\n"
|
||||
f"{prompt}"
|
||||
)
|
||||
has_injected_data = True
|
||||
|
||||
# Inject output from referenced cron jobs as context.
|
||||
context_from = job.get("context_from")
|
||||
@@ -1002,7 +1167,13 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
for source_job_id in context_from:
|
||||
# Guard against path traversal — valid job IDs are 12-char hex strings
|
||||
if not source_job_id or not all(c in "0123456789abcdef" for c in source_job_id):
|
||||
logger.warning("context_from: skipping invalid job_id %r", source_job_id)
|
||||
logger.warning(
|
||||
"context_from: skipping invalid job_id %r for job_id=%r name=%r%s",
|
||||
source_job_id,
|
||||
job.get("id"),
|
||||
job.get("name"),
|
||||
_cron_job_origin_log_suffix(job),
|
||||
)
|
||||
continue
|
||||
try:
|
||||
job_output_dir = OUTPUT_DIR / source_job_id
|
||||
@@ -1028,6 +1199,7 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
f"```\n{latest_output}\n```\n\n"
|
||||
f"{prompt}"
|
||||
)
|
||||
has_injected_data = True
|
||||
else:
|
||||
continue # silent skip — empty output
|
||||
except (OSError, PermissionError) as e:
|
||||
@@ -1056,14 +1228,46 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
|
||||
skill_names = [str(name).strip() for name in skills if str(name).strip()]
|
||||
if not skill_names:
|
||||
return _scan_assembled_cron_prompt(prompt, job)
|
||||
return _scan_assembled_cron_prompt(
|
||||
prompt,
|
||||
job,
|
||||
has_skills=False,
|
||||
has_injected_data=has_injected_data,
|
||||
user_prompt=user_prompt,
|
||||
)
|
||||
|
||||
from tools.skills_tool import skill_view
|
||||
from tools.skill_usage import bump_use
|
||||
from agent.skill_bundles import build_bundle_invocation_message, resolve_bundle_command_key
|
||||
|
||||
parts = []
|
||||
skipped: list[str] = []
|
||||
for skill_name in skill_names:
|
||||
# Cron jobs historically accepted only skill names here, but the CLI/gateway
|
||||
# slash-command path lets bundles shadow skills with the same slug. Mirror
|
||||
# that behavior so `skills: ["my-bundle"]` expands bundle members instead
|
||||
# of being treated as a missing skill.
|
||||
bundle_key = resolve_bundle_command_key(skill_name.lstrip("/"))
|
||||
if bundle_key:
|
||||
bundle_payload = build_bundle_invocation_message(
|
||||
bundle_key,
|
||||
user_instruction="",
|
||||
task_id=str(job.get("id") or "") or None,
|
||||
)
|
||||
if bundle_payload:
|
||||
bundle_message, _loaded_bundle_skills, _missing_bundle_skills = bundle_payload
|
||||
if parts:
|
||||
parts.append("")
|
||||
parts.append(bundle_message)
|
||||
continue
|
||||
logger.warning(
|
||||
"Cron job '%s': bundle '%s' could not load any skills, skipping",
|
||||
job.get("name", job.get("id")),
|
||||
skill_name,
|
||||
)
|
||||
skipped.append(skill_name)
|
||||
continue
|
||||
|
||||
try:
|
||||
loaded = json.loads(skill_view(skill_name))
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
@@ -1104,23 +1308,68 @@ def _build_job_prompt(job: dict, prerun_script: Optional[tuple] = None) -> str:
|
||||
|
||||
if prompt:
|
||||
parts.extend(["", f"The user has provided the following instruction alongside the skill invocation: {prompt}"])
|
||||
return _scan_assembled_cron_prompt("\n".join(parts), job)
|
||||
return _scan_assembled_cron_prompt("\n".join(parts), job, has_skills=True)
|
||||
|
||||
|
||||
def _scan_assembled_cron_prompt(assembled: str, job: dict) -> str:
|
||||
"""Scan the fully-assembled cron prompt (including skill content) for
|
||||
injection patterns. Raises ``CronPromptInjectionBlocked`` when a match
|
||||
fires so ``run_job`` can surface a clear refusal to the operator.
|
||||
def _scan_assembled_cron_prompt(
|
||||
assembled: str,
|
||||
job: dict,
|
||||
*,
|
||||
has_skills: bool = False,
|
||||
has_injected_data: bool = False,
|
||||
user_prompt: Optional[str] = None,
|
||||
) -> str:
|
||||
"""Scan the fully-assembled cron prompt for injection patterns. Raises
|
||||
``CronPromptInjectionBlocked`` when a match fires so ``run_job`` can
|
||||
surface a clear refusal to the operator.
|
||||
|
||||
Plugs the #3968 gap: ``_scan_cron_prompt`` runs on the user-supplied
|
||||
prompt at create/update, but skill content is loaded from disk at
|
||||
runtime and was never scanned. Since cron runs non-interactively
|
||||
(auto-approves tool calls), a malicious skill carrying an injection
|
||||
payload bypassed every gate.
|
||||
"""
|
||||
from tools.cronjob_tools import _scan_cron_prompt
|
||||
|
||||
scan_error = _scan_cron_prompt(assembled)
|
||||
Two pattern tiers, selected by what the assembled prompt CONTAINS,
|
||||
not just whether skills are attached:
|
||||
|
||||
- When the assembled prompt is essentially the user prompt + the cron
|
||||
hint (no skills, no injected data), the STRICT ``_scan_cron_prompt``
|
||||
patterns apply: a bare ``rm -rf /`` in a small directive prompt is a
|
||||
smoking gun, not prose.
|
||||
- When the assembled prompt includes runtime-loaded content — skill
|
||||
markdown (``has_skills=True``) or DATA injected from a job script's
|
||||
stdout / an upstream job's output (``has_injected_data=True``) — the
|
||||
LOOSER ``_scan_cron_skill_assembled`` pattern set is used: only
|
||||
unambiguous prompt-injection directives block; command-shape
|
||||
patterns are dropped and invisible unicode is sanitized (stripped +
|
||||
logged) rather than blocked, to avoid false-positives that
|
||||
permanently kill a job. Skill bodies are vetted at install time by
|
||||
``skills_guard.py``; script output is produced by operator-authored
|
||||
code, the same trust class — and data feeds (e.g. a triage bot
|
||||
ingesting bug reports) legitimately quote dangerous commands.
|
||||
|
||||
When the looser tier is selected because of injected data only,
|
||||
``user_prompt`` (the raw, pre-assembly prompt) is additionally scanned
|
||||
with the STRICT set so the user-authored surface keeps the full
|
||||
create/update-time guarantee at runtime (defense-in-depth for legacy
|
||||
jobs that predate the create-time scanner).
|
||||
"""
|
||||
from tools.cronjob_tools import _scan_cron_prompt, _scan_cron_skill_assembled
|
||||
|
||||
if has_skills or has_injected_data:
|
||||
# Runtime-loaded content (vetted skill markdown and/or data from
|
||||
# operator-authored scripts) legitimately contains command-shape
|
||||
# strings. Invisible unicode is sanitized (not blocked) so a stray
|
||||
# zero-width space can't permanently kill the job; the cleaned
|
||||
# prompt is what actually runs.
|
||||
cleaned, scan_error = _scan_cron_skill_assembled(assembled)
|
||||
assembled = cleaned
|
||||
if not scan_error and not has_skills and user_prompt:
|
||||
# Data-injection path: keep the strict guarantee on the
|
||||
# user-authored prompt itself.
|
||||
scan_error = _scan_cron_prompt(user_prompt)
|
||||
else:
|
||||
scan_error = _scan_cron_prompt(assembled)
|
||||
if scan_error:
|
||||
job_label = job.get("name") or job.get("id") or "<unknown>"
|
||||
logger.warning(
|
||||
@@ -1448,9 +1697,16 @@ def _run_job_impl(job: dict) -> tuple[bool, str, str, Optional[str]]:
|
||||
effort = str(_cfg.get("agent", {}).get("reasoning_effort", "")).strip()
|
||||
reasoning_config = parse_reasoning_effort(effort)
|
||||
|
||||
# Prefill messages from env or config.yaml
|
||||
# Prefill messages from env or config.yaml. The top-level
|
||||
# prefill_messages_file key is canonical; agent.prefill_messages_file is
|
||||
# retained as a legacy fallback for older CLI/godmode configs.
|
||||
prefill_messages = None
|
||||
prefill_file = os.getenv("HERMES_PREFILL_MESSAGES_FILE", "") or _cfg.get("prefill_messages_file", "")
|
||||
agent_cfg = _cfg.get("agent", {}) if isinstance(_cfg.get("agent", {}), dict) else {}
|
||||
prefill_file = (
|
||||
os.getenv("HERMES_PREFILL_MESSAGES_FILE", "")
|
||||
or _cfg.get("prefill_messages_file", "")
|
||||
or agent_cfg.get("prefill_messages_file", "")
|
||||
)
|
||||
if prefill_file:
|
||||
pfpath = Path(prefill_file).expanduser()
|
||||
if not pfpath.is_absolute():
|
||||
@@ -1572,7 +1828,7 @@ def _run_job_impl(job: dict) -> tuple[bool, str, str, Optional[str]]:
|
||||
provider_sort=pr.get("sort"),
|
||||
openrouter_min_coding_score=(_cfg.get("openrouter") or {}).get("min_coding_score"),
|
||||
enabled_toolsets=_resolve_cron_enabled_toolsets(job, _cfg),
|
||||
disabled_toolsets=["cronjob", "messaging", "clarify"],
|
||||
disabled_toolsets=_resolve_cron_disabled_toolsets(_cfg),
|
||||
quiet_mode=True,
|
||||
# Cron jobs should always inherit the user's SOUL.md identity from
|
||||
# HERMES_HOME. When a workdir is configured, also inject project
|
||||
@@ -1757,6 +2013,18 @@ def _run_job_impl(job: dict) -> tuple[bool, str, str, Optional[str]]:
|
||||
for _var_name in _cron_delivery_vars:
|
||||
_VAR_MAP[_var_name].set("")
|
||||
if _session_db:
|
||||
# Title the cron session from the job (name → short prompt → id) so
|
||||
# sidebars/history show a meaningful label instead of the injected
|
||||
# "[IMPORTANT: …]" hint that is the session's first message. Set here
|
||||
# (not at create time) so the agent's own INSERT keeps model /
|
||||
# system_prompt; this only UPDATEs the title column. The run-time
|
||||
# suffix keeps it unique against the sessions.title index across runs.
|
||||
try:
|
||||
_title_base = " ".join(job_name.split())[:60].strip() or f"cron {job_id}"
|
||||
_cron_title = f"{_title_base} · {_hermes_now().strftime('%b %d %H:%M')}"
|
||||
_session_db.set_session_title(_cron_session_id, _cron_title)
|
||||
except (Exception, KeyboardInterrupt) as e:
|
||||
logger.debug("Job '%s': failed to set cron session title: %s", job_id, e)
|
||||
try:
|
||||
_session_db.end_session(_cron_session_id, "cron_complete")
|
||||
except (Exception, KeyboardInterrupt) as e:
|
||||
@@ -1785,7 +2053,7 @@ def _run_job_impl(job: dict) -> tuple[bool, str, str, Optional[str]]:
|
||||
logger.debug("Job '%s': failed to reap stale auxiliary clients: %s", job_id, e)
|
||||
|
||||
|
||||
def tick(verbose: bool = True, adapters=None, loop=None) -> int:
|
||||
def tick(verbose: bool = True, adapters=None, loop=None, sync: bool = True) -> int:
|
||||
"""
|
||||
Check and run all due jobs.
|
||||
|
||||
@@ -1829,6 +2097,9 @@ def tick(verbose: bool = True, adapters=None, loop=None) -> int:
|
||||
|
||||
# Advance next_run_at for all recurring jobs FIRST, under the file lock,
|
||||
# before any execution begins. This preserves at-most-once semantics.
|
||||
# For parallel jobs that are already running, advance_next_run keeps
|
||||
# bumping next_run_at forward so the grace window never expires.
|
||||
# mark_job_run() overwrites next_run_at on completion.
|
||||
for job in due_jobs:
|
||||
advance_next_run(job["id"])
|
||||
|
||||
@@ -1920,36 +2191,103 @@ def tick(verbose: bool = True, adapters=None, loop=None) -> int:
|
||||
]
|
||||
|
||||
_results: list = []
|
||||
_all_futures: list = []
|
||||
|
||||
# Sequential pass for env/context-mutating jobs.
|
||||
for job in sequential_jobs:
|
||||
def _submit_with_guard(job: dict, pool: concurrent.futures.ThreadPoolExecutor):
|
||||
"""Submit a job fire-and-forget with the in-flight dedup guard.
|
||||
|
||||
Returns the future, or None if the job was skipped because a prior
|
||||
tick's run of the same job is still in flight. The running-set
|
||||
membership is released in the worker's finally block.
|
||||
"""
|
||||
job_id = job["id"]
|
||||
with _running_lock:
|
||||
if job_id in _running_job_ids:
|
||||
logger.info("Job '%s' already running — skipping", job.get("name", job_id))
|
||||
return None
|
||||
_running_job_ids.add(job_id)
|
||||
_ctx = contextvars.copy_context()
|
||||
_results.append(_ctx.run(_process_job, job))
|
||||
|
||||
# Parallel pass for the rest — same behaviour as before.
|
||||
def _run_and_release(j=job, ctx=_ctx):
|
||||
try:
|
||||
return ctx.run(_process_job, j)
|
||||
finally:
|
||||
with _running_lock:
|
||||
_running_job_ids.discard(j["id"])
|
||||
|
||||
return pool.submit(_run_and_release)
|
||||
|
||||
# Sequential pass for env/context-mutating (workdir/profile) jobs.
|
||||
# Queued to a persistent single-thread pool so they run one at a time
|
||||
# WITHOUT blocking the ticker thread — a long workdir/profile job no
|
||||
# longer starves the rest of the schedule (same fix as the parallel
|
||||
# pass, just serialized). The in-flight guard prevents a still-running
|
||||
# job from being re-queued on the next tick.
|
||||
if sequential_jobs:
|
||||
seq_pool = _get_sequential_pool()
|
||||
for job in sequential_jobs:
|
||||
fut = _submit_with_guard(job, seq_pool)
|
||||
if fut is None:
|
||||
continue
|
||||
_all_futures.append(fut)
|
||||
if not sync:
|
||||
_results.append(True) # optimistically counted
|
||||
|
||||
# Parallel pass — persistent pool, non-blocking dispatch.
|
||||
# Jobs that are already running (from a previous tick) are skipped.
|
||||
# mark_job_run() updates next_run_at on completion, so the next tick
|
||||
# after completion finds the job due again naturally. No catch-up
|
||||
# queue needed.
|
||||
if parallel_jobs:
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=_max_workers) as _tick_pool:
|
||||
_futures = []
|
||||
for job in parallel_jobs:
|
||||
_ctx = contextvars.copy_context()
|
||||
_futures.append(_tick_pool.submit(_ctx.run, _process_job, job))
|
||||
for f in concurrent.futures.as_completed(_futures, timeout=600):
|
||||
try:
|
||||
_results.append(f.result())
|
||||
except Exception as exc:
|
||||
logger.error("Parallel cron job future failed: %s", exc)
|
||||
_results.append(False)
|
||||
pool = _get_parallel_pool(_max_workers)
|
||||
for job in parallel_jobs:
|
||||
fut = _submit_with_guard(job, pool)
|
||||
if fut is None:
|
||||
continue
|
||||
_all_futures.append(fut)
|
||||
if not sync:
|
||||
_results.append(True) # optimistically counted
|
||||
|
||||
# Best-effort sweep of MCP stdio subprocesses that survived their
|
||||
# session teardown during this tick. Runs AFTER every job has
|
||||
# finished so active sessions (including live user chats) are
|
||||
# never touched — only PIDs explicitly detected as orphans in
|
||||
# tools.mcp_tool._run_stdio's finally block are reaped.
|
||||
try:
|
||||
from tools.mcp_tool import _kill_orphaned_mcp_children
|
||||
_kill_orphaned_mcp_children()
|
||||
except Exception as _e:
|
||||
logger.debug("Post-tick MCP orphan cleanup failed: %s", _e)
|
||||
# session teardown. Must run AFTER jobs finish so active sessions
|
||||
# (including live user chats) are never touched — only PIDs explicitly
|
||||
# detected as orphans in tools.mcp_tool._run_stdio's finally block are
|
||||
# reaped.
|
||||
def _sweep_mcp_orphans() -> None:
|
||||
try:
|
||||
from tools.mcp_tool import _kill_orphaned_mcp_children
|
||||
_kill_orphaned_mcp_children()
|
||||
except Exception as _e:
|
||||
logger.debug("Post-tick MCP orphan cleanup failed: %s", _e)
|
||||
|
||||
if sync:
|
||||
# Sync mode (tests / manual ticks): wait for all dispatched jobs,
|
||||
# collect results, then sweep once.
|
||||
for f in concurrent.futures.as_completed(_all_futures):
|
||||
try:
|
||||
_results.append(f.result())
|
||||
except Exception as exc:
|
||||
logger.error("Cron job future failed: %s", exc)
|
||||
_results.append(False)
|
||||
_sweep_mcp_orphans()
|
||||
return sum(_results)
|
||||
|
||||
# Async (gateway ticker) mode: don't block. Sweep orphans via a
|
||||
# done-callback fired after the LAST dispatched job completes, so the
|
||||
# sweep still happens after jobs finish without stalling the tick.
|
||||
if _all_futures:
|
||||
_remaining = [len(_all_futures)]
|
||||
|
||||
def _on_done(_f: concurrent.futures.Future) -> None:
|
||||
_remaining[0] -= 1
|
||||
if _remaining[0] <= 0:
|
||||
_sweep_mcp_orphans()
|
||||
|
||||
for _f in _all_futures:
|
||||
_f.add_done_callback(_on_done)
|
||||
else:
|
||||
# Nothing dispatched (all skipped / no due jobs) — sweep inline.
|
||||
_sweep_mcp_orphans()
|
||||
|
||||
return sum(_results)
|
||||
finally:
|
||||
|
||||
Reference in New Issue
Block a user