"""
Hermes Web UI -- Route handlers for GET and POST endpoints.
Extracted from server.py (Sprint 11) so server.py is a thin shell.
"""

import html as _html
import copy
import hashlib
import inspect
import errno
import io
import gzip
import json
from api.sse_chunked import end_sse_headers
import logging
import mimetypes
import os
import queue
import re
import platform
import shlex
import shutil
import sqlite3
import stat as _stat
import subprocess
import sys
import threading
import time
import uuid
import http.client
import socket as _socket
from collections import defaultdict, deque, OrderedDict
from pathlib import Path
from contextlib import closing
from urllib.parse import parse_qs, quote, unquote, urljoin, urlsplit
from urllib.error import HTTPError, URLError
from urllib.request import HTTPRedirectHandler, HTTPSHandler, ProxyHandler, Request, build_opener
from api.agent_runtime import (
    AgentRuntimeChangedError,
    agent_runtime_stale_payload,
    ensure_agent_runtime_current,
    require_ai_agent_class,
)
from api.agent_sessions import (
    MESSAGING_SOURCES,
    _looks_like_default_cli_title,
    is_cli_session_row,
    is_cli_session_row_visible,
    open_state_db_readonly,
    read_session_lineage_report,
)
from api.compression_anchor import visible_messages_for_anchor
from api.compression_recovery import (
    COMPRESSION_RECOVERY_ACTION_START_FOCUSED,
    clear_compression_recovery,
    compression_recovery_payload_for_session,
    is_generic_continuation_intent,
)
from api.session_events import (
    add_session_list_changed_listener,
    publish_session_list_changed,
    subscribe_session_events,
    unsubscribe_session_events,
)
from api.gateway_restart import restart_active_profile_gateway
from api.shares import create_or_refresh_share, load_share, revoke_share

logger = logging.getLogger(__name__)


def _publish_session_list_changed(
    reason: str,
    *,
    profile: str | None = None,
    session_id: str | None = None,
) -> None:
    """Publish scoped session changes while tolerating legacy test doubles."""
    if not profile and not session_id:
        publish_session_list_changed(reason)
        return
    try:
        publish_session_list_changed(reason, profile=profile, session_id=session_id)
    except TypeError:
        # Some focused tests monkeypatch the route-level publisher with the
        # historical one-argument or profile-only shape. Preserve the old signal instead of
        # turning unrelated session mutations into 500s.
        if profile:
            try:
                publish_session_list_changed(reason, profile=profile)
                return
            except TypeError:
                pass
        publish_session_list_changed(reason)


def _sync_session_title_to_insights(session) -> None:
    """Write title-only session metadata updates through to state.db when enabled."""
    try:
        if not load_settings().get("sync_to_insights"):
            return
        from api.state_sync import sync_session_usage

        messages = getattr(session, "messages", None) or []
        sync_session_usage(
            session_id=session.session_id,
            input_tokens=getattr(session, "input_tokens", None) or 0,
            output_tokens=getattr(session, "output_tokens", None) or 0,
            estimated_cost=getattr(session, "estimated_cost", 0.0),
            model=getattr(session, "model", ""),
            title=session.title,
            message_count=len(messages),
            profile=getattr(session, "profile", None),
            cache_read_tokens=getattr(session, "cache_read_tokens", None) or 0,
            cache_write_tokens=getattr(session, "cache_write_tokens", None) or 0,
        )
    except Exception:
        logger.debug("Failed to update session title in state.db", exc_info=True)


def _persist_generated_session_title(
    session,
    next_title: str,
    *,
    event_reason: str,
    require_default_title: bool = False,
) -> str:
    normalized_title = str(next_title or "").strip()[:80] or "Untitled"
    sid = str(getattr(session, "session_id", "") or "")
    original_session = session
    with _get_session_agent_lock(sid):
        with LOCK:
            latest = SESSIONS.get(sid)
            if latest is not None and str(getattr(latest, "session_id", "") or "") != sid:
                SESSIONS.pop(sid, None)
                latest = None
            elif latest is not None:
                SESSIONS.move_to_end(sid)
        if latest is None:
            latest = Session.load(sid)
            if latest is None:
                raise KeyError(sid)
        session = _ensure_full_session_before_mutation(sid, latest)
        if getattr(session, "read_only", False):
            raise PermissionError(f"Session {sid} is read-only")
        if require_default_title:
            latest_meta = {
                "title": getattr(session, "title", None),
                "source_tag": getattr(session, "source_tag", None),
                "raw_source": getattr(session, "raw_source", None),
                "session_source": getattr(session, "session_source", None),
                "source_label": getattr(session, "source_label", None),
            }
            if not _looks_like_default_cli_title(latest_meta):
                return session.title
        session.title = normalized_title
        from api.session_ops import mark_session_title_generated

        # mark_session_title_generated sets s.llm_title_generated = True and clears manual_title.
        mark_session_title_generated(session)
        session.save(touch_updated_at=False)
        with LOCK:
            SESSIONS[sid] = session
            SESSIONS.move_to_end(sid)
            _evict_sessions_over_cap()  # #4765: safe LRU eviction (never active/unsaved)
    _sync_session_title_to_insights(session)
    _publish_session_list_changed(
        event_reason,
        profile=getattr(session, "profile", None),
        session_id=sid,
    )
    if original_session is not session:
        original_session.title = session.title
        original_session.llm_title_generated = session.llm_title_generated
        original_session.manual_title = session.manual_title
    return session.title


def _queue_generated_title_for_imported_session(session, cli_meta: dict | None) -> None:
    try:
        cli_meta = dict(cli_meta or {})
        if not session or cli_meta.get("read_only") or not _looks_like_default_cli_title(cli_meta):
            return
        sid = str(getattr(session, "session_id", "") or "")
        if not sid:
            return

        def _run() -> None:
            try:
                current = Session.load(sid)
                if not current:
                    return
                current = _ensure_full_session_before_mutation(sid, current)
                if getattr(current, "read_only", False):
                    return
                current_meta = {
                    "title": getattr(current, "title", None),
                    "source_tag": getattr(current, "source_tag", None),
                    "raw_source": getattr(current, "raw_source", None),
                    "session_source": getattr(current, "session_source", None),
                    "source_label": getattr(current, "source_label", None),
                }
                if not _looks_like_default_cli_title(current_meta):
                    return
                next_title, _reason, _raw_preview = generate_session_title_for_session(current)
                normalized_current = str(getattr(current, "title", "") or "").strip()
                normalized_next = str(next_title or "").strip()
                if not normalized_next or normalized_next == normalized_current:
                    return
                _persist_generated_session_title(
                    current,
                    normalized_next,
                    event_reason="session_title_regenerate",
                    require_default_title=True,
                )
            except Exception:
                logger.debug("Failed to generate imported session title for %s", sid, exc_info=True)

        threading.Thread(target=_run, daemon=True, name=f"imported-title-{sid}").start()
    except Exception:
        logger.debug(
            "Failed to queue imported session title generation for %s",
            getattr(session, "session_id", None),
            exc_info=True,
        )


def _on_session_list_changed(profile: str | None = None) -> None:
    """Invalidate in-process /api/sessions cache when sidebar state mutates."""
    _clear_session_list_cache(profile)
    # #4842: also drop the inner CLI/cron projection cache. While a turn streams
    # that cache is frozen on a stable streaming marker (so per-token message
    # writes don't bust it), which means it no longer self-invalidates via the
    # state.db content fingerprint mid-stream. In-app structural mutations
    # (session create/rename/archive/delete/branch/pin/move/import, attention)
    # fire this listener — and never fire per streamed token — so clearing here
    # restores prompt freshness for those without reintroducing the per-poll
    # rebuild the freeze removed. Note: externally-driven changes that do NOT go
    # through this listener (a scheduled cron job completing, or an external CLI
    # writing rows directly) are not cleared here mid-stream; for those the 30s
    # streaming TTL is the backstop — they surface within one streaming-TTL
    # window (≤30s) rather than instantly. That bound is the deliberate
    # latency/CPU trade-off of the freeze.
    try:
        from api.models import clear_cli_sessions_cache
        clear_cli_sessions_cache()
    except Exception:
        logger.debug("Failed to clear CLI sessions cache on session list change", exc_info=True)


try:
    add_session_list_changed_listener(_on_session_list_changed)
except Exception:
    logger.debug("Failed to register session list cache invalidation listener", exc_info=True)


# ── Cron run tracking ────────────────────────────────────────────────────────
# Track job IDs currently being executed so the frontend can poll status.
_RUNNING_CRON_JOBS: dict[str, float] = {}  # job_id → start_timestamp
_RUNNING_CRON_LOCK = threading.Lock()
_CRON_CREATE_SNAPSHOT_LOCK = threading.Lock()
_MANUAL_COMPRESSION_JOBS: dict[str, dict] = {}
_MANUAL_COMPRESSION_JOBS_LOCK = threading.Lock()
_MANUAL_COMPRESSION_JOB_TTL_SECONDS = 10 * 60
_CRON_OUTPUT_CONTENT_LIMIT = 8000
_CRON_OUTPUT_HEADER_CONTEXT = 200
_MESSAGING_RAW_SOURCES = {str(s).strip().lower() for s in MESSAGING_SOURCES}
_MESSAGING_SESSION_METADATA_CACHE: dict[str, object] = {
    "path": None,
    "mtime": None,
    "identity": {},
}
_MESSAGING_SESSION_METADATA_LOCK = threading.Lock()
_STALE_MESSAGING_END_REASONS = {"session_reset", "session_switch"}
_CSP_REPORT_LOGGER = logging.getLogger("csp_report")
_CSP_REPORT_RATE_LIMIT: dict[str, list[float]] = {}
_CSP_REPORT_RATE_LIMIT_LOCK = threading.Lock()
_CSP_REPORT_RATE_LIMIT_WINDOW_SECONDS = 60
_CSP_REPORT_RATE_LIMIT_MAX = 100
_CSP_REPORT_MAX_BODY_BYTES = 64 * 1024
_CLIENT_EVENT_LOGGER = logging.getLogger("client_event")
_CLIENT_EVENT_RATE_LIMIT: dict[str, list[float]] = {}
_CLIENT_EVENT_RATE_LIMIT_LOCK = threading.Lock()
_CLIENT_EVENT_RATE_LIMIT_WINDOW_SECONDS = 60
_CLIENT_EVENT_RATE_LIMIT_MAX = 30
_CLIENT_EVENT_MAX_BODY_BYTES = 4 * 1024
_EXTENSION_SIDECAR_PROXY_MAX_RESPONSE_BYTES = 512 * 1024
_CLIENT_EVENT_ALLOWED_FIELDS = {
    "event": 64,
    "source": 80,
    "session_id": 128,
    "stream_id": 128,
    "visibility_state": 32,
    "url_path": 256,
    "reason": 160,
}


def _normalize_cron_job_ids(job_ids) -> list[str]:
    seen = set()
    normalized = []
    for job_id in job_ids or []:
        jid = str(job_id or "").strip()
        if not jid or jid in seen:
            continue
        seen.add(jid)
        normalized.append(jid)
    return normalized


def _latest_cron_session_info_for_jobs(
    job_ids, completed_job_ids=None
) -> dict[str, dict[str, int | str | None]]:
    """Return newest persisted cron session info keyed by completed cron job id."""
    normalized = _normalize_cron_job_ids(job_ids)
    requested = _normalize_cron_job_ids(completed_job_ids if completed_job_ids is not None else job_ids)
    if not requested:
        return {}
    if not normalized:
        return {jid: {"session_id": "", "message_count": None} for jid in requested}
    db_path = _active_state_db_path()
    if not db_path or not Path(db_path).exists():
        return {jid: {"session_id": "", "message_count": None} for jid in requested}
    try:
        with closing(open_state_db_readonly(db_path)) as conn:
            conn.row_factory = sqlite3.Row
            cur = conn.cursor()
            cur.execute("PRAGMA table_info(sessions)")
            session_cols = {row[1] for row in cur.fetchall()}
            if "id" not in session_cols or "source" not in session_cols:
                return {jid: {"session_id": "", "message_count": None} for jid in requested}
            select_message_count = (
                "s.message_count AS message_count"
                if "message_count" in session_cols
                else "NULL AS message_count"
            )
            if "started_at" in session_cols:
                query = f"""
                    SELECT s.id,
                           {select_message_count}
                    FROM sessions s
                    WHERE LOWER(COALESCE(s.source, '')) = 'cron'
                    ORDER BY COALESCE(s.started_at, 0) DESC, s.id DESC  -- newest start, not last activity
                """
            else:
                query = f"""
                    SELECT s.id,
                           {select_message_count}
                    FROM sessions s
                    WHERE LOWER(COALESCE(s.source, '')) = 'cron'
                    ORDER BY s.id DESC
                """
            cur.execute(query)
            results = {
                jid: {"session_id": "", "message_count": None} for jid in requested
            }
            requested_ids = set(requested)
            prefixes = {jid: f"cron_{jid}_" for jid in normalized}
            for row in cur.fetchall():
                sid = str(row["id"] or "")
                if not sid:
                    continue
                matches = [
                    jid
                    for jid in normalized
                    if sid.startswith(prefixes[jid])
                ]
                if matches:
                    jid = max(matches, key=len)
                    if jid not in requested_ids or results[jid]["session_id"]:
                        continue
                    results[jid] = {
                        "session_id": sid,
                        "message_count": (
                            int(row["message_count"])
                            if row["message_count"] is not None
                            else None
                        ),
                    }
                if all(info["session_id"] for info in results.values()):
                    break
            return results
    except sqlite3.Error:
        return {jid: {"session_id": "", "message_count": None} for jid in requested}



def _session_field(session, field, default=None):
    if isinstance(session, dict):
        return session.get(field, default)
    return getattr(session, field, default)


def _session_counts_toward_pin_quota(session) -> bool:
    """Return True when a pinned session should consume visible pin quota."""
    if not _session_field(session, "pinned", False):
        return False
    if _session_field(session, "archived", False):
        return False
    if isinstance(session, dict):
        row = session
    elif hasattr(session, "compact"):
        row = session.compact()
    else:
        row = {
            "pre_compression_snapshot": _session_field(session, "pre_compression_snapshot", False),
            "source_tag": _session_field(session, "source_tag", None),
            "default_hidden": _session_field(session, "default_hidden", False),
        }
    return not _hide_from_default_sidebar(row)


def _session_row_lineage_root_id(session, sessions_by_id) -> str:
    sid = str(_session_field(session, "session_id", "") or "")
    explicit = _session_field(session, "_lineage_root_id", None)
    if explicit:
        return str(explicit)
    # A branch/fork is an independent, separately-visible session (it carries a
    # parent_session_id purely for provenance), so it must count as its OWN pin
    # lineage — only compression/continuation rows should collapse to a shared
    # root. Without this, two pinned forks of the same parent would collapse to a
    # single quota lineage and let the user exceed pinned_sessions_limit (#3288).
    if _session_field(session, "session_source", None) == "fork":
        return sid
    current = sid
    seen = {sid} if sid else set()
    parent = _session_field(session, "parent_session_id", None)
    while parent:
        parent = str(parent)
        if parent in seen:
            break
        current = parent
        seen.add(parent)
        parent_row = sessions_by_id.get(parent)
        if not parent_row:
            break
        parent = _session_field(parent_row, "parent_session_id", None)
    return current or sid


def _visible_pinned_lineage_ids(session_rows) -> set[str]:
    sessions_by_id = {}
    for row in session_rows:
        sid = str(_session_field(row, "session_id", "") or "")
        if sid:
            sessions_by_id[sid] = row
    roots: set[str] = set()
    for row in session_rows:
        if not _session_counts_toward_pin_quota(row):
            continue
        root = _session_row_lineage_root_id(row, sessions_by_id)
        if root:
            roots.add(root)
    return roots


# ── Profile-scoped session/project filtering (#1611, #1614) ────────────────
#
# Sessions and projects are stored in the WebUI sidecar without per-row
# isolation by default — they're tagged with a `profile` field but every
# query saw all rows. The fix scopes both endpoints to the active profile
# by default, with `?all_profiles=1` opting into aggregate mode.
#
# Renamed-root profile handling (#1612): a row tagged `profile='default'`
# matches the active root regardless of the root's display name, and a row
# tagged with the renamed-root display name (e.g. 'kinni') likewise matches
# when the active profile is `'default'`. _is_root_profile() is the
# canonical check.

# Canonical helper now lives in api.profiles so out-of-process consumers
# (mcp_server.py) can import it without duplicating the visibility model.
# Re-exported here so existing `_profiles_match(...)` call sites in this
# module keep resolving without per-call-site refactors.
from api.profiles import (  # noqa: F401, E402  (re-export)
    _profiles_match,
    _is_isolated_profile_mode,
    _is_root_profile,
    _SKILLS_STATS_CACHE,
    get_active_profile_name,
    get_active_profile_name as _get_active_profile_name,
    get_active_hermes_home,
    list_profiles_api,
    profile_scope_for_detached_worker,
)


def _all_profiles_query_flag(parsed_url) -> bool:
    """Return True if the request URL has `?all_profiles=1` (or true/yes).

    Centralizes the opt-in parsing so /api/sessions and /api/projects use
    the same shape. Accepts 1/true/yes (case-insensitive) for ergonomics.
    """
    qs = parse_qs(parsed_url.query)
    raw = qs.get('all_profiles', [''])[0].strip().lower()
    return raw in ('1', 'true', 'yes', 'on')


def _all_profiles_enabled(parsed_url) -> bool:
    """Enable aggregate profile reads only when the request asks and mode allows it."""
    return _all_profiles_query_flag(parsed_url) and not _is_isolated_profile_mode()


def _query_flag(parsed_url, name: str) -> bool:
    """Return True for a truthy query flag value."""
    qs = parse_qs(parsed_url.query)
    raw = qs.get(name, [''])[0].strip().lower()
    return raw in ('1', 'true', 'yes', 'on')


def _query_positive_int(parsed_url, name: str, *, default=None, maximum: int | None = None):
    """Return a non-negative integer query parameter, or default when absent/invalid."""
    qs = parse_qs(parsed_url.query)
    raw = qs.get(name, [''])[0]
    try:
        value = int(str(raw).strip())
    except (TypeError, ValueError):
        return default
    if value < 0:
        return default
    if maximum is not None:
        value = min(value, int(maximum))
    return value


def _session_visible_to_active_profile(session_profile, handler=None) -> bool:
    """Return whether a detail-load session belongs to the active profile.

    Real request handlers must enforce the same profile boundary as
    /api/sessions, even when the request has no hermes_profile cookie and the
    process-level active profile is the default/root profile. Direct unit-callers
    without a request handler keep the historical metadata-load behavior.
    """
    if handler is None:
        return True
    active_profile = _get_active_profile_name()
    if not isinstance(session_profile, str):
        session_profile = None
    return _profiles_match(session_profile, active_profile)


def _is_profile_agnostic_foreign_session(cli_meta) -> bool:
    """Return whether a foreign-session row lives outside the Hermes profile tree.

    Claude Code transcripts are scanned straight out of ``~/.claude/projects``
    by ``get_claude_code_sessions()``, which stamps ``profile: None`` on every
    row because the JSONL files belong to no Hermes profile at all. The sidebar
    lists them under whichever profile is active, but ``_profiles_match``
    coerces ``None`` to ``'default'``, so the detail-load profile gate 404s
    every one of them as soon as the active profile is a named (non-root) one —
    the session shows in the list and then renders "Session not available in
    web UI." when clicked.

    Exempt these profile-less external-agent rows from the gate so opening one
    behaves identically on the root profile and on named profiles. Rows that
    DO carry a profile (every state.db-backed CLI/messaging/cron session) stay
    fully scoped.
    """
    if not isinstance(cli_meta, dict):
        return False
    if cli_meta.get("profile"):
        return False
    sources = {
        str(cli_meta.get("source_tag") or "").strip().lower(),
        str(cli_meta.get("raw_source") or "").strip().lower(),
    }
    # Profile-less external-agent rows that live outside the Hermes profile tree.
    # Claude Code: scanned from ~/.claude/projects; Codex: scanned from ~/.codex/
    profile_agnostic_sources = {CLAUDE_CODE_SOURCE}
    try:
        from api.codex_sessions import CODEX_SOURCE
        profile_agnostic_sources.add(CODEX_SOURCE)
    except ImportError:
        pass
    return bool(sources & profile_agnostic_sources)


def _request_session_visibility_exempt(method: str, path: str | None) -> bool:
    if not path:
        return False
    if method == "GET" and path == "/api/session":
        # Detail-load owns profile mismatch handling so the frontend can switch
        # to the session's profile instead of treating a valid cross-profile
        # deep link as a deleted/stale session.
        return True
    if method != "POST":
        return False
    # Import routes create/claim sessions before normal ownership exists, and
    # chat/start has inline placeholder-retag rules that must run before the
    # generic request-session guard.
    return path in {
        "/api/session/import",
        "/api/session/import_cli",
        "/api/chat/start",
    }


def _session_id_visible_to_request_profile(handler, sid, *, emit_error: bool = True) -> bool:
    """Return whether ``sid`` belongs to the active profile.

    On a profile mismatch, the helper mirrors the detail-load endpoint's
    contract (#13043, #13493): return ``409 session_profile_mismatch`` for
    a session owned by a KNOWN other profile, and keep ``404 Session
    not found`` only for the unknown/legacy None-profile case so the
    frontend's self-heal (clear stale URL + localStorage) keeps firing
    for actually-missing sids. ``#7710``.
    """
    if not isinstance(sid, str) or not sid:
        return True
    if not is_safe_session_id(sid):
        return True
    try:
        session = get_session(sid, metadata_only=True)
    except KeyError:
        return True
    session_profile = getattr(session, "profile", None) or None
    if not _session_visible_to_active_profile(session_profile, handler):
        if emit_error:
            if session_profile:
                j(handler, {
                    "error": "Session belongs to a different profile",
                    "code": "session_profile_mismatch",
                    "session_id": sid,
                    "profile": session_profile,
                }, status=409)
            else:
                # Unknown/legacy None-profile sidecar: keep the 404 so the
                # frontend's self-heal still fires. _profiles_match coerces
                # None->'default', so a truly missing/legacy session under a
                # non-default active profile would otherwise emit a useless
                # 409 with profile=null.
                bad(handler, "Session not found", 404)
        return False
    return True


def _stream_id_owner_session_id(stream_id: str | None) -> str | None:
    """Resolve stream owner session_id via active-run registry first, fallback to journal."""
    stream_id = str(stream_id or "").strip()
    if not stream_id:
        return None
    try:
        with ACTIVE_RUNS_LOCK:
            raw = (ACTIVE_RUNS or {}).get(stream_id)
        if isinstance(raw, dict):
            owner = str(raw.get("session_id") or "").strip()
            if owner:
                return owner
    except Exception:
        logger.debug("Failed reading ACTIVE_RUNS owner for stream %s", stream_id, exc_info=True)
    try:
        owner = stream_owner_session_id(stream_id)
        if owner:
            return owner
    except Exception:
        logger.debug("Failed reading registered owner for stream %s", stream_id, exc_info=True)
    if not is_safe_session_id(stream_id):
        return None
    try:
        summary = find_run_summary(stream_id)
        if isinstance(summary, dict):
            owner = str(summary.get("session_id") or "").strip()
            return owner or None
    except Exception:
        logger.debug("Failed reading run summary for stream %s", stream_id, exc_info=True)
    return None


def _stream_id_visible_to_request_profile(
    handler,
    stream_id: str | None,
    *,
    emit_error: bool = True,
) -> bool:
    """Return whether the stream owner is visible to the request's profile."""
    owner_session_id = _stream_id_owner_session_id(stream_id)
    if not owner_session_id:
        return True
    return _session_id_visible_to_request_profile(handler, owner_session_id, emit_error=emit_error)


def _guard_request_session_visibility(handler, parsed, body=None, method="GET") -> bool:
    """Apply request session-profile visibility check to request-supplied IDs.

    Covers top-level `session_id` in the query/body. Routes that accept session
    IDs under other keys must enforce their own visibility checks.
    """
    method = str(method).upper()
    if _request_session_visibility_exempt(method, getattr(parsed, "path", "")):
        return True
    sid = parse_qs(getattr(parsed, "query", "") or "").get("session_id", [None])[0]
    if not _session_id_visible_to_request_profile(handler, sid):
        return False
    if isinstance(body, dict) and not _session_id_visible_to_request_profile(handler, body.get("session_id")):
        return False
    return True


def _active_skills_dir() -> Path:
    """Return the skills directory for the request's active Hermes profile.

    WebUI profile switches are cookie/thread-local scoped, so the agent
    module-level ``tools.skills_tool.SKILLS_DIR`` can still point at the server
    startup profile. Skills UI endpoints must derive the directory from
    ``get_active_hermes_home()`` for every request instead of reading that
    process-global constant.
    """
    try:
        from api.profiles import get_active_hermes_home

        return Path(get_active_hermes_home()) / "skills"
    except Exception:
        try:
            from tools.skills_tool import SKILLS_DIR

            return Path(SKILLS_DIR)
        except Exception:
            return Path(os.getenv("HERMES_HOME", str(Path.home() / ".hermes"))).expanduser() / "skills"


def _skill_path_within(base_dir: Path, candidate: Path) -> bool:
    try:
        candidate.resolve().relative_to(base_dir.resolve())
        return True
    except (OSError, ValueError):
        return False


def _skill_category_from_path(
    skill_md: Path,
    skills_dirs: list[Path],
    local_skills_dir: Path | None = None,
) -> str | None:
    """Return the UI category for a discovered skill path.

    Flat skills directly under the active *local* skills root stay uncategorized,
    while flat skills under an *external* root use that root's directory name as
    their category. ``local_skills_dir`` identifies the local root explicitly; if
    omitted it falls back to ``skills_dirs[0]`` for backward compatibility, but
    callers should pass it directly because the local root can be filtered out of
    ``skills_dirs`` (e.g. when it does not exist yet on a host with only external
    skills configured), which would otherwise misclassify the first external root
    as local.
    """
    if local_skills_dir is None:
        local_skills_dir = skills_dirs[0] if skills_dirs else None
    for skills_dir in skills_dirs:
        try:
            rel_path = skill_md.relative_to(skills_dir)
        except ValueError:
            continue
        parts = rel_path.parts
        if len(parts) >= 3:
            return parts[0]
        if len(parts) >= 2 and local_skills_dir is not None and skills_dir != local_skills_dir:
            return skills_dir.name
        return None
    return None


def _active_skill_search_dirs(skills_dir: Path) -> list[Path]:
    dirs = [skills_dir]
    try:
        from agent.skill_utils import get_external_skills_dirs

        dirs.extend(Path(p) for p in get_external_skills_dirs())
    except Exception:
        pass
    return [p for p in dirs if p.exists()]


def _worktree_retained_payload(session) -> dict:
    """Return explicit no-cleanup metadata for worktree-backed session actions."""
    worktree_path = getattr(session, "worktree_path", None) if session else None
    if not worktree_path:
        return {}
    payload = {
        "worktree_retained": True,
        "worktree_path": worktree_path,
    }
    worktree_branch = getattr(session, "worktree_branch", None)
    worktree_repo_root = getattr(session, "worktree_repo_root", None)
    if worktree_branch:
        payload["worktree_branch"] = worktree_branch
    if worktree_repo_root:
        payload["worktree_repo_root"] = worktree_repo_root
    return payload


def _worktree_retained_payload_for_session_id(sid: str) -> dict:
    try:
        return _worktree_retained_payload(get_session(sid, metadata_only=True))
    except KeyError:
        return {}
    except Exception:
        logger.debug("Failed to read worktree metadata for deleted session %s", sid)
        return {}


def _active_profile_config_path() -> Path:
    """Return config.yaml for the request's active WebUI profile.

    Skills endpoints are profile-scoped UI actions: both the visible disabled
    toggle state and toggle writes must follow the cookie/thread-local active
    Hermes home, not process-global HERMES_HOME or HERMES_CONFIG_PATH values
    captured at server startup.
    """
    test_override_module = getattr(_get_config_path, "__module__", "")
    if test_override_module != "api.config":
        return _get_config_path()
    try:
        from api.profiles import get_active_hermes_home

        return Path(get_active_hermes_home()) / "config.yaml"
    except Exception:
        return _get_config_path()


def _get_disabled_skill_names_for_profile() -> set:
    """Read disabled skill names from the active profile's config.yaml.

    Unlike ``tools.skills_tool._get_disabled_skill_names`` which reads from
    the process-global ``HERMES_HOME``, this uses ``_get_config_path()`` which
    resolves against the WebUI's active profile.  Checks
    ``skills.platform_disabled.webui`` first, falling back to
    ``skills.disabled``.
    """
    config_path = _active_profile_config_path()
    if not config_path.exists():
        return set()
    try:
        cfg = _load_yaml_config_file(config_path)
    except Exception:
        return set()
    if not isinstance(cfg, dict):
        return set()
    skills_cfg = cfg.get("skills")
    if not isinstance(skills_cfg, dict):
        return set()
    # Check platform_disabled.webui first (mirrors agent platform resolution)
    platform_disabled = skills_cfg.get("platform_disabled")
    if isinstance(platform_disabled, dict) and "webui" in platform_disabled:
        return _normalize_disabled_set(platform_disabled["webui"])
    return _normalize_disabled_set(skills_cfg.get("disabled"))


def _parse_config_string_list(value) -> list:
    """Decode a config value that may hold a JSON-array string into a list.

    ``hermes config set`` (and JSON-mode editor saves) store lists as quoted
    JSON strings (``'[\"a\",\"b\"]'`` or the Python-literal ``\"['a']\"``), so a
    disabled list read from ``config.yaml`` can arrive as a single string
    instead of a YAML list. Treating it as one literal name makes the Skills
    panel show every skill as enabled and makes the toggle write a destructive
    single-entry list (hermes-webui#7120).

    Reuses ``agent.skill_utils.parse_config_string_list`` (hermes-agent #86661
    fix) when the bundled agent source is importable, and mirrors its logic
    otherwise so the two surfaces cannot drift. A scalar string still means one
    name.
    """
    try:
        from agent.skill_utils import parse_config_string_list

        return parse_config_string_list(value)
    except ImportError:
        pass
    import ast

    if value is None:
        return []
    if isinstance(value, str):
        stripped = value.strip()
        if stripped.startswith("["):
            try:
                parsed = ast.literal_eval(stripped)
            except (ValueError, SyntaxError):
                parsed = None
            if isinstance(parsed, list):
                return [str(item) for item in parsed]
        return [value]
    if isinstance(value, (list, tuple, set, frozenset)):
        return [str(item) for item in value]
    return []


def _normalize_disabled_set(values) -> set:
    """Normalize a YAML disabled list into a set of stripped strings."""
    if values is None:
        return set()
    if isinstance(values, str):
        values = _parse_config_string_list(values)
    return {str(v).strip() for v in values if str(v).strip()}


def _skills_list_from_dir(skills_dir: Path, category: str | None = None) -> dict:
    """List skills using an explicit local skills directory.

    This mirrors ``tools.skills_tool.skills_list`` closely, but keeps the local
    scan root explicit so per-client WebUI profile switches do not race on or
    leak through the skills tool's module-global ``SKILLS_DIR``.
    """
    from agent.skill_utils import iter_skill_index_files
    from tools.skills_tool import (
        MAX_DESCRIPTION_LENGTH,
        _EXCLUDED_SKILL_DIRS,
        _parse_frontmatter,
        _sort_skills,
        skill_matches_platform,
    )

    if not skills_dir.exists():
        skills_dir.mkdir(parents=True, exist_ok=True)
        return {
            "success": True,
            "skills": [],
            "categories": [],
            "message": f"No skills found. Skills directory created at {skills_dir}/",
        }

    all_skills = []
    seen_names: set[str] = set()
    disabled = _get_disabled_skill_names_for_profile()
    search_dirs = _active_skill_search_dirs(skills_dir)

    for scan_dir in search_dirs:
        for skill_md in iter_skill_index_files(scan_dir, "SKILL.md"):
            if any(part in _EXCLUDED_SKILL_DIRS for part in skill_md.parts):
                continue
            skill_dir = skill_md.parent
            try:
                content = skill_md.read_text(encoding="utf-8")[:4000]
                frontmatter, body = _parse_frontmatter(content)
                if not skill_matches_platform(frontmatter):
                    continue
                name = frontmatter.get("name", skill_dir.name)[:64]
                if name in seen_names:
                    continue
                description = frontmatter.get("description", "")
                if not description:
                    for line in body.strip().split("\n"):
                        line = line.strip()
                        if line and not line.startswith("#"):
                            description = line
                            break
                if len(description) > MAX_DESCRIPTION_LENGTH:
                    description = description[: MAX_DESCRIPTION_LENGTH - 3] + "..."
                seen_names.add(name)
                all_skills.append(
                    {
                        "name": name,
                        "description": description,
                        "category": _skill_category_from_path(
                            skill_md, search_dirs, local_skills_dir=skills_dir
                        ),
                        "disabled": name in disabled,
                    }
                )
            except (UnicodeDecodeError, PermissionError) as e:
                logger.debug("Failed to read skill file %s: %s", skill_md, e)
            except Exception as e:
                logger.debug(
                    "Skipping skill at %s: failed to parse: %s", skill_md, e, exc_info=True
                )

    if category:
        all_skills = [s for s in all_skills if s.get("category") == category]
    all_skills = _sort_skills(all_skills)
    categories = sorted(set(s.get("category") for s in all_skills if s.get("category")))
    result = {
        "success": True,
        "skills": all_skills,
        "categories": categories,
        "count": len(all_skills),
    }
    if all_skills:
        result["hint"] = "Use skill_view(name) to see full content, tags, and linked files"
    else:
        result["message"] = "No skills found in skills/ directory."
    return result


def _find_skill_in_dirs(name: str, skills_dirs: list[Path]) -> tuple[Path | None, Path | None]:
    """Resolve a WebUI skill name inside explicit skills directories."""
    from agent.skill_utils import iter_skill_index_files
    from tools.skills_tool import _EXCLUDED_SKILL_DIRS, _parse_frontmatter

    raw_name = str(name or "").strip().strip("/")
    if not raw_name:
        return None, None

    candidate_names = [raw_name]
    if ":" in raw_name:
        namespace, bare = raw_name.split(":", 1)
        if namespace and bare:
            candidate_names.append(f"{namespace}/{bare}")

    for skills_dir in skills_dirs:
        if not skills_dir.exists():
            continue
        for candidate_name in candidate_names:
            direct_path = skills_dir / candidate_name
            if not _skill_path_within(skills_dir, direct_path):
                continue
            if direct_path.is_dir() and (direct_path / "SKILL.md").exists():
                return direct_path, direct_path / "SKILL.md"
            legacy_md = direct_path.with_suffix(".md")
            if legacy_md.exists() and _skill_path_within(skills_dir, legacy_md):
                return legacy_md.parent, legacy_md

        for skill_md in iter_skill_index_files(skills_dir, "SKILL.md"):
            if any(part in _EXCLUDED_SKILL_DIRS for part in skill_md.parts):
                continue
            skill_dir = skill_md.parent
            if skill_dir.name == raw_name:
                return skill_dir, skill_md
            try:
                frontmatter, _ = _parse_frontmatter(skill_md.read_text(encoding="utf-8")[:4000])
                if frontmatter.get("name") == raw_name:
                    return skill_dir, skill_md
            except Exception:
                continue

        for legacy_md in skills_dir.rglob("*.md"):
            if legacy_md.name == "SKILL.md":
                continue
            if legacy_md.stem == raw_name and _skill_path_within(skills_dir, legacy_md):
                return legacy_md.parent, legacy_md
    return None, None


def _find_skill_in_dir(name: str, skills_dir: Path) -> tuple[Path | None, Path | None]:
    """Resolve a WebUI skill name inside an explicit skills directory."""
    return _find_skill_in_dirs(name, [skills_dir])


# Cap on the courtesy list of names carried by a skill-not-found reply. The
# bound stays; what it must never do is present a partial list as the whole
# set, because a caller that cannot find its skill in `available_skills` will
# conclude the skill is not installed.
_SKILL_NOT_FOUND_LIST_LIMIT = 20


def _skill_not_found_payload(name: str, skills_dir: Path) -> dict:
    all_names = [s["name"] for s in _skills_list_from_dir(skills_dir).get("skills", [])]
    total = len(all_names)
    available = all_names[:_SKILL_NOT_FOUND_LIST_LIMIT]
    truncated = total > len(available)
    hint = "Use skills_list to see all available skills"
    if truncated:
        hint = f"Showing {len(available)} of {total} skills. {hint}"
    return {
        "success": False,
        "error": f"Skill '{name}' not found.",
        "available_skills": available,
        "available_skills_truncated": truncated,
        "total_skills": total,
        "hint": hint,
    }


def _linked_files_for_skill(skill_dir: Path | None) -> dict:
    if not skill_dir or not (skill_dir / "SKILL.md").exists():
        return {}
    linked_files: dict[str, list[str]] = {}

    references_dir = skill_dir / "references"
    if references_dir.exists():
        refs = [str(f.relative_to(skill_dir)) for f in references_dir.glob("*.md")]
        if refs:
            linked_files["references"] = sorted(refs)

    templates_dir = skill_dir / "templates"
    if templates_dir.exists():
        templates = []
        for ext in ["*.md", "*.py", "*.yaml", "*.yml", "*.json", "*.tex", "*.sh"]:
            templates.extend(str(f.relative_to(skill_dir)) for f in templates_dir.rglob(ext))
        if templates:
            linked_files["templates"] = sorted(set(templates))

    assets_dir = skill_dir / "assets"
    if assets_dir.exists():
        assets = [str(f.relative_to(skill_dir)) for f in assets_dir.rglob("*") if f.is_file()]
        if assets:
            linked_files["assets"] = sorted(assets)

    scripts_dir = skill_dir / "scripts"
    if scripts_dir.exists():
        scripts = []
        for ext in ["*.py", "*.sh", "*.bash", "*.js", "*.ts", "*.rb"]:
            scripts.extend(str(f.relative_to(skill_dir)) for f in scripts_dir.glob(ext))
        if scripts:
            linked_files["scripts"] = sorted(set(scripts))

    return linked_files


def _skill_view_from_file(skill_dir: Path | None, skill_md: Path) -> dict:
    from tools.skills_tool import _parse_frontmatter, _parse_tags, skill_matches_platform

    content = skill_md.read_text(encoding="utf-8")
    frontmatter, _body = _parse_frontmatter(content)
    if not skill_matches_platform(frontmatter):
        return {"success": False, "error": "Skill is not available on this platform."}

    metadata = frontmatter.get("metadata")
    hermes_meta = metadata.get("hermes", {}) if isinstance(metadata, dict) else {}
    tags = _parse_tags(hermes_meta.get("tags") or frontmatter.get("tags", ""))
    related_skills = _parse_tags(
        hermes_meta.get("related_skills") or frontmatter.get("related_skills", "")
    )
    try:
        path = str(skill_md.relative_to((skill_dir or skill_md.parent).parent))
    except ValueError:
        path = str(skill_md)

    return {
        "success": True,
        "name": frontmatter.get("name", skill_md.stem if not skill_dir else skill_dir.name),
        "description": frontmatter.get("description", ""),
        "tags": tags,
        "related_skills": related_skills,
        "content": content,
        "path": path,
        "skill_dir": str(skill_dir) if skill_dir else None,
        "linked_files": _linked_files_for_skill(skill_dir),
    }


def _skill_view_from_active_dir(name: str) -> dict:
    from tools.skills_tool import skill_view as _skill_view

    skills_dir = _active_skills_dir()
    search_dirs = _active_skill_search_dirs(skills_dir)
    skill_dir, skill_md = _find_skill_in_dirs(name, search_dirs)
    if not skill_md:
        # Preserve plugin-qualified skill viewing without falling back to the
        # startup/root profile's local skills tree for ordinary missing skills.
        if ":" in str(name or ""):
            try:
                from agent.skill_utils import is_valid_namespace, parse_qualified_name
                from hermes_cli.plugins import discover_plugins, get_plugin_manager

                namespace, _bare = parse_qualified_name(name)
                if is_valid_namespace(namespace):
                    discover_plugins()
                    pm = get_plugin_manager()
                    if pm.find_plugin_skill(name) is not None or pm.list_plugin_skills(namespace):
                        raw = _skill_view(name)
                        return json.loads(raw) if isinstance(raw, str) else raw
            except Exception:
                pass
        return _skill_not_found_payload(name, skills_dir)
    return _skill_view_from_file(skill_dir, skill_md)

# ── SSE app-level heartbeat (#1623) ────────────────────────────────────────
#
# Kernel TCP keepalive (server.py setsockopt block) declares a peer dead at
# KEEPIDLE (10s) + KEEPINTVL (5s) * KEEPCNT (3) = 25s in the worst case. The
# app-level SSE heartbeat must fire well below that window so flaky-network
# probes never get the chance to kill an idle stream during long LLM thinking
# phases. 5s gives the kernel ~5x headroom: probe at 10s, heartbeat byte at
# every 5s of idle keeps the socket warm.
#
# Cost: ~12 bytes per heartbeat * 12 extra heartbeats/min = ~150B/min idle.
# Trivial; many production SSE deployments run 5-15s heartbeats specifically
# to handle proxies and mobile NAT.
_SSE_HEARTBEAT_INTERVAL_SECONDS = 5
_SESSION_SSE_SENT_EVENT_ID_LIMIT = 4096


def _normalize_messaging_source(raw_source) -> str:
    return str(raw_source or "").strip().lower()


def _is_known_messaging_source(raw_source) -> bool:
    return _normalize_messaging_source(raw_source) in _MESSAGING_RAW_SOURCES


def _safe_first(*values):
    for value in values:
        if value is None:
            continue
        text = str(value).strip()
        if text:
            return text
    return ""


def _gateway_session_metadata_path():
    try:
        from api.profiles import get_active_hermes_home
        hermes_home = Path(get_active_hermes_home()).expanduser().resolve()
    except Exception:
        hermes_home = Path(os.getenv("HERMES_HOME", str(Path.home() / ".hermes"))).expanduser().resolve()
    return hermes_home / "sessions" / "sessions.json"


def _load_gateway_session_identity_map() -> dict[str, dict]:
    path = _gateway_session_metadata_path()
    if not path.exists():
        return {}

    try:
        st = path.stat()
        cache = _MESSAGING_SESSION_METADATA_CACHE
        with _MESSAGING_SESSION_METADATA_LOCK:
            if cache["path"] == str(path) and cache["mtime"] == st.st_mtime:
                return cache["identity"].copy()
    except Exception:
        return {}

    try:
        raw_sessions = json.loads(path.read_text(encoding="utf-8"))
    except Exception as _json_err:
        logger.debug("Failed to parse gateway sessions metadata from %s: %s", path, _json_err)
        return {}

    mapping: dict[str, dict] = {}
    if isinstance(raw_sessions, dict):
        for _entry in raw_sessions.values():
            if not isinstance(_entry, dict):
                continue
            session_id = _safe_first(_entry.get("session_id"))
            if not session_id:
                continue
            origin = _entry.get("origin") if isinstance(_entry.get("origin"), dict) else {}
            platform = _safe_first(origin.get("platform"), _entry.get("platform"))
            mapping[session_id] = {
                "session_key": _safe_first(_entry.get("session_key"), _entry.get("key")),
                "chat_id": _safe_first(origin.get("chat_id"), _entry.get("chat_id")),
                "thread_id": _safe_first(origin.get("thread_id"), _entry.get("thread_id")),
                "chat_type": _safe_first(origin.get("chat_type"), _entry.get("chat_type")),
                "user_id": _safe_first(origin.get("user_id"), _entry.get("user_id")),
                "platform": platform,
                "raw_source": platform,
            }

    with _MESSAGING_SESSION_METADATA_LOCK:
        _MESSAGING_SESSION_METADATA_CACHE["path"] = str(path)
        _MESSAGING_SESSION_METADATA_CACHE["mtime"] = st.st_mtime
        _MESSAGING_SESSION_METADATA_CACHE["identity"] = mapping
    return mapping.copy()


def _gateway_status_payload() -> dict:
    import datetime

    identity_map = _load_gateway_session_identity_map()
    sessions_path = _gateway_session_metadata_path()

    # Detect whether the gateway process is alive, independent of connected
    # messaging platforms. An empty identity_map means zero connected
    # platforms, not necessarily a stopped gateway.
    health = build_agent_health_payload()
    alive = health.get("alive")
    details = health.get("details") if isinstance(health.get("details"), dict) else {}
    health_reason = details.get("reason")
    health_state = details.get("state")
    health_gateway_state = details.get("gateway_state")
    if alive is True:
        running = True
        configured = True
    elif alive is False:
        running = False
        configured = True
    else:
        gateway_running_metadata = (
            health_reason == "gateway_stale_running_state"
            or health_gateway_state == "running"
        )
        configured = True if gateway_running_metadata else bool(identity_map)
        running = bool(identity_map)

    platforms_set: set[str] = set()
    for meta in identity_map.values():
        raw = meta.get("raw_source") or meta.get("platform") or ""
        norm = _normalize_messaging_source(raw)
        if norm:
            platforms_set.add(norm)
    platform_labels = {
        "telegram": "Telegram",
        "discord": "Discord",
        "slack": "Slack",
        "email": "Email",
        "web": "Web",
        "api": "API",
    }
    platforms = sorted(
        [{"name": p, "label": platform_labels.get(p, p.title())} for p in platforms_set],
        key=lambda x: x["label"],
    )
    last_active = ""
    if running and sessions_path.exists():
        try:
            mtime = sessions_path.stat().st_mtime
            last_active = datetime.datetime.fromtimestamp(mtime).isoformat()
        except Exception:
            pass
    return {
        "running": running,
        "configured": configured,
        "platforms": platforms,
        "last_active": last_active,
        "session_count": len(identity_map),
        "health": {
            "state": health_state,
            "reason": health_reason,
            "gateway_state": health_gateway_state,
        },
    }


_GATEWAY_LIFECYCLE_TIMEOUT_SECONDS = 60

# Server-side single-flight guard for gateway lifecycle actions. The client
# disables its button while a request is in flight, but a scripted authed
# client could still fire overlapping start/stop/restart calls, spawning
# concurrent `hermes gateway` subprocesses. Serialize them here (mirrors the
# self-update _apply_lock pattern): a non-blocking acquire returns 409 on
# contention rather than launching a second overlapping subprocess.
_GATEWAY_ACTION_LOCK = threading.Lock()


def _run_gateway_lifecycle_command(action: str) -> subprocess.CompletedProcess:
    if action not in {"start", "stop", "restart"}:
        raise ValueError("unsupported gateway action")

    from api import config as api_config
    from api.profiles import get_active_profile_name

    agent_dir = getattr(api_config, "_AGENT_DIR", None)
    if not agent_dir:
        raise FileNotFoundError("Hermes agent checkout not found")
    agent_dir = Path(agent_dir).expanduser().resolve()
    main_py = agent_dir / "hermes_cli" / "main.py"
    if not main_py.exists():
        raise FileNotFoundError("Hermes agent CLI entrypoint not found")

    cmd = [str(getattr(api_config, "PYTHON_EXE", sys.executable)), str(main_py)]
    profile_name = ""
    try:
        profile_name = str(get_active_profile_name() or "").strip()
    except Exception as exc:
        logger.debug("Could not resolve active profile for gateway lifecycle: %s", exc)
    if profile_name and profile_name != "default":
        cmd.extend(["--profile", profile_name])
    cmd.extend(["gateway", action])

    env = os.environ.copy()
    env.setdefault("PYTHONUTF8", "1")
    env.setdefault("BROWSER", "echo")
    return subprocess.run(
        cmd,
        cwd=str(agent_dir),
        env=env,
        capture_output=True,
        text=True,
        timeout=_GATEWAY_LIFECYCLE_TIMEOUT_SECONDS,
    )


def _handle_gateway_lifecycle(handler, action: str, body: dict):
    del body  # Reserved for future per-gateway naming without changing the route contract.
    # Reject overlapping lifecycle actions instead of spawning concurrent
    # `hermes gateway` subprocesses (a non-blocking acquire — the action holds
    # the lock for at most _GATEWAY_LIFECYCLE_TIMEOUT_SECONDS).
    if action not in {"start", "stop", "restart"}:
        return bad(handler, "unsupported gateway action", 400)
    if not _GATEWAY_ACTION_LOCK.acquire(blocking=False):
        return j(
            handler,
            {
                "ok": False,
                "error": "Another gateway action is already in progress; try again shortly.",
                "action": action,
            },
            status=409,
        )
    try:
        result = _run_gateway_lifecycle_command(action)
    except ValueError as exc:
        return bad(handler, str(exc), 400)
    except FileNotFoundError as exc:
        return j(handler, {"ok": False, "error": _sanitize_error(exc), "action": action}, status=500)
    except subprocess.TimeoutExpired as exc:
        logger.warning(
            "Gateway %s command timed out after %ss; stdout=%r stderr=%r",
            action,
            _GATEWAY_LIFECYCLE_TIMEOUT_SECONDS,
            exc.stdout,
            exc.stderr,
        )
        return j(
            handler,
            {
                "ok": False,
                "error": f"Gateway {action} timed out after {_GATEWAY_LIFECYCLE_TIMEOUT_SECONDS} seconds",
                "action": action,
            },
            status=504,
        )
    except Exception as exc:
        logger.exception("Gateway %s command failed before completion", action)
        return j(handler, {"ok": False, "error": _sanitize_error(exc), "action": action}, status=500)
    finally:
        _GATEWAY_ACTION_LOCK.release()

    stdout = (result.stdout or "").strip()
    stderr = (result.stderr or "").strip()
    if result.returncode != 0:
        logger.warning(
            "Gateway %s command failed with exit code %s; stdout=%r stderr=%r",
            action,
            result.returncode,
            stdout,
            stderr,
        )
        return j(
            handler,
            {
                "ok": False,
                "error": f"Gateway {action} failed with exit code {result.returncode}",
                "action": action,
                "returncode": result.returncode,
            },
            status=500,
        )

    return j(
        handler,
        {
            "ok": True,
            "action": action,
            # Do NOT return captured stdout/stderr — the `hermes gateway` CLI
            # prints service/PID/status details the browser shouldn't receive
            # (mirrors the failure path, which already suppresses them). The
            # frontend localizes its own success copy; the refreshed status
            # payload carries the user-facing state.
            "message": f"Gateway {action} completed.",
            "status": _gateway_status_payload(),
        },
    )


def _mark_cron_running(job_id: str):
    with _RUNNING_CRON_LOCK:
        _RUNNING_CRON_JOBS[job_id] = time.time()


def _mark_cron_done(job_id: str):
    with _RUNNING_CRON_LOCK:
        _RUNNING_CRON_JOBS.pop(job_id, None)


def _is_cron_running(job_id: str) -> tuple[bool, float]:
    """Return (is_running, elapsed_seconds)."""
    with _RUNNING_CRON_LOCK:
        t = _RUNNING_CRON_JOBS.get(job_id)
        if t is None:
            return False, 0.0
        return True, time.time() - t


def _cron_response_marker_index(text: str) -> int:
    """Return the start index of a markdown Response heading, if present."""
    candidates = []
    for heading in ("## Response", "# Response"):
        if text.startswith(heading):
            candidates.append(0)
        idx = text.find(f"\n{heading}")
        if idx >= 0:
            candidates.append(idx + 1)
    return min(candidates) if candidates else -1


def _cron_output_content_window(text: str, limit: int = _CRON_OUTPUT_CONTENT_LIMIT) -> str:
    """Return a bounded cron output window that preserves useful response text.

    Cron output files can contain large skill dumps in the Prompt section. The
    UI already extracts ``## Response`` when present, so keep that section in
    the API payload instead of blindly returning the first ``limit`` chars.
    """
    if limit <= 0:
        return ""
    if len(text) <= limit:
        return text

    response_idx = _cron_response_marker_index(text)
    if response_idx >= 0:
        header = text[:min(_CRON_OUTPUT_HEADER_CONTEXT, response_idx)].rstrip()
        response = text[response_idx:].lstrip("\n")
        content = f"{header}\n...\n{response}" if header else response
        return content[:limit]

    return text[-limit:]




def _cron_job_for_api(job: dict) -> dict:
    """Return a cron job payload with optional UI settings normalized.

    Legacy jobs intentionally persist without ``profile`` so they keep the
    scheduler's server-default behavior. The API still returns ``profile: None``
    so the UI can label that state explicitly instead of guessing.

    ``toast_notifications`` is a WebUI preference for completion toasts. Legacy
    jobs default to enabled so existing behavior is preserved unless a job is
    explicitly muted.
    """
    payload = dict(job or {})
    payload.setdefault("profile", None)
    payload["toast_notifications"] = payload.get("toast_notifications") is not False
    return payload


def _cron_jobs_for_api(jobs) -> list[dict]:
    return [_cron_job_for_api(job) for job in (jobs or [])]


_AGENT_CRON_IMPORT_PATH_LOCK = threading.Lock()
_AGENT_CRON_IMPORT_PATH_READY: str | None = None


def _ensure_agent_cron_import_path() -> None:
    """Prefer the agent's cron package over unrelated top-level cron packages."""
    try:
        from api import config as api_config
    except Exception:
        return

    agent_dir = getattr(api_config, "_AGENT_DIR", None)
    if not agent_dir:
        return
    agent_path = str(Path(agent_dir).expanduser().resolve())
    agent_cron_path = str(Path(agent_path) / "cron")

    global _AGENT_CRON_IMPORT_PATH_READY
    with _AGENT_CRON_IMPORT_PATH_LOCK:
        cron_mod = sys.modules.get("cron")
        cron_file = str(getattr(cron_mod, "__file__", "") or "") if cron_mod else ""
        cron_is_agent = bool(cron_mod is not None and cron_file.startswith(agent_cron_path + os.sep))
        if _AGENT_CRON_IMPORT_PATH_READY == agent_path and (cron_mod is None or cron_is_agent):
            return

        while agent_path in sys.path:
            sys.path.remove(agent_path)
        shadow_indexes = [
            idx
            for idx, path_entry in enumerate(sys.path)
            if path_entry
            and Path(path_entry).resolve() != Path(agent_path)
            and (Path(path_entry) / "cron" / "__init__.py").exists()
        ]
        if shadow_indexes:
            sys.path.insert(min(shadow_indexes), agent_path)
        else:
            sys.path.append(agent_path)
        _AGENT_CRON_IMPORT_PATH_READY = agent_path

        # Keep in-memory test doubles or namespace stubs intact; only evict a
        # real on-disk shadow package so the agent's cron package can import.
        if cron_mod is not None and cron_file and not cron_is_agent:
            for name in list(sys.modules):
                if name == "cron" or name.startswith("cron."):
                    sys.modules.pop(name, None)


def _cron_jobs_cross_profile(active_profile: str) -> tuple[list[dict], list[dict]]:
    """Return active-profile rows plus foreign rows for the Tasks panel.

    Row ownership is intentionally distinct from a cron job's persisted
    ``profile`` field. The persisted field controls where the job executes;
    ``owner_profile`` tells the UI which profile home the row came from.
    """
    from cron.jobs import list_jobs
    from api.profiles import (
        cron_profile_context_for_home,
        get_hermes_home_for_profile,
        list_profiles_api,
    )

    def _home_key(path: Path) -> str:
        try:
            return str(Path(path).expanduser().resolve(strict=False))
        except Exception:
            return str(Path(path).expanduser())

    names: list[str] = []
    seen_names: set[str] = set()

    def _add_name(raw_name) -> None:
        name = str(raw_name or "").strip()
        if not name:
            return
        folded = name.casefold()
        if folded in seen_names:
            return
        seen_names.add(folded)
        names.append(name)

    _add_name(active_profile)
    for row in list_profiles_api():
        if not isinstance(row, dict):
            continue
        name = str(row.get("name") or "").strip()
        if not name:
            continue
        if row.get("visible") is False and not _profiles_match(name, active_profile):
            continue
        _add_name(name)

    active_jobs: list[dict] = []
    other_jobs: list[dict] = []
    seen_homes: set[str] = set()
    for owner_profile in names:
        home = Path(get_hermes_home_for_profile(owner_profile))
        home_key = _home_key(home)
        if home_key in seen_homes:
            continue
        seen_homes.add(home_key)
        is_active = _profiles_match(owner_profile, active_profile)
        try:
            with cron_profile_context_for_home(home):
                jobs = _cron_jobs_for_api(list_jobs(include_disabled=True))
        except Exception:
            if not is_active:
                continue
            raise
        for job in jobs:
            row = dict(job)
            row["owner_profile"] = owner_profile
            row["read_only"] = not is_active
            if is_active:
                active_jobs.append(row)
            else:
                other_jobs.append(row)
    return active_jobs, other_jobs


def _available_cron_profile_names() -> set[str]:
    from api.profiles import list_profiles_api

    names = {"default"}
    for profile in list_profiles_api():
        try:
            name = str(profile.get("name") or "").strip()
        except AttributeError:
            continue
        if name:
            names.add(name)
    return names


def _normalize_cron_profile_value(value) -> str | None:
    if value is None:
        return None
    profile = str(value).strip()
    if not profile:
        return None
    if profile not in _available_cron_profile_names():
        raise ValueError(f"Unknown profile: {profile}")
    return profile


def _profile_home_for_cron_job(job: dict):
    """Resolve the execution profile for a cron job, with graceful fallback.

    A missing/blank profile preserves legacy server-default behavior. If a job
    points at a profile that was deleted after save, fall back to the active
    server profile and log a warning instead of crashing the Run Now path.
    """
    from api.profiles import get_active_hermes_home, get_hermes_home_for_profile

    raw = str((job or {}).get("profile") or "").strip()
    if not raw:
        return get_active_hermes_home()
    if raw not in _available_cron_profile_names():
        logger.warning(
            "Cron job %s references missing profile %r; falling back to server default",
            (job or {}).get("id", "?"), raw,
        )
        return get_active_hermes_home()
    return get_hermes_home_for_profile(raw)


def _event_profile_for_cron_job(job: dict) -> str | None:
    """Return the profile identity browsers should refresh for a manual cron run."""
    raw = str((job or {}).get("profile") or "").strip()
    if not raw:
        return None
    if raw not in _available_cron_profile_names():
        return None
    return raw


def _cron_job_subprocess_main(job, execution_profile_home, result_queue):
    """Run one cron job inside a child process pinned to a profile home."""
    try:
        def _run():
            from cron.scheduler import run_job

            return run_job(job)

        if execution_profile_home is None:
            result = _run()
        else:
            from api.profiles import cron_profile_context_for_home

            with cron_profile_context_for_home(execution_profile_home):
                result = _run()
        result_queue.put(("ok", result))
    except BaseException as exc:  # pragma: no cover - surfaced in parent
        import traceback

        result_queue.put(("error", f"{type(exc).__name__}: {exc}", traceback.format_exc()))


def _cron_subprocess_result_timeout_seconds(job):
    """Return how long the manual-run parent waits for child result payloads."""
    for key in ("timeout_seconds", "max_runtime_seconds", "timeout"):
        raw = (job or {}).get(key)
        if raw in (None, ""):
            continue
        try:
            value = float(raw)
        except (TypeError, ValueError):
            continue
        if value > 0:
            return max(60.0, value + 30.0)
    # Manual cron jobs can legitimately run for a long time.  Keep a recovery
    # path for wedged children without truncating normal long-running jobs.
    return 6 * 60 * 60.0


def _run_cron_job_in_profile_subprocess(job, execution_profile_home):
    """Execute cron.scheduler.run_job without holding the parent cron env lock.

    cron.scheduler/cron.jobs still rely on process-global HERMES_HOME and module
    constants, so running the job body in a child process gives each long cron
    execution its own globals. The parent process only uses cron_profile_context
    for short metadata reads/writes and remains responsive to unrelated cron UI
    and API calls while the job runs.
    """
    import multiprocessing
    import queue

    ctx = multiprocessing.get_context("spawn")
    result_queue = ctx.Queue(maxsize=1)
    process = ctx.Process(
        target=_cron_job_subprocess_main,
        args=(job, execution_profile_home, result_queue),
    )
    process.start()

    result_timeout = _cron_subprocess_result_timeout_seconds(job)
    status = "error"
    payload = ["cron run subprocess failed before producing a result", ""]
    try:
        try:
            # Drain the potentially large pickled result before joining.  If the
            # child puts >~64 KiB on a multiprocessing.Queue, joining first can
            # deadlock while the child's feeder thread waits for the parent to
            # read from the pipe.
            status, *payload = result_queue.get(timeout=result_timeout)
        except queue.Empty:
            status = "error"
            if process.is_alive():
                process.terminate()
                process.join(timeout=5)
                payload = [
                    f"cron run subprocess produced no result within {result_timeout:g}s and was terminated",
                    "",
                ]
            else:
                payload = [
                    f"cron run subprocess exited with code {process.exitcode} without producing a result",
                    "",
                ]
        finally:
            process.join(timeout=5)
            if process.is_alive():
                process.terminate()
                process.join(timeout=5)
                if status == "ok":
                    status = "error"
                    payload = [
                        "cron run subprocess did not exit after returning a result",
                        "",
                    ]
    finally:
        result_queue.close()
        result_queue.join_thread()

    if status == "ok":
        return payload[0]

    message = payload[0]
    traceback_text = payload[1] if len(payload) > 1 else ""
    if traceback_text:
        logger.error("Manual cron subprocess failed:\n%s", traceback_text)
    raise RuntimeError(message)


def _run_cron_tracked(
    job,
    profile_home=None,
    execution_profile_home=None,
    event_profile=None,
):
    """Wrapper that tracks running state around cron.scheduler.run_job.

    ``profile_home`` is the cron store that owns the job row/output metadata.
    ``execution_profile_home`` is the selected per-job profile used to load
    agent config/.env while running. When no job profile is selected, both homes
    are the same and legacy server-default behavior is preserved.
    """
    import importlib

    from cron.jobs import mark_job_run, save_job_output

    _cron_scheduler = importlib.import_module("cron.scheduler")

    _silent_marker = getattr(_cron_scheduler, "SILENT_MARKER", "[SILENT]")
    _deliver_result = getattr(_cron_scheduler, "_deliver_result", None)

    job_id = job.get("id", "")
    execution_profile_home = execution_profile_home or profile_home

    def _with_cron_home(home, fn):
        if home is None:
            return fn()
        from api.profiles import cron_profile_context_for_home

        with cron_profile_context_for_home(home):
            return fn()

    try:
        success, output, final_response, error = _run_cron_job_in_profile_subprocess(
            job, execution_profile_home
        )

        # Persist output, deliver the same content the scheduled cron path would
        # send, and write run metadata back to the job's owning cron store even
        # when the selected execution profile is different.
        def _persist_success():
            save_job_output(job_id, output)

            deliver_content = (
                final_response
                if success
                else f"⚠️ Cron job '{job.get('name', job_id)}' failed:\n{error}"
            )
            should_deliver = bool(deliver_content)
            if should_deliver and success and _silent_marker in deliver_content.strip().upper():
                should_deliver = False

            delivery_error = None
            if should_deliver and _deliver_result is not None:
                try:
                    delivery_error = _deliver_result(job, deliver_content)
                except Exception as de:
                    delivery_error = str(de)
                    logger.error("Delivery failed for manual cron job %s: %s", job_id, de)

            # Match the scheduled cron path: an apparently successful run with no
            # final response should not leave the job looking healthy.
            _success, _error = success, error
            if _success and not final_response:
                _success = False
                _error = "Agent completed but produced empty response (model error, timeout, or misconfiguration)"

            try:
                mark_job_run(job_id, _success, _error, delivery_error=delivery_error)
            except TypeError:
                # Older/fake cron.jobs modules used by focused WebUI tests may
                # not expose the newer delivery_error parameter. Real Hermes
                # scheduler builds do, so this is only a compatibility shim for
                # legacy test doubles and deployments.
                mark_job_run(job_id, _success, _error)

        _with_cron_home(profile_home, _persist_success)
    except Exception as e:
        logger.exception("Manual cron run failed for job %s", job_id)
        try:
            _with_cron_home(profile_home, lambda: mark_job_run(job_id, False, str(e)))  # noqa: F821  e is bound by the enclosing `except ... as e` and the lambda runs synchronously here
        except Exception:
            logger.debug("Failed to mark manual cron run failure for %s", job_id)
    finally:
        _mark_cron_done(job_id)
        _publish_session_list_changed("cron_complete", profile=event_profile)

_PROVIDER_ALIASES = {
    "claude": "anthropic",
    "gpt": "openai",
    "gemini": "google",
    "openai-codex": "openai",
    "openai-api": "openai",
    "google-gemini": "google",
    "google-ai-studio": "google",
    "claude-code": "anthropic",
}

# OpenAI-compatible /v1/models endpoints for live model discovery.
# Used as fallback when hermes_cli.provider_model_ids() is unavailable or
# returns [] for a provider (#871).  Kept at module level so the dict is
# built once, not reconstructed per request.
_OPENAI_COMPAT_ENDPOINTS = {
    "zai": "https://api.z.ai/v1",
    "minimax": "https://api.minimax.chat/v1",
    "mistralai": "https://api.mistral.ai/v1",
    "xai": "https://api.x.ai/v1",
    "deepseek": "https://api.deepseek.com",
    "gemini": "https://generativelanguage.googleapis.com/v1beta/openai",
    "nvidia": "https://integrate.api.nvidia.com/v1",
}
# NOTE: "openai-codex" is excluded because it maps to the same endpoint as
# the base "openai" provider (api.openai.com/v1).  When both are configured
# the openai provider is already wired through provider_model_ids(); codex-
# specific model filtering happens downstream in hermes_cli.
#
_LIVE_MODELS_CACHE_TTL = 60.0
_LIVE_MODELS_CACHE: dict[tuple[str, str], tuple[float, dict]] = {}
_LIVE_MODELS_CACHE_LOCK = threading.RLock()


def _active_profile_for_live_models_cache() -> str:
    try:
        from api.profiles import get_active_profile_name

        return get_active_profile_name() or "default"
    except Exception as _e:
        # A transient profile-resolution error mis-scopes the cache for up to
        # 60s ("default" gets the wrong payload). Log so we can detect it; the
        # blast radius stays small because the TTL caps the bad-cache window.
        logger.debug("_active_profile_for_live_models_cache fell back to 'default': %s", _e)
        return "default"


def _live_models_cache_key(provider: str) -> tuple[str, str]:
    return (_active_profile_for_live_models_cache(), provider)


def _get_cached_live_models(key: tuple[str, str]) -> dict | None:
    now = time.monotonic()
    with _LIVE_MODELS_CACHE_LOCK:
        cached = _LIVE_MODELS_CACHE.get(key)
        if not cached:
            return None
        ts, payload = cached
        if now - ts >= _LIVE_MODELS_CACHE_TTL:
            _LIVE_MODELS_CACHE.pop(key, None)
            return None
        return copy.deepcopy(payload)


def _set_cached_live_models(key: tuple[str, str], payload: dict) -> None:
    with _LIVE_MODELS_CACHE_LOCK:
        _LIVE_MODELS_CACHE[key] = (time.monotonic(), copy.deepcopy(payload))


def _clear_live_models_cache() -> None:
    with _LIVE_MODELS_CACHE_LOCK:
        _LIVE_MODELS_CACHE.clear()


from api import route_session_list_cache as _route_session_list_cache

_SESSIONS_CACHE = _route_session_list_cache._SESSIONS_CACHE
_SESSIONS_CACHE_INFLIGHT = _route_session_list_cache._SESSIONS_CACHE_INFLIGHT
_SESSIONS_CACHE_LOCK = _route_session_list_cache._SESSIONS_CACHE_LOCK
_SESSIONS_CACHE_MAX_ENTRIES = _route_session_list_cache._SESSIONS_CACHE_MAX_ENTRIES
_SESSIONS_CACHE_PROFILE_INVALIDATION_VERSION = (
    _route_session_list_cache._SESSIONS_CACHE_PROFILE_INVALIDATION_VERSION
)
_SESSIONS_CACHE_STALE_WAIT_SECONDS = _route_session_list_cache._SESSIONS_CACHE_STALE_WAIT_SECONDS
_SESSIONS_CACHE_STREAMING_TTL_SECONDS = (
    _route_session_list_cache._SESSIONS_CACHE_STREAMING_TTL_SECONDS
)
_SESSIONS_CACHE_TTL_SECONDS = _route_session_list_cache._SESSIONS_CACHE_TTL_SECONDS
_SESSIONS_CACHE_WAIT_SECONDS = _route_session_list_cache._SESSIONS_CACHE_WAIT_SECONDS
_clear_session_list_cache = _route_session_list_cache._clear_session_list_cache
_session_list_cache_clear = _route_session_list_cache._session_list_cache_clear
_session_list_cache_claim_rebuild = _route_session_list_cache._session_list_cache_claim_rebuild
_session_list_cache_done = _route_session_list_cache._session_list_cache_done
_session_list_cache_get = _route_session_list_cache._session_list_cache_get
_session_list_cache_invalidation_stamp = _route_session_list_cache._session_list_cache_invalidation_stamp
_route_session_list_cache_key = _route_session_list_cache._session_list_cache_key
_session_list_cache_overlay_runtime_rows = _route_session_list_cache._session_list_cache_overlay_runtime_rows
_session_list_cache_path_stamp = _route_session_list_cache._session_list_cache_path_stamp
_session_list_cache_profile_scope = _route_session_list_cache._session_list_cache_profile_scope
_session_list_row_is_runtime_active = _route_session_list_cache._session_list_row_is_runtime_active
_session_list_row_numeric_value = _route_session_list_cache._session_list_row_numeric_value
_session_list_row_timestamp = _route_session_list_cache._session_list_row_timestamp
_session_list_runtime_sort_key = _route_session_list_cache._session_list_runtime_sort_key
_session_list_cache_set = _route_session_list_cache._session_list_cache_set
_session_list_cache_source_stamp = _route_session_list_cache._session_list_cache_source_stamp
_session_list_cache_state_db_fingerprint = _route_session_list_cache._session_list_cache_state_db_fingerprint
_session_list_cache_stale_reason = _route_session_list_cache._session_list_cache_stale_reason
_session_list_cache_streaming_freeze_marker = _route_session_list_cache._session_list_cache_streaming_freeze_marker


def _callable_accepts_kwarg(callable_obj, kwarg_name: str) -> bool:
    try:
        signature = inspect.signature(callable_obj)
    except (TypeError, ValueError):
        return True
    if kwarg_name in signature.parameters:
        return True
    return any(
        parameter.kind == inspect.Parameter.VAR_KEYWORD
        for parameter in signature.parameters.values()
    )


def _session_list_cache_key(
    active_profile: str | None,
    all_profiles: bool,
    show_cli_sessions: bool,
    show_previous_messaging_sessions: bool,
    show_cron_sessions: bool,
    include_archived: bool = False,
    exclude_hidden: bool = False,
    visible_only: bool = False,
    show_webhook_sessions: bool = False,
    show_kanban_sessions: bool = False,
    source_filter: str | None = None,
    sidebar_source: str | None = None,
    archived_limit: int | None = None,
    archived_offset: int = 0,
    show_claude_code_sessions: bool = True,
) -> tuple:
    return _route_session_list_cache_key(
        active_profile=active_profile,
        all_profiles=all_profiles,
        show_cli_sessions=show_cli_sessions,
        show_previous_messaging_sessions=show_previous_messaging_sessions,
        show_cron_sessions=show_cron_sessions,
        include_archived=include_archived,
        exclude_hidden=exclude_hidden,
        visible_only=visible_only,
        show_webhook_sessions=show_webhook_sessions,
        show_kanban_sessions=show_kanban_sessions,
        source_filter=source_filter,
        sidebar_source=sidebar_source,
        archived_limit=archived_limit,
        archived_offset=archived_offset,
    ) + (bool(show_claude_code_sessions),)

_ROUTE_SESSION_LIST_CACHE_DYNAMIC_EXPORTS = {
    "_SESSIONS_CACHE_ALL_PROFILES_INVALIDATION_VERSION",
    "_SESSIONS_CACHE_GLOBAL_INVALIDATION_VERSION",
    "_session_list_cache_settings_write_version",
}


def __getattr__(name):
    if name in _ROUTE_SESSION_LIST_CACHE_DYNAMIC_EXPORTS:
        return getattr(_route_session_list_cache, name)
    raise AttributeError(f"module {__name__!r} has no attribute {name!r}")


def _prune_orphaned_webui_zero_message_sessions(rows, *, diag_stage=None):
    """#4985 second-pass orphan prune for native-WebUI rows whose ``state.db.messages`` is empty.

    Takes the post-``#3238`` ``webui_sessions`` list (i.e. rows that already
    survived the #3238/#4591 CLI/API-server prune) and returns a NEW list with
    any row whose backing ``state.db.messages`` table is empty removed.
    Removed sids are also persisted to the tombstone via
    ``_record_webui_zero_message_orphan_tombstone`` so
    ``recover_missing_index_sidecars`` does not re-add them to the sidebar
    index on the next poll (avoids the cache-thrash loop where every poll
    does one fsync'd index write + one state.db probe per orphan, forever).

    Invariants preserved:

    - Rows with ``active_stream_id`` / ``has_pending_user_message`` /
      ``worktree_path`` set are NEVER pruned — the inflight / worktree-bound
      / pending safety contract from ``IC_kwDOR1LuPM8AAAABHrkF1Q``.
    - Rows whose ``state.db.messages`` is empty AND that survived the
      upstream ``all_sessions()`` ``#1171`` keep-filter (i.e. titled OR has
      positive ``message_count``) ARE pruned — the post-#1171-survivor
      shape #4985 actually describes (a row that lingers VISIBLY in the
      sidebar because of a stale positive ``message_count`` or a title set
      before the first turn committed).

    This helper is intentionally extracted out of the ``if show_cli_sessions:``
    branch so the prune fires in BOTH branches of
    ``_build_session_list_cache_payload``. Established installs have
    ``settings.show_cli_sessions`` pinned to ``False`` (per
    ``api/config.py:7637-7648``) and those are exactly the long-time users
    who have accumulated the #4985 404 orphans — without hoisting, the
    ``else:`` branch silently skipped the prune and the sidebar kept
    dangling rows that 404 on click (review
    ``IC_kwDOR1LuPM8AAAABHsyFGg``).
    """
    if not rows:
        return list(rows) if rows is not None else []
    _diag = diag_stage if callable(diag_stage) else (lambda *_a, **_k: None)
    # #4985 self-healing: the tombstone is NOT a blind-drop filter at the
    # top of the helper. A row whose sid is in the tombstone is allowed
    # into the gate predicate like any other row — and the post-probe
    # logic below explicitly distinguishes four cases:
    #
    #   1. probe says NOT empty AND sid IS tombstoned → SELF-HEAL: the row
    #      has actually gained messages, so clear the tombstone and keep
    #      the row (do NOT add to missing_webui_orphan_ids). This is the
    #      primary fix for review IC_kwDOR1LuPM8AAAABHvY-dw.
    #   2. probe says empty AND sid IS tombstoned → tombstone persists
    #      (orphan shape unchanged), but the row is excluded from the
    #      returned list so the tombstone continues to suppress it on
    #      this poll too. Do NOT redundantly prune+tombstone (would
    #      cycle).
    #   3. probe says empty AND sid is NOT tombstoned → new orphan: prune
    #      from index, record tombstone, diag_stage.
    #   4. probe says NOT empty AND sid is NOT tombstoned → row has
    #      messages, retain (gate already passes anyway).
    #
    # A blind-drop at the top (the previous behavior) is strictly worse
    # than the orphan it suppresses — it would silently swallow a
    # legitimately-resurfaced row forever, even after the user actually
    # sent messages. The self-healing case is what makes the tombstone a
    # recoverable "this sid is currently empty" signal rather than a
    # permanent hide-list.
    if not rows:
        return []
    # Gate predicate mirrors the inline block that lived here before the
    # helper extract. The (title!='Untitled' OR count>0) clause is what makes
    # this gate actually reach a row #1171 kept — without it, the gate is a
    # no-op because ``all_sessions()`` at ``api/models.py:3892-3898`` (and
    # its full-scan fallback at 3946-3952) has already stripped every
    # (Untitled ∧ count==0 ∧ ¬active_stream_id ∧ ¬has_pending_user_message ∧
    # ¬worktree_path) row before our prune block runs.
    _webui_orphan_probe_rows = [
        s for s in rows
        if _session_source_is_webui(s)
        and not s.get("active_stream_id")
        and not s.get("has_pending_user_message")
        and not s.get("worktree_path")
        and (
            s.get("title", "Untitled") != "Untitled"
            or _numeric_count(s.get("message_count")) > 0
        )
    ]
    if not _webui_orphan_probe_rows:
        return list(rows)
    rows_by_profile_webui: dict[object, list[dict]] = defaultdict(list)
    for row in _webui_orphan_probe_rows:
        rows_by_profile_webui[row.get("profile")].append(row)
    _tombstoned = _load_webui_zero_message_orphan_tombstone()
    self_healed_ids: set[str] = set()
    missing_webui_orphan_ids: set[str] = set()
    still_hidden_ids: set[str] = set()
    for profile_key, profile_rows in rows_by_profile_webui.items():
        probe_ids = [
            str(row.get("session_id")).strip()
            for row in profile_rows
            if str(row.get("session_id") or "").strip()
        ]
        zero_message_sids = agent_session_zero_message_sids(
            probe_ids,
            profile=profile_key if isinstance(profile_key, str) and profile_key else None,
        )
        # Iterate over the actual rows (not just probe_ids) so each sid
        # decision can probe the sidecar for real ``messages``. The r5
        # signal keyed off the row's cached ``message_count`` (which is
        # stale-positive on the very phantom rows #4985 exists to prune:
        # sidecar ``messages`` empty but cached count > 0), so the r5
        # retain branch kept the phantom and re-opened the bug (maintainer
        # review 4584722701, supersedes the r5 cached-count signal). The
        # r6 signal probes ``Session.load(sid).messages`` directly — but
        # ONLY for ``state.db``-empty candidates (the small set; the
        # common live-row path takes the ``else`` branch and pays
        # nothing). Full ``Session.load`` is intentional (vs
        # ``load_metadata_only`` which zeroes the messages array at
        # ``api/models.py:1210``).
        for row in profile_rows:
            sid = str(row.get("session_id") or "").strip()
            if not sid:
                continue
            is_empty = sid in zero_message_sids
            is_tombstoned = sid in _tombstoned
            if is_empty:
                # ``state.db.messages`` is empty. Probe the sidecar JSON
                # for real messages — the cached ``message_count`` alone
                # is stale-positive on phantom rows (sidecar ``messages``
                # empty but cached count > 0) and would retain the very
                # phantom this feature exists to prune (maintainer review
                # 4584722701, supersedes the r5 cached-count signal).
                # Full ``Session.load`` is intentional (vs
                # ``load_metadata_only`` which zeros the messages array
                # at ``api/models.py:1210``); the common live-row path
                # pays nothing because it skips the load via the
                # ``else`` branch below.
                try:
                    from api.models import Session as _Session
                    _loaded = _Session.load(sid)
                    sidecar_has_messages = bool(
                        _loaded is not None and len(_loaded.messages or []) > 0
                    )
                except Exception:
                    logger.debug(
                        "Failed to load sidecar for webui orphan decision %s; "
                        "treating as empty for prune purposes",
                        sid,
                        exc_info=True,
                    )
                    sidecar_has_messages = False
            else:
                # ``state.db.messages`` is non-empty — the conversation is real.
                sidecar_has_messages = True
            if sidecar_has_messages:
                # Real transcript (state.db OR loaded sidecar). Retain; if
                # tombstoned, self-heal so it stops thrashing on recovery.
                if is_tombstoned:
                    self_healed_ids.add(sid)
                continue
            if not is_empty and is_tombstoned:
                # Case 1: SELF-HEAL — clear tombstone, keep row.
                self_healed_ids.add(sid)
            elif is_empty and is_tombstoned:
                # Case 2: still-empty tombstoned row stays hidden this
                # poll (do not add to missing_webui_orphan_ids — would
                # cycle through prune_session_from_index + record).
                still_hidden_ids.add(sid)
            elif is_empty and not is_tombstoned:
                # Case 3: new orphan.
                missing_webui_orphan_ids.add(sid)
            # Case 4 (not empty + not tombstoned): row has messages, retain.
    if self_healed_ids:
        for _sid in self_healed_ids:
            try:
                _clear_webui_zero_message_orphan_tombstone(_sid)
                logger.debug(
                    "self-heal: cleared webui zero-message orphan tombstone "
                    "for %s (state.db.messages now non-empty)",
                    _sid,
                )
            except Exception:
                logger.debug(
                    "Failed to clear webui zero-message orphan tombstone for %s",
                    _sid,
                    exc_info=True,
                )
        _diag("self_heal_webui_zero_message_orphan")
    if missing_webui_orphan_ids:
        for _sid in missing_webui_orphan_ids:
            try:
                prune_session_from_index(_sid)
                _diag("prune_orphaned_webui_zero_message")
            except Exception:
                logger.debug(
                    "Failed to prune orphaned webui zero-message row %s",
                    _sid,
                    exc_info=True,
                )
            # Tombstone the sid in a SECOND step so a tombstone-write failure
            # never blocks the prune itself (the prune still removes the row
            # from the sidebar; only the re-prune avoidance would degrade).
            try:
                _record_webui_zero_message_orphan_tombstone(_sid)
            except Exception:
                logger.debug(
                    "Failed to tombstone webui zero-message orphan %s",
                    _sid,
                    exc_info=True,
                )
    # Return rows excluding both the freshly-pruned orphans AND the
    # tombstoned rows that the probe confirmed are still empty (case 2).
    # Self-healed rows (case 1) and live rows (case 4) stay in the result.
    _hidden = missing_webui_orphan_ids | still_hidden_ids
    return [
        s for s in rows
        if str(s.get("session_id") or "").strip() not in _hidden
    ]


def _build_session_list_cache_payload(
    active_profile: str | None,
    all_profiles: bool,
    show_cli_sessions: bool,
    show_previous_messaging_sessions: bool,
    show_cron_sessions: bool,
    show_claude_code_sessions: bool = True,
    include_archived: bool = False,
    exclude_hidden: bool = False,
    visible_only: bool = False,
    show_webhook_sessions: bool = False,
    show_kanban_sessions: bool = False,
    source_filter: str | None = None,
    sidebar_source: str | None = None,
    archived_limit: int | None = None,
    archived_offset: int = 0,
    diag=None,
) -> dict:
    diag_stage = diag.stage if diag is not None else lambda *_a, **_k: None

    def _session_has_server_visible_messages(session: dict) -> bool:
        """Return True when a non-active sidebar row has a visibility signal.

        Keep this mirror of the non-active server filter narrow and local to
        route behavior so model-layer behavior remains unchanged.
        """
        if not isinstance(session, dict):
            return False
        if _numeric_count(session.get("message_count")) > 0:
            return True

        attention = session.get("attention")
        if not (isinstance(attention, dict) and attention.get("kind")):
            attention = _session_attention_summary(str(session.get("session_id") or ""))
        if isinstance(attention, dict) and attention.get("kind"):
            if _numeric_count(attention.get("count")) > 0:
                return True

        return bool(
            session.get("is_streaming")
            or session.get("active_stream_id")
            or session.get("pending_user_message")
            or session.get("has_pending_user_message")
        )

    def _all_sessions_for_sidebar():
        kwargs = {"diag": diag, "include_lineage_metadata": False}
        if _callable_accepts_kwarg(all_sessions, "sidebar_metadata_only"):
            kwargs["sidebar_metadata_only"] = True
        if _callable_accepts_kwarg(all_sessions, "include_lineage_metadata"):
            return all_sessions(**kwargs)
        # Focused tests and third-party callers sometimes monkeypatch
        # routes.all_sessions with the historical diag-only signature.
        return all_sessions(diag=diag)

    diag_stage("all_sessions")
    webui_sessions = _all_sessions_for_sidebar()
    diag_stage("reconcile_stale_stream_state")
    if _reconcile_stale_stream_state_for_session_rows(webui_sessions):
        diag_stage("all_sessions_after_stale_stream_reconcile")
        webui_sessions = _all_sessions_for_sidebar()
    diag_stage("normalize_cli_rows")
    show_cli_sessions = bool(show_cli_sessions)
    show_previous_messaging_sessions = bool(show_previous_messaging_sessions)
    show_cron_sessions = bool(show_cron_sessions)
    show_webhook_sessions = bool(show_webhook_sessions)
    show_kanban_sessions = bool(show_kanban_sessions)
    webui_sessions = [_normalize_sidebar_source_flags(s) for s in webui_sessions]
    if show_cli_sessions:
        diag_stage("get_cli_sessions")
        if _callable_accepts_kwarg(get_cli_sessions, "include_claude_code"):
            cli = get_cli_sessions(
                source_filter=source_filter,
                all_profiles=all_profiles,
                include_claude_code=show_claude_code_sessions,
            )
        else:
            # Focused tests sometimes monkeypatch routes.get_cli_sessions with
            # the historical two-keyword signature.
            cli = get_cli_sessions(
                source_filter=source_filter,
                all_profiles=all_profiles,
            )
        diag_stage("merge_cli_sessions")
        cli_by_id = {s["session_id"]: s for s in cli}
        # #3238/#4591: reconcile orphaned imported sidecars. When a CLI or
        # API-server session is clicked in WebUI it gets a WebUI-owned sidecar
        # that all_sessions() returns independently of state.db. If the user
        # later deletes the backing agent session outside WebUI, the sidecar is
        # never pruned and the stale row lingers in the sidebar forever (there
        # is no WebUI delete affordance for read-only imported rows).
        # Drop rows whose backing agent row is genuinely gone. We probe
        # state.db directly (agent_session_rows_existing) rather than trust
        # cli_by_id absence, because get_cli_sessions() caps at
        # CLI_VISIBLE_SESSION_LIMIT (20) — an existing session can fall
        # out of that window and look deleted. Native WebUI sessions
        # (source == "webui") that merely have a CLI ancestor are never
        # pruned by this path.
        #
        # #4985: parallel pass for native-WebUI rows that have a backing
        # agent row in state.db but zero messages (a `+`-click that opened a
        # row but the first turn never committed, or a sidebar nav that
        # opened then closed before any message landed). The same #3238
        # helper doesn't catch these because source == "webui" is excluded
        # above, and the WebUI delete affordance isn't exposed for them,
        # so they would otherwise linger forever. Inflight first-turn
        # safety is preserved by gating on `active_stream_id` (after
        # _reconcile_stale_stream_state has cleared stale stream ids).
        _orphan_probe_rows = []
        _kept_after_orphan_prune = []
        for s in webui_sessions:
            _sid = s.get("session_id")
            if (
                _sid
                and (is_cli_session_row(s) or _is_api_server_sidecar_row(s))
                and not _session_source_is_webui(s)
                and _sid not in cli_by_id
            ):
                _orphan_probe_rows.append(s)
            else:
                _kept_after_orphan_prune.append(s)
        if _orphan_probe_rows:
            rows_by_profile: dict[object, list[dict]] = defaultdict(list)
            for row in _orphan_probe_rows:
                rows_by_profile[row.get("profile")].append(row)
            missing_orphan_ids: set[str] = set()
            for profile_key, rows in rows_by_profile.items():
                probe_ids = [
                    str(row.get("session_id")).strip()
                    for row in rows
                    if str(row.get("session_id") or "").strip()
                ]
                existing = agent_session_rows_existing(
                    probe_ids,
                    profile=profile_key if isinstance(profile_key, str) and profile_key else None,
                )
                for row in rows:
                    _sid = str(row.get("session_id") or "").strip()
                    if _sid and _sid not in existing:
                        missing_orphan_ids.add(_sid)
            for s in _orphan_probe_rows:
                _sid = str(s.get("session_id") or "").strip()
                if _sid in missing_orphan_ids:
                    try:
                        prune_session_from_index(_sid)
                    except Exception:
                        logger.debug(
                            "Failed to prune orphaned agent sidecar %s",
                            _sid,
                            exc_info=True,
                        )
                    diag_stage("prune_orphaned_agent_sidecar")
                    continue
                _kept_after_orphan_prune.append(s)
        # #4985 second pass — probe state.db.messages for native-WebUI rows
        # that *survived* the upstream all_sessions() #1171 keep-filter (so
        # the row is TITLED or has a POSITIVE message_count, meaning it IS
        # shown in the sidebar — and the 404 click reported in #4985 happens),
        # BUT whose actual state.db.messages table is empty (the ground-truth
        # probe). This is the orphan shape #4985 actually describes: a row
        # that lingers VISIBLY in the sidebar because of a stale positive
        # message_count or a title set before the first turn committed.
        #
        # The (title!='Untitled' OR count>0) clause is the part that makes
        # this gate actually reach a row #1171 kept. Without it, the gate is
        # a no-op because all_sessions() at api/models.py:3892-3898 and
        # 3946-3952 has already stripped every (Untitled ∧ count==0 ∧
        # ¬active_stream_id ∧ ¬has_pending_user_message ∧ ¬worktree_path)
        # row before this point — making the earlier 6-condition gate a
        # no-op against the real pipeline (review IC_kwDOR1LuPM8AAAABHrkF1Q).
        #
        # Implementation lives in ``_prune_orphaned_webui_zero_message_sessions``
        # above so the prune runs in BOTH branches of this function
        # (``if show_cli_sessions:`` AND ``else:``). Established installs
        # have ``settings.show_cli_sessions`` pinned to False (per
        # api/config.py:7637-7648) and those are exactly the long-time
        # users who accumulated the #4985 404 orphans — without hoisting,
        # the ``else:`` branch silently skipped the prune
        # (review IC_kwDOR1LuPM8AAAABHsyFGg).
        #
        # Inflight / worktree / pending safety: same as before — any row
        # still carrying active_stream_id / has_pending_user_message /
        # worktree_path is never pruned, even if its messages table is
        # momentarily empty. _reconcile_stale_stream_state_for_session_rows
        # at line 2224 has already cleared stale stream ids above this point.
        webui_sessions = _prune_orphaned_webui_zero_message_sessions(
            _kept_after_orphan_prune,
            diag_stage=diag_stage,
        )
        for s in webui_sessions:
            meta = cli_by_id.get(s.get("session_id"))
            if not meta:
                continue
            if _is_messaging_session_record(meta):
                s.update(_merge_cli_sidebar_metadata(s, meta))
                if s.get("session_id") != meta.get("session_id"):
                    s["session_id"] = meta.get("session_id")
            else:
                for key in ("source_tag", "raw_source", "session_source", "source_label"):
                    if not s.get(key) and meta.get(key):
                        s[key] = meta[key]
        webui_sessions = [_normalize_sidebar_source_flags(s) for s in webui_sessions]
        # Apply the same CLI visibility semantics to imported local copies so
        # low-value imported artifacts do not leak into the sidebar.
        webui_sessions = [s for s in webui_sessions if is_cli_session_row_visible(s)]
        represented_webui_ids = set()
        for s in webui_sessions:
            represented_webui_ids.update(_session_lineage_ids(s))
        deduped_cli = _dedupe_cli_sidebar_sessions_for_api(
            cli,
            represented_webui_ids,
            show_cron_sessions=show_cron_sessions,
            show_webhook_sessions=show_webhook_sessions,
            show_kanban_sessions=show_kanban_sessions,
            source_filter=source_filter,
        )
    else:
        diag_stage("filter_webui_sessions")
        webui_sessions = [s for s in webui_sessions if not _is_cli_session_for_settings(s)]
        # #4985 second pass — see _prune_orphaned_webui_zero_message_sessions
        # for the gate predicate and the post-#1171-survivor rationale. The
        # prune MUST run here too: established installs have
        # ``settings.show_cli_sessions`` pinned to False
        # (api/config.py:7637-7648) and those are exactly the long-time
        # users who accumulated the 404 orphans — review
        # IC_kwDOR1LuPM8AAAABHsyFGg. Without this call the else branch
        # silently skipped the prune and the sidebar kept dangling rows.
        webui_sessions = _prune_orphaned_webui_zero_message_sessions(
            webui_sessions,
            diag_stage=diag_stage,
        )
        deduped_cli = []
    diag_stage("sort_sessions")
    merged = webui_sessions + deduped_cli
    merged.sort(
        key=lambda s: s.get("last_message_at") or s.get("updated_at", 0) or 0,
        reverse=True,
    )
    # ── Profile scoping (#1611) ────────────────────────────────────────
    # Default: filter to the active profile. ?all_profiles=1 opts into
    # the aggregate view used by the "All profiles" sidebar toggle.
    # The other_profile_count is always returned so the UI can render
    # the "Show N from other profiles" affordance without sending the
    # cross-profile rows by default.
    #
    # IMPORTANT: scope BEFORE _keep_latest_messaging_session_per_source.
    # _messaging_source_key is profile-blind (#1614 follow-up): if the
    # same Slack/Telegram identity has sessions in profiles A and B, a
    # profile-blind dedupe would discard the older one even when scoped
    # to its own profile, leaving that profile with zero rows for that
    # source. Filter first so the dedupe operates only within the active
    # profile's rows.
    diag_stage("profile_scope")
    if all_profiles:
        scoped = merged
        other_profile_count = 0
    else:
        scoped = [s for s in merged if _profiles_match(s.get("profile"), active_profile)]
        other_profile_count = 0 if _is_isolated_profile_mode() else len(merged) - len(scoped)
    diag_stage("messaging_dedupe")
    archived_scoped = _keep_latest_messaging_session_per_source(
        list(scoped),
        show_previous_messaging_sessions=show_previous_messaging_sessions,
    )
    visible_scoped = _keep_latest_messaging_session_per_source(
        [s for s in scoped if not s.get("archived")],
        show_previous_messaging_sessions=show_previous_messaging_sessions,
    )
    if show_cli_sessions:
        diag_stage("cli_cap")
        archived_scoped = _cap_recent_cli_sessions(archived_scoped)
        visible_scoped = _cap_recent_cli_sessions(visible_scoped)
    if visible_only:
        archived_scoped = [
            s for s in archived_scoped if _session_has_server_visible_messages(s)
        ]
        visible_scoped = [
            s for s in visible_scoped if _session_has_server_visible_messages(s)
        ]
    if exclude_hidden:
        archived_scoped = [s for s in archived_scoped if not s.get("default_hidden")]
        visible_scoped = [s for s in visible_scoped if not s.get("default_hidden")]
    archived_webui_count = sum(
        1 for s in archived_scoped
        if s.get("archived") and not _is_cli_session_for_settings(s)
    )
    archived_cli_count = sum(
        1 for s in archived_scoped
        if s.get("archived") and _is_cli_session_for_settings(s)
    )
    archived_count = archived_webui_count + archived_cli_count
    def _filter_sidebar_source(rows: list[dict]) -> list[dict]:
        if sidebar_source == "webui":
            return [s for s in rows if not _is_cli_session_for_settings(s)]
        if sidebar_source == "cli":
            return [s for s in rows if _is_cli_session_for_settings(s)]
        return list(rows)

    full_scoped_all_sources = archived_scoped if include_archived else visible_scoped
    webui_session_count = sum(
        1 for s in full_scoped_all_sources
        if not _is_cli_session_for_settings(s)
    )
    cli_session_count = sum(
        1 for s in full_scoped_all_sources
        if _is_cli_session_for_settings(s)
    )
    visible_scoped_filtered = _filter_sidebar_source(visible_scoped)
    archived_scoped_filtered = _filter_sidebar_source(archived_scoped)
    scoped = _filter_sidebar_source(full_scoped_all_sources)
    if include_archived and archived_limit is not None:
        try:
            normalized_archived_limit = max(0, int(archived_limit))
        except (TypeError, ValueError):
            normalized_archived_limit = None
        try:
            normalized_archived_offset = max(0, int(archived_offset or 0))
        except (TypeError, ValueError):
            normalized_archived_offset = 0
        if normalized_archived_limit is not None:
            visible_rows_for_page = [s for s in visible_scoped_filtered if not s.get("archived")]
            archived_rows_for_page = [s for s in archived_scoped_filtered if s.get("archived")]
            scoped = visible_rows_for_page + archived_rows_for_page[
                normalized_archived_offset: normalized_archived_offset + normalized_archived_limit
            ]
    sidebar_reference_sessions: list[dict] = []
    if not include_archived:
        sidebar_reference_sessions = _hidden_archived_sidebar_reference_sessions(
            visible_scoped_filtered,
            archived_scoped_filtered,
        )
    if not include_archived:
        diag_stage("filter_archived_sessions")
    diag_stage("visible_lineage_metadata")
    _enrich_sidebar_lineage_metadata(scoped)
    # Delegated subagent children (#5307) are view-only, owned by the delegate
    # runner. The model-layer batch overlay above has already applied the
    # authoritative source and view-only flags before this route runs.
    def _coerce_subagent_rows(_rows):
        for _r in _rows:
            if not isinstance(_r, dict):
                continue
            _src = (
                str(_r.get("source_tag") or _r.get("raw_source")
                    or _r.get("session_source") or _r.get("source") or "").strip().lower()
            )
            if _src == "subagent":
                _r["read_only"] = True
                _r["is_cli_session"] = False
    _coerce_subagent_rows(scoped)
    _coerce_subagent_rows(sidebar_reference_sessions)
    return {
        "sessions": [
            dict(s) if isinstance(s, dict) else {}
            for s in scoped
        ],
        "sidebar_reference_sessions": [
            dict(s) if isinstance(s, dict) else {}
            for s in sidebar_reference_sessions
        ],
        "cli_count": len(deduped_cli),
        "archived_count": archived_count,
        "archived_webui_count": archived_webui_count,
        "archived_cli_count": archived_cli_count,
        "webui_session_count": webui_session_count,
        "cli_session_count": cli_session_count,
        "include_archived": include_archived,
        "archived_limit": archived_limit,
        "archived_offset": archived_offset,
        "all_profiles": all_profiles,
        "active_profile": active_profile,
        "other_profile_count": other_profile_count,
        "settings": {
            "show_cli_sessions": show_cli_sessions,
            "show_previous_messaging_sessions": show_previous_messaging_sessions,
            "show_cron_sessions": show_cron_sessions,
            "show_claude_code_sessions": show_claude_code_sessions if show_cli_sessions else False,
            "show_webhook_sessions": show_webhook_sessions,
            "show_kanban_sessions": show_kanban_sessions,
        },
    }


def _session_list_payload_to_response(payload: dict) -> dict:
    safe_merged = []
    runtime_rows = _session_list_cache_overlay_runtime_rows(payload.get("sessions", []) or [])
    # Read the redaction setting ONCE for the whole response and thread it through
    # every row, instead of letting each row's _redact_text() re-read settings.json
    # from disk (per title). The _sidebar_session_response_item -> _redact_text(_enabled=...)
    # plumbing already exists; this wires the caller so the sidebar list path gets the
    # same read-once optimization redact_session_data() already uses. On a large list
    # this was the multi-second response_write stage in /api/sessions diagnostics. (#4662 Phase 3)
    # load_settings is imported at module scope (below); this function only runs at
    # request time, well after module load, so no lazy import is needed.
    try:
        _redact_enabled = bool(load_settings().get("api_redact_enabled", True))
    except Exception:
        _redact_enabled = True  # fail safe: redact when settings are unreadable
    for s in runtime_rows:
        item = _sidebar_session_response_item(s, redact_enabled=_redact_enabled) if isinstance(s, dict) else {}
        safe_merged.append(item)
    safe_reference = []
    for s in payload.get("sidebar_reference_sessions", []) or []:
        item = _sidebar_session_response_item(s, redact_enabled=_redact_enabled) if isinstance(s, dict) else {}
        if item:
            item["_sidebar_reference_only"] = True
        safe_reference.append(item)
    response = {
        "sessions": safe_merged,
        "sidebar_reference_sessions": safe_reference,
        "cli_count": int(payload.get("cli_count", 0)),
        "archived_count": int(payload.get("archived_count", 0)),
        "archived_webui_count": int(payload.get("archived_webui_count", 0)),
        "archived_cli_count": int(payload.get("archived_cli_count", 0)),
        "include_archived": bool(payload.get("include_archived", False)),
        "all_profiles": bool(payload.get("all_profiles", False)),
        "active_profile": payload.get("active_profile"),
        "other_profile_count": int(payload.get("other_profile_count", 0)),
        "server_time": time.time(),
        "server_tz": time.strftime("%z"),
    }
    if "webui_session_count" in payload:
        response["webui_session_count"] = int(payload.get("webui_session_count", 0))
    if "cli_session_count" in payload:
        response["cli_session_count"] = int(payload.get("cli_session_count", 0))
    if payload.get("archived_limit") is not None:
        response["archived_limit"] = int(payload.get("archived_limit") or 0)
        response["archived_offset"] = int(payload.get("archived_offset") or 0)
    return response


def _hidden_archived_sidebar_reference_sessions(
    visible_rows: list[dict],
    archived_rows: list[dict],
) -> list[dict]:
    """Return hidden archived ancestors needed for client-side sidebar nesting.

    The default sidebar payload intentionally omits archived sessions. The
    browser still needs a tiny reference row for an archived parent/ancestor so
    `_attachChildSessionsToSidebarRows()` can suppress its visible child rows
    instead of rendering them as orphan top-level conversations (#4293).
    """
    archived_by_id = {
        str(row.get("session_id")): row
        for row in archived_rows
        if isinstance(row, dict) and row.get("archived") and row.get("session_id")
    }
    if not archived_by_id:
        return []

    references: list[dict] = []
    added: set[str] = set()
    visible_ids = {
        str(row.get("session_id"))
        for row in visible_rows
        if isinstance(row, dict) and row.get("session_id")
    }

    for row in visible_rows:
        if not isinstance(row, dict):
            continue
        parent_id = str(row.get("parent_session_id") or "").strip()
        seen: set[str] = set()
        while parent_id and parent_id not in seen:
            seen.add(parent_id)
            if parent_id in visible_ids:
                break
            parent = archived_by_id.get(parent_id)
            if not parent:
                break
            if parent_id not in added:
                references.append(parent)
                added.add(parent_id)
            parent_id = str(parent.get("parent_session_id") or "").strip()

    return references


def _get_cached_session_list_payload(
    *,
    key: tuple,
    builder,
    diag=None,
) -> dict:
    if diag is not None:
        try:
            diag.stage("session_list_cache_lookup")
        except Exception:
            pass

    cached, is_fresh = _session_list_cache_get(key, allow_stale=True)
    if cached is not None and is_fresh:
        if diag is not None:
            try:
                diag.stage("session_list_cache_hit")
            except Exception:
                pass
        return cached

    stale = cached  # now actually a stale payload when one exists, else None
    stale_reason = _session_list_cache_stale_reason(key) if stale is not None else None
    if stale is not None and stale_reason != "source":
        event, is_owner = _session_list_cache_claim_rebuild(key)
        if is_owner:
            if diag is not None:
                try:
                    diag.stage("session_list_cache_stale_background_rebuild")
                except Exception:
                    pass

            def _rebuild_stale_session_list_cache():
                try:
                    rebuild_attempts = 0
                    while True:
                        invalidation_stamp = _session_list_cache_invalidation_stamp(key)
                        try:
                            payload = builder()
                        except Exception:
                            logger.exception(
                                "session list stale-cache background rebuild failed"
                            )
                            return
                        if (
                            _session_list_cache_invalidation_stamp(key) == invalidation_stamp
                            and _session_list_cache_set(
                                key,
                                payload,
                                expected_invalidation_stamp=invalidation_stamp,
                            )
                        ):
                            return
                        rebuild_attempts += 1
                        if rebuild_attempts >= 3:
                            return
                finally:
                    _session_list_cache_done(key, event)

            try:
                thread = threading.Thread(
                    target=_rebuild_stale_session_list_cache,
                    name="session-list-cache-rebuild",
                    daemon=True,
                )
                thread.start()
            except Exception:
                _session_list_cache_done(key, event)
        elif diag is not None:
            try:
                diag.stage("session_list_cache_stale_return")
            except Exception:
                pass
        return stale

    event, is_owner = _session_list_cache_claim_rebuild(key)
    if is_owner:
        if diag is not None:
            try:
                diag.stage("session_list_cache_rebuild_owner")
            except Exception:
                pass
        try:
            rebuild_attempts = 0
            while True:
                invalidation_stamp = _session_list_cache_invalidation_stamp(key)
                payload = builder()
                if (
                    _session_list_cache_invalidation_stamp(key) == invalidation_stamp
                    and _session_list_cache_set(
                        key,
                        payload,
                        expected_invalidation_stamp=invalidation_stamp,
                    )
                ):
                    if diag is not None:
                        try:
                            diag.stage("session_list_cache_stored")
                        except Exception:
                            pass
                    return payload
                rebuild_attempts += 1
                if diag is not None:
                    try:
                        diag.stage("session_list_cache_invalidated_during_rebuild")
                    except Exception:
                        pass
                if rebuild_attempts >= 3:
                    return payload
        finally:
            _session_list_cache_done(key, event)

    if diag is not None:
        try:
            if stale is not None:
                diag.stage("session_list_cache_wait_stale")
            else:
                diag.stage("session_list_cache_wait")
        except Exception:
            pass

    if stale is not None:
        timeout = _SESSIONS_CACHE_STALE_WAIT_SECONDS
    else:
        timeout = _SESSIONS_CACHE_WAIT_SECONDS
    event.wait(timeout)

    latest, is_fresh = _session_list_cache_get(key, allow_stale=False)
    if latest is not None:
        if diag is not None:
            try:
                diag.stage("session_list_cache_wait_hit")
            except Exception:
                pass
        return latest

    if stale is not None:
        if diag is not None:
            try:
                diag.stage("session_list_cache_wait_stale_fallback")
            except Exception:
                pass
        return stale

    # Safety path if the owner died before storing anything.
    if diag is not None:
        try:
            diag.stage("session_list_cache_fallback_rebuild")
        except Exception:
            pass
    invalidation_stamp = _session_list_cache_invalidation_stamp(key)
    payload = builder()
    if _session_list_cache_invalidation_stamp(key) == invalidation_stamp:
        _session_list_cache_set(
            key,
            payload,
            expected_invalidation_stamp=invalidation_stamp,
        )
    return payload

from api.config import (
    STATE_DIR,
    SESSION_DIR,
    DEFAULT_WORKSPACE,
    DEFAULT_MODEL,
    SESSIONS,
    SESSIONS_MAX,
    LOCK,
    STREAMS,
    STREAMS_LOCK,
    CANCEL_FLAGS,
    STREAM_LAST_EVENT_ID,
    SERVER_START_TIME,
    _resolve_cli_toolsets,
    get_available_models,
    get_available_models_for_session_visit,
    _provider_is_known_or_configured,
    IMAGE_EXTS,
    MD_EXTS,
    MIME_MAP,
    MAX_FILE_BYTES,
    MAX_UPLOAD_BYTES,
    ACTIVE_RUNS,
    ACTIVE_RUNS_LOCK,
    register_stream_owner,
    register_session_writeback_owner,
    clear_session_writeback_owner_if_owned,
    stream_owner_session_id,
    peek_stream,
    unregister_stream_owner,
    CHAT_LOCK,
    _get_session_agent_lock,
    CUSTOM_MODELS_ENDPOINT_TIMEOUT_SECONDS,
    load_settings,
    persisted_speech_settings_keys,
    save_settings,
    SETTINGS_FILE,
    set_hermes_default_model,
    canonical_model_provider_lane,
    model_with_provider_context,
    get_reasoning_status,
    set_reasoning_display,
    set_reasoning_effort,
    create_stream_channel,
    get_config,
    get_webui_session_save_mode,
    get_config_snapshot,
    STREAM_GOAL_RELATED,
    PENDING_GOAL_CONTINUATION,
    _get_config_path,
    _load_yaml_config_file,
    _save_yaml_config_file,
    reload_config,
    get_config_for_profile_home,
    _cfg_lock,
    PENDING_BG_TASK_COMPLETIONS,
    _parse_provider_qualified_model_id,
)
from api import config as api_config
from api.helpers import (
    require,
    bad,
    safe_resolve,
    arm_connection_close_if_body_pending,
    j,
    t,
    read_body,
    MAX_BODY_BYTES,
    _security_headers,
    _sanitize_error,
    redact_session_data,
    public_session_projection,
    strip_public_internal_fields,
    _redact_text,
    _CLIENT_DISCONNECT_ERRORS,
)
from api.agent_health import build_agent_health_payload
from api.gateway_chat import gateway_chat_config_status
from api.request_diagnostics import RequestDiagnostics
from api.system_health import build_system_health_payload


# ── Non-streaming custom-provider connection authority ───────────────────────
#
# Several WebUI consumers build their own AIAgent outside the streaming path
# (POST /api/chat, manual compression, update summary, git commit message,
# handoff summary). They resolve a model/provider, ask the Hermes runtime
# provider for a connection, then had to apply the named ``custom:<slug>``
# record's authority by hand.
#
# ``apply_custom_provider_connection_authority`` returns only the THREE
# connection fields, and three fields are not a complete constructor contract:
# AIAgent also takes ``api_mode`` (wire protocol), ``credential_pool``
# (credential source) and ``acp_command``/``acp_args`` (subprocess transport).
# A consumer that replaced the endpoint and credential while leaving those to
# default — or, worse, to the ambient runtime — built a mixed-authority agent:
# an exact row's ``api_mode: anthropic_messages`` or its own pool never reached
# the constructor at all. These helpers carry the WHOLE bundle instead, exactly
# as ``api/streaming.py`` does for the streaming path.

# The constructor-routing fields that travel with provider/base_url/api_key as
# one authority. Mirrors ``api.streaming._RUNTIME_BUNDLE_FIELDS`` and
# ``api.config.CUSTOM_CONNECTION_SIDE_FIELDS``.
_AGENT_BUNDLE_SIDE_FIELDS = ("api_mode", "acp_command", "acp_args", "credential_pool")


def _resolve_agent_connection_bundle(
    resolved_provider,
    resolved_api_key,
    resolved_base_url,
    runtime_provider=None,
    *,
    lookup_provider=None,
):
    """Return the COMPLETE constructor-routing bundle for a non-streaming send.

    Keys: ``provider``, ``base_url``, ``api_key`` plus every field in
    :data:`_AGENT_BUNDLE_SIDE_FIELDS`. Pass the whole dict to the constructor
    via :func:`_agent_bundle_kwargs` — the endpoint/credential and the
    transport/protocol/pool fields are ONE authority.

    ``runtime_provider`` is the dict ``resolve_runtime_provider`` returned. It
    matters: the merge seeds the side fields from it and then decides, by
    endpoint provenance, whether they are same-authority (keep) or the ambient
    provider's (clear). Omitting it silently drops that signal.

    ``lookup_provider`` preserves the pre-canonicalization ``custom:<slug>``
    identity, since the merge rewrites a resolved bundle's provider to the
    generic ``custom``.

    Raises :class:`api.config.CustomProviderRouteError` when the merge returns a
    TERMINAL route verdict — a named ``custom:<slug>`` that resolved no complete
    ``(api_key, base_url)`` pair. This is the single chokepoint for every
    non-streaming and auxiliary consumer precisely because an incomplete bundle
    is NOT a refusal at the constructor: AIAgent's ``_init_openai_client()``
    only honours an explicit pair when BOTH fields are truthy and otherwise
    calls ``_routed_client_kwargs()``, which re-resolves a provider and can
    reach the ambient endpoint or the init-time fallback chain. Returning the
    bundle with a hole in it would therefore route the send somewhere the user
    never asked for; raising here keeps the refusal terminal for all five call
    sites (POST /api/chat, manual compression, update summary, git commit
    message, handoff summary) without each having to remember to check.

    The exception subclasses ``ValueError``, so the existing ``except
    ValueError`` / broad-``except`` handlers at those call sites already turn it
    into a controlled 400 or a deterministic non-LLM fallback.
    """
    return api_config.raise_for_custom_provider_route(
        api_config.merge_custom_provider_runtime_bundle(
            resolved_provider,
            resolved_api_key,
            resolved_base_url,
            runtime_provider,
            lookup_provider=lookup_provider or resolved_provider,
        )
    )


def _agent_bundle_kwargs(agent_cls, bundle):
    """Return the bundle's side-field kwargs supported by ``agent_cls``.

    ``api_mode``/``acp_command``/``acp_args``/``credential_pool`` were added to
    AIAgent over several releases, so gate each on the constructor signature the
    way the streaming path does rather than raising TypeError against an older
    hermes-agent build. Values come from the BUNDLE, never from the runtime
    provider dict: a custom-provider override clears these, and reading them off
    the runtime would re-introduce the authority the merge just replaced.
    """
    import inspect as _inspect

    try:
        params = set(_inspect.signature(agent_cls.__init__).parameters)
    except (TypeError, ValueError):
        return {}
    return {
        field: bundle[field]
        for field in _AGENT_BUNDLE_SIDE_FIELDS
        if field in params
    }


def _auxiliary_main_runtime(bundle, model):
    """Return the ``main_runtime`` an auxiliary client must receive for a bundle.

    When the auxiliary client answers, AIAgent is never built, so this dict is
    the ONLY place the resolved authority reaches the wire. It therefore carries
    the same whole bundle :func:`_agent_bundle_kwargs` hands the constructor —
    endpoint and credential plus every field in
    :data:`_AGENT_BUNDLE_SIDE_FIELDS`. Sending only provider/model/base_url/
    api_key silently downgraded an exact row's ``api_mode``
    (``anthropic_messages`` fell back to chat completions) and dropped the
    credential pool/ACP transport that belong to the same record.
    """
    runtime = {
        "provider": bundle["provider"],
        "model": model,
        "base_url": bundle["base_url"],
        "api_key": bundle["api_key"],
    }
    for field in _AGENT_BUNDLE_SIDE_FIELDS:
        runtime[field] = bundle[field]
    return runtime


def _kanban_unknown_endpoint(handler, parsed, method: str) -> bool:
    """Return a Kanban-specific 404 for stale clients/obsolete endpoint shapes."""
    return bad(
        handler,
        (
            f"unknown Kanban endpoint: {method} {parsed.path}. "
            "If this appeared after a WebUI update, your browser may be running "
            "a stale cached bundle; use Hard refresh now, then reopen Kanban."
        ),
        status=404,
    ) or True


# A cancelled worker that stays in ACTIVE_RUNS longer than this is treated as
# stuck (e.g. blocked in C-level provider I/O and never reaching its finally).
# Once the cancel has been outstanding past this grace window, the run row can
# no longer protect the session's active_stream_id/pending_* from stale
# cleanup: _clear_stale_stream_state() clears them, and every delayed cancel
# finalizer (api/streaming.py _finalize_cancelled_turn) is generation-guarded
# under the session lock — it no-ops unless the session still points at the
# cancelled stream — so clearing early cannot clobber a newer turn (#6623).
_STALE_CANCELLED_RUN_GRACE_SECONDS = 60.0


def _cancelled_run_is_stale(run_entry) -> bool:
    """Return True when an ACTIVE_RUNS row belongs to a cancel that has been
    outstanding longer than the stale grace window.

    ``cancelled_at`` is stamped by cancel_stream() when it flips the run to
    phase="cancelling". ``started_at`` is accepted as a fallback anchor so runs
    cancelled before the stamp was introduced are still reclaimed eventually.
    """
    try:
        from api import config as _live_config

        return _live_config.active_run_cancel_is_stale(
            run_entry,
            grace_seconds=_STALE_CANCELLED_RUN_GRACE_SECONDS,
        )
    except Exception:
        return False


def _clear_stale_stream_state(session) -> bool:
    """Clear persisted streaming flags when the in-memory stream no longer exists.

    A server restart or worker crash can leave active_stream_id/pending_* in the
    session JSON while STREAMS is empty. The frontend then keeps reconnecting to
    a dead stream and shows a permanent running/thinking state.

    SAFETY (#1558): If ``session`` was loaded with ``metadata_only=True``, its
    ``messages`` array is empty by design and calling ``save()`` would
    atomically overwrite the on-disk JSON, wiping the conversation. In that
    case we re-load the full session before mutating, so the persisted
    write carries the real messages forward.
    """
    stream_id = getattr(session, "active_stream_id", None)
    if not stream_id:
        return False
    with STREAMS_LOCK:
        stream_alive = stream_id in STREAMS
    if stream_alive:
        return False
    try:
        from api import config as _live_config
        with _live_config.ACTIVE_RUNS_LOCK:
            worker_alive = stream_id in (_live_config.ACTIVE_RUNS or {})
    except Exception:
        worker_alive = False
    if worker_alive:
        # #6623: a worker stuck in C-level I/O may never reach its finally to
        # unregister the run, so ACTIVE_RUNS could hold the row forever and
        # block stale cleanup indefinitely. A *cancelled* run (cancel_stream()
        # stamped phase="cancelling" + cancelled_at) that has not unwound past
        # the grace window is treated as stale — clear the session anyway. The
        # _stream_writeback_is_current() guard rejects any eventual writeback
        # from the stuck worker, so this cannot clobber a newer turn.
        try:
            with _live_config.ACTIVE_RUNS_LOCK:
                run_entry = dict((_live_config.ACTIVE_RUNS or {}).get(stream_id) or {})
        except Exception:
            run_entry = {}
        if not _cancelled_run_is_stale(run_entry):
            logger.debug(
                "_clear_stale_stream_state: stream %s for session %s missing SSE channel "
                "but worker bookkeeping is still active; deferring stale cleanup",
                stream_id,
                getattr(session, "session_id", "?"),
            )
            return False
        logger.info(
            "_clear_stale_stream_state: stream %s for session %s missing SSE channel and "
            "cancelled run is stale (cancelled_at=%s); clearing stale stream state (#6623)",
            stream_id,
            getattr(session, "session_id", "?"),
            run_entry.get("cancelled_at"),
        )
    grace_seconds = 30.0
    try:
        from api.models import _REPAIR_STALE_PENDING_GRACE_SECONDS
        grace_seconds = float(_REPAIR_STALE_PENDING_GRACE_SECONDS)
        pending_started_at = getattr(session, "pending_started_at", None)
        pending_age = time.time() - float(pending_started_at) if pending_started_at else None
    except Exception:
        pending_age = None
    if (
        getattr(session, "pending_user_message", None)
        and pending_age is not None
        and pending_age < grace_seconds
    ):
        logger.debug(
            "_clear_stale_stream_state: stream %s for session %s missing SSE channel "
            "but pending turn is %.1fs old; waiting for %.1fs stale-repair grace",
            stream_id,
            getattr(session, "session_id", "?"),
            pending_age,
            grace_seconds,
        )
        return False

    # ── #1558 P0 safety: if we were handed a metadata-only stub, reload the
    # full session before touching persisted state. The original
    # metadata-only object is left untouched so the caller's read path is
    # unaffected.
    original_stub = session  # SHOULD-FIX #1 (Opus): keep reference so we can
                             # patch the caller's in-memory copy after a
                             # successful clear, avoiding one ghost SSE
                             # reconnect on the very next /api/session GET.
    if getattr(session, "_loaded_metadata_only", False):
        try:
            from api.models import get_session as _get_session
            session = _get_session(session.session_id, metadata_only=False)
        except Exception:
            # If we cannot upgrade to a full load (file gone, decode error,
            # etc.) bail without clearing — better to leave a stale
            # active_stream_id than to wipe the conversation.
            logger.warning(
                "_clear_stale_stream_state: refused to clear stale stream %s "
                "for session %s — full reload failed and we will not save a "
                "metadata-only stub. See #1558.",
                stream_id, getattr(session, "session_id", "?"),
            )
            return False
        if session is None:
            return False
        # The full-load path may have already repaired stale pending fields
        # via _repair_stale_pending(); only re-assert if still set.
        if not getattr(session, "active_stream_id", None):
            # Patch the caller's stub so its read path also sees the cleared
            # field (matches the Opus SHOULD-FIX #1 — without this, /api/session
            # would briefly return the stale active_stream_id and the frontend
            # would attempt one ghost SSE reconnect before recovering).
            try:
                original_stub.active_stream_id = None
                if hasattr(original_stub, "pending_user_message"):
                    original_stub.pending_user_message = None
                if hasattr(original_stub, "pending_attachments"):
                    original_stub.pending_attachments = []
                if hasattr(original_stub, "pending_started_at"):
                    original_stub.pending_started_at = None
                if hasattr(original_stub, "pending_user_source"):
                    original_stub.pending_user_source = None
            except Exception:
                pass
            return False

    # ── #1533 race fix: acquire the per-session lock and re-read
    # active_stream_id under it. A concurrent chat_start may have already
    # registered a new stream after our STREAMS_LOCK check above; in that
    # case we must NOT clobber its session.active_stream_id.
    with _get_session_agent_lock(session.session_id):
        if getattr(session, "active_stream_id", None) != stream_id:
            return False
        if getattr(session, "pending_user_message", None):
            try:
                from api.models import _apply_core_sync_or_error_marker, _get_profile_home
                profile_home = _get_profile_home(getattr(session, "profile", None))
                core_path = profile_home / "sessions" / f"session_{session.session_id}.json"
                repaired = _apply_core_sync_or_error_marker(
                    session,
                    core_path,
                    stream_id_for_recheck=stream_id,
                    touch_updated_at=False,
                )
            except Exception:
                logger.exception(
                    "_clear_stale_stream_state: failed to repair stale pending stream %s "
                    "for session %s",
                    stream_id, getattr(session, "session_id", "?"),
                )
                repaired = False
            if repaired:
                if original_stub is not session:
                    try:
                        original_stub.active_stream_id = None
                        if hasattr(original_stub, "pending_user_message"):
                            original_stub.pending_user_message = None
                        if hasattr(original_stub, "pending_attachments"):
                            original_stub.pending_attachments = []
                        if hasattr(original_stub, "pending_started_at"):
                            original_stub.pending_started_at = None
                        if hasattr(original_stub, "pending_user_source"):
                            original_stub.pending_user_source = None
                    except Exception:
                        pass
                return True
            if getattr(session, "active_stream_id", None) != stream_id:
                return False
        _materialize_pending_user_turn_before_error(session)
        session.active_stream_id = None
        if hasattr(session, "pending_user_message"):
            session.pending_user_message = None
        if hasattr(session, "pending_attachments"):
            session.pending_attachments = []
        if hasattr(session, "pending_started_at"):
            session.pending_started_at = None
        if hasattr(session, "pending_user_source"):
            session.pending_user_source = None
        try:
            # Runtime cleanup is not user activity; do not bubble old sessions
            # to the top of the sidebar just because a stale stream flag was
            # repaired during a read/list path.
            session.save(touch_updated_at=False)
        except Exception:
            logger.exception(
                "_clear_stale_stream_state: save() failed for session %s",
                getattr(session, "session_id", "?"),
            )
    # Patch the caller's stub (if different from the full-load object) so
    # its in-memory active_stream_id matches what just got persisted.
    if original_stub is not session:
        try:
            original_stub.active_stream_id = None
            if hasattr(original_stub, "pending_user_message"):
                original_stub.pending_user_message = None
            if hasattr(original_stub, "pending_attachments"):
                original_stub.pending_attachments = []
            if hasattr(original_stub, "pending_started_at"):
                original_stub.pending_started_at = None
            if hasattr(original_stub, "pending_user_source"):
                original_stub.pending_user_source = None
        except Exception:
            pass
    return True


def _run_journal_status_payload(summary: dict, *, active: bool = False) -> dict:
    terminal = bool(summary.get("terminal"))
    terminal_state = summary.get("terminal_state")
    if not active and not terminal:
        terminal_state = "lost-worker-bookkeeping"
    return {
        "session_id": summary.get("session_id"),
        "run_id": summary.get("run_id"),
        "last_seq": summary.get("last_seq"),
        "last_event_id": summary.get("last_event_id"),
        "last_event": summary.get("last_event"),
        "terminal": terminal,
        "terminal_state": terminal_state,
    }


_RUN_JOURNAL_TOOL_ID_KEYS = ("tid", "id", "tool_call_id", "tool_use_id", "call_id")


def _run_journal_snapshot_tool_id(payload: dict | None) -> str:
    if not isinstance(payload, dict):
        return ""
    for key in _RUN_JOURNAL_TOOL_ID_KEYS:
        value = str(payload.get(key) or "").strip()
        if value:
            return value
    return ""


def _truncate_journal_snapshot_value(value, *, limit: int = 120):
    if isinstance(value, str):
        return value if len(value) <= limit else value[:limit] + "..."
    if isinstance(value, dict):
        return {str(k): _truncate_journal_snapshot_value(v, limit=limit) for k, v in value.items()}
    if isinstance(value, list):
        return [_truncate_journal_snapshot_value(v, limit=limit) for v in value[:20]]
    return value


def _run_journal_snapshot_recovery_args(payload: dict | None):
    if not isinstance(payload, dict):
        return {}
    args = payload.get("args")
    return bound_run_journal_snapshot_args(args)


def _run_journal_snapshot_arg_detail_score(value) -> int:
    if value in (None, "", [], {}):
        return 0
    if isinstance(value, str):
        return len(value)
    if isinstance(value, dict):
        return sum(
            len(str(key)) + _run_journal_snapshot_arg_detail_score(item)
            for key, item in value.items()
        )
    if isinstance(value, list):
        return sum(_run_journal_snapshot_arg_detail_score(item) for item in value)
    return 1


def _run_journal_snapshot_merge_args(existing, incoming):
    if not incoming:
        return existing, False
    if not isinstance(existing, dict) or not existing:
        return incoming, True
    if not isinstance(incoming, dict):
        return existing, False
    merged = copy.deepcopy(existing)
    changed = False
    for key, value in incoming.items():
        current = merged.get(key)
        if key not in merged or (
            _run_journal_snapshot_arg_detail_score(value)
            > _run_journal_snapshot_arg_detail_score(current)
        ):
            merged[key] = value
            changed = True
    return merged, changed


def _run_journal_envelope_run_id_result(event: dict) -> tuple[str | None, bool]:
    raw_run_id = event.get("run_id")
    if raw_run_id is None:
        return None, False
    if not isinstance(raw_run_id, str):
        return None, True
    run_id = raw_run_id.strip()
    if not run_id:
        return None, True
    raw_event_id = event.get("event_id")
    event_id = str(raw_event_id or "").strip()
    if event_id:
        event_run_id, event_seq = _shared_parse_run_journal_event_id(event_id)
        if event_run_id and event_seq is not None and event_run_id != run_id:
            return None, True
    return run_id, False


def _run_journal_snapshot_event_id_for_run(
    event: dict,
    run_id: str,
    event_seq: int,
) -> str | None:
    raw_event_id = event.get("event_id")
    event_id = str(raw_event_id or "").strip()
    if event_id:
        event_run_id, parsed_seq = _shared_parse_run_journal_event_id(event_id)
        if event_run_id == run_id and parsed_seq is not None:
            return event_id
    return f"{run_id}:{event_seq}" if event_seq else None


def _run_journal_live_snapshot(stream_id: str | None, *, handler=None) -> dict | None:
    stream_id = str(stream_id or "").strip()
    if not stream_id:
        return None
    if handler is not None and not _stream_id_visible_to_request_profile(
        handler,
        stream_id,
        emit_error=False,
    ):
        return None
    # Locate the run journal WITHOUT parsing it first. find_run_summary reads
    # and parses the whole file (tens of thousands of rows on a long live
    # run), and read_run_events below then parsed it AGAIN — two full passes
    # cost ~0.6s per rebuild. Derive the durable summary from the single
    # parse below instead; its last_seq / last_event_id are rebuilt from the
    # same rows the snapshot consumes.
    located = find_run_file(stream_id)
    if located:
        session_id, _journal_path = located
    else:
        # No journal file on disk for this run id: fall back to the
        # historical summary-based lookup seam (defensive — a real miss
        # returns None exactly like before; this is also the seam that
        # long-standing tests stub with a handler-side summary).
        fallback_summary = find_run_summary(stream_id)
        if not fallback_summary:
            return None
        session_id = str(fallback_summary.get("session_id") or "")
        if not session_id:
            return None
    journal = read_run_events(session_id, stream_id)
    events = [event for event in (journal.get("events") or []) if isinstance(event, dict)]
    if not events:
        return None
    summary = _summary_from_events(session_id, stream_id, events)
    event_run_ids: set[str] = set()
    malformed_envelope_run_id = False
    for event in events:
        event_run_id, event_run_id_malformed = _run_journal_envelope_run_id_result(event)
        if event_run_id is not None:
            event_run_ids.add(event_run_id)
        if event_run_id_malformed:
            malformed_envelope_run_id = True
    # The event envelope is the durable identity authority. Older summaries
    # are keyed by the transport id, so only use that fallback when the journal
    # does not provide one unambiguous run id.
    run_id = (
        next(iter(event_run_ids))
        if not malformed_envelope_run_id and len(event_run_ids) == 1
        else str(summary.get("run_id") or stream_id).strip()
    )

    assistant_text = ""
    reasoning_text = ""
    # Reasoning deltas arrive as thousands of tiny append events. Growing the
    # transcript with ``reasoning_text += chunk`` copies the full (multi-
    # hundred-KB) string on every append inside these closures — quadratic
    # (~2.3s of a 4.6s rebuild on a 9.7k-chunk run). Collect chunks and
    # materialize lazily; readers (echo probes, strip, final message) join at
    # most once per interim segment.
    reasoning_parts: list[str] = []
    reasoning_dirty = False
    # Incremental folded index over the reasoning transcript: the per-interim
    # echo probe consults this instead of re-walking the raw text (see
    # ``_CompactEchoIndex``).
    reasoning_index = _CompactEchoIndex()
    messages: list[dict] = []
    tool_calls: list[dict] = []
    activity_burst_anchors: list[dict] = []
    current_activity_burst_id = 0
    fresh_segment = True
    last_ts = None
    reasoning_first_tool_count: int | None = None

    def _materialize_reasoning_text() -> str:
        nonlocal reasoning_text, reasoning_dirty
        if reasoning_dirty:
            reasoning_text = "".join(reasoning_parts)
            reasoning_dirty = False
        return reasoning_text

    def mark_boundary() -> int:
        nonlocal current_activity_burst_id
        text_end = len(assistant_text)
        if text_end <= 0:
            return current_activity_burst_id
        last_end = max(
            [int(anchor.get("textEnd") or 0) for anchor in activity_burst_anchors]
            or [0]
        )
        if text_end > last_end:
            current_activity_burst_id += 1
            activity_burst_anchors.append(
                {"id": current_activity_burst_id, "textEnd": text_end}
            )
        return current_activity_burst_id

    def update_completed_tool(payload: dict) -> None:
        tool_id = _run_journal_snapshot_tool_id(payload)
        name = str(payload.get("name") or "").strip()
        for call in reversed(tool_calls):
            if call.get("done"):
                continue
            call_id = _run_journal_snapshot_tool_id(call)
            if (tool_id and call_id == tool_id) or (not tool_id and name and call.get("name") == name):
                call["done"] = True
                merged_args, args_changed = _run_journal_snapshot_merge_args(
                    call.get("args"),
                    _run_journal_snapshot_recovery_args(payload),
                )
                if args_changed:
                    call["args"] = merged_args
                if payload.get("preview") is not None:
                    call["snippet"] = str(payload.get("preview") or "")
                    call["preview"] = call.get("preview") or call["snippet"]
                if payload.get("duration") is not None:
                    call["duration"] = payload.get("duration")
                if payload.get("is_error") is not None:
                    call["is_error"] = bool(payload.get("is_error"))
                return

        if not name or name == "clarify":
            return
        call = {
            "name": name,
            "preview": str(payload.get("preview") or ""),
            "snippet": str(payload.get("preview") or ""),
            "args": _run_journal_snapshot_recovery_args(payload),
            "done": True,
            "_live": True,
            "_journal_snapshot": True,
            "_journal_stream_id": stream_id,
        }
        tool_id = _run_journal_snapshot_tool_id(payload)
        if tool_id:
            call["tid"] = tool_id
        for key in _RUN_JOURNAL_TOOL_ID_KEYS:
            if payload.get(key):
                call[key] = str(payload.get(key))
        if current_activity_burst_id:
            call["activityBurstId"] = current_activity_burst_id
            call["activitySegmentSeq"] = current_activity_burst_id
        tool_calls.append(call)

    def reasoning_echo_tail_matches(text: str) -> bool:
        # Indexed tail match (no raw-text walk): the incremental index folds
        # each reasoning chunk once as it is appended, so this probe costs
        # O(len(text)) regardless of how much interior whitespace stretches
        # the raw span. The previous raw backward walk re-walked that span
        # once per interim event, which turned the replay quadratic on a
        # whitespace-heavy transcript (#7569 review, ~55x slower than master
        # on a production-shaped journal).
        return reasoning_index.matches_tail(text)

    def strip_reasoning_echo_tail(text: str) -> bool:
        nonlocal reasoning_text, reasoning_dirty, reasoning_first_tool_count
        cut = reasoning_index.cut_to(text)
        if cut is None:
            return False
        next_reasoning = _materialize_reasoning_text()[:cut].rstrip()
        reasoning_text = next_reasoning
        reasoning_parts.clear()
        reasoning_parts.append(next_reasoning)
        reasoning_dirty = False
        # Re-index the truncated transcript so later probes match against it
        # instead of the pre-strip tail (the index is the match authority).
        reasoning_index.reset()
        reasoning_index.append(next_reasoning)
        if not _compact_for_echo_compare(reasoning_text):
            reasoning_first_tool_count = None
        return True

    for event in events:
        event_name = str(event.get("event") or event.get("type") or "")
        if event_name == "metering":
            # Metering rows are the bulk of a long live journal (~10k rows,
            # ~60% of its bytes) and project nothing onto the snapshot; skip
            # their per-row work while preserving the last_ts watermark they
            # carry.
            last_ts = event.get("created_at", last_ts)
            continue
        payload = event.get("payload") if isinstance(event.get("payload"), dict) else {}
        last_ts = event.get("created_at", last_ts)
        if event_name == "token":
            text = str(payload.get("text") or "")
            if text:
                assistant_text += text
                fresh_segment = False
            continue
        if event_name == "reasoning":
            text = str(payload.get("text") or "")
            if text:
                if reasoning_first_tool_count is None:
                    reasoning_first_tool_count = len(tool_calls)
                reasoning_parts.append(text)
                reasoning_index.append(text)
                reasoning_dirty = True
            continue
        if event_name == "interim_assistant":
            visible = str(payload.get("text") or "").strip()
            if visible:
                if payload.get("reasoning_echo") or reasoning_echo_tail_matches(visible):
                    strip_reasoning_echo_tail(visible)
                if payload.get("already_streamed"):
                    if not assistant_text:
                        assistant_text = visible
                else:
                    assistant_text = f"{assistant_text}\n\n{visible}" if assistant_text else visible
                mark_boundary()
                fresh_segment = True
            continue
        if event_name == "tool":
            name = str(payload.get("name") or "").strip()
            if not name or name == "clarify":
                continue
            boundary_id = mark_boundary()
            tool_id = _run_journal_snapshot_tool_id(payload)
            call = {
                "name": name,
                "preview": str(payload.get("preview") or ""),
                "args": _run_journal_snapshot_recovery_args(payload),
                "done": False,
                "_live": True,
                "_journal_snapshot": True,
                "_journal_stream_id": stream_id,
            }
            if tool_id:
                call["tid"] = tool_id
            for key in _RUN_JOURNAL_TOOL_ID_KEYS:
                if payload.get(key):
                    call[key] = str(payload.get(key))
            if boundary_id:
                call["activityBurstId"] = boundary_id
                call["activitySegmentSeq"] = boundary_id
            tool_calls.append(call)
            fresh_segment = True
            continue
        if event_name == "tool_complete":
            update_completed_tool(payload)
            fresh_segment = True

    reasoning_text = _materialize_reasoning_text()

    if assistant_text or reasoning_text:
        message = {
            "role": "assistant",
            "content": assistant_text,
            "_live": True,
            "_journal_snapshot": True,
            "_journal_stream_id": stream_id,
        }
        if reasoning_text:
            message["reasoning"] = reasoning_text
        if last_ts is not None:
            message["_ts"] = last_ts
        messages.append(message)

    def scene_group(segment_seq: int | None = None, burst_id: int | None = None) -> dict:
        group: dict = {}
        if segment_seq:
            group["group_key"] = f"segment:{segment_seq}"
            group["activity_segment_seq"] = segment_seq
        elif burst_id:
            group["group_key"] = f"burst:{burst_id}"
            group["activity_burst_id"] = burst_id
        else:
            group["group_key"] = "activity:0"
        if burst_id:
            group["activity_burst_id"] = burst_id
        return group

    def scene_prose_row(text: str, *, burst_id: int | None, segment_seq: int, status: str) -> dict | None:
        clean = str(text or "").strip()
        if not clean:
            return None
        local_id = f"live-prose:{stream_id}:{segment_seq}"
        return {
            "row_id": local_id,
            "order_index": len(anchor_activity_rows),
            "kind": "process_prose",
            "role": "prose",
            "display_hint": "main_prose",
            "display_hints": {
                "compact_worklog": "main_prose",
                "transparent_stream": "chronological_activity",
            },
            "source_event_type": "token",
            "event_id": None,
            "local_id": local_id,
            "run_id": run_id,
            "stream_id": stream_id,
            "seq": None,
            "status": status,
            "created_at": last_ts,
            "identity": {
                "event_id": None,
                "local_id": local_id,
                "run_id": run_id,
                "stream_id": stream_id,
                "seq": None,
            },
            "group": scene_group(segment_seq, burst_id),
            "text": clean,
            "thinking": None,
            "tool_call_id": "",
            "tool": None,
            "payload": {
                "text": clean,
                "activitySegmentSeq": segment_seq,
                "activityBurstId": burst_id or 0,
            },
        }

    def scene_thinking_row(text: str, *, status: str) -> dict | None:
        clean = str(text or "").strip()
        if not clean:
            return None
        preview = " ".join(clean.split())
        local_id = f"live-thinking:{stream_id}:1"
        return {
            "row_id": local_id,
            "order_index": len(anchor_activity_rows),
            "kind": "reasoning",
            "role": "thinking",
            "display_hint": "collapsed_thinking",
            "display_hints": {
                "compact_worklog": "collapsed_thinking",
                "transparent_stream": "chronological_activity",
            },
            "source_event_type": "reasoning",
            "event_id": None,
            "local_id": local_id,
            "run_id": run_id,
            "stream_id": stream_id,
            "seq": None,
            "status": status,
            "created_at": last_ts,
            "identity": {
                "event_id": None,
                "local_id": local_id,
                "run_id": run_id,
                "stream_id": stream_id,
                "seq": None,
            },
            "group": scene_group(),
            "text": clean,
            "thinking": {
                "text": clean,
                "preview": (preview[:177] + "...") if len(preview) > 180 else preview,
                "dedupe_key": f"thinking:{preview.lower()}" if preview else "",
            },
            "tool_call_id": "",
            "tool": None,
            "payload": {
                "text": clean,
            },
        }

    def scene_tool_row(call: dict, *, fallback_order: int) -> dict | None:
        if not isinstance(call, dict):
            return None
        name = str(call.get("name") or "").strip()
        if not name:
            return None
        tool_id = _run_journal_snapshot_tool_id(call)
        burst_id = int(call.get("activityBurstId") or 0) or None
        segment_seq = int(call.get("activitySegmentSeq") or burst_id or 0) or None
        status = "error" if call.get("is_error") else ("completed" if call.get("done") else "running")
        row_id = f"tool:{tool_id or name}:{fallback_order}"
        args = call.get("args") if isinstance(call.get("args"), dict) else {}
        preview = str(call.get("preview") or "")
        snippet = str(call.get("snippet") or "")
        tool = {
            "id": tool_id,
            "tid": tool_id,
            "name": name,
            "args": args,
            "preview": preview,
            "snippet": snippet,
            "done": bool(call.get("done")),
            "is_error": bool(call.get("is_error")),
            "duration": call.get("duration"),
            "started_at": call.get("started_at"),
        }
        payload = {
            "name": name,
            "args": args,
            "preview": preview,
            "snippet": snippet,
            "tid": tool_id,
            "id": tool_id,
            "is_error": bool(call.get("is_error")),
            "duration": call.get("duration"),
            "activitySegmentSeq": segment_seq,
            "activityBurstId": burst_id or 0,
        }
        return {
            "row_id": row_id,
            "order_index": len(anchor_activity_rows),
            "kind": "tool_completed" if call.get("done") else "tool_started",
            "role": "tool",
            "display_hint": "tool_row",
            "display_hints": {
                "compact_worklog": "tool_row",
                "transparent_stream": "chronological_activity",
            },
            "source_event_type": "tool_complete" if call.get("done") else "tool",
            "event_id": None,
            "local_id": tool_id or row_id,
            "run_id": run_id,
            "stream_id": stream_id,
            "seq": None,
            "status": status,
            "created_at": last_ts,
            "identity": {
                "event_id": None,
                "local_id": tool_id or row_id,
                "run_id": run_id,
                "stream_id": stream_id,
                "seq": None,
            },
            "group": scene_group(segment_seq, burst_id),
            "text": snippet or preview,
            "thinking": None,
            "tool_call_id": tool_id,
            "tool": tool,
            "payload": payload,
        }

    anchor_activity_rows: list[dict] = []
    thinking_row_inserted = False
    tool_rows_rendered = 0

    def append_thinking_row(*, force: bool = False) -> None:
        nonlocal thinking_row_inserted
        if thinking_row_inserted:
            return
        if not force and reasoning_first_tool_count and tool_rows_rendered < reasoning_first_tool_count:
            return
        row = scene_thinking_row(reasoning_text, status="running")
        if not row:
            return
        row["order_index"] = len(anchor_activity_rows)
        anchor_activity_rows.append(row)
        thinking_row_inserted = True

    tool_rows_by_burst: dict[int, list[tuple[int, dict]]] = {}
    ungrouped_tool_rows: list[tuple[int, dict]] = []
    for order, call in enumerate(tool_calls):
        burst_id = int(call.get("activityBurstId") or 0) if isinstance(call, dict) else 0
        row = scene_tool_row(call, fallback_order=order)
        if not row:
            continue
        if burst_id:
            tool_rows_by_burst.setdefault(burst_id, []).append((order, row))
        else:
            ungrouped_tool_rows.append((order, row))

    consumed_tools: set[int] = set()
    text_start = 0
    sorted_anchors = sorted(
        [
            anchor
            for anchor in activity_burst_anchors
            if int(anchor.get("textEnd") or 0) > 0
        ],
        key=lambda anchor: int(anchor.get("textEnd") or 0),
    )
    for anchor in sorted_anchors:
        burst_id = int(anchor.get("id") or 0) or None
        text_end = min(len(assistant_text), int(anchor.get("textEnd") or 0))
        segment_seq = burst_id or (len(anchor_activity_rows) + 1)
        prose = scene_prose_row(
            assistant_text[text_start:text_end],
            burst_id=burst_id,
            segment_seq=segment_seq,
            status="completed",
        )
        if prose:
            anchor_activity_rows.append(prose)
            append_thinking_row()
        for order, row in tool_rows_by_burst.get(burst_id or 0, []):
            row["order_index"] = len(anchor_activity_rows)
            anchor_activity_rows.append(row)
            consumed_tools.add(order)
            tool_rows_rendered += 1
            append_thinking_row()
        text_start = max(text_start, text_end)

    if text_start < len(assistant_text):
        segment_seq = max(len(sorted_anchors) + 1, 1)
        tail = scene_prose_row(
            assistant_text[text_start:],
            burst_id=None,
            segment_seq=segment_seq,
            status="running",
        )
        if tail:
            anchor_activity_rows.append(tail)
            append_thinking_row()

    if not assistant_text:
        append_thinking_row()

    for order, row in sorted(ungrouped_tool_rows, key=lambda item: item[0]):
        if order in consumed_tools:
            continue
        row["order_index"] = len(anchor_activity_rows)
        anchor_activity_rows.append(row)
        tool_rows_rendered += 1
        append_thinking_row()

    append_thinking_row(force=True)

    # Keep a live anchor shell during session-switch replay even before the
    # journal has projected visible prose or tool rows from the first events.
    if not anchor_activity_rows and events:
        anchor_activity_rows.append(
            {
                "row_id": f"lifecycle:{stream_id}:running",
                "order_index": 0,
                "kind": "lifecycle_status",
                "role": "lifecycle",
                "display_hint": "quiet_lifecycle_row",
                "display_hints": {
                    "compact_worklog": "quiet_lifecycle_row",
                    "transparent_stream": "chronological_activity",
                },
                "source_event_type": "runtime_journal_snapshot",
                "event_id": None,
                "local_id": f"lifecycle:{stream_id}:running",
                "run_id": run_id,
                "stream_id": stream_id,
                "seq": None,
                "status": "running",
                "created_at": last_ts,
                "identity": {
                    "event_id": None,
                    "local_id": f"lifecycle:{stream_id}:running",
                    "run_id": run_id,
                    "stream_id": stream_id,
                    "seq": None,
                },
                "group": scene_group(),
                "text": "Working",
                "thinking": None,
                "tool_call_id": "",
                "tool": None,
                "payload": {},
            }
        )

    visible_anchors = [
        anchor
        for anchor in activity_burst_anchors
        if int(anchor.get("textEnd") or 0) < len(assistant_text)
    ]
    segment_count = len(visible_anchors) + (1 if assistant_text else 0)
    current_live_segment_seq = max(segment_count, len(activity_burst_anchors), 0)
    try:
        summary_last_seq = max(0, int(summary.get("last_seq") or 0))
    except (TypeError, ValueError):
        summary_last_seq = 0
    try:
        event_last_seq = max(0, int(events[-1].get("seq") or 0))
    except (TypeError, ValueError):
        event_last_seq = 0
    if event_last_seq >= summary_last_seq:
        last_seq = event_last_seq
        last_event_id = _run_journal_snapshot_event_id_for_run(
            events[-1],
            run_id,
            event_last_seq,
        ) or summary.get("last_event_id")
    else:
        last_seq = summary_last_seq
        last_event_id = summary.get("last_event_id") or events[-1].get("event_id")

    # Keep returning a live snapshot even when the journal has events but no
    # projected message/tool rows yet. The frontend treats the empty activity
    # scene as "nothing renderable yet" while preserving the live cursor.
    return {
        "session_id": session_id,
        "stream_id": stream_id,
        "last_seq": last_seq,
        "last_event_id": last_event_id,
        "event_count": len(events),
        "fresh_segment": fresh_segment,
        "messages": messages,
        "tool_calls": tool_calls,
        "last_assistant_text": assistant_text,
        "last_reasoning_text": reasoning_text,
        "runtime_model": runtime_model_from_events(session_id, stream_id, events),
        "activity_burst_anchors": activity_burst_anchors,
        "current_activity_burst_id": current_activity_burst_id,
        "current_live_segment_seq": current_live_segment_seq,
        "anchor_activity_scene": {
            "version": "activity_scene_v1",
            "mode": "compact_worklog",
            "identity": {
                "session_id": session_id,
                "stream_id": stream_id,
                "run_id": run_id,
                "source_message_refs": [],
            },
            "lifecycle": {
                "status": "running",
                "terminal_state": None,
            },
            "final_answer": "",
            "final_message_ref": None,
            "terminal_state": None,
            "activity_rows": anchor_activity_rows,
        },
    }


def _runtime_journal_snapshot_for_session_payload(snapshot: dict | None) -> dict | None:
    """Return the non-mutating, display-equivalent transport form of a live snapshot.

    The canonical recovery snapshot intentionally keeps fallback representations.
    Sending all of them duplicates each tool result in top-level ``tool_calls``
    and row ``text``, ``tool``, and ``payload`` fields. Keep one authoritative
    display source for each value at the HTTP boundary; the durable journal and
    the canonical in-process snapshot remain unchanged.
    """
    if not isinstance(snapshot, dict):
        return snapshot

    projected = dict(snapshot)
    # The frontend reconstructs this single live assistant row from the
    # authoritative last_* strings when messages is empty. Preserve the row's
    # timestamp separately so the synthesized message keeps stable identity.
    live_messages = [
        message
        for message in (projected.get("messages") or [])
        if isinstance(message, dict) and message.get("role") == "assistant"
    ]
    if live_messages:
        last_message_ts = live_messages[-1].get("_ts")
        if last_message_ts is not None:
            projected["last_message_ts"] = last_message_ts
    if projected.get("last_assistant_text") or projected.get("last_reasoning_text"):
        projected["messages"] = []

    scene = projected.get("anchor_activity_scene")
    compact_calls = []
    for raw_call in projected.get("tool_calls") or []:
        if not isinstance(raw_call, dict):
            continue
        call = dict(raw_call)
        if call.get("preview") == call.get("snippet"):
            call.pop("preview", None)
        compact_calls.append(call)
    # Retain the compact top-level list as the degraded-render fallback. The
    # Anchor scene is normally authoritative, but session reattach deliberately
    # falls back to INFLIGHT.toolCalls when scene rendering is unavailable.
    projected["tool_calls"] = compact_calls

    if not isinstance(scene, dict):
        return projected
    compact_scene = dict(scene)
    compact_rows = []
    row_keys = (
        "row_id", "local_id", "kind", "role", "source_event_type", "status",
        "created_at", "group", "text", "thinking", "tool_call_id", "tool",
    )
    for raw_row in scene.get("activity_rows") or []:
        if not isinstance(raw_row, dict):
            continue
        row = {
            key: raw_row[key]
            for key in row_keys
            if key in raw_row and raw_row[key] not in (None, "")
        }
        if str(raw_row.get("role") or "") == "tool":
            tool = raw_row.get("tool") if isinstance(raw_row.get("tool"), dict) else {}
            compact_tool = dict(tool)
            if compact_tool.get("preview") == compact_tool.get("snippet"):
                compact_tool.pop("preview", None)
            if compact_tool.get("tid") == compact_tool.get("id"):
                compact_tool.pop("tid", None)
            row["tool"] = compact_tool
            # Tool cards consume args/snippet from row.tool. row.payload and
            # row.text are byte-for-byte fallbacks of those same values.
            row.pop("text", None)
        else:
            thinking = row.get("thinking")
            if isinstance(thinking, dict) and thinking.get("text") == row.get("text"):
                row.pop("thinking", None)
        compact_rows.append(row)
    compact_scene["activity_rows"] = compact_rows
    projected["anchor_activity_scene"] = compact_scene
    return projected


def _ensure_full_session_before_mutation(sid: str, session):
    """Reload cached metadata-only sessions before mutating persisted fields.

    Session.save() intentionally refuses metadata-only stubs (#1558) because
    their messages list is empty by design. Mutation routes that save session
    metadata must upgrade the cached stub first so they do not trip that guard
    or risk writing an incomplete object.
    """
    if not getattr(session, "_loaded_metadata_only", False):
        return session
    full_session = Session.load(sid)
    if full_session is None:
        raise KeyError(sid)
    with LOCK:
        SESSIONS[sid] = full_session
        SESSIONS.move_to_end(sid)
        _evict_sessions_over_cap()  # #4765: safe LRU eviction (never active/unsaved)
    return full_session


_ANCHOR_ACTIVITY_SCENE_MAX_BYTES = 256_000
_ANCHOR_ACTIVITY_SCENE_MAX_ROWS = 1_000


def _assistant_anchor_scene_message_ref(message) -> str:
    if not isinstance(message, dict):
        return ""
    payload = _assistant_anchor_scene_message_ref_payload(message)
    return _anchor_scene_message_ref_digest(payload)


def _assistant_anchor_scene_message_ref_payload(message) -> dict:
    role = str(message.get("role") or "")
    content = message.get("content")
    if isinstance(content, list):
        parts = []
        for part in content:
            if isinstance(part, dict):
                parts.append(str(part.get("text") or part.get("content") or part.get("input_text") or ""))
            else:
                parts.append(str(part or ""))
        content_text = "\n".join(parts)
    else:
        content_text = str(content or "")
    payload = {
        "role": role,
        "content": " ".join(content_text.split()),
        "timestamp": message.get("_ts") or message.get("timestamp") or "",
    }
    return payload


def _anchor_scene_message_ref_digest(payload: dict) -> str:
    raw = json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
    return hashlib.sha256(raw.encode("utf-8")).hexdigest()


def _sanitize_anchor_activity_scene(scene):
    if not isinstance(scene, dict):
        raise ValueError("scene must be an object")
    if str(scene.get("version") or "") != "activity_scene_v1":
        raise ValueError("scene.version must be activity_scene_v1")
    rows = scene.get("activity_rows")
    if not isinstance(rows, list):
        raise ValueError("scene.activity_rows must be a list")
    if len(rows) > _ANCHOR_ACTIVITY_SCENE_MAX_ROWS:
        raise ValueError("scene.activity_rows is too large")
    scene_copy = copy.deepcopy(scene)
    encoded = json.dumps(scene_copy, ensure_ascii=False, separators=(",", ":"), default=str).encode("utf-8")
    if len(encoded) > _ANCHOR_ACTIVITY_SCENE_MAX_BYTES:
        raise ValueError("scene payload is too large")
    return json.loads(encoded.decode("utf-8"))


def _anchor_scene_int_or_none(value):
    try:
        return int(value)
    except (TypeError, ValueError):
        return None


def _anchor_scene_message_index_from_request(body):
    if not isinstance(body, dict):
        return None
    message_index = _anchor_scene_int_or_none(body.get("message_index"))
    message_offset = _anchor_scene_int_or_none(body.get("message_offset"))
    message_window_index = _anchor_scene_int_or_none(body.get("message_window_index"))
    if (
        message_window_index is not None
        and message_offset is not None
        and message_offset > 0
        and (message_index is None or message_index == message_window_index)
    ):
        return message_window_index + message_offset
    return message_index


def _anchor_scene_candidate_matches_scene(candidate, scene) -> bool:
    if not isinstance(scene, dict):
        return True
    final_key = _anchor_scene_text_key(scene.get("final_answer") or "")
    if not final_key:
        return True
    candidate_key = _anchor_scene_text_key(_anchor_scene_message_text(candidate))
    if not candidate_key:
        return False
    if candidate_key == final_key:
        return True
    if len(final_key) >= 16 and final_key in candidate_key:
        return True
    if len(candidate_key) >= 16 and candidate_key in final_key:
        return True
    return _anchor_scene_text_has_long_overlap(candidate_key, final_key)


def _find_anchor_scene_message(messages, *, message_index=None, message_ref="", scene=None):
    if not isinstance(messages, list):
        return None, None
    normalized_message_ref = _normalize_anchor_scene_message_ref(message_ref)
    candidate = None
    if isinstance(message_index, int) and 0 <= message_index < len(messages):
        maybe_candidate = messages[message_index]
        if isinstance(maybe_candidate, dict) and maybe_candidate.get("role") == "assistant":
            candidate = maybe_candidate
    if normalized_message_ref:
        matches = [
            (idx, message)
            for idx, message in enumerate(messages)
            if isinstance(message, dict)
            and message.get("role") == "assistant"
            and _assistant_anchor_scene_message_ref(message) == normalized_message_ref
        ]
        if len(matches) == 1:
            return matches[0]
        if len(matches) > 1:
            return None, None
        if candidate is None:
            return None, None
    if candidate is not None:
        # Content can be normalized or split during settlement; use the explicit
        # index as the durability fallback only after a unique ref did not pick a
        # different assistant message. The index is the full transcript index.
        if normalized_message_ref and not _anchor_scene_candidate_matches_scene(candidate, scene):
            return None, None
        return message_index, candidate
    for idx in range(len(messages) - 1, -1, -1):
        message = messages[idx]
        if isinstance(message, dict) and message.get("role") == "assistant":
            return idx, message
    return None, None


def _normalize_anchor_scene_message_ref(message_ref) -> str:
    ref = str(message_ref or "").strip()
    if not ref:
        return ""
    if re.fullmatch(r"[0-9a-fA-F]{64}", ref):
        return ref.lower()
    try:
        payload = json.loads(ref)
    except (TypeError, ValueError):
        return ref
    if not isinstance(payload, dict):
        return ref
    canonical = {
        "role": str(payload.get("role") or ""),
        "content": " ".join(str(payload.get("content") or "").split()),
        "timestamp": payload.get("timestamp") or "",
    }
    return _anchor_scene_message_ref_digest(canonical)


def _anchor_scene_records(session) -> dict:
    records = getattr(session, "anchor_activity_scenes", None)
    return records if isinstance(records, dict) else {}


def _anchor_scene_message_text(message) -> str:
    if not isinstance(message, dict):
        return ""
    content = message.get("content", "")
    if isinstance(content, list):
        parts = []
        for part in content:
            if isinstance(part, dict):
                parts.append(str(part.get("text") or part.get("content") or part.get("input_text") or ""))
            else:
                parts.append(str(part or ""))
        return "\n".join(parts)
    return str(content or "")


def _anchor_scene_content_text(part) -> str:
    if part is None:
        return ""
    if isinstance(part, str):
        return part
    if not isinstance(part, dict):
        return str(part or "")
    return str(
        part.get("text")
        or part.get("content")
        or part.get("input_text")
        or part.get("output_text")
        or part.get("thinking")
        or part.get("reasoning")
        or part.get("summary")
        or ""
    )


def _anchor_scene_content_visible_text(part) -> str:
    if part is None:
        return ""
    if isinstance(part, str):
        return part
    if not isinstance(part, dict):
        return str(part or "")
    part_type = str(part.get("type") or "")
    if part_type in ("thinking", "reasoning"):
        return ""
    content_text = part.get("content") if part_type in ("text", "input_text", "output_text") else ""
    return str(part.get("text") or part.get("input_text") or part.get("output_text") or content_text or "")


def _anchor_scene_message_has_content_tool_use(message) -> bool:
    content = message.get("content") if isinstance(message, dict) else None
    return isinstance(content, list) and any(
        isinstance(part, dict) and part.get("type") == "tool_use" for part in content
    )


def _anchor_scene_final_answer_text(message) -> str:
    if not _anchor_scene_message_has_content_tool_use(message):
        return _anchor_scene_message_text(message)
    content = message.get("content") if isinstance(message, dict) else []
    last_tool_index = -1
    for idx, part in enumerate(content):
        if isinstance(part, dict) and part.get("type") == "tool_use":
            last_tool_index = idx
    tail_text = "\n".join(
        text
        for text in (_anchor_scene_content_visible_text(part) for part in content[last_tool_index + 1 :])
        if _anchor_scene_clean_text(text)
    )
    return tail_text if _anchor_scene_clean_text(tail_text) else ""


def _anchor_scene_message_reasoning_text(message) -> str:
    if not isinstance(message, dict):
        return ""
    for key in ("reasoning", "_reasoning", "reasoning_content", "thinking"):
        value = message.get(key)
        if not value:
            continue
        if isinstance(value, list):
            parts = []
            for part in value:
                if isinstance(part, dict):
                    parts.append(
                        str(
                            part.get("text")
                            or part.get("content")
                            or part.get("reasoning")
                            or part.get("summary")
                            or ""
                        )
                    )
                else:
                    parts.append(str(part or ""))
            return "\n".join(parts)
        if isinstance(value, dict):
            return str(
                value.get("text")
                or value.get("content")
                or value.get("reasoning")
                or value.get("summary")
                or ""
            )
        return str(value or "")
    return ""


def _anchor_scene_clean_text(value) -> str:
    return " ".join(str(value or "").split()).strip()


def _anchor_scene_text_key(value) -> str:
    return _anchor_scene_clean_text(value).lower()


_ANCHOR_SCENE_SETTLED_SNIPPET_CAP = 4000


def _anchor_scene_string_payload(value) -> str:
    if value is None:
        return ""
    if isinstance(value, str):
        return value
    try:
        return json.dumps(value)
    except Exception:
        return str(value)


def _anchor_scene_is_bounded_tool_body_preview(settled, full) -> bool:
    settled_text = _anchor_scene_string_payload(settled)
    full_text = _anchor_scene_string_payload(full)
    return bool(
        settled_text
        and full_text
        and len(full_text) > len(settled_text)
        and len(settled_text) >= _ANCHOR_SCENE_SETTLED_SNIPPET_CAP
        and full_text.startswith(settled_text)
    )


def _anchor_scene_row_looks_like_final_answer(row_text_key: str, final_key: str) -> bool:
    if not row_text_key or not final_key:
        return False
    if row_text_key == final_key:
        return True
    # #4587: align with the renderer's _anchorSceneProseMatchesFinalAnswer — a
    # prefix-like overlap only counts as "the final answer" when it's a
    # NEAR-complete match (ratio >= 0.9). A shorter intermediate-prose row that
    # is merely a prefix of the final answer is legitimate progress narration and
    # must be preserved in the persisted scene, not filtered out.
    if not (final_key.startswith(row_text_key) or row_text_key.startswith(final_key)):
        return False
    shorter = min(len(row_text_key), len(final_key))
    longer = max(len(row_text_key), len(final_key))
    return shorter >= 80 and longer > 0 and (shorter / longer) >= 0.9


def _anchor_scene_text_has_long_overlap(text_key: str, final_key: str) -> bool:
    if len(text_key) < 80 or len(final_key) < 80:
        return False
    shorter, longer = (text_key, final_key) if len(text_key) <= len(final_key) else (final_key, text_key)
    window = 64
    scan_limit = min(len(shorter), 1400)
    if scan_limit < window:
        return False
    for start in range(0, scan_limit - window + 1, 24):
        chunk = shorter[start : start + window].strip()
        if len(chunk) >= 48 and chunk in longer:
            return True
    text_tokens = set(re.findall(r"[a-z0-9_./:-]{3,}", text_key))
    final_tokens = set(re.findall(r"[a-z0-9_./:-]{3,}", final_key))
    if text_tokens and final_tokens:
        common = text_tokens & final_tokens
        shorter_count = min(len(text_tokens), len(final_tokens))
        if shorter_count >= 3 and len(common) >= min(5, shorter_count) and (len(common) / shorter_count) >= 0.5:
            return True
    text_compact = re.sub(r"[\s`*_#|\[\](){}<>.,;:!?，。；：！？、/\\-]+", "", text_key)
    final_compact = re.sub(r"[\s`*_#|\[\](){}<>.,;:!?，。；：！？、/\\-]+", "", final_key)
    if len(text_compact) >= 40 and len(final_compact) >= 40:
        text_grams = {text_compact[idx : idx + 4] for idx in range(0, len(text_compact) - 3)}
        final_grams = {final_compact[idx : idx + 4] for idx in range(0, len(final_compact) - 3)}
        common_grams = text_grams & final_grams
        shorter_grams = min(len(text_grams), len(final_grams))
        if shorter_grams and len(common_grams) >= 12 and (len(common_grams) / shorter_grams) >= 0.35:
            return True
    return False


def _anchor_scene_row_is_stale_token_answer(row, row_text_key: str, final_key: str) -> bool:
    if not isinstance(row, dict) or row.get("role") not in ("prose", "thinking"):
        return False
    source_type = str(row.get("source_event_type") or row.get("source") or "")
    if source_type != "token":
        return False
    return _anchor_scene_text_has_long_overlap(row_text_key, final_key)


def _anchor_scene_message_turn_duration(message):
    if not isinstance(message, dict):
        return None
    for key in ("_turnDuration", "_turn_duration", "turn_duration"):
        value = message.get(key)
        if isinstance(value, (int, float)) and value >= 0:
            return value
    return None


def _anchor_scene_tool_id(tool) -> str:
    if not isinstance(tool, dict):
        return ""
    return str(
        tool.get("tid")
        or tool.get("id")
        or tool.get("tool_call_id")
        or tool.get("tool_use_id")
        or tool.get("call_id")
        or ""
    ).strip()


def _anchor_scene_tool_name(tool) -> str:
    if not isinstance(tool, dict):
        return "tool"
    fn = tool.get("function") if isinstance(tool.get("function"), dict) else {}
    return str(tool.get("name") or tool.get("tool_name") or fn.get("name") or "tool").strip() or "tool"


def _anchor_scene_tool_args(tool):
    if not isinstance(tool, dict):
        return {}
    for key in ("args", "input"):
        value = tool.get(key)
        if isinstance(value, dict):
            return copy.deepcopy(value)
    fn = tool.get("function") if isinstance(tool.get("function"), dict) else {}
    raw = fn.get("arguments")
    if isinstance(raw, str) and raw.strip():
        try:
            parsed = json.loads(raw)
            return parsed if isinstance(parsed, dict) else {}
        except Exception:
            return {}
    return {}


def _anchor_scene_content_tool(part):
    if not isinstance(part, dict):
        return {}
    fn = part.get("function") if isinstance(part.get("function"), dict) else {}
    tool_id = (
        part.get("id")
        or part.get("tid")
        or part.get("tool_call_id")
        or part.get("tool_use_id")
        or part.get("call_id")
    )
    return {
        "id": tool_id,
        "tid": part.get("tid") or tool_id,
        "tool_call_id": part.get("tool_call_id"),
        "tool_use_id": part.get("tool_use_id"),
        "call_id": part.get("call_id"),
        "name": part.get("name") or part.get("tool_name") or fn.get("name") or "tool",
        "tool_name": part.get("tool_name"),
        "args": part.get("args"),
        "input": part.get("input"),
        "function": copy.deepcopy(part.get("function")) if isinstance(part.get("function"), dict) else None,
        "command": part.get("command") or part.get("raw_command") or part.get("original_command") or part.get("display_command"),
        "preview": part.get("preview") or part.get("summary"),
        "snippet": part.get("snippet") or part.get("result") or part.get("output"),
        "result": copy.deepcopy(part.get("result")),
        "output": copy.deepcopy(part.get("output")),
        "is_error": part.get("is_error"),
        "error": part.get("error"),
        "duration": part.get("duration"),
        "started_at": part.get("started_at"),
    }


def _anchor_scene_row_base(role, kind, source_event_type, order_index, message_index, stream_id=""):
    return {
        "row_id": f"hydrated:{stream_id or 'stream'}:{role}:{message_index}:{order_index}",
        "order_index": order_index,
        "kind": kind,
        "role": role,
        "display_hint": {
            "prose": "main_prose",
            "thinking": "collapsed_thinking",
            "tool": "tool_row",
            "terminal": "terminal_status_row",
        }.get(role, "activity_row"),
        "display_hints": {
            "compact_worklog": {
                "prose": "main_prose",
                "thinking": "collapsed_thinking",
                "tool": "tool_row",
                "terminal": "terminal_status_row",
            }.get(role, "activity_row"),
            "transparent_stream": "chronological_activity",
        },
        "source_event_type": source_event_type,
        "event_id": None,
        "local_id": None,
        "run_id": None,
        "stream_id": stream_id or None,
        "seq": order_index,
        "status": "completed",
        "created_at": None,
        "identity": {"event_id": None, "local_id": None, "run_id": None, "stream_id": stream_id or None, "seq": order_index},
        "group": {
            "group_key": f"assistant:{message_index}" if isinstance(message_index, int) else f"activity:{order_index}",
            "activity_burst_id": None,
            "activity_segment_seq": None,
            "assistant_msg_idx": message_index if isinstance(message_index, int) else None,
        },
        "text": "",
        "thinking": None,
        "tool_call_id": None,
        "tool": None,
        "payload": {"assistant_msg_idx": message_index if isinstance(message_index, int) else None},
    }


def _anchor_scene_prose_row(text, order_index, message_index, stream_id=""):
    row = _anchor_scene_row_base("prose", "process_prose", "settled_message", order_index, message_index, stream_id)
    row["text"] = str(text or "")
    row["payload"]["text"] = row["text"]
    return row


def _anchor_scene_thinking_row(text, order_index, message_index, stream_id=""):
    row = _anchor_scene_row_base("thinking", "reasoning", "reasoning", order_index, message_index, stream_id)
    row["text"] = str(text or "")
    preview = _anchor_scene_clean_text(text)
    row["thinking"] = {
        "text": row["text"],
        "preview": (preview[:177] + "...") if len(preview) > 180 else preview,
        "dedupe_key": f"thinking:{preview.lower()}" if preview else "",
    }
    row["payload"]["text"] = row["text"]
    return row


def _anchor_scene_tool_row(tool, order_index, message_index, stream_id=""):
    row = _anchor_scene_row_base("tool", "tool_completed", "tool_complete", order_index, message_index, stream_id)
    tid = _anchor_scene_tool_id(tool)
    name = _anchor_scene_tool_name(tool)
    args = _anchor_scene_tool_args(tool)
    preview = str((tool or {}).get("preview") or (tool or {}).get("summary") or "")
    snippet = str((tool or {}).get("snippet") or (tool or {}).get("result") or (tool or {}).get("output") or "")
    row["row_id"] = f"hydrated:{stream_id or 'stream'}:tool:{tid}" if tid else row["row_id"]
    row["tool_call_id"] = tid or None
    row["tool"] = {
        "id": tid or None,
        "name": name,
        "args": args,
        "preview": preview,
        "snippet": snippet,
        "result": copy.deepcopy((tool or {}).get("result")) if isinstance(tool, dict) else None,
        "output": copy.deepcopy((tool or {}).get("output")) if isinstance(tool, dict) else None,
        "done": True,
        "is_error": bool((tool or {}).get("is_error") or (tool or {}).get("error")),
        "duration": (tool or {}).get("duration") if isinstance(tool, dict) else None,
        "started_at": (tool or {}).get("started_at") if isinstance(tool, dict) else None,
        "signature": f"{name}|{tid}|{json.dumps(args, sort_keys=True, default=str)}",
    }
    row["payload"].update({"tid": tid, "id": tid, "name": name, "args": args, "preview": preview, "snippet": snippet})
    return row


def _anchor_scene_tool_row_id(row) -> str:
    if not isinstance(row, dict):
        return ""
    tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
    payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
    return str(
        row.get("tool_call_id")
        or tool.get("id")
        or tool.get("tid")
        or tool.get("tool_call_id")
        or tool.get("tool_use_id")
        or tool.get("call_id")
        or payload.get("tid")
        or payload.get("id")
        or ""
    ).strip()


def _anchor_scene_tool_row_name(row) -> str:
    if not isinstance(row, dict):
        return ""
    tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
    payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
    return str(tool.get("name") or payload.get("name") or "tool").strip().lower()


def _anchor_scene_tool_rows_have_compatible_names(existing, incoming) -> bool:
    existing_name = _anchor_scene_tool_row_name(existing)
    incoming_name = _anchor_scene_tool_row_name(incoming)
    return (
        not existing_name
        or not incoming_name
        or existing_name == "tool"
        or incoming_name == "tool"
        or existing_name == incoming_name
    )


def _anchor_scene_tool_row_args(row):
    if not isinstance(row, dict):
        return None
    tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
    payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
    args = tool.get("args") if isinstance(tool.get("args"), dict) else payload.get("args")
    return args if isinstance(args, dict) and args else None


def _anchor_scene_object_contains_subset(base, subset) -> bool:
    if not isinstance(base, dict) or not isinstance(subset, dict):
        return False
    for key, value in subset.items():
        if key not in base:
            return False
        if json.dumps(base[key], sort_keys=True, separators=(",", ":")) != json.dumps(
            value,
            sort_keys=True,
            separators=(",", ":"),
        ):
            return False
    return True


def _anchor_scene_tool_rows_have_compatible_invocation(existing, incoming) -> bool:
    if not isinstance(existing, dict) or not isinstance(incoming, dict):
        return False
    existing_tool = existing.get("tool") if isinstance(existing.get("tool"), dict) else {}
    incoming_tool = incoming.get("tool") if isinstance(incoming.get("tool"), dict) else {}
    existing_payload = existing.get("payload") if isinstance(existing.get("payload"), dict) else {}
    incoming_payload = incoming.get("payload") if isinstance(incoming.get("payload"), dict) else {}
    existing_command = str(existing_tool.get("command") or existing_payload.get("command") or "").strip()
    incoming_command = str(incoming_tool.get("command") or incoming_payload.get("command") or "").strip()
    if existing_command and incoming_command:
        return existing_command == incoming_command
    existing_args = _anchor_scene_tool_row_args(existing)
    incoming_args = _anchor_scene_tool_row_args(incoming)
    if not existing_args or not incoming_args:
        return False
    return _anchor_scene_object_contains_subset(
        existing_args,
        incoming_args,
    ) or _anchor_scene_object_contains_subset(incoming_args, existing_args)


def _anchor_scene_tool_row_has_invocation_evidence(row) -> bool:
    if not isinstance(row, dict):
        return False
    tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
    payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
    command = str(tool.get("command") or payload.get("command") or "").strip()
    args = _anchor_scene_tool_row_args(row)
    return bool(command or args)


def _anchor_scene_tool_rows_can_name_match(existing, incoming) -> bool:
    if not _anchor_scene_tool_rows_have_compatible_names(existing, incoming):
        return False
    if _anchor_scene_tool_row_has_invocation_evidence(existing) and _anchor_scene_tool_row_has_invocation_evidence(incoming):
        return _anchor_scene_tool_rows_have_compatible_invocation(existing, incoming)
    return True


def _anchor_scene_tool_rows_have_different_explicit_ids(existing, incoming) -> bool:
    existing_id = _anchor_scene_tool_row_id(existing)
    incoming_id = _anchor_scene_tool_row_id(incoming)
    return bool(existing_id and incoming_id and existing_id != incoming_id)


def _anchor_scene_tool_row_started_at(row) -> str:
    if not isinstance(row, dict):
        return ""
    tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
    payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
    value = tool.get("started_at")
    if value is None or value == "":
        value = payload.get("started_at")
    return str(value) if value is not None and value != "" else ""


def _anchor_scene_tool_rows_have_same_started_at(existing, incoming) -> bool:
    existing_started_at = _anchor_scene_tool_row_started_at(existing)
    incoming_started_at = _anchor_scene_tool_row_started_at(incoming)
    return bool(existing_started_at and incoming_started_at and existing_started_at == incoming_started_at)


def _anchor_scene_tool_row_body_text(row) -> str:
    if not isinstance(row, dict):
        return ""
    tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
    payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
    for value in (
        tool.get("snippet"),
        payload.get("snippet"),
        tool.get("output"),
        payload.get("output"),
        tool.get("result"),
        payload.get("result"),
        tool.get("preview"),
        payload.get("preview"),
    ):
        text = str(value or "").strip()
        if text:
            return text
    return ""


def _anchor_scene_tool_rows_have_compatible_body(existing, incoming) -> bool:
    existing_body = _anchor_scene_tool_row_body_text(existing)
    incoming_body = _anchor_scene_tool_row_body_text(incoming)
    return bool(
        existing_body
        and incoming_body
        and (
            existing_body == incoming_body
            or existing_body.startswith(incoming_body)
            or incoming_body.startswith(existing_body)
        )
    )


def _anchor_scene_matching_content_tool_row_index(
    rows,
    content_tool_indexes,
    incoming_row,
    ordinal,
    used_indexes,
    incoming_total=0,
    id_flexible_indexes=None,
):
    if not isinstance(rows, list) or not isinstance(content_tool_indexes, list) or not isinstance(incoming_row, dict):
        return None
    incoming_id = _anchor_scene_tool_row_id(incoming_row)
    for index in content_tool_indexes:
        if index in used_indexes or index < 0 or index >= len(rows):
            continue
        existing_id = _anchor_scene_tool_row_id(rows[index])
        if existing_id and incoming_id and existing_id == incoming_id:
            return index
    if len(content_tool_indexes) == 1 and incoming_total == 1:
        index = content_tool_indexes[0]
        if (
            index not in used_indexes
            and 0 <= index < len(rows)
            and _anchor_scene_tool_rows_can_name_match(rows[index], incoming_row)
        ):
            return index
    available_indexes = [
        index for index in content_tool_indexes if index not in used_indexes and 0 <= index < len(rows)
    ]
    if len(available_indexes) == 1:
        index = available_indexes[0]
        if incoming_total == 1 and _anchor_scene_tool_rows_can_name_match(rows[index], incoming_row):
            return index
        if _anchor_scene_tool_rows_have_compatible_names(
            rows[index],
            incoming_row,
        ) and _anchor_scene_tool_rows_have_compatible_invocation(rows[index], incoming_row):
            return index
    reusable_indexes = [
        index for index in content_tool_indexes if index in used_indexes and 0 <= index < len(rows)
    ]
    if len(reusable_indexes) == 1 and incoming_total == 1:
        index = reusable_indexes[0]
        existing_id = _anchor_scene_tool_row_id(rows[index])
        id_flexible = isinstance(id_flexible_indexes, set) and index in id_flexible_indexes
        if _anchor_scene_tool_rows_have_compatible_names(
            rows[index],
            incoming_row,
        ) and (
            (existing_id and incoming_id and existing_id == incoming_id)
            or (
                id_flexible
                and _anchor_scene_tool_rows_have_same_started_at(rows[index], incoming_row)
                and _anchor_scene_tool_rows_have_compatible_body(rows[index], incoming_row)
            )
        ) and _anchor_scene_tool_rows_have_compatible_invocation(rows[index], incoming_row):
            return index
    for index in content_tool_indexes:
        if index in used_indexes or index < 0 or index >= len(rows):
            continue
        existing_id = _anchor_scene_tool_row_id(rows[index])
        if not existing_id and not incoming_id and _anchor_scene_tool_rows_can_name_match(rows[index], incoming_row):
            return index
    return None


def _anchor_scene_content_rows(message, order_index, message_index, stream_id="", *, is_final_message=False):
    if not _anchor_scene_message_has_content_tool_use(message):
        return None
    rows = []
    content = message.get("content") if isinstance(message, dict) else []
    last_tool_index = -1
    for idx, part in enumerate(content):
        if isinstance(part, dict) and part.get("type") == "tool_use":
            last_tool_index = idx
    for idx, part in enumerate(content):
        if not isinstance(part, dict):
            if is_final_message and idx > last_tool_index:
                continue
            text = _anchor_scene_content_text(part)
            if _anchor_scene_clean_text(text):
                rows.append(_anchor_scene_prose_row(text, order_index + len(rows), message_index, stream_id))
            continue
        part_type = part.get("type")
        if part_type in ("text", "input_text", "output_text"):
            if is_final_message and idx > last_tool_index and _anchor_scene_content_visible_text(part):
                continue
            text = _anchor_scene_content_text(part)
            if _anchor_scene_clean_text(text):
                rows.append(_anchor_scene_prose_row(text, order_index + len(rows), message_index, stream_id))
            continue
        if part_type in ("thinking", "reasoning"):
            text = _anchor_scene_content_text(part)
            if _anchor_scene_clean_text(text):
                rows.append(_anchor_scene_thinking_row(text, order_index + len(rows), message_index, stream_id))
            continue
        if part_type == "tool_use":
            rows.append(
                _anchor_scene_tool_row(
                    _anchor_scene_content_tool(part),
                    order_index + len(rows),
                    message_index,
                    stream_id,
                )
            )
    return rows


def _anchor_scene_row_key(row) -> str:
    if not isinstance(row, dict):
        return ""
    if row.get("role") == "tool":
        tool = row.get("tool") if isinstance(row.get("tool"), dict) else {}
        return "tool:" + str(
            row.get("tool_call_id")
            or tool.get("id")
            or tool.get("tid")
            or tool.get("tool_call_id")
            or tool.get("tool_use_id")
            or tool.get("call_id")
            or row.get("row_id")
            or ""
        )
    if row.get("role") in ("prose", "thinking"):
        return f"{row.get('role')}:{_anchor_scene_text_key(row.get('text'))}"
    if row.get("role") == "lifecycle":
        source_type = str(row.get("source_event_type") or row.get("source") or "")
        if source_type in ("compressing", "compressed"):
            return "lifecycle:compression"
    return f"{row.get('role') or row.get('kind')}:{row.get('source_event_type') or ''}:{row.get('status') or ''}:{row.get('row_id') or ''}"


def _anchor_scene_row_has_live_identity(row) -> bool:
    if not isinstance(row, dict):
        return False
    values = [row.get("row_id"), row.get("local_id"), row.get("event_id")]
    identity = row.get("identity") if isinstance(row.get("identity"), dict) else {}
    values.extend([identity.get("local_id"), identity.get("event_id")])
    if any(str(value or "").startswith("live-") for value in values):
        return True
    group = row.get("group") if isinstance(row.get("group"), dict) else {}
    has_stream_owner = bool(
        row.get("stream_id")
        or row.get("run_id")
        or identity.get("stream_id")
        or identity.get("run_id")
    )
    has_assistant_message_index = group.get("assistant_msg_idx") is not None
    return has_stream_owner and not has_assistant_message_index


def _anchor_scene_settle_live_running_row(row, *, has_settled_thinking: bool):
    if not isinstance(row, dict):
        return row
    role = row.get("role")
    if role not in ("thinking", "prose", "tool"):
        return row
    if str(row.get("status") or "").lower() != "running":
        return row
    if not _anchor_scene_row_has_live_identity(row):
        return row
    if role == "thinking" and has_settled_thinking:
        return None
    next_row = copy.deepcopy(row)
    next_row["status"] = "completed"
    payload = next_row.get("payload")
    if isinstance(payload, dict):
        payload["status"] = "completed"
        if role == "tool":
            payload["done"] = True
    tool = next_row.get("tool")
    if role == "tool" and isinstance(tool, dict):
        tool["done"] = True
    return next_row


def _complete_hydrated_anchor_scene(messages, scene, message_index, *, message_offset=0, tool_calls=None, stream_id=""):
    if not isinstance(messages, list) or not isinstance(scene, dict) or not isinstance(message_index, int):
        return scene
    local_final_idx = message_index - int(message_offset or 0)
    if local_final_idx < 0 or local_final_idx >= len(messages):
        return scene
    final_message = messages[local_final_idx]
    if not isinstance(final_message, dict) or final_message.get("role") != "assistant":
        return scene
    turn_start = -1
    for idx in range(local_final_idx - 1, -1, -1):
        message = messages[idx]
        if isinstance(message, dict) and message.get("role") == "user":
            turn_start = idx
            break
    message_final_answer = _anchor_scene_final_answer_text(final_message)
    scene_final_answer = scene.get("final_answer") if isinstance(scene.get("final_answer"), str) else ""
    final_answer = message_final_answer if _anchor_scene_clean_text(message_final_answer) else scene_final_answer
    final_key = _anchor_scene_text_key(final_answer)
    rows = []
    seen = {}

    def merge_duplicate_tool_row(existing, incoming, *, prefer_incoming_body=False):
        if not isinstance(existing, dict) or not isinstance(incoming, dict):
            return existing
        merged = copy.deepcopy(existing)
        merged_tool = merged.get("tool") if isinstance(merged.get("tool"), dict) else {}
        incoming_tool = incoming.get("tool") if isinstance(incoming.get("tool"), dict) else {}
        merged_payload = merged.get("payload") if isinstance(merged.get("payload"), dict) else {}
        incoming_payload = incoming.get("payload") if isinstance(incoming.get("payload"), dict) else {}

        def empty(value):
            return value is None or value == "" or value == {}

        def merge_missing_args(existing_args, incoming_args):
            if not isinstance(incoming_args, dict) or not incoming_args:
                return existing_args, False
            base = copy.deepcopy(existing_args) if isinstance(existing_args, dict) else {}
            changed = not isinstance(existing_args, dict)
            for key, value in incoming_args.items():
                if key not in base:
                    base[key] = copy.deepcopy(value)
                    changed = True
            return base, changed

        for key in ("snippet", "result", "output"):
            incoming_value = incoming_tool.get(key)
            if not empty(incoming_value) and (
                empty(merged_tool.get(key))
                or (
                    prefer_incoming_body
                    and _anchor_scene_is_bounded_tool_body_preview(merged_tool.get(key), incoming_value)
                )
            ):
                merged_tool[key] = copy.deepcopy(incoming_value)
            incoming_value = incoming_payload.get(key)
            if not empty(incoming_value) and (
                empty(merged_payload.get(key))
                or (
                    prefer_incoming_body
                    and _anchor_scene_is_bounded_tool_body_preview(merged_payload.get(key), incoming_value)
                )
            ):
                merged_payload[key] = copy.deepcopy(incoming_value)
        for key in ("preview", "command", "duration", "started_at"):
            incoming_value = incoming_tool.get(key)
            if not empty(incoming_value) and empty(merged_tool.get(key)):
                merged_tool[key] = copy.deepcopy(incoming_value)
            incoming_value = incoming_payload.get(key)
            if not empty(incoming_value) and empty(merged_payload.get(key)):
                merged_payload[key] = copy.deepcopy(incoming_value)
        merged_args, args_changed = merge_missing_args(merged_tool.get("args"), incoming_tool.get("args"))
        if args_changed:
            merged_tool["args"] = merged_args
        merged_payload_args, payload_args_changed = merge_missing_args(
            merged_payload.get("args"),
            incoming_payload.get("args"),
        )
        if payload_args_changed:
            merged_payload["args"] = merged_payload_args
        merged["tool"] = merged_tool
        merged["payload"] = merged_payload
        return merged

    def push(row, *, prefer_incoming_tool_body=False):
        if not isinstance(row, dict):
            return
        row = _anchor_scene_settle_live_running_row(
            row,
            has_settled_thinking=any(existing.get("role") == "thinking" for existing in rows),
        )
        if row is None or not isinstance(row, dict):
            return
        text_key = _anchor_scene_text_key(row.get("text"))
        if row.get("role") in ("prose", "thinking") and _anchor_scene_row_looks_like_final_answer(text_key, final_key):
            return
        if _anchor_scene_row_is_stale_token_answer(row, text_key, final_key):
            return
        key = _anchor_scene_row_key(row)
        if key and key in seen:
            if key.startswith("tool:"):
                index = seen[key]
                rows[index] = merge_duplicate_tool_row(
                    rows[index],
                    row,
                    prefer_incoming_body=prefer_incoming_tool_body,
                )
                return
            if key == "lifecycle:compression":
                index = seen[key]
                next_row = copy.deepcopy(row)
                next_row["order_index"] = index
                next_row["seq"] = index
                rows[index] = next_row
            return
        if key:
            seen[key] = len(rows)
        next_row = copy.deepcopy(row)
        next_row["order_index"] = len(rows)
        next_row["seq"] = len(rows)
        rows.append(next_row)

    order = 0
    content_tool_indexes_by_idx = {}
    used_content_tool_indexes_by_idx = {}
    id_flexible_content_tool_indexes_by_idx = {}
    for local_idx in range(turn_start + 1, local_final_idx + 1):
        message = messages[local_idx]
        if not isinstance(message, dict) or message.get("role") != "assistant":
            continue
        absolute_idx = int(message_offset or 0) + local_idx
        text = _anchor_scene_message_text(message)
        content_rows = _anchor_scene_content_rows(
            message,
            order,
            absolute_idx,
            stream_id,
            is_final_message=local_idx == local_final_idx,
        )
        content_tool_indexes = []
        used_content_tool_indexes = set()
        id_flexible_content_tool_indexes = set()
        if content_rows:
            for row in content_rows:
                previous_len = len(rows)
                push(row)
                if row.get("role") == "tool" and len(rows) > previous_len:
                    content_tool_indexes.append(len(rows) - 1)
                order += 1
            if content_tool_indexes:
                content_tool_indexes_by_idx[absolute_idx] = content_tool_indexes
                used_content_tool_indexes_by_idx[absolute_idx] = used_content_tool_indexes
                id_flexible_content_tool_indexes_by_idx[absolute_idx] = id_flexible_content_tool_indexes
        elif _anchor_scene_clean_text(text):
            push(_anchor_scene_prose_row(text, order, absolute_idx, stream_id))
            order += 1
        reasoning = _anchor_scene_message_reasoning_text(message)
        if _anchor_scene_clean_text(reasoning) and _anchor_scene_text_key(reasoning) != _anchor_scene_text_key(text):
            push(_anchor_scene_thinking_row(reasoning, order, absolute_idx, stream_id))
            order += 1
        for key in ("tool_calls", "_partial_tool_calls"):
            calls = message.get(key)
            if isinstance(calls, list):
                for tool_ordinal, call in enumerate(calls):
                    row = _anchor_scene_tool_row(call, order, absolute_idx, stream_id)
                    content_match_index = _anchor_scene_matching_content_tool_row_index(
                        rows,
                        content_tool_indexes,
                        row,
                        tool_ordinal,
                        used_content_tool_indexes,
                        len(calls),
                        id_flexible_content_tool_indexes,
                    )
                    if content_match_index is not None:
                        if _anchor_scene_tool_rows_have_different_explicit_ids(
                            rows[content_match_index],
                            row,
                        ):
                            id_flexible_content_tool_indexes.add(content_match_index)
                        rows[content_match_index] = merge_duplicate_tool_row(rows[content_match_index], row)
                        incoming_key = _anchor_scene_row_key(row)
                        if incoming_key:
                            seen[incoming_key] = content_match_index
                        used_content_tool_indexes.add(content_match_index)
                        order += 1
                        continue
                    push(row)
                    order += 1
    external_tool_counts = {}
    for call in tool_calls or []:
        if not isinstance(call, dict):
            continue
        try:
            absolute_idx = int(call.get("assistant_msg_idx"))
        except (TypeError, ValueError):
            continue
        external_tool_counts[absolute_idx] = external_tool_counts.get(absolute_idx, 0) + 1
    external_tool_ordinals = {}
    for call in tool_calls or []:
        if not isinstance(call, dict):
            continue
        try:
            absolute_idx = int(call.get("assistant_msg_idx"))
        except (TypeError, ValueError):
            continue
        local_idx = absolute_idx - int(message_offset or 0)
        if not (turn_start < local_idx <= local_final_idx):
            continue
        row = _anchor_scene_tool_row(call, order, absolute_idx, stream_id)
        tool_ordinal = external_tool_ordinals.get(absolute_idx, 0)
        external_tool_ordinals[absolute_idx] = tool_ordinal + 1
        content_match_index = _anchor_scene_matching_content_tool_row_index(
            rows,
            content_tool_indexes_by_idx.get(absolute_idx, []),
            row,
            tool_ordinal,
            used_content_tool_indexes_by_idx.setdefault(absolute_idx, set()),
            external_tool_counts.get(absolute_idx, 0),
            id_flexible_content_tool_indexes_by_idx.setdefault(absolute_idx, set()),
        )
        if content_match_index is not None:
            if _anchor_scene_tool_rows_have_different_explicit_ids(rows[content_match_index], row):
                id_flexible_content_tool_indexes_by_idx.setdefault(absolute_idx, set()).add(content_match_index)
            rows[content_match_index] = merge_duplicate_tool_row(
                rows[content_match_index],
                row,
                prefer_incoming_body=True,
            )
            incoming_key = _anchor_scene_row_key(row)
            if incoming_key:
                seen[incoming_key] = content_match_index
            used_content_tool_indexes_by_idx[absolute_idx].add(content_match_index)
            order += 1
            continue
        push(row, prefer_incoming_tool_body=True)
        order += 1
    for row in scene.get("activity_rows") or []:
        if isinstance(row, dict) and row.get("role") != "terminal":
            push(row)
    for row in scene.get("activity_rows") or []:
        if isinstance(row, dict) and row.get("role") == "terminal":
            push(row)
    repaired = copy.deepcopy(scene)
    repaired["version"] = "activity_scene_v1"
    repaired["mode"] = repaired.get("mode") or "compact_worklog"
    repaired["final_answer"] = final_answer if _anchor_scene_clean_text(final_answer) else repaired.get("final_answer", "")
    repaired["final_message_ref"] = _assistant_anchor_scene_message_ref(final_message)
    if repaired.get("turn_duration") is None:
        duration = _anchor_scene_message_turn_duration(final_message)
        if duration is not None:
            repaired["turn_duration"] = duration
    repaired["activity_rows"] = rows
    identity = repaired.get("identity") if isinstance(repaired.get("identity"), dict) else {}
    identity = dict(identity)
    identity["source_message_refs"] = [
        _assistant_anchor_scene_message_ref(message)
        for message in messages[turn_start + 1 : local_final_idx + 1]
        if isinstance(message, dict) and message.get("role") == "assistant"
    ]
    repaired["identity"] = identity
    return repaired


def _hydrate_anchor_activity_scenes(messages, records, *, message_offset=0, tool_calls=None):
    if not isinstance(messages, list) or not isinstance(records, dict) or not records:
        return messages
    by_ref = {}
    by_index = {}
    for key, record in records.items():
        if not isinstance(record, dict):
            continue
        scene = record.get("scene")
        if not isinstance(scene, dict):
            continue
        ref = str(record.get("message_ref") or key or "")
        if ref:
            by_ref[ref] = record
        try:
            idx = int(record.get("message_index"))
        except (TypeError, ValueError):
            idx = None
        if idx is not None:
            by_index[idx] = record
    out = list(messages)
    # Read-side ref-ambiguity guard (parity with the write-side
    # _find_anchor_scene_message, which returns None when a ref matches >1
    # message). If two assistant messages ever share a ref (byte-identical
    # whitespace-normalized content + identical _ts), attaching the same scene
    # to both would render duplicate worklog groups. Count ref occurrences and
    # fall through to the index-based match (which is positional, unambiguous)
    # for any ref that resolves to more than one assistant message.
    _ref_counts: dict[str, int] = {}
    for _m in messages:
        if isinstance(_m, dict) and _m.get("role") == "assistant":
            _r = _assistant_anchor_scene_message_ref(_m)
            if _r:
                _ref_counts[_r] = _ref_counts.get(_r, 0) + 1
    for local_idx, message in enumerate(messages):
        if not isinstance(message, dict) or message.get("role") != "assistant":
            continue
        absolute_idx = int(message_offset or 0) + local_idx
        _msg_ref = _assistant_anchor_scene_message_ref(message)
        record = by_ref.get(_msg_ref) if _ref_counts.get(_msg_ref, 0) <= 1 else None
        if not record:
            candidate = by_index.get(absolute_idx)
            if candidate and _anchor_scene_candidate_matches_scene(message, candidate.get("scene") or {}):
                record = candidate
        if not record:
            continue
        scene = record.get("scene")
        if not isinstance(scene, dict):
            continue
        next_message = dict(message)
        stream_id = record.get("stream_id")
        next_message["_anchor_activity_scene"] = _complete_hydrated_anchor_scene(
            messages,
            scene,
            absolute_idx,
            message_offset=message_offset,
            tool_calls=tool_calls,
            stream_id=str(stream_id or ""),
        )
        if stream_id:
            next_message["_anchor_stream_id"] = str(stream_id)
        out[local_idx] = next_message
    return out


def _handle_session_anchor_scene(handler, body):
    try:
        require(body, "session_id", "scene")
    except ValueError as exc:
        return bad(handler, str(exc))
    sid = str(body.get("session_id") or "").strip()
    if not sid:
        return bad(handler, "session_id is required", 400)
    message_index = _anchor_scene_message_index_from_request(body)
    message_ref = str(body.get("message_ref") or "")
    try:
        scene = _sanitize_anchor_activity_scene(body.get("scene"))
    except ValueError as exc:
        return bad(handler, str(exc), 400)
    try:
        s = _get_or_materialize_session(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    except PermissionError:
        return bad(handler, "Read-only imported sessions cannot persist anchor scenes", 403)
    # Active-profile visibility guard (parity with GET /api/session, routes.py:~8922).
    # _get_or_materialize_session loads by id with no profile scoping, so without
    # this an authenticated request under profile A could persist anchor scenes
    # onto a session owned by profile B (cross-profile write). Reject as 404 —
    # same shape the read path uses — and leave anchor_activity_scenes untouched.
    # #7710: cross-profile writes are rejected with 409
    # ``session_profile_mismatch`` so the client can offer to switch
    # to the owning profile (mirrors the detail-load endpoint's
    # contract at #13043 / #13493). 404 is preserved for the
    # None-profile (unknown/legacy) case so the frontend self-heal
    # path still fires for actually-missing sids.
    _anchor_session_profile = getattr(s, "profile", None) or None
    if not _session_visible_to_active_profile(_anchor_session_profile, handler):
        if _anchor_session_profile:
            return j(handler, {
                "error": "Session belongs to a different profile",
                "code": "session_profile_mismatch",
                "session_id": sid,
                "profile": _anchor_session_profile,
            }, status=409)
        return bad(handler, "Session not found", 404)
    with _get_session_agent_lock(sid):
        idx, message = _find_anchor_scene_message(
            getattr(s, "messages", None) or [],
            message_index=message_index,
            message_ref=message_ref,
            scene=scene,
        )
        if message is None or idx is None:
            return bad(handler, "Assistant message not found", 404)
        if scene.get("turn_duration") is None:
            duration = _anchor_scene_message_turn_duration(message)
            if duration is not None:
                scene["turn_duration"] = duration
        ref = _assistant_anchor_scene_message_ref(message)
        records = dict(_anchor_scene_records(s))
        records[ref or f"index:{idx}"] = {
            "version": "anchor_activity_scene_record_v1",
            "message_index": idx,
            "message_ref": ref,
            "stream_id": str(body.get("stream_id") or ""),
            "scene": scene,
            "updated_at": time.time(),
        }
        if len(records) > 256:
            ordered = sorted(
                records.items(),
                key=lambda item: float((item[1] or {}).get("updated_at") or 0),
            )
            records = dict(ordered[-256:])
        s.anchor_activity_scenes = records
        s.save(touch_updated_at=False, skip_index=True)
    return j(handler, {"ok": True, "message_index": idx, "message_ref": ref})


def _get_or_materialize_session(sid: str, *, refresh_cli_messages: bool = False):
    """Get a session, materializing from CLI/agent metadata if not in WebUI store.

    Mirrors the fallback logic in /api/session/archive (routes.py:~8530).
    Raises:
        KeyError: session not found in any store
        PermissionError: session is read-only (messaging/Claude Code)
    """
    try:
        s = get_session(sid)
        s = _ensure_full_session_before_mutation(sid, s)
        # Read-only guard on the happy path too: an already-stored read-only /
        # imported session must not be mutated via rename/update/move
        # (Session.save() does not enforce this). Scope this to the explicit
        # read_only flag — a stored messaging session already owns its sidecar,
        # so the messaging-fork concern only applies to the materialize fallback
        # below (and the heuristic record-check would mis-trip on mock sessions).
        if getattr(s, "read_only", False):
            raise PermissionError("read-only imported session")
        # A previously-persisted subagent sidecar (#5307) is view-only and
        # owned by the delegate runner — even if it was stored with
        # read_only=False (e.g. materialized before this fix), it must not be
        # mutated / used as a writable chat session. This mirrors the
        # missing-sidecar subagent guard below on the happy path.
        if (
            (getattr(s, "source_tag", "") or getattr(s, "raw_source", "") or "").strip().lower() == "subagent"
            or _is_subagent_child_session_id(sid)
        ):
            raise PermissionError("read-only subagent child session")
        if refresh_cli_messages and getattr(s, "is_cli_session", False):
            latest_messages = get_cli_session_messages(
                sid,
                profile=getattr(s, "profile", None),
            )
            current_messages = list(getattr(s, "messages", None) or [])
            if (
                latest_messages
                and len(latest_messages) >= len(current_messages)
                and _session_messages_have_prefix(latest_messages, current_messages)
            ):
                # Keep the stitched CLI transcript authoritative on the first
                # WebUI continuation path without clobbering later divergent
                # WebUI-owned turns.
                s.messages = list(latest_messages)
        return s
    except KeyError:
        pass

    # Fallback: try to materialize from CLI/agent session metadata
    cli_meta = _lookup_cli_session_metadata(sid)

    # Delegated subagent children (#5307) are view-only: their transcript lives
    # in state.db and ownership belongs to the delegate runner, not WebUI. They
    # must never be materialized as a writable sidecar here — this is the shared
    # chokepoint reached by POST /api/chat/start (_get_or_materialize_session),
    # so gating it closes the write path that bypasses the GET/import_cli guards.
    # Checked via state.db source (independent of cli_meta, which is often empty
    # for a server-side subagent child).
    _mat_source_tag = (
        (cli_meta or {}).get("source_tag") or (cli_meta or {}).get("raw_source") or ""
    ).strip().lower()
    if _mat_source_tag == "subagent" or _is_subagent_child_session_id(sid):
        raise PermissionError("read-only subagent child session")

    if not cli_meta:
        raise KeyError(sid)

    # Read-only guard: messaging sessions and Claude Code imports cannot be
    # mutated. Reject BOTH an explicit read_only flag AND any messaging-source
    # record — agent rows normalize messaging sources without setting read_only,
    # and state.db (not a WebUI sidecar) is the source of truth for them, so
    # materializing a writable sidecar would fork the title/state.
    if cli_meta.get("read_only") or _is_messaging_session_record(cli_meta):
        raise PermissionError("read-only imported session")

    # Preserve source metadata fields
    def _apply_source_meta(s):
        s.is_cli_session = is_cli_session_row(cli_meta)
        s.source_tag = cli_meta.get("source_tag")
        s.raw_source = cli_meta.get("raw_source") or cli_meta.get("source_tag")
        s.session_source = cli_meta.get("session_source")
        s.source_label = cli_meta.get("source_label")
        s.user_id = cli_meta.get("user_id")
        s.chat_id = cli_meta.get("chat_id")
        s.chat_type = cli_meta.get("chat_type")
        s.thread_id = cli_meta.get("thread_id")
        s.session_key = cli_meta.get("session_key")
        s.platform = cli_meta.get("platform")

    if _is_messaging_session_record(cli_meta):
        # Messaging sessions: lightweight Session with no messages (state.db is source of truth)
        s = Session(
            session_id=sid,
            title=cli_meta.get("title") or title_from(get_cli_session_messages(sid), "CLI Session"),
            workspace=get_last_workspace(),
            model=cli_meta.get("model") or "unknown",
            created_at=cli_meta.get("created_at"),
            updated_at=cli_meta.get("updated_at"),
        )
        _apply_source_meta(s)
        s.save(touch_updated_at=False)
    else:
        # Regular CLI/agent sessions: import full message history
        msgs = get_cli_session_messages(sid)
        if not msgs:
            raise KeyError(sid)
        s = import_cli_session(
            sid,
            cli_meta.get("title") or title_from(msgs, "CLI Session"),
            msgs,
            cli_meta.get("model") or "unknown",
            profile=cli_meta.get("profile"),
            created_at=cli_meta.get("created_at"),
            updated_at=cli_meta.get("updated_at"),
        )
        _apply_source_meta(s)

    return s


def _share_snapshot_messages_for_session(session, *, cli_meta: dict | None = None) -> list:
    """Return the visible transcript that a public share should snapshot.

    External sessions (Telegram/Discord/Slack/CLI/etc.) may have no WebUI sidecar
    or may persist only local metadata in the sidecar while the transcript lives
    in state.db. Public sharing should snapshot the same visible conversation the
    session page renders, not the bare local sidecar payload.
    """
    sid = str(getattr(session, "session_id", "") or "").strip()
    current_messages = list(getattr(session, "messages", None) or [])
    if not sid:
        return current_messages
    profile = getattr(session, "profile", None)
    is_messaging = (
        _is_messaging_session_record(session)
        or _is_messaging_session_record(cli_meta)
    )
    if is_messaging or not current_messages:
        cli_messages = get_cli_session_messages(sid, profile=profile)
        if cli_messages:
            if is_messaging:
                return _merged_session_messages_for_display(session, cli_messages)
            return list(cli_messages)
    return current_messages


def _build_share_metadata_sidecar(
    sid: str,
    snapshot_session,
    *,
    cli_meta: dict | None = None,
):
    """Create a minimal WebUI sidecar for share metadata on external sessions."""
    cli_meta = dict(cli_meta or {})
    workspace = (
        cli_meta.get("workspace")
        or cli_meta.get("cwd")
        or getattr(snapshot_session, "workspace", None)
    )
    if not workspace:
        workspace = get_last_workspace()
    session = Session(
        session_id=sid,
        title=(
            cli_meta.get("title")
            or getattr(snapshot_session, "title", None)
            or title_from(getattr(snapshot_session, "messages", None) or [], "CLI Session")
        ),
        workspace=workspace,
        messages=[],
        model=cli_meta.get("model") or getattr(snapshot_session, "model", None) or "unknown",
        model_provider=(
            cli_meta.get("model_provider")
            or getattr(snapshot_session, "model_provider", None)
        ),
        created_at=cli_meta.get("created_at") or getattr(snapshot_session, "created_at", None),
        updated_at=cli_meta.get("updated_at") or getattr(snapshot_session, "updated_at", None),
        profile=cli_meta.get("profile") or getattr(snapshot_session, "profile", None),
    )
    session.is_cli_session = bool(
        getattr(snapshot_session, "is_cli_session", False)
        or is_cli_session_row(cli_meta)
    )
    session.source_tag = cli_meta.get("source_tag") or getattr(snapshot_session, "source_tag", None)
    session.raw_source = (
        cli_meta.get("raw_source")
        or getattr(snapshot_session, "raw_source", None)
        or session.source_tag
    )
    session.session_source = (
        cli_meta.get("session_source")
        or getattr(snapshot_session, "session_source", None)
    )
    session.source_label = (
        cli_meta.get("source_label")
        or getattr(snapshot_session, "source_label", None)
    )
    session.read_only = bool(
        cli_meta.get("read_only") or getattr(snapshot_session, "read_only", False)
    )
    for attr in (
        "user_id",
        "chat_id",
        "chat_type",
        "thread_id",
        "session_key",
        "platform",
        "origin_chat_id",
        "origin_user_id",
        "parent_session_id",
    ):
        value = cli_meta.get(attr)
        if value is None:
            value = getattr(snapshot_session, attr, None)
        if value is not None:
            setattr(session, attr, value)
    return session


def _resolve_share_session_pair(sid: str, handler):
    """Resolve a shareable session plus the sidecar that stores share metadata.

    Returns ``(snapshot_session, stored_session_or_none, cli_meta)``. The
    snapshot session always carries the transcript that should become the public
    share payload. ``stored_session`` is the WebUI-owned sidecar to mutate for
    share_token/share_created_at persistence; it may be absent for pure external
    sessions that have not yet created local metadata.
    """
    try:
        stored_session = get_session(sid)
        cli_meta = (
            _lookup_cli_session_metadata(sid)
            if _session_requires_cli_metadata_lookup(stored_session)
            else {}
        )
        effective_profile = (
            (cli_meta or {}).get("profile")
            or getattr(stored_session, "profile", None)
            or None
        )
        if not _session_visible_to_active_profile(effective_profile, handler):
            raise KeyError(sid)
        stored_session = _ensure_full_session_before_mutation(sid, stored_session)
        snapshot_session = copy.copy(stored_session)
        snapshot_session.messages = _share_snapshot_messages_for_session(
            stored_session,
            cli_meta=cli_meta,
        )
        return snapshot_session, stored_session, cli_meta or {}
    except KeyError:
        cli_meta = _lookup_cli_session_metadata(sid) or {}
        effective_profile = cli_meta.get("profile") or None
        if not _session_visible_to_active_profile(effective_profile, handler):
            raise KeyError(sid) from None
        synth, reason = _claim_or_synthesize_cli_session(sid, cli_meta=cli_meta)
        if reason == "was_webui" or synth is None:
            raise KeyError(sid) from None
        return synth, None, cli_meta


def _reconcile_stale_stream_state_for_session_rows(session_rows) -> bool:
    """Clear stale persisted stream fields before /api/sessions serializes rows."""
    changed = False
    for row in session_rows:
        if not isinstance(row, dict):
            continue
        sid = row.get("session_id")
        if not sid or not row.get("active_stream_id"):
            continue
        if row.get("is_streaming") is True:
            continue
        try:
            session = get_session(sid, metadata_only=True)
        except Exception:
            logger.debug(
                "Failed to load session %s while reconciling stale stream state",
                sid,
                exc_info=True,
            )
            continue
        if session is None:
            continue
        changed = _clear_stale_stream_state(session) or changed
    return changed

# ── CSRF: validate Origin/Referer on POST ────────────────────────────────────
import re as _re


def _normalize_host_port(value: str) -> tuple[str, str | None]:
    """Split a host or host:port string into (hostname, port|None).
    Handles IPv6 bracket notation, e.g. [::1]:8080."""
    value = value.strip().lower()
    if not value:
        return '', None
    if value.startswith('['):
        end = value.find(']')
        if end != -1:
            host = value[1:end]
            rest = value[end + 1 :]
            if rest.startswith(':') and rest[1:].isdigit():
                return host, rest[1:]
            return host, None
    if value.count(':') == 1:
        host, port = value.rsplit(':', 1)
        if port.isdigit():
            return host, port
    return value, None


def _ports_match(origin_scheme: str, origin_port: str | None, allowed_port: str | None) -> bool:
    """Return True when two ports should be considered equivalent, scheme-aware.

    Treats an absent port as the scheme default: port 80 for http, port 443 for https.
    Port 80 is NOT treated as equivalent to 443 (different protocols = different origins).
    """
    if origin_port == allowed_port:
        return True
    # Determine the default port for the origin's scheme
    default = '443' if origin_scheme == 'https' else '80'
    if not origin_port and allowed_port == default:
        return True
    if not allowed_port and origin_port == default:
        return True
    return False


def _allowed_public_origins() -> set[str]:
    """Parse HERMES_WEBUI_ALLOWED_ORIGINS env var (comma-separated) into a set.

    Each entry must include the scheme, e.g. https://myapp.example.com:8000.
    Entries without a scheme are silently skipped and a warning is printed.
    """
    raw = os.getenv('HERMES_WEBUI_ALLOWED_ORIGINS', '')
    result = set()
    for value in raw.split(','):
        value = value.strip().rstrip('/').lower()
        if not value:
            continue
        if not (value.startswith('http://') or value.startswith('https://')):
            import sys
            print(
                f"[webui] WARNING: HERMES_WEBUI_ALLOWED_ORIGINS entry {value!r} is missing "
                f"the scheme (expected https://hostname or http://hostname). Entry ignored.",
                flush=True, file=sys.stderr,
            )
            continue
        result.add(value)
    return result


def _is_browser_unsafe_request(handler) -> bool:
    """Return True when request headers identify a browser unsafe request.

    Non-browser API clients, including the MCP bridge and curl-style scripts,
    normally send no Origin/Referer and remain compatible with the existing
    same-machine API contract. Browsers send Origin for unsafe fetch/form POSTs;
    Referer is retained for older paths and proxies.
    """
    return bool(handler.headers.get("Origin") or handler.headers.get("Referer"))


def _check_same_origin_browser_request(handler, *, require_provenance: bool = False) -> bool:
    _clear_csrf_failure_reason(handler)
    origin = handler.headers.get("Origin", "")
    referer = handler.headers.get("Referer", "")
    host = handler.headers.get("Host", "")
    sec_fetch_site = handler.headers.get("Sec-Fetch-Site", "").strip().lower()
    if not (origin or referer or sec_fetch_site):
        return not require_provenance or _set_csrf_failure_reason(handler, "origin_mismatch")
    if sec_fetch_site == "cross-site":
        return _set_csrf_failure_reason(handler, "origin_mismatch")
    target = origin or referer
    if not target:
        if sec_fetch_site == "none":
            return True
        if sec_fetch_site == "same-origin":
            return not require_provenance or _set_csrf_failure_reason(
                handler, "origin_mismatch"
            )
        return _set_csrf_failure_reason(handler, "origin_mismatch")
    m = _re.match(r"^https?://([^/]+)", target)
    if not m:
        return _set_csrf_failure_reason(handler, "origin_mismatch")
    origin_host = m.group(1)
    origin_scheme = m.group(0).split('://')[0].lower()
    origin_name, origin_port = _normalize_host_port(origin_host)
    origin_allowed = False
    origin_value = m.group(0).rstrip('/').lower()
    if origin_value in _allowed_public_origins():
        origin_allowed = True
    if not origin_allowed:
        allowed_hosts = [h.strip() for h in [host] if h.strip()]
        trust_forwarded_host = os.getenv("HERMES_WEBUI_TRUST_FORWARDED_HOST", "").strip().lower()
        if trust_forwarded_host in ("1", "true", "yes", "on"):
            allowed_hosts.extend(
                h.strip()
                for h in [
                    handler.headers.get("X-Forwarded-Host", ""),
                    handler.headers.get("X-Real-Host", ""),
                ]
                if h.strip()
            )
        for allowed in allowed_hosts:
            allowed_name, allowed_port = _normalize_host_port(allowed)
            if origin_name == allowed_name and _ports_match(origin_scheme, origin_port, allowed_port):
                origin_allowed = True
                break
    if not origin_allowed:
        return _set_csrf_failure_reason(handler, "origin_mismatch")
    return True


def apply_cors_preflight_headers(handler) -> None:
    """Emit CORS preflight headers on ``handler`` for a same-origin/allowlisted
    request; emit nothing for a disallowed origin (browser treats the header-less
    200 as a preflight denial).

    Echoes the request Origin only when it is same-origin or explicitly
    allowlisted via HERMES_WEBUI_ALLOWED_ORIGINS — the exact policy the CSRF gate
    enforces for real requests. Reuses _check_same_origin_browser_request so the
    preflight can never advertise wider access (`*`) than an actual request would
    be granted. A wildcard here would let any site read authenticated responses
    on a deployment with no password set. Kept in api/ so server.py stays a thin
    dispatcher.
    """
    origin = handler.headers.get("Origin", "").strip()
    if not origin or not _check_same_origin_browser_request(handler):
        return
    handler.send_header("Access-Control-Allow-Origin", origin)
    handler.send_header("Vary", "Origin")
    handler.send_header("Access-Control-Allow-Methods", "GET, POST, PUT, PATCH, DELETE, OPTIONS")
    handler.send_header("Access-Control-Allow-Headers", "Content-Type, Authorization")


def _csrf_exempt_path(path: str) -> bool:
    """Paths that cannot or must not carry a session CSRF token."""
    return path in {
        "/api/auth/login",
        "/api/auth/passkey/options",
        "/api/auth/passkey/login",
        "/api/csp-report",
    }


_CSRF_FAILURE_ATTR = "_hermes_csrf_failure_reason"


def _set_csrf_failure_reason(handler, reason: str) -> bool:
    try:
        setattr(handler, _CSRF_FAILURE_ATTR, reason)
    except Exception:
        pass
    return False


def _clear_csrf_failure_reason(handler) -> None:
    try:
        if hasattr(handler, _CSRF_FAILURE_ATTR):
            delattr(handler, _CSRF_FAILURE_ATTR)
    except Exception:
        pass


def _csrf_rejection_error(handler) -> str:
    reason = getattr(handler, _CSRF_FAILURE_ATTR, "")
    if reason == "origin_mismatch":
        return "Cross-origin mismatch - check reverse proxy headers"
    if reason == "token_mismatch":
        return "Session expired - reload the page"
    return "Cross-origin request rejected"


def _check_csrf(handler) -> bool:
    """Reject cross-origin or tokenless authenticated browser unsafe requests."""
    if not _check_same_origin_browser_request(handler):
        # CSRF checks run before read_body(), so close rather than reusing an
        # HTTP/1.1 connection whose unread body would corrupt the next request --
        # but ONLY when the framing says bytes are really queued. Arming
        # unconditionally dropped a healthy pooled connection on a body-less
        # write: verified on the wire, `POST /api/session/new` with
        # `Origin: http://evil.invalid` and no `Content-Length` (and with
        # `Content-Length: 0`) answered 403 + `Connection: close` and the
        # pipelined `GET /api/health/agent` was never served.
        arm_connection_close_if_body_pending(handler)
        return False
    if not _is_browser_unsafe_request(handler):
        return True  # non-browser clients (curl, MCP, agent) have no Origin/Referer

    from api.auth import CSRF_HEADER_NAME, is_auth_enabled, parse_cookie, verify_csrf_token

    if not is_auth_enabled():
        return True
    cookie_val = parse_cookie(handler)
    submitted = handler.headers.get(CSRF_HEADER_NAME) or handler.headers.get("X-CSRF-Token")
    if verify_csrf_token(cookie_val or "", submitted or ""):
        return True
    # Same framing rule as the origin rejection above: a token mismatch on a
    # body-less write has nothing unread to protect. Verified on the wire with an
    # authenticated same-origin `POST /api/session/new` carrying no CSRF token and
    # no `Content-Length`: 403 + `Connection: close`, follow-up dropped.
    arm_connection_close_if_body_pending(handler)
    return _set_csrf_failure_reason(handler, "token_mismatch")


_EXTENSION_SIDECAR_PROXY_RE = _re.compile(
    r"^/api/extensions/(?P<extension_id>[^/]+)/sidecar(?:/(?P<proxy_path>.*))?$"
)
_HOP_BY_HOP_HEADERS = {
    "connection",
    "keep-alive",
    "proxy-connection",
    "proxy-authenticate",
    "proxy-authorization",
    "te",
    "trailer",
    "transfer-encoding",
    "upgrade",
}


def _connection_bound_header_names(headers) -> set[str]:
    names = set(_HOP_BY_HOP_HEADERS)
    if not headers or not hasattr(headers, "items"):
        return names
    connection_values = []
    if hasattr(headers, "get_all"):
        connection_values.extend(headers.get_all("Connection", []))
    else:
        for name, value in headers.items():
            if str(name).lower() == "connection":
                connection_values.append(value)
    for value in connection_values:
        for token in str(value).split(","):
            normalized = token.strip().lower()
            if normalized:
                names.add(normalized)
    return names


def _match_extension_sidecar_proxy_path(path: str) -> tuple[str, str] | None:
    match = _EXTENSION_SIDECAR_PROXY_RE.match(path or "")
    if not match:
        return None
    return match.group("extension_id"), match.group("proxy_path") or ""


def _read_body_bytes(handler) -> bytes:
    raw_length = handler.headers.get("Content-Length", 0)
    try:
        length = int(raw_length)
    except (TypeError, ValueError):
        try:
            handler.close_connection = True
        except Exception:
            pass
        raise ValueError(f"Invalid Content-Length: {raw_length!r}") from None
    if length < 0:
        try:
            handler.close_connection = True
        except Exception:
            pass
        raise ValueError(f"Invalid Content-Length: {length}")
    if length > MAX_BODY_BYTES:
        try:
            handler.close_connection = True
        except Exception:
            pass
        raise ValueError(f"Request body too large ({length} bytes, max {MAX_BODY_BYTES})")
    return handler.rfile.read(length) if length else b""


def _extension_sidecar_proxy_request_headers(handler) -> dict[str, str]:
    headers = {}
    raw_headers = getattr(handler, "headers", None)
    if not raw_headers or not hasattr(raw_headers, "items"):
        return headers
    blocked_headers = _connection_bound_header_names(raw_headers)
    for name, value in raw_headers.items():
        lower = str(name).lower()
        if (
            lower in blocked_headers
            or lower in {"authorization", "cookie", "content-length", "host", "origin", "referer"}
            or lower.startswith("x-csrf")
            or lower.startswith("x-hermes-")
        ):
            continue
        headers[str(name)] = str(value)
    return headers


def _send_extension_sidecar_proxy_response(handler, status: int, body: bytes, headers) -> bool:
    handler.send_response(status)
    sent_content_type = False
    blocked_headers = _connection_bound_header_names(headers)
    if headers and hasattr(headers, "items"):
        for name, value in headers.items():
            lower = str(name).lower()
            if (
                lower in blocked_headers
                or lower in {"content-length", "set-cookie"}
                or lower.startswith("x-hermes-")
            ):
                continue
            if lower == "content-type":
                sent_content_type = True
            handler.send_header(str(name), str(value))
    if not sent_content_type:
        handler.send_header("Content-Type", "application/octet-stream")
    handler.send_header("Content-Length", str(len(body)))
    handler.send_header("Cache-Control", "no-store")
    _security_headers(handler)
    handler.end_headers()
    handler.wfile.write(body)
    return True


def _read_extension_sidecar_proxy_body(stream) -> bytes:
    body = stream.read(_EXTENSION_SIDECAR_PROXY_MAX_RESPONSE_BYTES + 1)
    if len(body) > _EXTENSION_SIDECAR_PROXY_MAX_RESPONSE_BYTES:
        raise ValueError("Extension sidecar response too large")
    return body


def _extension_sidecar_proxy_redirect_url(
    allowed_origin: str,
    request_url: str,
    redirect_url: str,
) -> str | None:
    resolved = urljoin(request_url, redirect_url or "")
    allowed = urlsplit(allowed_origin or "")
    parts = urlsplit(resolved)
    if not allowed.scheme or not allowed.netloc or not parts.scheme or not parts.netloc:
        return None
    allowed_scheme = allowed.scheme.lower()
    redirect_scheme = parts.scheme.lower()
    if redirect_scheme != allowed_scheme:
        return None
    allowed_name, allowed_port = _normalize_host_port(allowed.netloc)
    redirect_name, redirect_port = _normalize_host_port(parts.netloc)
    if redirect_name != allowed_name or not _ports_match(
        allowed_scheme,
        redirect_port,
        allowed_port,
    ):
        return None
    return resolved


def _extension_sidecar_proxy_same_origin_opener(allowed_origin: str):
    class _SameOriginRedirectHandler(HTTPRedirectHandler):
        def redirect_request(self, req, fp, code, msg, headers, newurl):
            resolved = _extension_sidecar_proxy_redirect_url(
                allowed_origin,
                req.full_url,
                newurl,
            )
            if not resolved:
                raise URLError("Extension sidecar redirect crossed declared origin")
            return super().redirect_request(req, fp, code, msg, headers, resolved)

    return build_opener(ProxyHandler({}), _SameOriginRedirectHandler)


def _handle_extension_sidecar_proxy(
    handler,
    parsed,
    method: str,
    *,
    read_request_body: bool = False,
):
    matched = _match_extension_sidecar_proxy_path(parsed.path)
    if matched is None:
        return False
    # Require same-origin browser provenance on EVERY proxied method, not just
    # GET. Browser extensions (the only legitimate caller) always send Origin/
    # Referer/Sec-Fetch-Site, so this costs nothing on the real path while
    # closing the GET-vs-unsafe-method asymmetry: without it, POST/PATCH/PUT/
    # DELETE fell through the CSRF compatibility path that intentionally admits
    # non-browser clients, giving unsafe methods weaker provenance than GET.
    if not _check_same_origin_browser_request(handler, require_provenance=True):
        # Provenance rejection runs before read_body(), so close-and-advertise
        # whenever the request DECLARED a body (Content-Length non-zero, or any
        # Transfer-Encoding): those bytes are still queued in rfile and a reused
        # HTTP/1.1 connection would parse them as the next request line (same
        # class as _check_csrf). Gating on read_request_body instead was wrong in
        # both directions — it missed a GET that carries a declared body, and it
        # closed a healthy connection on a body-less DELETE/PUT/PATCH.
        arm_connection_close_if_body_pending(handler)
        return j(handler, {"error": _csrf_rejection_error(handler)}, status=403)
    try:
        request_body = _read_body_bytes(handler) if read_request_body else None
    except ValueError as exc:
        status = 413 if "too large" in str(exc).lower() else 400
        return bad(handler, str(exc), status=status)
    from api.extensions import (
        ExtensionSidecarProxyError,
        resolve_extension_sidecar_proxy_target,
    )

    extension_id, proxy_path = matched
    try:
        target = resolve_extension_sidecar_proxy_target(
            extension_id,
            proxy_path,
            query=parsed.query,
        )
        proxied_headers = _extension_sidecar_proxy_request_headers(handler)
        # token-v1: inject the per-extension shared secret core minted. The
        # inbound x-hermes-* strip above guarantees the client cannot have
        # forged this header.
        _auth_token = target.get("auth_token")
        if _auth_token:
            proxied_headers["X-Hermes-Sidecar-Token"] = _auth_token
        request = Request(
            target["upstream_url"],
            data=request_body,
            headers=proxied_headers,
            method=method,
        )
        opener = _extension_sidecar_proxy_same_origin_opener(target["origin"])
        with opener.open(request, timeout=10) as response:
            body = _read_extension_sidecar_proxy_body(response)
            return _send_extension_sidecar_proxy_response(
                handler,
                getattr(response, "status", 200),
                body,
                response.headers,
            )
    except ExtensionSidecarProxyError as exc:
        return bad(handler, str(exc), status=exc.status)
    except ValueError as exc:
        return bad(handler, str(exc), status=502)
    except HTTPError as exc:
        try:
            body = _read_extension_sidecar_proxy_body(exc)
        except ValueError as read_exc:
            return bad(handler, str(read_exc), status=502)
        return _send_extension_sidecar_proxy_response(
            handler,
            exc.code,
            body,
            exc.headers,
        )
    except (TimeoutError, URLError, OSError):
        logger.warning(
            "extension sidecar proxy failed for %s %s",
            method,
            parsed.path,
            exc_info=True,
        )
        return bad(handler, "Failed to reach extension sidecar", status=502)


def _client_ip_for_rate_limit(handler) -> str:
    try:
        address = getattr(handler, "client_address", None)
        if address:
            return str(address[0])
    except Exception:
        pass
    return "unknown"


def _truthy_env(name: str) -> bool:
    return os.getenv(name, "").strip().lower() in {"1", "true", "yes", "on"}


def _request_client_ip(handler) -> str:
    try:
        address = getattr(handler, "client_address", None)
        if address:
            return str(address[0] or "")
    except Exception:
        pass
    return ""


def _ip_is_loopback_or_private(raw: str):
    """Parse an IP string; return (parsed_ok, is_loopback_or_private).

    Returns (False, False) for empty/malformed input so callers fail closed.
    """
    import ipaddress

    raw = (raw or "").strip()
    if not raw:
        return (False, False)
    try:
        addr = ipaddress.ip_address(raw)
    except ValueError:
        return (False, False)
    return (True, bool(addr.is_loopback or addr.is_private))


def _trusted_proxy_networks():
    """Networks whose socket peer is allowed to assert a forwarded client IP.

    Loopback is ALWAYS trusted implicitly (the common same-host reverse-proxy
    deployment). Operators fronting the WebUI with a LAN/remote proxy add its
    address(es) via HERMES_WEBUI_TRUSTED_PROXY_CIDRS (comma-separated CIDRs or
    bare IPs). Malformed entries are skipped, never widening trust.
    """
    import ipaddress

    nets = [
        ipaddress.ip_network("127.0.0.0/8"),
        ipaddress.ip_network("::1/128"),
        ipaddress.ip_network("::ffff:127.0.0.0/104"),
    ]
    raw = os.getenv("HERMES_WEBUI_TRUSTED_PROXY_CIDRS", "") or ""
    for token in raw.replace(";", ",").split(","):
        token = token.strip()
        if not token:
            continue
        try:
            nets.append(ipaddress.ip_network(token, strict=False))
        except ValueError:
            # Invalid CIDR/IP → skip (fail closed: never widens trust).
            continue
    return nets


def _ip_in_networks(addr, networks) -> bool:
    """Family-aware membership test.

    Checks the parsed address against each network, and — for an IPv4-mapped
    IPv6 address (e.g. ``::ffff:10.9.9.9``) — ALSO checks its embedded IPv4 form
    against IPv4 networks. Without this, a mapped-IPv6 proxy peer would never
    match an IPv4 CIDR allowlist: the trusted proxy would be treated as
    untrusted (locking out legitimate clients behind it) and, inside an XFF
    chain, a mapped trusted hop would be mis-returned as the client (admitting a
    public client that preceded it). See #5764.
    """
    candidates = [addr]
    mapped = getattr(addr, "ipv4_mapped", None)
    if mapped is not None:
        candidates.append(mapped)
    for cand in candidates:
        for net in networks:
            try:
                if cand in net:
                    return True
            except TypeError:
                # IPv4/IPv6 family mismatch between candidate and net → skip.
                continue
    return False


def _raw_peer_is_trusted_proxy(handler) -> bool:
    """True when the immediate socket peer is loopback or an allowlisted proxy.

    Only such a peer is allowed to assert a forwarded client IP. Judged on the
    RAW socket address (never a header), so it cannot be spoofed.
    """
    import ipaddress

    raw = _request_client_ip(handler)
    if not raw:
        return False
    try:
        addr = ipaddress.ip_address(raw)
    except ValueError:
        return False
    return _ip_in_networks(addr, _trusted_proxy_networks())


def _forwarded_client_ip_from_trusted_proxy(handler):
    """Resolve the real client IP from a chain fronted by a trusted proxy.

    Precondition: the caller has verified the raw socket peer is a trusted proxy.
    Consumes ALL X-Forwarded-For values (across repeated headers), preserves wire
    order, walks RIGHT-TO-LEFT skipping hops that are themselves trusted-proxy
    addresses, and returns the first non-trusted (i.e. real-client) hop. Falls
    back to X-Real-IP, then the raw socket peer. Returns None when the chain is
    present-but-empty / malformed so the caller fails closed.
    """
    import ipaddress

    try:
        xff_values = handler.headers.get_all("X-Forwarded-For") or []
    except AttributeError:
        single = handler.headers.get("X-Forwarded-For", "")
        xff_values = [single] if single else []

    hops: list[str] = []
    for header_value in xff_values:
        for token in str(header_value or "").split(","):
            hops.append(token.strip())

    if xff_values:
        # A present-but-empty / all-blank XFF is malformed → fail closed.
        if not any(hops):
            return None
        trusted_nets = _trusted_proxy_networks()

        def _is_trusted_hop(ip_str: str) -> bool:
            try:
                addr = ipaddress.ip_address(ip_str)
            except ValueError:
                return False
            return _ip_in_networks(addr, trusted_nets)

        for hop in reversed(hops):
            if not hop:
                # An empty hop inside the chain is malformed → fail closed
                # rather than skip past it (an attacker could inject blanks).
                return None
            try:
                ipaddress.ip_address(hop)
            except ValueError:
                # Non-IP token in the chain → malformed → fail closed.
                return None
            if _is_trusted_hop(hop):
                continue
            return hop
        # Every hop was a trusted proxy → no distinct client; treat as the proxy
        # tier itself (loopback/private), i.e. resolve to the raw peer below.
        return _request_client_ip(handler)

    real_ip = handler.headers.get("X-Real-IP", "").strip()
    if real_ip:
        return real_ip
    # No forwarded header at all → the trusted proxy is speaking for itself.
    return _request_client_ip(handler)


def _onboarding_request_is_local(handler) -> bool:
    """Return True when an unauthenticated onboarding request is local/private.

    Trust model (single, symmetric — see the full truth table in
    tests/test_cvd3_terminal_local_origin_gate.py):

    * Forwarded client-IP headers are honored ONLY when the RAW socket peer is a
      trusted proxy (loopback, or an address in HERMES_WEBUI_TRUSTED_PROXY_CIDRS).
      This is checked on the un-spoofable socket address, so a direct client
      cannot promote itself to "local" by sending X-Forwarded-For: 127.0.0.1.
    * When the peer is NOT a trusted proxy, forwarded headers are ignored and the
      request is classified by the raw socket peer directly. A direct loopback or
      private/LAN client (no proxy) is therefore still correctly local — so
      onboarding, first-password/passkey setup, and passwordless embedded-terminal
      access keep working on the common direct-LAN deployment.
    * HERMES_WEBUI_TRUST_FORWARDED_FOR=1 is the opt-in that makes us CONSULT the
      forwarded chain at all; without it the raw peer is authoritative. Either
      way the classification fails closed on malformed/empty chains.
    """
    trust_forwarded = _truthy_env("HERMES_WEBUI_TRUST_FORWARDED_FOR")
    peer_is_trusted_proxy = _raw_peer_is_trusted_proxy(handler)

    if trust_forwarded and peer_is_trusted_proxy:
        client_ip = _forwarded_client_ip_from_trusted_proxy(handler)
        if client_ip is None:
            # Malformed/empty forwarded chain from a trusted proxy → fail closed.
            return False
        parsed_ok, is_local = _ip_is_loopback_or_private(client_ip)
        return parsed_ok and is_local

    # Not consulting the forwarded chain (either the opt-in is off, or the raw
    # peer is not a trusted proxy). Classify by the raw socket peer — it cannot
    # be spoofed by a header. A public peer sending X-Forwarded-For: 127.0.0.1 is
    # therefore correctly rejected (its raw peer is public).
    raw = _request_client_ip(handler)
    parsed_ok, is_local = _ip_is_loopback_or_private(raw)
    if not parsed_ok:
        return False

    import ipaddress

    addr = ipaddress.ip_address(raw.strip())
    if addr.is_loopback:
        # A loopback TCP source is genuinely same-host and unspoofable → local
        # even if a (ignored) forwarded header is present.
        return True

    # Non-loopback raw peer. A forwarded header being PRESENT here means the
    # request most likely arrived through a proxy we have NOT been told to trust
    # (no trusted-proxy env, or the peer isn't in the allowlist) — so a
    # private/LAN raw peer could be an untrusted proxy relaying an arbitrary
    # (public) client we can't see. Deny in that case; require the operator to
    # opt in via HERMES_WEBUI_TRUST_FORWARDED_FOR (+ HERMES_WEBUI_TRUSTED_PROXY_CIDRS
    # for a non-loopback proxy). With NO forwarded header, a direct private/LAN
    # client (the common direct-LAN deployment) stays local so onboarding,
    # first-password/passkey setup, and passwordless terminal keep working.
    forwarded_present = bool(
        (handler.headers.get("X-Forwarded-For", "") or "").strip()
        or (handler.headers.get("X-Real-IP", "") or "").strip()
    )
    if forwarded_present:
        return False
    return bool(is_local)


def _onboarding_gate_allows(handler, auth_enabled: bool | None = None) -> bool:
    from api.auth import is_auth_enabled

    auth_enabled = is_auth_enabled() if auth_enabled is None else auth_enabled
    if auth_enabled or _truthy_env("HERMES_WEBUI_ONBOARDING_OPEN"):
        return True
    return _onboarding_request_is_local(handler)


# Operator-facing copy reused by every embedded-terminal endpoint refusal.
_EMBEDDED_TERMINAL_GATE_DENIED_MESSAGE = (
    "Embedded terminal is only available from local networks when authentication "
    "is not configured. Configure a password/passkey, or set "
    "HERMES_WEBUI_ONBOARDING_OPEN=1 to allow it on a deliberately-exposed server."
)


def _embedded_terminal_gate_allows(handler) -> bool:
    """Local-origin gate for the embedded-terminal endpoints.

    The embedded terminal spawns a PTY shell that runs arbitrary commands as the
    server-process user, so admitting an unauthenticated remote caller is remote
    code execution. When auth is enabled, ``check_auth()`` has already verified
    the session cookie before the request reaches these handlers, so this returns
    True. When auth is DISABLED (the default out-of-the-box state) ``check_auth()``
    admits every caller unconditionally, so restrict the terminal to local/private
    origins — the same trust model the onboarding/bootstrap endpoints use, ignoring
    spoofable forwarded headers unless an operator has opted into trusting them.
    A deliberately-exposed passwordless server (access secured at another layer)
    opts out with ``HERMES_WEBUI_ONBOARDING_OPEN=1``.
    """
    return _onboarding_gate_allows(handler)


# Above this many distinct client keys, sweep out entries whose timestamps have
# all aged past the window on the next update. Behind a reverse proxy the map
# holds a single key (the proxy IP) and never trips this; a directly-exposed
# deployment would otherwise keep one entry forever for every IP that ever hit
# the endpoint, since a key is only revisited when that same IP calls again.
_RATE_LIMIT_MAP_SWEEP_THRESHOLD = 4096


def _prune_stale_rate_limit_keys(mapping: dict, cutoff: float) -> None:
    """Drop keys whose newest timestamp has aged out of the window. Caller holds
    the map's lock. Size-gated so the common (few-key) path stays O(1)."""
    if len(mapping) <= _RATE_LIMIT_MAP_SWEEP_THRESHOLD:
        return
    stale = [k for k, ts in mapping.items() if not ts or ts[-1] < cutoff]
    for k in stale:
        del mapping[k]


def _csp_report_rate_limited(handler, *, now: float | None = None) -> bool:
    now = time.time() if now is None else now
    key = _client_ip_for_rate_limit(handler)
    cutoff = now - _CSP_REPORT_RATE_LIMIT_WINDOW_SECONDS
    with _CSP_REPORT_RATE_LIMIT_LOCK:
        _prune_stale_rate_limit_keys(_CSP_REPORT_RATE_LIMIT, cutoff)
        timestamps = [ts for ts in _CSP_REPORT_RATE_LIMIT.get(key, []) if ts >= cutoff]
        if len(timestamps) >= _CSP_REPORT_RATE_LIMIT_MAX:
            _CSP_REPORT_RATE_LIMIT[key] = timestamps
            return True
        timestamps.append(now)
        _CSP_REPORT_RATE_LIMIT[key] = timestamps
    return False


def _client_event_rate_limited(handler, *, now: float | None = None) -> bool:
    now = time.time() if now is None else now
    key = _client_ip_for_rate_limit(handler)
    cutoff = now - _CLIENT_EVENT_RATE_LIMIT_WINDOW_SECONDS
    with _CLIENT_EVENT_RATE_LIMIT_LOCK:
        _prune_stale_rate_limit_keys(_CLIENT_EVENT_RATE_LIMIT, cutoff)
        timestamps = [ts for ts in _CLIENT_EVENT_RATE_LIMIT.get(key, []) if ts >= cutoff]
        if len(timestamps) >= _CLIENT_EVENT_RATE_LIMIT_MAX:
            _CLIENT_EVENT_RATE_LIMIT[key] = timestamps
            return True
        timestamps.append(now)
        _CLIENT_EVENT_RATE_LIMIT[key] = timestamps
    return False


def _send_no_content(handler, status: int = 204) -> bool:
    handler.send_response(status)
    handler.send_header("Content-Length", "0")
    handler.end_headers()
    return True


def _safe_content_length(handler, max_bytes: int) -> int:
    raw_length = handler.headers.get("Content-Length", 0)
    try:
        length = int(raw_length)
    except (TypeError, ValueError):
        try:
            handler.close_connection = True
        except Exception:
            pass
        raise ValueError(f"Invalid Content-Length: {raw_length!r}") from None
    if length < 0:
        try:
            handler.close_connection = True
        except Exception:
            pass
        raise ValueError(f"Invalid Content-Length: {length}")
    if length > max_bytes:
        try:
            handler.close_connection = True
        except Exception:
            pass
        raise OverflowError(f"Request body too large ({length} bytes, max {max_bytes})")
    return length


def _read_csp_report_payload(handler):
    try:
        length = _safe_content_length(handler, _CSP_REPORT_MAX_BODY_BYTES)
    except OverflowError as exc:
        try:
            handler.rfile.read(_CSP_REPORT_MAX_BODY_BYTES)
        except Exception:
            pass
        return {"discarded": "body_too_large", "error": str(exc)}
    except ValueError as exc:
        return {"discarded": "invalid_content_length", "error": str(exc)}
    raw = handler.rfile.read(length) if length else b"{}"
    try:
        return json.loads(raw.decode("utf-8"))
    except Exception:
        return {"invalid": True, "bytes": len(raw)}


def _handle_csp_report(handler) -> bool:
    """Collect browser CSP report-only violations without requiring auth."""
    if _csp_report_rate_limited(handler):
        _CSP_REPORT_LOGGER.warning(
            "Dropped CSP report from %s: rate limit exceeded",
            _client_ip_for_rate_limit(handler),
        )
        # Rate-limit rejection runs before the body is read; close-and-advertise
        # so the unread report can't corrupt the next pooled request -- but only
        # when a body was really declared. A body-less report answered 204 WITH
        # `Connection: close` once the 100-per-60s limiter tripped (reproduced on
        # the wire at request 101; the pipelined `GET /api/health/agent` was
        # dropped), so a browser that keeps reporting loses its socket each time.
        arm_connection_close_if_body_pending(handler)
        return _send_no_content(handler)

    payload = _read_csp_report_payload(handler)
    _CSP_REPORT_LOGGER.info("CSP report from %s: %s", _client_ip_for_rate_limit(handler), payload)
    return _send_no_content(handler)


def _bounded_client_event_string(value, limit: int) -> str | None:
    if value is None:
        return None
    text = str(value).strip()
    if not text:
        return None
    return text[:limit]


def _sanitize_client_event_url_path(value) -> str | None:
    text = _bounded_client_event_string(value, 1024)
    if not text:
        return None
    try:
        parsed = urlsplit(text)
        path = parsed.path or "/"
    except Exception:
        path = text.split("?", 1)[0] or "/"
    if not path.startswith("/"):
        path = "/" + path.lstrip("/")
    return path[: _CLIENT_EVENT_ALLOWED_FIELDS["url_path"]]


def _sanitize_client_event_payload(payload: dict | None) -> dict:
    """Whitelist tiny browser diagnostic events and discard sensitive content.

    Client-side SSE diagnostics should explain transport failures without
    persisting prompts, cookies, query strings, headers, or arbitrary browser
    payloads. This helper intentionally keeps only bounded scalar metadata.
    """
    if not isinstance(payload, dict):
        return {"event": "unknown"}
    sanitized: dict[str, object] = {}
    for field, limit in _CLIENT_EVENT_ALLOWED_FIELDS.items():
        if field == "url_path":
            value = _sanitize_client_event_url_path(payload.get(field))
        else:
            value = _bounded_client_event_string(payload.get(field), limit)
        if value is not None:
            sanitized[field] = value
    ready_state = payload.get("ready_state")
    if isinstance(ready_state, bool):
        pass
    elif isinstance(ready_state, int) and 0 <= ready_state <= 3:
        sanitized["ready_state"] = ready_state
    online = payload.get("online")
    if isinstance(online, bool):
        sanitized["online"] = online
    elif isinstance(online, str):
        lowered = online.strip().lower()
        if lowered in {"true", "1", "yes", "on"}:
            sanitized["online"] = True
        elif lowered in {"false", "0", "no", "off"}:
            sanitized["online"] = False
    if "event" not in sanitized:
        sanitized["event"] = "unknown"
    return sanitized


def _read_client_event_payload(handler) -> dict:
    try:
        length = _safe_content_length(handler, _CLIENT_EVENT_MAX_BODY_BYTES)
    except OverflowError:
        try:
            handler.rfile.read(_CLIENT_EVENT_MAX_BODY_BYTES)
        except Exception:
            pass
        return {"event": "discarded", "reason": "body_too_large"}
    except ValueError:
        return {"event": "invalid", "reason": "invalid_content_length"}
    raw = handler.rfile.read(length) if length else b"{}"
    try:
        decoded = raw.decode("utf-8")
        payload = json.loads(decoded)
    except Exception:
        return {"event": "invalid", "reason": "invalid_json"}
    return payload if isinstance(payload, dict) else {"event": "invalid", "reason": "not_object"}


def _handle_client_event_log(handler, body: dict) -> bool:
    if _client_event_rate_limited(handler):
        _CLIENT_EVENT_LOGGER.warning(
            "Dropped client event from %s: rate limit exceeded",
            _client_ip_for_rate_limit(handler),
        )
        return j(handler, {"ok": False, "error": "rate_limited"}, status=429) or True
    payload = _sanitize_client_event_payload(body)
    _CLIENT_EVENT_LOGGER.info("Client event from %s: %s", _client_ip_for_rate_limit(handler), payload)
    return j(handler, {"ok": True, "event": payload.get("event")}) or True


def _starts_token(raw: str, prefix: str) -> bool:
    if not raw.startswith(prefix):
        return False
    rest = raw[len(prefix):]
    return rest == "" or rest[0] in ":/"


def _normalize_provider_id(value: str | None) -> str:
    raw = str(value or "").strip().lower()
    if not raw:
        return ""
    if raw in _PROVIDER_ALIASES:
        return _PROVIDER_ALIASES[raw]
    for prefix, normalized in (
        ("openai-codex", "openai"),
        ("openai", "openai"),
        ("anthropic", "anthropic"),
        ("claude", "anthropic"),
        ("google", "google"),
        ("gemini", "google"),
        ("openrouter", "openrouter"),
        ("custom", "custom"),
    ):
        if _starts_token(raw, prefix):
            return normalized
    # Unknown prefix — return empty so callers treat it as "no match" and pass
    # the model through unchanged rather than incorrectly stripping it.
    return "" 


def _catalog_provider_id_sets(catalog: dict) -> tuple[set[str], set[str]]:
    raw_provider_ids: set[str] = set()
    normalized_provider_ids: set[str] = set()
    for group in catalog.get("groups") or []:
        raw = str(group.get("provider_id") or "").strip().lower()
        if not raw:
            continue
        raw_provider_ids.add(raw)
        normalized = _normalize_provider_id(raw)
        if normalized:
            normalized_provider_ids.add(normalized)
    return raw_provider_ids, normalized_provider_ids


def _catalog_has_provider(
    provider_raw: str,
    provider_normalized: str,
    raw_provider_ids: set[str],
    normalized_provider_ids: set[str],
) -> bool:
    return (
        provider_raw in raw_provider_ids
        or (provider_normalized and provider_normalized in raw_provider_ids)
        or (provider_normalized and provider_normalized in normalized_provider_ids)

    )


def _model_matches_active_provider_family(
    model: str,
    active_provider: str,
) -> bool:
    model_lower = model.lower()
    for bare_prefix in ("gpt", "claude", "gemini"):
        if model_lower.startswith(bare_prefix):
            return _normalize_provider_id(bare_prefix) == active_provider
    return False


def _catalog_model_id_matches(candidate: str, model: str) -> bool:
    candidate = str(candidate or "").strip()
    if candidate.startswith("@") and ":" in candidate:
        candidate = candidate.rsplit(":", 1)[1]
    if "/" in candidate:
        candidate = candidate.split("/", 1)[1]
    return candidate.replace("-", ".").lower() == model.replace("-", ".").lower()


def _catalog_group_owns_exact_model(group: dict, model: str) -> bool:
    provider_id = str(group.get("provider_id") or "").strip()
    wrapper = f"@{provider_id}:"
    for bucket in ("models", "extra_models"):
        for entry in group.get(bucket) or []:
            if not isinstance(entry, dict):
                continue
            candidate = str(entry.get("id") or "").strip()
            if candidate.lower().startswith(wrapper.lower()):
                candidate = candidate[len(wrapper):]
            if candidate == model or _catalog_model_id_matches(candidate, model):
                return True
    return False


def _repair_foreign_session_model_provider(
    session,
    *,
    requested_model: str,
    requested_provider: str | None,
    resolved_model: str,
    resolved_provider: str | None,
    explicit_model_pick: bool,
    profile_provider: str | None,
) -> str | None:
    """Repair a stale provider only when the cached catalog names one owner."""
    stored_model = str(getattr(session, "model", "") or "").strip()
    stored_provider = _clean_session_model_provider(getattr(session, "model_provider", None))
    requested_provider = _clean_session_model_provider(requested_provider)
    resolved_provider = _clean_session_model_provider(resolved_provider)
    profile_provider = _clean_session_model_provider(profile_provider)
    _, qualified_provider = _split_provider_qualified_model(requested_model)
    if (
        explicit_model_pick
        or qualified_provider
        or not stored_model
        or not stored_provider
        or (
            str(requested_model or "").strip() != stored_model
            and not _catalog_model_id_matches(str(requested_model or "").strip(), stored_model)
        )
        or requested_provider != stored_provider
        or (
            resolved_model != stored_model
            and not _catalog_model_id_matches(resolved_model, stored_model)
        )
        or resolved_provider != stored_provider
        or not profile_provider
        or profile_provider == stored_provider
    ):
        return resolved_provider

    try:
        catalog = get_available_models(prefer_cache=True)
    except Exception:
        return resolved_provider
    groups = [group for group in catalog.get("groups") or [] if isinstance(group, dict)]
    stored_groups = [
        group
        for group in groups
        if str(group.get("provider_id") or "").strip().lower() == stored_provider
    ]
    if (
        not stored_groups
        or any(group.get("models_endpoint_error") for group in stored_groups)
        or any(_catalog_group_owns_exact_model(group, stored_model) for group in stored_groups)
    ):
        return resolved_provider
    owners = [
        group
        for group in groups
        if str(group.get("provider_id") or "").strip().lower() != stored_provider
        and _catalog_group_owns_exact_model(group, stored_model)
    ]
    if len(owners) != 1:
        return resolved_provider
    return str(owners[0].get("provider_id") or "").strip() or resolved_provider


def _clean_session_model_provider(value: str | None) -> str | None:
    """Normalize a stored/requested provider value to a bare provider ID.

    An ``@``-prefixed value is a provider-qualified *model* hint, so the
    provider is resolved with the shared
    ``config._parse_provider_qualified_model_id()`` grammar rather than a
    positional colon split — that keeps multi-segment custom provider IDs
    (``custom:<slug>``, ``custom:<host>:<port>``) whole while still dropping a
    trailing model segment (#6722). Values without the ``@`` marker are already
    plain provider IDs, whose colons belong to the ID itself, so they are
    preserved verbatim.
    """
    provider = str(value or "").strip().lower()
    if not provider or provider == "default":
        return None
    if provider.startswith("@"):
        parsed = _parse_provider_qualified_model_id(provider)
        provider = parsed[1].strip() if parsed else provider[1:]
    return provider or None


def _split_provider_qualified_model(model: str) -> tuple[str, str | None]:
    """Split an ``@provider:model`` hint into ``(bare_model, provider)``.

    Delegates the grammar to ``config._parse_provider_qualified_model_id()``,
    the shared parser that already knows how to keep a multi-segment custom
    provider ID (``custom:<slug>``, ``custom:<host>:<port>``) intact while
    still letting the model segment carry its own colons for tags such as
    ``:free``. Keeping one parser here means every caller in this module and
    the gateway request path resolve the same provider/model pair (#6722).
    """
    model = str(model or "").strip()
    parsed = _parse_provider_qualified_model_id(model)
    if parsed:
        bare_model, provider_hint = parsed
        provider = _clean_session_model_provider(provider_hint)
        bare = str(bare_model or "").strip()
        if provider and bare:
            return bare, provider
    return model, None


def _model_matches_configured_default(
    session_model: str | None,
    cfg_default: str | None,
    provider: str | None = None,
) -> bool:
    """Return True when ``session_model`` refers to the configured ``model.default``.

    The global ``model.context_length`` cap applies ONLY to the default model
    (#3256/#3263). An exact string compare is not enough because ``model.default``
    and the session model can be stored in different but equivalent shapes:
      - bare:            ``claude-opus-4.8``
      - slash-prefixed:  ``anthropic/claude-opus-4.8``  (OpenRouter-style)
      - @provider:model: ``@anthropic:claude-opus-4.8``

    Matching rule (correct in both directions):
      1. Identical strings → match.
      2. Otherwise compare BARE model ids — BUT only after a provider-compatibility
         check: if BOTH sides carry an identifiable provider (from a ``provider/``
         prefix, an ``@provider:`` qualifier, or the explicit ``provider`` arg for
         the session side) and those providers DIFFER, it is NOT a match. This
         stops a non-default model on a different provider that happens to share a
         bare name (``openai/gpt-4o`` vs default ``openrouter/gpt-4o``) from being
         treated as the default and wrongly receiving its cap.
      3. When a provider can't be identified on one side, fall through to the bare
         comparison (lenient-when-unknown — a bare default config still matches a
         bare/prefixed session model).
    Empty default → no match.
    """
    sess = str(session_model or "").strip()
    default = str(cfg_default or "").strip()
    if not sess or not default:
        return False
    if sess == default:
        return True

    def _split(value: str) -> tuple[str, str | None]:
        """Return (bare_model, provider_or_None) for any of the 3 shapes."""
        value = str(value or "").strip()
        # @provider:model
        unq, q_prov = _split_provider_qualified_model(value)
        if q_prov:
            return unq.strip(), str(q_prov).strip().lower()
        # provider/model (single leading slash segment)
        if "/" in value:
            prefix, rest = value.split("/", 1)
            return rest.strip(), prefix.strip().lower()
        return value, None

    sess_bare, sess_prov = _split(sess)
    default_bare, default_prov = _split(default)
    # The explicit provider arg is the session side's provider when the model
    # string itself didn't carry one.
    if not sess_prov and provider:
        sess_prov = str(provider).strip().lower() or None

    if not sess_bare or not default_bare or sess_bare != default_bare:
        return False
    # Bare ids match. Reject only when both sides name DIFFERENT providers.
    if sess_prov and default_prov and sess_prov != default_prov:
        return False
    return True


class _ContextLengthLookupInputs:
    __slots__ = ("config_context_length", "custom_providers", "base_url", "provider", "api_key")

    def __init__(
        self,
        *,
        config_context_length: int | None = None,
        custom_providers: list | None = None,
        base_url: str = "",
        provider: str = "",
        api_key: str = "",
    ) -> None:
        self.config_context_length = config_context_length
        self.custom_providers = custom_providers
        self.base_url = base_url
        self.provider = provider
        self.api_key = api_key


def _positive_context_length(value) -> int | None:
    try:
        parsed = int(value)
    except (TypeError, ValueError):
        return None
    return parsed if parsed > 0 else None


def _model_lookup_candidates(model: str) -> tuple[str, ...]:
    raw = str(model or "").strip()
    candidates = []
    for candidate in (raw, _split_provider_qualified_model(raw)[0]):
        if candidate and candidate not in candidates:
            candidates.append(candidate)
        if "/" in candidate:
            bare = candidate.split("/", 1)[1].strip()
            if bare and bare not in candidates:
                candidates.append(bare)
    return tuple(candidates)


def _models_config_context_length(models_cfg, model: str) -> int | None:
    candidates = _model_lookup_candidates(model)
    if isinstance(models_cfg, dict):
        for candidate in candidates:
            entry = models_cfg.get(candidate)
            raw_ctx = entry.get("context_length") if isinstance(entry, dict) else entry
            ctx = _positive_context_length(raw_ctx)
            if ctx is not None:
                return ctx
    if isinstance(models_cfg, list):
        for entry in models_cfg:
            if not isinstance(entry, dict):
                continue
            entry_model = str(entry.get("id") or entry.get("model") or entry.get("name") or "").strip()
            if entry_model in candidates:
                ctx = _positive_context_length(entry.get("context_length"))
                if ctx is not None:
                    return ctx
    return None


def _canonical_context_provider(value: str | None) -> str:
    provider = _clean_session_model_provider(value) or ""
    if not provider:
        return ""
    try:
        from api.config import _resolve_provider_alias

        provider = _resolve_provider_alias(provider)
    except Exception:
        pass
    return str(provider or "").strip().lower()


def _custom_provider_slug_for_context(name: object) -> str:
    try:
        from api.config import _custom_provider_slug_from_name

        return _custom_provider_slug_from_name(name)
    except Exception:
        raw = str(name or "").strip().lower()
        if not raw:
            return ""
        if raw.startswith("custom:"):
            return raw
        slug = re.sub(r"[^a-z0-9._-]+", "-", raw).strip("-")
        slug = re.sub(r"-{2,}", "-", slug)
        return f"custom:{slug}" if slug else ""


def _providers_match_for_context(config_key: object, requested_provider: str) -> bool:
    if not requested_provider:
        return False
    raw_key = str(config_key or "").strip().lower()
    key = _canonical_context_provider(raw_key)
    requested = _canonical_context_provider(requested_provider)
    return bool(
        requested
        and (
            raw_key == requested
            or key == requested
            or raw_key == str(requested_provider or "").strip().lower()
        )
    )


def _custom_provider_api_key_for_context(entry: dict, provider: str) -> str:
    """Resolve the API key for a matched ``custom_providers`` entry.

    Static session hydration/update routes already have a per-profile config
    snapshot. Resolve from the matched entry instead of re-reading global config,
    while preserving the same literal, ``${ENV_VAR}``, ``key_env``, and
    sanitized-env shapes used by streaming/provider resolution.
    """
    raw_api_key = entry.get("api_key")
    if raw_api_key is not None:
        api_key_text = str(raw_api_key).strip()
        if api_key_text.startswith("${") and api_key_text.endswith("}") and len(api_key_text) > 3:
            env_name = api_key_text[2:-1]
            resolved = os.getenv(env_name, "").strip()
            if resolved:
                return resolved
            logger.debug(
                "Custom provider %s api_key references %s, but the environment variable is unset or empty",
                provider,
                api_key_text,
            )
        elif api_key_text:
            return api_key_text

    key_env = str(entry.get("key_env") or "").strip()
    if key_env:
        resolved = os.getenv(key_env, "").strip()
        if resolved:
            return resolved

    try:
        from api.config import _lookup_custom_api_key_env

        return _lookup_custom_api_key_env(provider) or ""
    except Exception:
        return ""


def _context_length_config_api_key_for_provider(
    provider: str | None,
    cfg: dict | None,
) -> str:
    """Return a config/env API key usable for context-window metadata lookup."""
    cfg = cfg if isinstance(cfg, dict) else {}
    provider = _canonical_context_provider(provider)

    def _resolve_key(raw_api_key, raw_key_env=None) -> str:
        api_key_text = str(raw_api_key or "").strip()
        if (
            api_key_text.startswith("${")
            and api_key_text.endswith("}")
            and len(api_key_text) > 3
        ):
            resolved = os.getenv(api_key_text[2:-1], "").strip()
            if resolved:
                return resolved
        elif api_key_text:
            return api_key_text
        key_env = str(raw_key_env or "").strip()
        if key_env:
            resolved = os.getenv(key_env, "").strip()
            if resolved:
                return resolved
        return ""

    providers_cfg = cfg.get("providers") or {}
    if isinstance(providers_cfg, dict):
        for provider_key, provider_cfg in providers_cfg.items():
            if not isinstance(provider_cfg, dict):
                continue
            if not _providers_match_for_context(provider_key, provider):
                continue
            api_key = _resolve_key(provider_cfg.get("api_key"), provider_cfg.get("key_env"))
            if api_key:
                return api_key

    model_cfg = cfg.get("model", {})
    if isinstance(model_cfg, dict):
        model_provider = _canonical_context_provider(model_cfg.get("provider"))
        if not provider or _providers_match_for_context(model_provider, provider):
            api_key = _resolve_key(model_cfg.get("api_key"), model_cfg.get("key_env"))
            if api_key:
                return api_key
    return ""


def _context_length_lookup_inputs_for_model(
    model: str | None,
    provider: str | None = None,
    *,
    base_url: str | None = None,
    api_key: str | None = None,
    cfg: dict | None = None,
) -> _ContextLengthLookupInputs:
    """Return the effective metadata resolver inputs for a WebUI model.

    ``agent.model_metadata.get_model_context_length`` understands global
    ``config_context_length`` and custom-provider overrides, but only when the
    matching base URL is supplied. WebUI also owns ``providers.<provider>.models``
    overrides, so normalize those here and keep route/session-save/SSE aligned.
    """
    model_for_lookup = str(model or "").strip()
    if not model_for_lookup:
        return _ContextLengthLookupInputs()

    if cfg is None:
        try:
            from api.config import get_config as _get_config_for_cl

            cfg = _get_config_for_cl()
        except Exception:
            cfg = {}
    cfg = cfg if isinstance(cfg, dict) else {}

    bare_model, explicit_provider = _split_provider_qualified_model(model_for_lookup)
    effective_provider = _canonical_context_provider(provider or explicit_provider)
    effective_base_url = str(base_url or "").strip()

    model_cfg = cfg.get("model", {}) if isinstance(cfg, dict) else {}
    if isinstance(model_cfg, dict):
        if not effective_provider:
            effective_provider = _canonical_context_provider(model_cfg.get("provider"))
        if not effective_base_url:
            # #7535: the global model.base_url may only fill an empty slot when
            # the session provider IS the configured model.provider owner
            # (mirror the ownership predicate used for model_cfg's API key in
            # _context_length_config_api_key_for_provider). A built-in registry
            # provider (empty base_url by design) must keep the slot empty so
            # the registry endpoint resolves instead of another provider's URL.
            #
            # Two shapes cannot own the slot and therefore cannot conflict, so
            # they keep master's backfill: a config that declares no provider
            # at all (the profile-setup path writes model.base_url without one)
            # and the two spellings of the same built-in id (opencode_go ==
            # opencode-go), folded through api.config._canonicalise_provider_id
            # so distinct custom:* slugs stay distinct.
            _model_cfg_provider = _canonical_context_provider(model_cfg.get("provider"))
            _owner_provider = _model_cfg_provider
            _session_provider = effective_provider
            try:
                from api.config import _canonicalise_provider_id as _canon_provider_id

                _owner_provider = _canon_provider_id(_owner_provider) or _owner_provider
                _session_provider = _canon_provider_id(_session_provider) or _session_provider
            except Exception:
                pass
            if (
                not effective_provider
                or not _model_cfg_provider
                or _providers_match_for_context(_owner_provider, _session_provider)
            ):
                effective_base_url = str(model_cfg.get("base_url") or "").strip()

    custom_providers = cfg.get("custom_providers") if isinstance(cfg, dict) else None
    if not isinstance(custom_providers, list):
        custom_providers = None

    provider_context_length = None
    providers_cfg = (cfg.get("providers") or {}) if isinstance(cfg, dict) else {}
    if isinstance(providers_cfg, dict):
        for provider_key, provider_cfg in providers_cfg.items():
            if not isinstance(provider_cfg, dict):
                continue
            if not _providers_match_for_context(provider_key, effective_provider):
                continue
            if not effective_base_url:
                effective_base_url = str(provider_cfg.get("base_url") or "").strip()
            provider_context_length = _models_config_context_length(
                provider_cfg.get("models"),
                bare_model or model_for_lookup,
            )
            break

    custom_context_length = None
    effective_api_key = str(api_key or "").strip()
    if custom_providers:
        target_base = effective_base_url.rstrip("/")
        model_candidates = set(_model_lookup_candidates(bare_model or model_for_lookup))
        for entry in custom_providers:
            if not isinstance(entry, dict):
                continue
            entry_name = str(entry.get("name") or "").strip()
            entry_slug = _custom_provider_slug_for_context(entry_name)
            entry_base = str(entry.get("base_url") or "").strip()
            entry_base_norm = entry_base.rstrip("/")
            provider_matches = bool(
                effective_provider
                and (
                    effective_provider == entry_slug
                    or effective_provider == entry_name.lower()
                    or (effective_provider == "custom" and len(custom_providers) == 1)
                )
            )
            base_matches = bool(target_base and entry_base_norm and target_base == entry_base_norm)
            model_matches = bool(model_candidates.intersection(set(_model_lookup_candidates(entry.get("model")))))
            models_cfg = entry.get("models")
            if isinstance(models_cfg, dict):
                model_matches = model_matches or any(candidate in models_cfg for candidate in model_candidates)
            if not (provider_matches or base_matches or (not effective_provider and model_matches)):
                continue
            if not effective_provider and entry_slug:
                effective_provider = entry_slug
            if not effective_base_url and entry_base:
                effective_base_url = entry_base
            if not effective_api_key:
                effective_api_key = _custom_provider_api_key_for_context(entry, effective_provider or entry_slug)
            custom_context_length = _models_config_context_length(models_cfg, bare_model or model_for_lookup)
            break

    global_context_length = None
    if isinstance(model_cfg, dict):
        cfg_default_model = str(model_cfg.get("default") or "").strip()
        raw_cfg_ctx = model_cfg.get("context_length")
        if raw_cfg_ctx is not None and (
            not cfg_default_model
            or _model_matches_configured_default(
                model_for_lookup,
                cfg_default_model,
                effective_provider,
            )
        ):
            global_context_length = _positive_context_length(raw_cfg_ctx)

    if not effective_api_key:
        effective_api_key = _context_length_config_api_key_for_provider(effective_provider, cfg)

    return _ContextLengthLookupInputs(
        config_context_length=provider_context_length or custom_context_length or global_context_length,
        custom_providers=custom_providers,
        base_url=effective_base_url,
        provider=effective_provider,
        api_key=effective_api_key,
    )



def _should_attach_codex_provider_context(model: str, raw_active_provider: str, catalog: dict) -> bool:
    """Return True when a bare Codex model needs separate provider context.

    OpenAI, OpenAI Codex, Copilot, and OpenRouter can all expose GPT-looking
    bare names. If a session stores only ``gpt-...`` while Codex is active, a
    later provider-list/default-model round trip can lose the user's Codex
    choice. Store the provider separately instead of converting the persisted
    model to ``@openai-codex:model``.
    """
    if raw_active_provider != "openai-codex":
        return False
    if not model.lower().startswith("gpt"):
        return False
    for group in catalog.get("groups") or []:
        if str(group.get("provider_id") or "").strip().lower() != "openai-codex":
            continue
        return any(
            _catalog_model_id_matches(entry.get("id"), model)
            for entry in group.get("models", [])
            if isinstance(entry, dict)
        )
    return False

def _read_profile_model_config(
    session,
    requested_provider: str | None,
) -> tuple[str | None, str | None, dict | None]:
    """Read model.provider, model.default, and the full profile config dict.

    Returns (profile_provider, profile_default_model, profile_config_dict).
    The first two are None when the session has no profile or the profile config
    is unreadable; profile_config_dict is None in the same cases so callers only
    pay for one YAML parse.

    When the session already has an explicit ``requested_provider``, the profile
    ``model.provider`` is not returned (first tuple element is None) so profile
    does not override the session provider. ``profile_default_model`` is still
    returned for suffix repair (#5127) only when the profile's configured
    provider matches ``requested_provider`` after normalization.

    perf(webui/session-load-latency) tier2a: the parse is wrapped in a
    per-process LRU keyed by (profile_name, config_mtime, size). The
    function fires on every chat-open for sessions under a named
    profile (resolve_model=1 path), and the YAML parse alone is
    hundreds of µs to single-digit ms on the Chromebook. Cache TTL
    60s is a backstop in case mtime resolution is poor on a given
    filesystem; under normal edits the mtime changes and invalidates
    immediately.
    """
    if not getattr(session, "profile", None):
        return None, None, None

    try:
        from api.profiles import get_hermes_home_for_profile

        _profile_name = str(session.profile or "")
        _profile_home = get_hermes_home_for_profile(_profile_name)
        _profile_cfg_path = os.path.join(str(_profile_home), "config.yaml")
        if not os.path.isfile(_profile_cfg_path):
            return None, None, None
        _pcfg = _read_profile_config_cached(_profile_name, _profile_cfg_path)
        if _pcfg is None:
            return None, None, None
        _model_cfg = _pcfg.get("model") or {}
        if not isinstance(_model_cfg, dict):
            return None, None, _pcfg
        _provider = (_model_cfg.get("provider") or "").strip() or None
        _default = (_model_cfg.get("default") or "").strip() or None
    except Exception:
        logger.warning(
            "profile provider read failed for %r",
            getattr(session, "profile", None),
            exc_info=True,
        )
        return None, None, None

    _requested = _clean_session_model_provider(requested_provider)
    if _requested:
        _profile_prov = _clean_session_model_provider(_provider)
        if _profile_prov != _requested:
            return None, None, _pcfg
        return None, _default, _pcfg
    return _provider, _default, _pcfg


# perf(webui/session-load-latency) tier2a: process-wide cache for parsed
# profile config.yaml. Key = (profile_name, inode, mtime, size); value = parsed
# dict. inode tracks atomic-rename edits (most editors replace files, giving a
# new inode on Linux). mtime+size auto-invalidates on in-place edits; a 60s TTL
# is the backstop in case of coarse mtime resolution (some network filesystems
# round mtime to whole seconds — the size guard catches a write of equal-length
# bytes within the same second). On cache hit, a full-content comparison catches
# any in-place rewrite that inode+mtime+size missed (Greptile P1, PR#5803
# discussion_r3548477915). Reading and comparing the full file content (~1-10KB)
# is much cheaper than yaml.safe_load(). Reads are guarded by a single Lock to
# keep the hot path simple; the underlying yaml.safe_load is the slow step, not
# the lock, so contention is bounded.
_PROFILE_CONFIG_CACHE: "dict[tuple, tuple[float, str, dict]]" = {}
_PROFILE_CONFIG_CACHE_TTL_SECONDS = 60.0
_PROFILE_CONFIG_CACHE_LOCK = threading.Lock()


def _read_profile_config_cached(profile_name: str, cfg_path: str) -> dict | None:
    """Return parsed profile config, caching by (inode, mtime, size) with
    TTL backstop and full-content verification.

    The full-content comparison reads the current file and compares it to a
    copy stored in the cache entry. This catches any in-place rewrite where
    inode+mtime+size are identical, regardless of where in the file the change
    occurs — unlike a fixed-length prefix comparison, edits to fields after the
    first N characters are always detected. Reading and comparing the full file
    content (~1-10KB for a typical config.yaml) is much cheaper than
    yaml.safe_load().

    NOTE: The cache key uses inode+mtime+size to handle the common cases
    (atomic-rename editors -> new inode; in-place editors -> mtime/size
    change). The full-content comparison is a backstop for the rare case where
    all three collide (e.g., sed -i on a filesystem with coarse mtime
    resolution, writing the same byte count).
    """
    try:
        st = os.stat(cfg_path)
    except OSError:
        return None
    mtime = float(getattr(st, "st_mtime", 0.0) or 0.0)
    size = int(getattr(st, "st_size", 0) or 0)
    inode = int(getattr(st, "st_ino", 0) or 0)
    key = (str(profile_name or ""), inode, mtime, size)
    now = time.monotonic()
    with _PROFILE_CONFIG_CACHE_LOCK:
        cached = _PROFILE_CONFIG_CACHE.get(key)
        if cached is not None:
            cached_at, cached_content, cached_dict = cached
            if (now - cached_at) <= _PROFILE_CONFIG_CACHE_TTL_SECONDS:
                # Full content comparison catches any in-place rewrite where
                # inode+mtime+size are identical but the file content changed.
                # Reading and comparing the full file (~1-10KB) is cheaper than
                # yaml.safe_load(). Unlike a fixed-length prefix, this detects
                # edits anywhere in the file. Greptile P1 (PR#5803).
                _current_content = None
                try:
                    with open(cfg_path, "r", encoding="utf-8") as _f:
                        _current_content = _f.read()
                except Exception:
                    pass
                if _current_content == cached_content:
                    return cached_dict
                # Content changed while key collided — fall through to re-parse
    import yaml
    try:
        with open(cfg_path, encoding="utf-8") as _f:
            content = _f.read()
            parsed = yaml.safe_load(content) or {}
    except Exception:
        return None
    if not isinstance(parsed, dict):
        return None
    with _PROFILE_CONFIG_CACHE_LOCK:
        _PROFILE_CONFIG_CACHE[key] = (now, content, parsed)
        # Cap the cache at 32 entries; profiles are bounded in practice
        # and unbounded growth would be a leak.
        if len(_PROFILE_CONFIG_CACHE) > 32:
            # Drop the oldest entry by insertion order (dict is ordered).
            for old_key in list(_PROFILE_CONFIG_CACHE.keys())[:max(0, len(_PROFILE_CONFIG_CACHE) - 32)]:
                _PROFILE_CONFIG_CACHE.pop(old_key, None)
    return parsed


def _load_profile_config_dict(session) -> dict | None:
    """Load the session profile's config.yaml as a dict, or None."""
    if not getattr(session, "profile", None):
        return None
    try:
        from api.profiles import get_hermes_home_for_profile

        _profile_cfg_path = os.path.join(
            str(get_hermes_home_for_profile(session.profile)),
            "config.yaml",
        )
        if not os.path.isfile(_profile_cfg_path):
            return None
        import yaml

        with open(_profile_cfg_path, encoding="utf-8") as _f:
            _pcfg = yaml.safe_load(_f) or {}
        return _pcfg if isinstance(_pcfg, dict) else None
    except Exception:
        logger.warning(
            "profile config read failed for %r",
            getattr(session, "profile", None),
            exc_info=True,
        )
        return None


def _ordered_custom_provider_model_ids(entry: dict) -> list[str]:
    """Model ids from a custom_providers entry (default model + dict/list models)."""
    ordered: list[str] = []
    _cp_model = str(entry.get("model") or "").strip()
    if _cp_model:
        ordered.append(_cp_model)
    _cp_models = entry.get("models")
    if isinstance(_cp_models, dict):
        for _key in _cp_models.keys():
            if isinstance(_key, str):
                _kid = _key.strip()
                if _kid and _kid not in ordered:
                    ordered.append(_kid)
    elif isinstance(_cp_models, list):
        for _item in _cp_models:
            if isinstance(_item, str):
                _mid = _item.strip()
                if _mid and _mid not in ordered:
                    ordered.append(_mid)
            elif isinstance(_item, dict):
                _mid = str(
                    _item.get("id") or _item.get("model") or _item.get("name") or ""
                ).strip()
                if _mid and _mid not in ordered:
                    ordered.append(_mid)
    return ordered


def _repair_bare_custom_provider_model(
    bare_model: str,
    provider: str | None,
    *,
    config_obj: dict | None = None,
) -> str | None:
    """Re-qualify a bare model ID using the named custom provider's config (#5314).

    Returns the fully namespaced model id when ``bare_model`` matches the suffix
    of a registered id on ``custom_providers``; otherwise None. Model ids are
    scanned in config declaration order (default ``model`` first, then
    ``models`` dict keys or list entries) so repair is deterministic when
    suffixes collide.

    When ``config_obj`` is set (typically the session profile's config.yaml),
    only that object's ``custom_providers`` are scanned. Otherwise uses
    ``get_config()`` for the active global config (not the raw ``cfg`` alias).
    """
    try:
        model = str(bare_model or "").strip()
        prov = _clean_session_model_provider(provider)
        if not model or "/" in model or not prov:
            return None
        if prov != "custom" and not str(prov).startswith("custom:"):
            return None
        from api.config import (
            _custom_provider_entries,
            _custom_provider_slug_from_name,
            get_config,
        )

        if isinstance(config_obj, dict):
            _entries = _custom_provider_entries(config_obj)
        else:
            _cfg = get_config()
            _entries = _custom_provider_entries(
                _cfg if isinstance(_cfg, dict) else None
            )
        prov_norm = str(prov).strip().lower()
        raw_suffix = prov_norm.removeprefix("custom:")
        _matching_cp = None
        for _entry in _entries:
            entry_name = str(_entry.get("name") or "").strip().lower()
            slug = _custom_provider_slug_from_name(_entry.get("name"))
            if not slug:
                continue
            if (
                prov_norm in {entry_name, slug}
                or raw_suffix == slug.removeprefix("custom:")
            ):
                _matching_cp = _entry
                break
        if not _matching_cp:
            return None
        for _id in _ordered_custom_provider_model_ids(_matching_cp):
            if "/" in _id and _id.rsplit("/", 1)[-1] == model:
                return _id
        return None
    except Exception:
        return None


def _moa_fast_path_model_state(model: str) -> tuple[str, str, bool]:
    """Strip an optional ``@moa:``/``moa/`` prefix from an MoA-routed model.

    Split out of ``_resolve_compatible_session_model_state`` so the MoA
    fast-path stays a single-line call in that function body (see
    ``test_issue1855_resolve_model_provider_fast_path.py``, the fast-path/
    catalog-call ordering check scans a bounded window of that function's
    source, and inlining this here previously pushed the catalog call just
    past that window).
    """
    if model.startswith("@moa:"):
        return model.split(":", 1)[1].strip(), "moa", True
    if model.lower().startswith("moa/"):
        return model.split("/", 1)[1].strip(), "moa", True
    return model, "moa", False


def _resolve_compatible_session_model_state(
    model_id: str | None,
    model_provider: str | None = None,
    *,
    profile_provider: str | None = None,
    profile_default_model: str | None = None,
    profile_config: dict | None = None,
    explicit_model_pick: bool = False,
    prefer_cached_catalog: bool = False,
) -> tuple[str, str | None, bool]:
    """Return (effective_model, effective_provider, model_was_normalized).

    Sessions can outlive provider changes. When an older session still points at
    a different provider namespace (for example `gemini/...` after switching the
    agent to OpenAI Codex), reusing that stale model causes chat startup to hit
    the wrong backend and fail. Normalize only obvious cross-provider mismatches.
    When a model has an explicit provider context, keep the model string itself
    in its picker/API shape and carry the provider as separate state.

    Fast path (#1855): when the caller supplies both a model and an explicit
    ``model_provider`` AND the model is not itself ``@provider:model``-qualified,
    we can return the inputs verbatim without calling ``get_available_models()``.
    The slow path below would arrive at the same answer via
    ``if requested_provider and not explicit_provider: return model, requested_provider, False``
    after paying the full catalog-build cost. Avoiding the catalog here keeps
    ``POST /api/chat/start`` snappy even when the model catalog is cold and the
    rebuild has to make network calls (custom OpenAI-compat endpoints,
    OpenRouter ``/models``, LM Studio ``/models``, credential pool refresh),
    those used to wedge the handler for >100s and trigger 502s on default-60s
    reverse proxies, even though the WebUI itself eventually responded.

    ``prefer_cached_catalog=True`` (ours-original) makes the catalog lookup
    non-blocking: it resolves from the warm/disk cache or a network-free
    minimal catalog and NEVER triggers a live per-provider rebuild (the
    Copilot token-exchange HTTPS call that hangs a server-initiated wakeup
    turn, see rebase report §1/§3/model-resolve-hang). Human-initiated
    chat/start leaves this False to keep full live discovery; a session that
    already has a persisted model still resolves correctly because the
    persisted model wins over the catalog and the catalog is only consulted
    for the default-model backstop.
    """
    model = str(model_id or "").strip()
    requested_provider = _clean_session_model_provider(model_provider)
    if model and requested_provider == "moa":
        return _moa_fast_path_model_state(model)
    if model and requested_provider and model.startswith(f"@{requested_provider}:"):
        try:
            from api.config import cfg as _active_cfg

            providers_cfg = _active_cfg.get("providers") if isinstance(_active_cfg, dict) else {}
        except Exception:
            providers_cfg = {}
        if isinstance(providers_cfg, dict) and requested_provider in providers_cfg:
            return model, requested_provider, False
    if model and requested_provider:
        # Only safe when the model itself does not carry an ``@provider:model``
        # qualifier — qualified strings require the catalog to decide whether
        # the qualifier matches the active provider (see slow path below).
        bare_model, explicit_provider = _split_provider_qualified_model(model)
        model_prefix = model.split("/", 1)[0].strip().lower() if "/" in model else ""
        stale_codex_openai_slash_id = (
            requested_provider == "openai-codex"
            and model_prefix == "openai"
        )
        if not explicit_provider and not stale_codex_openai_slash_id:
            _profile_default = str(profile_default_model or "").strip()
            _profile_prov = _clean_session_model_provider(profile_provider)
            _providers_match_for_repair = (
                _profile_prov is None or _profile_prov == requested_provider
            )
            if (
                _profile_default
                and "/" in _profile_default
                and "/" not in model
                and _profile_default.rsplit("/", 1)[-1] == model
                and _providers_match_for_repair
                and (
                    requested_provider == "custom"
                    or str(requested_provider).startswith("custom:")
                )
            ):
                return _profile_default, requested_provider, True

            _repaired_model = _repair_bare_custom_provider_model(
                model,
                requested_provider,
                config_obj=profile_config,
            )
            if _repaired_model:
                return _repaired_model, requested_provider, True

            return model, requested_provider, False

    # Default (human chat/start) path calls get_available_models() with NO
    # kwargs so it stays signature-compatible with the many tests that stub
    # get_available_models as a zero-arg callable. Only the server-side wakeup
    # path (prefer_cached_catalog=True) opts into the cache-only mode. Some
    # tests monkeypatch get_available_models as a zero-arg callable, so probe
    # the (possibly monkeypatched) signature for ``prefer_cache`` rather than
    # catching TypeError — a blanket ``except TypeError`` would also swallow a
    # genuine TypeError raised *inside* get_available_models(prefer_cache=True)
    # and silently fall back to the slow live provider rebuild that
    # prefer_cached_catalog=True is meant to avoid.
    if prefer_cached_catalog:
        import inspect as _inspect

        try:
            _gam_accepts_prefer_cache = (
                "prefer_cache" in _inspect.signature(get_available_models).parameters
            )
        except (TypeError, ValueError):
            # Builtins / C-callables can refuse introspection; assume the
            # zero-arg stub shape in that case.
            _gam_accepts_prefer_cache = False
        if _gam_accepts_prefer_cache:
            catalog = get_available_models(prefer_cache=True)
        else:
            catalog = get_available_models()
    else:
        catalog = get_available_models()
    default_model = str(catalog.get("default_model") or DEFAULT_MODEL or "").strip()

    # Profile-aware resolution: when the caller supplies profile context
    # (not an explicit per-chat override), use the profile's provider and
    # default model as the resolution context instead of the catalog's
    # active_provider / default_model. This preserves the repair path
    # (stale models still get normalized) but normalizes to the profile's
    # default model under the profile's provider rather than the global default.
    bare_model, explicit_provider = _split_provider_qualified_model(model) if model else ("", None)
    if profile_provider and not explicit_provider:
        _profile_provider_normalized = _normalize_provider_id(profile_provider)
        _profile_default = str(profile_default_model or "").strip()
        if not model:
            _fallback = _profile_default or default_model
            return _fallback, profile_provider, bool(_fallback)

        model_prefix = model.split("/", 1)[0].strip().lower() if "/" in model else ""
        model_provider_from_name = _normalize_provider_id(model_prefix) if "/" in model else ""

        model_family = ""
        if "/" not in model:
            model_lower = model.lower()
            for bare_prefix in ("gpt", "claude", "gemini"):
                if model_lower.startswith(bare_prefix):
                    model_family = _normalize_provider_id(bare_prefix)
                    break

        if model_family and model_family != _profile_provider_normalized:
            if explicit_model_pick:
                # User explicitly chose a cross-family model; honor it (#3737)
                return model, profile_provider, False
            _target = _profile_default or default_model
            return _target, profile_provider, True

        if (
            "/" in model
            and str(profile_provider).strip().lower() == "openai-codex"
            and model_provider_from_name == "openai"
        ):
            _target = _profile_default or default_model
            return _target, profile_provider, True

        # Slash-qualified models (e.g. openai/gpt-5.4-mini) are native IDs on
        # OpenRouter and custom providers, not cross-provider artifacts. Only
        # repair when the profile provider actually requires a different family.
        if "/" in model and _profile_provider_normalized in {"openrouter", "custom", ""}:
            return model, profile_provider, False

        if "/" in model and model_provider_from_name and model_provider_from_name != _profile_provider_normalized:
            _target = _profile_default or default_model
            return _target, profile_provider, True

        # Async server-side continuations (for example delegate_task completion
        # re-entry) can arrive here with profile context but without a usable
        # requested_provider, bypassing the fast-path custom-provider repair
        # above. If the profile's configured custom-provider default is a
        # slash-qualified model whose suffix matches the bare session model,
        # repair back to the profile default before the provider call (#5225).
        if (
            "/" not in model
            and _profile_default
            and "/" in _profile_default
            and _profile_default.rsplit("/", 1)[-1] == model
            and (
                _profile_provider_normalized == "custom"
                or str(profile_provider).startswith("custom:")
            )
        ):
            return _profile_default, profile_provider, True

        _repaired_model = _repair_bare_custom_provider_model(
            model,
            profile_provider,
            config_obj=profile_config,
        )
        if _repaired_model:
            return _repaired_model, profile_provider, True

        return model, profile_provider, False

    if not model:
        return default_model, requested_provider, bool(default_model)

    active_provider = _normalize_provider_id(catalog.get("active_provider"))
    # Also keep the raw active_provider slug for cross-provider detection with
    # non-listed providers (ollama-cloud, deepseek, xai, etc.) that _normalize_provider_id
    # returns "" for. If the raw provider is set but normalization returned "", we still
    # want to detect that a session model from a known provider (e.g. openai/gpt-5.4-mini)
    # is stale relative to this unknown active provider. (#1023)
    raw_active_provider = str(catalog.get("active_provider") or "").strip().lower()
    if not active_provider and not raw_active_provider:
        bare_model, explicit_provider = _split_provider_qualified_model(model)
        return model, explicit_provider or requested_provider, False

    bare_for_context, explicit_provider = _split_provider_qualified_model(model)
    if requested_provider and not explicit_provider:
        model_prefix = model.split("/", 1)[0].strip().lower() if "/" in model else ""
        stale_codex_openai_slash_id = (
            raw_active_provider == "openai-codex"
            and requested_provider == "openai-codex"
            and model_prefix == "openai"
        )
        if not stale_codex_openai_slash_id:
            return model, requested_provider, False

    if model.startswith("@") and ":" in model:
        provider_raw = explicit_provider or ""
        provider_normalized = _normalize_provider_id(provider_raw)
        bare_model = bare_for_context.strip()
        if not provider_raw or not bare_model:
            return model, requested_provider, False

        # A fresh, explicit user pick is by definition not a stale artifact, so
        # honor the @provider:model exactly as chosen — never reroute it via the
        # active-provider family repair or the cold-catalog fallback below (a bare
        # id like "gpt-oss-120b" under an OpenAI-active agent would otherwise get
        # pulled to OpenAI by the family-match branch). If the named provider is
        # unreachable the user sees a clear run-time error rather than a silent
        # model swap. Must sit above the family-match repair (#3737 principle).
        if explicit_model_pick:
            return model, provider_raw, False

        raw_provider_ids, normalized_provider_ids = _catalog_provider_id_sets(catalog)
        hint_matches_active = (
            provider_raw == raw_active_provider
            or provider_raw == active_provider
            or (provider_normalized and provider_normalized == active_provider)
        )
        if hint_matches_active:
            # The @provider:model hint explicitly names the active provider, so this
            # selection is intentional — not a stale cross-provider artifact. Return
            # the full @provider:model string unchanged so downstream (resolve_model_provider
            # in config.py) can route through the correct provider. Stripping the prefix
            # here would collapse duplicate model IDs from different providers back to the
            # bare ID, causing the first matching provider to win on the next UI render
            # and the wrong provider to be used for the agent run. (#1253)
            return model, provider_raw, False

        if _catalog_has_provider(
            provider_raw,
            provider_normalized,
            raw_provider_ids,
            normalized_provider_ids,
        ):
            return model, provider_raw, False

        if _model_matches_active_provider_family(bare_model, active_provider):
            provider_context = (
                raw_active_provider
                if _should_attach_codex_provider_context(bare_model, raw_active_provider, catalog)
                else None
            )
            return bare_model, provider_context, True
        # On NON-explicit resolves (2nd+ turn, chat switch — explicit picks already
        # returned above), preserve the selection only when all three hold:
        #
        #   * provider_normalized == "" — a non-first-party provider hint
        #     (ollama-cloud / deepseek / xai / a named custom proxy). First-party
        #     families fall through to the stale-cross-provider repair below.
        #
        #   * the BARE model is not a first-party family id (does not start with
        #     gpt/claude/gemini), i.e. not a misrouted first-party model that a
        #     vanished provider used to host (e.g. "@copilot:claude-opus-4.6").
        #
        #   * the provider is KNOWN or CONFIGURED. This is the load-bearing
        #     distinction: catalog-absence has two causes —
        #       (a) a cold live-discovery provider (ollama-cloud is configured; its
        #           group just isn't in this cached snapshot yet) → preserve, and
        #       (b) a genuinely removed/unknown provider ("@removed:mistral-large"
        #           configured nowhere) → fall through to the default so chat/start
        #           doesn't route to an unreachable provider.
        #     _provider_is_known_or_configured() decides this from the static
        #     provider registry + config state, NOT from the cold catalog snapshot
        #     (re-deriving that live would defeat the prefer_cached_catalog win).
        #
        # DELIBERATE: the registry test treats a KNOWN built-in (deepseek, minimax,
        # ollama-cloud, …) as preservable even when the user has no key configured
        # for it. We accept this on purpose. The only fully-reliable "is this
        # provider authenticated" signal is the live auth store / catalog rebuild —
        # exactly the cost this hot path avoids — and a cheap config/env-only check
        # would mis-classify providers configured via OAuth/auth-store (ollama-cloud
        # among them), re-introducing the original silent-revert bug for them. So a
        # known-but-unconfigured pick is kept; the user gets a clear run-time auth
        # error instead of a silent swap to the default. Pinned by
        # test_at_provider_known_unconfigured_builtin_is_intentionally_preserved.
        #
        # KNOWN LIMITATION: the first-party-family test is a bare-name prefix match
        # (the same approximation _model_matches_active_provider_family uses). A
        # genuine third-party model whose name merely *starts* with gpt/claude/
        # gemini (e.g. "@ollama:gpt4all-mini") is therefore still mis-classified as
        # first-party and reverted on non-explicit paths. A name-based check cannot
        # disambiguate that; the behavior is pinned by
        # test_at_provider_first_party_named_third_party_model_known_limitation.
        _bare_is_first_party_family = any(
            bare_model.lower().startswith(_p) for _p in ("gpt", "claude", "gemini")
        )
        if (
            not provider_normalized
            and not _bare_is_first_party_family
            and _provider_is_known_or_configured(provider_raw)
        ):
            return model, provider_raw, False
        if default_model:
            provider_context = (
                raw_active_provider
                if _should_attach_codex_provider_context(default_model, raw_active_provider, catalog)
                else None
            )
            return default_model, provider_context, True
        return model, provider_raw, False

    slash = model.find("/")
    if slash < 0:
        if explicit_model_pick:
            # User explicitly chose this model; don't second-guess (#3737)
            return model, requested_provider, False
        model_lower = model.lower()
        for bare_prefix in ("gpt", "claude", "gemini"):
            if model_lower.startswith(bare_prefix):
                model_provider = _normalize_provider_id(bare_prefix)
                if model_provider and model_provider != active_provider and default_model:
                    provider_context = (
                        raw_active_provider
                        if _should_attach_codex_provider_context(default_model, raw_active_provider, catalog)
                        else None
                    )
                    return default_model, provider_context, True
                provider_context = (
                    raw_active_provider
                    if _should_attach_codex_provider_context(model, raw_active_provider, catalog)
                    else requested_provider
                )
                return model, provider_context, False
        return model, requested_provider, False

    model_provider = _normalize_provider_id(model[:slash])

    # For custom/openrouter active providers: only skip normalization when the
    # model's namespace prefix is actually routable by a group in the catalog.
    # A user who only has custom_providers configured (active_provider="custom")
    # with a stale session model like "openai/gpt-5.4-mini" would otherwise
    # never get cleaned up, causing "(unavailable)" to appear in the picker.
    if active_provider in {"custom", "openrouter"}:
        # These namespaces are always routable as-is — preserve them.
        if model_provider in {"", "custom", "openrouter"}:
            return model, requested_provider, False
        # Check if any catalog group can actually route this model's prefix.
        groups = catalog.get("groups") or []
        routable_provider_ids = {
            _normalize_provider_id(g.get("provider_id") or "") for g in groups
        }
        # openrouter group can route any provider/model namespace
        has_openrouter_group = any(
            (g.get("provider_id") or "") == "openrouter" for g in groups
        )
        if model_provider in routable_provider_ids or has_openrouter_group:
            return model, requested_provider, False
        # Model prefix is not routable — stale cross-provider reference, clear it.
        if default_model:
            return default_model, requested_provider, True
        return model, requested_provider, False

    # Skip normalization for models on custom/openrouter namespaces — these are
    # user-controlled and should never be silently replaced.
    #
    # OpenAI Codex is intentionally normalized to the OpenAI family above so bare
    # GPT IDs survive provider switches. Slash-qualified OpenAI IDs are different:
    # ``openai/gpt-...`` is the OpenRouter shape for OpenAI models, and
    # resolve_model_provider() routes that through OpenRouter when Codex is the
    # configured provider. Legacy sessions can carry that stale slash ID without
    # a saved model_provider, so repair it to the active Codex default unless the
    # session/request explicitly says it is an OpenRouter selection. (#1734)
    if (
        raw_active_provider == "openai-codex"
        and model_provider == "openai"
        and requested_provider in {None, "openai-codex"}
        and default_model
    ):
        # Persist provider_context = "openai-codex" unconditionally on this
        # repair path so the resolved shape is stable across resolutions
        # (Opus stage-303 SHOULD-FIX: avoid redundant repair-writes per
        # chat-start when the catalog-coverage check fails — e.g. if a
        # future Codex default is itself slash-prefixed). Once we've
        # decided the session belongs to Codex, persist that decision.
        return default_model, raw_active_provider, True

    # Also normalize when the model is from a known provider but the active provider
    # is an unlisted one (e.g. ollama-cloud) — active_provider is "" in that case
    # but raw_active_provider is set. If model_provider doesn't start with the raw
    # active provider name, the session model is stale. (#1023)
    _active_for_compare = active_provider or raw_active_provider
    if model_provider and model_provider not in {"", "custom", "openrouter"} and model_provider != _active_for_compare and default_model:
        return default_model, requested_provider, True
    return model, requested_provider, False


def _resolve_compatible_session_model(model_id: str | None) -> tuple[str, bool]:
    """Return (effective_model, model_was_normalized) for legacy callers."""
    effective_model, _provider, changed = _resolve_compatible_session_model_state(model_id)
    return effective_model, changed


def _normalize_session_model_in_place(session) -> str:
    original_model = getattr(session, "model", None) or ""
    original_provider = _clean_session_model_provider(
        getattr(session, "model_provider", None)
    )
    effective_model, effective_provider, changed = _resolve_compatible_session_model_state(
        original_model or None,
        original_provider,
    )
    provider_changed = effective_provider != original_provider
    # Only persist the correction if the session had an explicit model that needed changing.
    # Sessions with no model stored (empty/None) get the effective default returned without
    # a disk write — no need to rebuild the index for a fill-in-blank operation.
    if original_model and effective_model and (
        (changed and original_model != effective_model) or provider_changed
    ):
        if changed and original_model != effective_model:
            session.model = effective_model
        session.model_provider = effective_provider
        session.save(touch_updated_at=False)
    return effective_model


def _resolve_effective_session_model_for_display(session) -> str:
    """Resolve the model a session should display without mutating persisted state.

    `GET /api/session` should stay side-effect free. If a stale persisted model
    needs normalization for the current provider configuration, return the
    effective model for the response payload only and leave disk state alone.
    """
    original_model = getattr(session, "model", None) or ""
    requested_provider = getattr(session, "model_provider", None)
    _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(session, requested_provider)
    effective_model, _provider, _changed = _resolve_compatible_session_model_state(
        original_model or None,
        requested_provider,
        profile_provider=_pp_provider,
        profile_default_model=_pp_default,
        profile_config=_pp_cfg,
        # GET /api/session is a hot, side-effect-free per-tab/per-poll path.
        # It must never pay the cold live provider-catalog rebuild (a
        # botocore IMDS probe that cannot resolve on a non-AWS / WSL / corp
        # network, plus anthropic/openrouter /models). That rebuild is
        # un-cacheable here (auth.json fingerprint churn) so every cold call
        # cost ~10s and, run concurrently across browser tabs, serialized on
        # the models-cache lock and starved SSE/streaming -> BrokenPipe storm
        # (#multi-tab-streaming-interlock). The persisted session model is
        # authoritative; the catalog is only a default-model backstop, which
        # the network-free minimal catalog already provides.
        prefer_cached_catalog=True,
    )
    return effective_model or original_model

def _resolve_effective_session_model_provider_for_display(session) -> str | None:
    original_model = getattr(session, "model", None) or ""
    requested_provider = getattr(session, "model_provider", None)
    _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(session, requested_provider)
    _model, provider, _changed = _resolve_compatible_session_model_state(
        original_model or None,
        requested_provider,
        profile_provider=_pp_provider,
        profile_default_model=_pp_default,
        profile_config=_pp_cfg,
        # See _resolve_effective_session_model_for_display: same hot
        # side-effect-free GET /api/session path; must not trigger the cold
        # live rebuild. prefer_cached_catalog resolves from warm/disk cache
        # or the network-free minimal catalog.
        prefer_cached_catalog=True,
    )
    return provider


def _resolve_context_length_for_session_model(
    model: str | None,
    provider: str | None = None,
    *,
    base_url: str | None = None,
    api_key: str | None = None,
) -> int:
    """Best-effort current context window for a session model.

    Persisted session context metadata is a snapshot from a prior model call.
    During session hydration/model switching, the current model metadata should
    be allowed to replace that stale snapshot.
    """
    model_for_lookup = str(model or "").strip()
    if not model_for_lookup:
        return 0
    try:
        from agent.model_metadata import get_model_context_length as _get_cl
        from api.config import get_config as _get_config_for_cl

        _cfg_for_cl = _get_config_for_cl()
        _ctx_lookup = _context_length_lookup_inputs_for_model(
            model_for_lookup,
            provider,
            base_url=base_url,
            api_key=api_key,
            cfg=_cfg_for_cl if isinstance(_cfg_for_cl, dict) else {},
        )
        try:
            return _get_cl(
                model_for_lookup,
                _ctx_lookup.base_url,
                api_key=_ctx_lookup.api_key,
                config_context_length=_ctx_lookup.config_context_length,
                provider=_ctx_lookup.provider or provider or "",
                custom_providers=_ctx_lookup.custom_providers,
            ) or 0
        except TypeError:
            # Older hermes-agent builds: legacy 2-arg form.
            return _get_cl(model_for_lookup, _ctx_lookup.base_url) or 0
    except Exception:
        return 0


def _session_context_length_lookup_state(
    model: str | None,
    provider: str | None,
) -> tuple[str, str, str, str]:
    """Return model/provider/base_url/api_key inputs for session context lookup.

    This stays config-based and side-effect-free for GET /api/session. It avoids
    a live provider catalog rebuild while still aligning the reload path with
    the base URL / custom-provider key shape used by streaming saves. (#4248)
    """
    model_for_lookup = str(model or "").strip()
    provider_for_lookup = str(provider or "").strip()
    base_url_for_lookup = ""
    api_key_for_lookup = ""
    if not model_for_lookup:
        return "", provider_for_lookup, "", ""
    try:
        from api.config import resolve_model_provider

        model_for_resolution = model_with_provider_context(model_for_lookup, provider_for_lookup or None)
        resolved_model, resolved_provider, resolved_base_url = resolve_model_provider(model_for_resolution)
        model_for_lookup = str(resolved_model or model_for_lookup).strip()
        provider_for_lookup = str(resolved_provider or provider_for_lookup or "").strip()
        base_url_for_lookup = str(resolved_base_url or "").strip()
    except Exception:
        logger.debug("session context-length lookup state resolution failed", exc_info=True)
    if provider_for_lookup.startswith("custom:"):
        try:
            from api.config import resolve_custom_provider_connection

            custom_key, custom_base = resolve_custom_provider_connection(provider_for_lookup)
            api_key_for_lookup = str(custom_key or "").strip()
            if not base_url_for_lookup:
                base_url_for_lookup = str(custom_base or "").strip()
        except Exception:
            logger.debug("custom provider context-length connection resolution failed", exc_info=True)
    return model_for_lookup, provider_for_lookup, base_url_for_lookup, api_key_for_lookup


def _session_model_identity_matches(
    stored_model: str | None,
    stored_provider: str | None,
    resolved_model: str | None,
    resolved_provider: str | None,
) -> bool:
    stored = str(stored_model or "").strip()
    resolved = str(resolved_model or "").strip()
    if not stored or not resolved:
        return False

    def _split_model_identity(value: str) -> tuple[str, str | None]:
        # Handle BOTH provider-qualified shapes so a slash-prefixed session model
        # (e.g. ``deepseek/deepseek-v4-1m``, OpenRouter-style) compares equal to its
        # resolved bare id. ``_split_provider_qualified_model`` only handles the
        # ``@provider:model`` form; without the slash case a reload of a
        # slash-stored model is wrongly treated as a model change, bypassing the
        # #4248 256k-clobber guard (Codex regression gate, v0.51.x).
        bare, prov = _split_provider_qualified_model(value)
        if prov is None and "/" in value:
            prefix, rest = value.split("/", 1)
            prefix = prefix.strip()
            rest = rest.strip()
            if prefix and rest:
                return rest, prefix
        return bare, prov

    stored_bare, stored_explicit_provider = _split_model_identity(stored)
    resolved_bare, resolved_explicit_provider = _split_model_identity(resolved)
    stored_provider_norm = _canonical_context_provider(stored_explicit_provider or stored_provider)
    resolved_provider_norm = _canonical_context_provider(resolved_explicit_provider or resolved_provider)
    if stored == resolved and stored_provider_norm == resolved_provider_norm:
        return True
    if stored_bare != resolved_bare:
        return False
    if stored_provider_norm and resolved_provider_norm:
        return stored_provider_norm == resolved_provider_norm
    return True


def _should_accept_session_context_length_refresh(
    persisted: int,
    resolved: int,
    *,
    model_changed: bool = False,
) -> bool:
    if not resolved:
        return False
    if not persisted:
        return True
    # #4248: an anonymous reload resolver can still fall through to the agent
    # metadata default fallback. Do not let that lower-confidence 256k value
    # clobber a larger context window persisted by the streaming path. If the
    # effective model changed, though, a 256k result may be the real new model
    # window and should replace the old snapshot.
    return model_changed or not (resolved == 256_000 and persisted > resolved)


def _rescale_threshold_tokens_for_context_window(
    threshold: int,
    old_window: int,
    new_window: int,
) -> int:
    try:
        threshold = int(threshold or 0)
        old_window = int(old_window or 0)
        new_window = int(new_window or 0)
    except (TypeError, ValueError):
        return 0
    if threshold <= 0 or old_window <= 0 or new_window <= 0:
        return 0
    return max(1, int(threshold * new_window / old_window))


def _worktree_default_from_config(profile: str | None) -> bool:
    """Return the agent's config-level ``worktree:`` default for *profile*.

    The agent CLI honors ``worktree: true`` in config.yaml for every session
    it creates (``use_worktree = worktree or w or CLI_CONFIG.get("worktree",
    False)``).  /api/session/new consults this only when the request body has
    no explicit ``worktree`` key, so both entry points to the same repo agree
    on isolation (#6022).  Explicit body values always win.

    Profile-aware on purpose: the WebUI serves multiple profiles from one
    process, and a user with ``worktree: true`` in one profile but not another
    expects per-profile behavior.  ``get_config_for_profile_home`` handles the
    ambient/common case via the mtime-tracked cache and reads a diverging
    profile's config.yaml directly off disk (see #3294).
    """
    try:
        if profile:
            from api.profiles import get_hermes_home_for_profile

            cfg_dict = get_config_for_profile_home(get_hermes_home_for_profile(profile))
        else:
            cfg_dict = get_config_for_profile_home(None)
        # Strict boolean: only a real YAML `true` opts in.  Any other shape
        # ("true", 1, [], {}, null, ...) is malformed for this key and must
        # fall to the safe no-worktree default rather than truthiness-coerce
        # into minting worktrees.
        return (cfg_dict or {}).get("worktree", False) is True
    except Exception:
        # Config resolution must never break session creation.
        logger.warning("failed to read worktree config default", exc_info=True)
        return False


def _session_model_state_from_request(
    model: str | None,
    requested_provider: str | None,
    current_provider: str | None = None,
) -> tuple[str | None, str | None]:
    model_value = str(model).strip() if model is not None else None
    provider = (
        _clean_session_model_provider(requested_provider)
        if requested_provider is not None
        else None
    )
    if model_value:
        _bare, explicit_provider = _split_provider_qualified_model(model_value)
        if explicit_provider:
            provider = explicit_provider
        elif requested_provider is None:
            provider = _clean_session_model_provider(current_provider)
        model_value, provider, _changed = _resolve_compatible_session_model_state(
            model_value,
            provider,
        )
    return model_value, provider


def _lookup_gateway_session_identity(session_id: str) -> dict:
    if not session_id:
        return {}
    metadata = _load_gateway_session_identity_map().get(str(session_id))
    return metadata if isinstance(metadata, dict) else {}


def _lookup_cli_session_metadata(session_id: str, *, all_profiles: bool = False) -> dict:
    if not session_id:
        return {}
    try:
        for row in get_cli_sessions(all_profiles=all_profiles):
            if row.get("session_id") == session_id:
                return row
    except Exception:
        return {}
    return {}


def _session_index_marks_was_webui(sid: str) -> bool:
    """Return True iff ``sid`` is in the WebUI session index as a WebUI- or
    fork-origin row whose sidecar is now gone.

    The WebUI session index (``SESSION_INDEX_FILE``) is the canonical registry
    of sessions the WebUI ever owned. A row there tagged with ``webui`` or
    ``fork`` means the session once had a sidecar that has since been deleted
    (or never materialised on this profile). Returning 404 to the client on
    these ids is what lets the browser self-heal: strip the stale
    ``/session/<id>`` URL and clear localStorage instead of silently
    re-attaching to a now-empty session (#2782).

    Foreign-origin rows (CLI, TUI, Desktop, claude_code, gateway, telegram,
    etc.) — those with explicit non-webui source tags, OR blank sources with
    ``is_cli_session``/``read_only`` markers — are NOT treated as deleted
    WebUI sessions, even when the sidecar is absent.
    """
    if not SESSION_INDEX_FILE.exists():
        return False
    try:
        entries = json.loads(SESSION_INDEX_FILE.read_bytes())
    except Exception:
        return False
    for entry in entries if isinstance(entries, list) else []:
        if entry.get("session_id") != sid:
            continue
        # Classify per source field, not on a collapsed `a or b or c` — a
        # legacy CLI/imported row can carry is_cli_session:true with BLANK
        # source fields, and collapsing-then-defaulting-to-WebUI would wrongly
        # 404 it (it should keep its read-only CLI stub).
        srcs = [
            str(entry.get("source_tag") or "").strip().lower(),
            str(entry.get("raw_source") or "").strip().lower(),
            str(entry.get("session_source") or "").strip().lower(),
        ]
        explicit = [s for s in srcs if s]
        if any(s in ("webui", "fork") for s in explicit):
            # Explicit WebUI-origin (incl. forks, which /api/session/branch
            # stamps session_source="fork") — a deleted sidecar bricks
            # identically. 404.
            return True
        if explicit:
            # Explicit non-WebUI source (cli, telegram, claude_code, ...) —
            # genuine foreign session, keep the existing CLI/read-only stub.
            return False
        # All source fields blank: WebUI-origin UNLESS the row is a legacy
        # CLI/imported session marked only by is_cli_session / read_only.
        is_cli = entry.get("is_cli_session") is True
        is_read_only = bool(entry.get("read_only") or entry.get("is_read_only"))
        return not (is_cli or is_read_only)
    return False


def _session_deleted_tombstone_marks_was_webui(sid: str) -> bool:
    try:
        return sid in _load_webui_deleted_session_tombstone()
    except Exception:
        return False


def _state_db_session_source(sid: str) -> str:
    """Return the lowercased ``sessions.source`` for ``sid`` from state.db.

    Cheap single-row lookup used to distinguish delegated ``subagent`` children
    (which have a recoverable state.db transcript) from genuinely-deleted WebUI
    sessions.  Returns "" on any error / missing row so callers fall back to
    their existing behaviour.
    """
    if not sid or not is_safe_session_id(sid):
        return ""
    try:
        from api.models import _active_state_db_path
        db_path = _active_state_db_path()
        if not db_path or not Path(db_path).exists():
            return ""
        import sqlite3 as _sqlite
        with closing(_sqlite.connect(str(db_path))) as _conn:
            row = _conn.execute(
                "SELECT source FROM sessions WHERE id = ?", (sid,)
            ).fetchone()
    except Exception:
        return ""
    if not row:
        return ""
    return str(row[0] or "").strip().lower()


def _is_subagent_child_session_id(sid: str) -> bool:
    """Return True when ``sid`` is a delegated subagent child in state.db.

    Delegated ``delegate_task`` children are recorded in Hermes state.db with
    ``source='subagent'`` and a ``parent_session_id``. They frequently have no
    WebUI sidecar (they ran server-side), so opening one from the sidebar must
    recover the transcript from state.db rather than 404 as a deleted WebUI
    session (#5307).
    """
    return _state_db_session_source(sid) == "subagent"


def _session_is_subagent_view_only(sid: str) -> bool:
    """Return True when ``sid`` is a delegated subagent child by ANY signal —
    state.db source OR a persisted WebUI sidecar tagged subagent.

    Delegated children are view-only and owned by the delegate runner. Direct
    transcript/metadata mutation routes (delete / truncate / clear / pin /
    rename / move) must refuse them so a stray WebUI action can't delete or
    fork the child's state.db transcript (#5307). This is the shared
    defense-in-depth guard for routes that bypass
    ``_get_or_materialize_session()``.
    """
    if _is_subagent_child_session_id(sid):
        return True
    try:
        s = get_session(sid)
    except Exception:
        return False
    src = (
        str(getattr(s, "source_tag", "") or getattr(s, "raw_source", "")
            or getattr(s, "session_source", "") or "").strip().lower()
    )
    return src == "subagent"


def _is_claimable_cli_source(cli_meta: dict, state_db_source: str = "") -> tuple[bool, str]:
    """Decide whether a foreign-origin session is safe to claim writeable
    in WebUI. Returns ``(claimable, reason_if_not)``.

    Policy mirrors ``/api/session/import_cli``:
    sessions explicitly marked ``read_only`` in their foreign store are
    surfaced as read-only stubs but never materialised as writable
    WebUI sidecars. We extend that with a denylist of foreign-source
    families whose ownership belongs to a non-WebUI process
    (messaging channels, external agents, Claude Code, scheduled
    cron runs, and platformless gateway fallbacks), so a WebUI POST
    cannot accidentally turn them into writable sidecars and violate
    their ownership boundary (#4911 review + the residual
    gateway/unknown gap flagged in the follow-up + the cron-claim
    policy flagged in the Greptile 4/5 review).

    The check is denylist-based: if a source is in any of the
    refused families below, it is non-claimable. Everything else
    (CLI, TUI, Desktop, plus future local agent sources) is allowed.
    TUI/Desktop sessions whose cli_meta is empty (they don't appear
    in ``get_cli_sessions()`` due to the CLI cap) fall through to
    ``state_db_source``; state.db has a ``source`` column with values
    like ``tui``, ``desktop``, ``cli``, ``cron``, ``claude_code``,
    ``messaging``, ``external_agent``, ``gateway`` (platform-tagged
    gateways land in ``_MESSAGING_RAW_SOURCES`` and are caught by
    the messaging check above; bare ``"gateway"`` / ``"unknown"``
    literals are caught here).
    """
    cm = cli_meta or {}
    if bool(cm.get("read_only")):
        return False, "explicit_readonly"
    session_source = (cm.get("session_source") or "").strip().lower()
    if session_source in {"messaging", "external_agent"}:
        return False, f"session_source={session_source}"
    # Track cli_meta-sourced and state.db-sourced source_tag values
    # separately so the diagnostic reason string never mislabels a
    # cli_meta-sourced denial as a state.db-sourced one (Greptile P2,
    # #4911 follow-up).  The reason is currently discarded by the
    # caller, but it is exported in the return tuple and may surface
    # in a future log / user-visible diagnostic.
    cli_meta_source_tag = (cm.get("source_tag") or cm.get("raw_source") or "").strip().lower()
    if cli_meta_source_tag in {"claude_code", "cron", "external_agent",
                                "gateway", "messaging", "subagent", "unknown"}:
        # gateway/unknown are the platformless gateway fallbacks
        # (gateway/run.py, gateway/slash_commands.py) — they own the
        # conversation in the gateway, not in WebUI.
        # cron sessions are scheduled and owned by the cron runner
        # process; claiming them into a writable WebUI sidecar would
        # let a stray POST break the next scheduled run.
        # messaging / external_agent can be supplied by the foreign
        # store directly in source_tag (in addition to the
        # session_source / messaging-record checks above), and they
        # need the same provenance-correct refusal.
        return False, f"cli_meta_source={cli_meta_source_tag}"
    if _is_messaging_session_record(cm):
        return False, "messaging_record"
    # Empty cli_meta is the common case for TUI/Desktop; fall through
    # to state.db's source column.  Refuse known-foreign state.db sources.
    if not cli_meta_source_tag and state_db_source:
        state_db_source_tag = state_db_source.strip().lower()
        if state_db_source_tag in {"claude_code", "cron", "messaging",
                                    "external_agent", "gateway", "subagent", "unknown"}:
            return False, f"state_db_source={state_db_source_tag}"
    return True, ""


def _claim_or_synthesize_cli_session(sid: str, cli_meta: dict = None):
    """Resolve a session_id that has no WebUI sidecar.

    Returns ``(session_or_None, reason)``. Reasons:

      ``'materialized'``
        A state.db row with messages exists AND the foreign source is
        claimable per :func:`_is_claimable_cli_source` (CLI / TUI /
        Desktop, no explicit read_only, not a messaging / claude_code
        session). ``session`` is a fully populated
        :class:`api.models.Session` ready for writeable use; the caller
        MUST call ``session.save()`` to persist a WebUI-owned sidecar
        before the first write.  The Session carries the source-tag
        metadata from the CLI/state.db lookup (``is_cli_session=True``,
        ``read_only=False``) so the sidebar still renders the original
        source badge.

      ``'not_claimable'``
        The sid has recoverable state.db messages but the foreign
        source is owned by a non-WebUI process (messaging channel,
        claude_code, external_agent, or explicit read_only). ``session``
        is still returned, but with ``read_only=True`` preserved and
        the foreign source tag intact, so the GET stub continues to
        render the original badge and the read-only banner. The POST
        path must return 403 (not 404) so the user sees a clear
        refusal instead of the empty-state self-heal that the 404
        handler triggers (#4911 review).

      ``'was_webui'``
        The sid is in the WebUI session index as a webui/fork origin row but
        its sidecar is gone.  Callers MUST return 404 so the browser clears
        its stale ``/session/<id>`` URL and localStorage instead of silently
        re-attaching to a now-empty session (#2782).

      ``'no_foreign_state'``
        The sid has no WebUI sidecar AND no recoverable messages in
        state.db.  Callers MUST return 404.

      ``'invalid_sid'``
        ``sid`` failed :func:`is_safe_session_id`.  Callers MUST return 404.

    ``cli_meta`` is an optional pass-through.  Callers that already
    computed ``_lookup_cli_session_metadata(sid)`` (e.g. the GET path
    building a sidebar dict) can pass it in to avoid the redundant
    lookup; callers without it (POST path, tests) pass nothing and the
    helper does the lookup itself.

    Closing the GET-vs-POST asymmetry for foreign-origin sessions: GET
    ``/api/session`` and POST ``/api/chat/start`` both call this helper
    on a missing-sidecar KeyError, so a TUI/Desktop/CLI session can be
    loaded read-only AND continued writeable from the WebUI.
    """
    def build_workspace(sid, cli_meta):
        """Coalesce workspace with sane fallbacks so _start_run doesn't
        trip on a missing field. state.db's cwd is the canonical workspace for
        agent sessions; CLI metadata is the fallback (handles Telegram/etc).
        """
        workspace = (cli_meta or {}).get("workspace") or (cli_meta or {}).get("cwd")
        if not workspace:
            try:
                from api.workspace import get_last_workspace
                workspace = get_last_workspace()
            except Exception:
                workspace = None
        if not workspace:
            try:
                from api.models import DEFAULT_WORKSPACE
                workspace = DEFAULT_WORKSPACE
            except Exception:
                workspace = "/"
        return workspace

    def build_session(sid, cli_meta, msgs, read_only_flag, is_cli_flag=True):
        return Session(
            session_id=sid,
            title=(cli_meta or {}).get("title") or "CLI Session",
            workspace=build_workspace(sid, cli_meta),
            model=(cli_meta or {}).get("model") or "unknown",
            model_provider=(cli_meta or {}).get("model_provider"),
            messages=msgs,
            created_at=(cli_meta or {}).get("created_at") or 0,
            updated_at=(cli_meta or {}).get("updated_at") or 0,
            profile=(cli_meta or {}).get("profile"),
            # ``is_cli_flag`` is True for genuine CLI/TUI/Desktop sessions so the
            # sidebar renders the source badge and the client's external-session
            # gating applies. It is False for delegated subagent children (#5307):
            # they are recovered read-only and must NOT be CLI-classified, or they
            # would pass the frontend ``_isExternalSession`` poll-skip /
            # active-refresh gates that #3603 keeps narrow.
            is_cli_session=is_cli_flag,
            source_tag=(cli_meta or {}).get("source_tag"),
            raw_source=(cli_meta or {}).get("raw_source"),
            session_source=(cli_meta or {}).get("session_source"),
            source_label=(cli_meta or {}).get("source_label"),
            # ``read_only_flag`` is True for not_claimable sources (foreign
            # store marked them read-only / messaging / claude_code) and
            # False for genuine CLI / TUI / Desktop sessions.  Only the
            # POST claim path with a verified-claimable source can
            # actually write; the GET stub always reflects whatever the
            # helper returns so the read-only banner stays accurate.
            read_only=read_only_flag,
        )

    if not is_safe_session_id(sid):
        return None, "invalid_sid"
    if (
        (
            _session_index_marks_was_webui(sid)
            or (
                _session_deleted_tombstone_marks_was_webui(sid)
                and _state_db_session_source(sid) in ("", "webui", "fork")
            )
        )
        and not _is_subagent_child_session_id(sid)
    ):
        # A delegated subagent child (source='subagent' in state.db) can be
        # registered in the WebUI index as a webui/fork/blank-source row (it
        # shares the parent's lineage) yet have no WebUI sidecar of its own.
        # Those must recover their transcript from state.db below rather than
        # 404 as a genuinely-deleted WebUI session (#5307). Every other
        # index-marked-WebUI id keeps the #2782 self-heal 404 contract.
        #
        # The durable delete tombstone only 404s a row that is WebUI-owned
        # (source webui/fork, or blank = a stale URL with no surviving row).
        # A foreign-source row (messaging/cli/tui/desktop) that happens to
        # carry a tombstone — e.g. a WebUI delete of an imported session whose
        # state.db row the external writer later re-created — must still
        # materialize its transcript, never be self-healed to a 404 (#5504).
        return None, "was_webui"
    if cli_meta is None:
        cli_meta = _lookup_cli_session_metadata(sid) or {}
    msgs = get_cli_session_messages(sid)
    if not msgs:
        return None, "no_foreign_state"
    # TUI/Desktop sessions often have empty cli_meta (they don't appear in
    # get_cli_sessions() because of the cap).  Fall back to the state.db
    # ``source`` column to make the claim-eligibility check robust and to
    # populate the Session's source-tag metadata so the sidebar still
    # renders the correct badge for these sessions.
    state_db_source = ""
    state_db_row = None
    try:
        from api.models import _active_state_db_path
        db_path = _active_state_db_path()
        if db_path and Path(db_path).exists():
            import sqlite3 as _sqlite
            with closing(_sqlite.connect(str(db_path))) as _conn:
                _conn.row_factory = _sqlite.Row
                _row = _conn.execute(
                    "SELECT source, title, model, cwd, started_at, ended_at "
                    "FROM sessions WHERE id = ?", (sid,)
                ).fetchone()
                if _row is not None:
                    state_db_row = dict(_row)
                    state_db_source = str(_row["source"] or "").strip().lower()
    except Exception:
        state_db_source = ""
    # Populate source metadata from state.db when cli_meta is empty so the
    # synthesized Session carries the right source_tag/source_label.  Only
    # fill fields that are actually missing from cli_meta; the foreign store
    # always wins when both are present.
    #
    # No-mutation contract (Greptile #4911 follow-up): the GET path passes
    # a pre-computed cli_meta dict and expects it to be unchanged after
    # this helper returns.  We use a single copy-on-write rebind at the
    # top of the block and then plain subscript assignment so the
    # caller's dict is never touched in place.
    if state_db_row:
        cli_meta = dict(cli_meta or {})
        if not cli_meta.get("source_tag") and state_db_source:
            cli_meta["source_tag"] = state_db_source
        if not cli_meta.get("raw_source") and state_db_source:
            cli_meta["raw_source"] = state_db_source
        if not cli_meta.get("title") and state_db_row.get("title"):
            cli_meta["title"] = state_db_row["title"]
        if not cli_meta.get("model") and state_db_row.get("model"):
            cli_meta["model"] = state_db_row["model"]
        if not cli_meta.get("workspace") and state_db_row.get("cwd"):
            cli_meta["workspace"] = state_db_row["cwd"]
        # Map state.db timestamps to created_at/updated_at on the
        # synthesized Session.  Without this, the first POST writes
        # epoch (0) timestamps into the permanent sidecar and the
        # sidebar sorts/dates the session as "Jan 1 1970" (Greptile
        # #4911 follow-up, P1).  created_at always comes from
        # started_at; Session.save() never touches created_at (it's
        # not in the metadata touch list), so the mapping is
        # load-bearing on both the GET stub and the POST claim path.
        # updated_at prefers ended_at (last activity) and falls back
        # to started_at — note that on the POST claim path
        # Session.save() defaults to touch_updated_at=True and stamps
        # updated_at to wall-clock now, so this value is only the
        # GET-stub display value; the claimed sidecar's updated_at
        # reflects the moment of claim (the desired "just now" UX).
        if not cli_meta.get("created_at") and state_db_row.get("started_at"):
            cli_meta["created_at"] = state_db_row["started_at"]
        if not cli_meta.get("updated_at"):
            _ended = state_db_row.get("ended_at")
            _started = state_db_row.get("started_at")
            if _ended or _started:
                cli_meta["updated_at"] = _ended or _started
    claimable, _reason = _is_claimable_cli_source(cli_meta, state_db_source)
    if not claimable:
        # The session is real and viewable, but the foreign source forbids
        # the WebUI from taking write ownership.  Build the Session with
        # readonly=True so the GET stub keeps rendering the original
        # read-only badge, and return 'not_claimable' so the POST path
        # 403s instead of bare-404ing.
        #
        # Delegated subagent children (#5307) additionally must NOT be
        # CLI-classified: they are recovered read-only for viewing, but
        # is_cli_session=True would let them pass the frontend
        # _isExternalSession poll-skip / active-refresh gates that #3603
        # keeps narrow. Every other non-claimable foreign source keeps the
        # CLI classification so its source badge renders.
        _sa_child = _is_subagent_child_session_id(sid)
        return (
            build_session(sid, cli_meta, msgs, read_only_flag=True,
                          is_cli_flag=not _sa_child),
            "not_claimable",
        )
    return build_session(sid, cli_meta, msgs, read_only_flag=False), "materialized"


def _request_wants_all_profiles_import(body) -> bool:
    if not isinstance(body, dict):
        return False
    value = body.get("all_profiles")
    if isinstance(value, str):
        return value.strip().lower() in {"1", "true", "yes", "on"}
    return bool(value)


def _normalize_import_profile_value(value):
    profile = str(value or "").strip()
    if not profile:
        return None
    try:
        from api.profiles import _PROFILE_ID_RE
        if profile != "default" and not _PROFILE_ID_RE.fullmatch(profile):
            return ""
    except Exception:
        pass
    return profile


def _load_branch_source_or_refuse(handler, sid: str):
    if _session_is_subagent_view_only(sid):
        bad(handler, "Subagent sessions are view-only and cannot be branched from WebUI", 400)
        return None
    try:
        source = get_session(sid)
    except KeyError:
        _foreign_session, _reason = _claim_or_synthesize_cli_session(sid)
        _source_kind = str((getattr(_foreign_session, "source_tag", None) or getattr(_foreign_session, "raw_source", None) or getattr(_foreign_session, "source", None) or "")).strip().lower() if _foreign_session is not None else ""
        if _reason == "not_claimable" and _foreign_session is not None and _source_kind == "cron":
            _foreign_session._branch_source_readonly = True; return _foreign_session
        if _reason == "not_claimable": bad(handler, "Read-only sessions cannot be branched from WebUI", 403); return None
        bad(handler, "Session not found", 404)
        return None
    # A PERSISTED (stored) session can also be read-only (e.g. a cron-owned or
    # messaging-sourced sidecar). Apply the SAME read-only branch gate as the
    # synthesized path: allow forking only a canonical-cron read-only source
    # (server-authoritative source kind, not the id prefix), marking it so the fork
    # never .save()s the read-only source; refuse every other read-only source.
    if bool(getattr(source, "read_only", False)):
        _source_kind = str((getattr(source, "source_tag", None) or getattr(source, "raw_source", None) or getattr(source, "source", None) or "")).strip().lower()
        if _source_kind == "cron":
            source._branch_source_readonly = True
            return source
        bad(handler, "Read-only sessions cannot be branched from WebUI", 403)
        return None
    return source


def _resolve_cli_import_metadata(session_id: str, *, requested_profile=None, allow_all_profiles: bool = False) -> dict:
    cli_meta = _lookup_cli_session_metadata(session_id)
    if cli_meta and (not requested_profile or _profiles_match(cli_meta.get("profile"), requested_profile)):
        return cli_meta
    if not allow_all_profiles:
        return {}
    cli_meta = _lookup_cli_session_metadata(session_id, all_profiles=True)
    if cli_meta and requested_profile and not _profiles_match(cli_meta.get("profile"), requested_profile):
        return {}
    return cli_meta or {}


def _messaging_session_identity(session: dict, raw_source: str) -> str:
    sid = _safe_first(session.get("session_id"))
    if sid and _is_pre_compression_continuation_row(session):
        return f"{raw_source}|session_id:{sid}"

    metadata = _lookup_gateway_session_identity(session.get("session_id"))
    session_key = _safe_first(
        metadata.get("session_key"),
        session.get("session_key"),
        session.get("gateway_session_key"),
    )
    if session_key:
        return f"{raw_source}|session_key:{session_key}"

    chat_id = _safe_first(
        metadata.get("chat_id"),
        session.get("chat_id"),
        session.get("origin_chat_id"),
    )
    thread_id = _safe_first(metadata.get("thread_id"), session.get("thread_id"))
    chat_type = _safe_first(metadata.get("chat_type"), session.get("chat_type"))
    user_id = _safe_first(
        metadata.get("user_id"),
        session.get("user_id"),
        session.get("origin_user_id"),
    )

    identity_parts = []
    if chat_type:
        identity_parts.append(f"chat_type:{chat_type}")
    if chat_id:
        identity_parts.append(f"chat_id:{chat_id}")
    if thread_id:
        identity_parts.append(f"thread_id:{thread_id}")
    if user_id:
        identity_parts.append(f"user_id:{user_id}")

    if identity_parts:
        return f"{raw_source}|" + "|".join(identity_parts)

    return raw_source


def _is_pre_compression_snapshot_id(session_id: str) -> bool:
    sid = _safe_first(session_id)
    if not sid or not all(c in "0123456789abcdefghijklmnopqrstuvwxyz_" for c in sid):
        return False
    try:
        path = SESSION_DIR / f"{sid}.json"
        if not path.exists():
            return False
        data = json.loads(path.read_text(encoding="utf-8"))
        return bool(data.get("pre_compression_snapshot"))
    except Exception:
        return False


def _is_pre_compression_continuation_row(session: dict) -> bool:
    parent_sid = _safe_first(session.get("parent_session_id"))
    return bool(parent_sid and _is_pre_compression_snapshot_id(parent_sid))


def _session_messaging_raw_source(session: dict) -> str:
    raw = _safe_first(
        session.get("raw_source"),
        session.get("source_tag"),
        session.get("source"),
        session.get("platform"),
    )
    if not raw:
        raw = session.get("source_label") or "messaging"
    return _normalize_messaging_source(raw)


def _has_durable_messaging_identity(session: dict) -> bool:
    metadata = _lookup_gateway_session_identity(session.get("session_id"))
    return bool(_safe_first(
        metadata.get("session_key"),
        session.get("session_key"),
        session.get("gateway_session_key"),
        metadata.get("chat_id"),
        session.get("chat_id"),
        session.get("origin_chat_id"),
        metadata.get("thread_id"),
        session.get("thread_id"),
    ))


def _numeric_count(value) -> int:
    try:
        return int(float(_safe_first(value, 0) or 0))
    except (TypeError, ValueError):
        return 0


def _should_hide_stale_messaging_session(
    session: dict,
    active_gateway_session_ids: set[str],
    active_gateway_sources: set[str],
) -> bool:
    """Hide stale Gateway-owned internal rows after an external chat moved on.

    Hermes Gateway keeps the external conversation identity in sessions.json.
    Compression/session-reset can leave old Agent state.db rows behind; those
    rows are implementation segments, not distinct conversations users chose.
    Only apply this aggressive hiding when Gateway is currently advertising an
    active session for the same messaging source. Without that source-of-truth
    file we keep the old fallback behavior.
    """
    raw_source = _session_messaging_raw_source(session)
    if not _is_known_messaging_source(raw_source):
        return False
    if not active_gateway_session_ids or raw_source not in active_gateway_sources:
        return False

    sid = _safe_first(session.get("session_id"))
    if sid and sid in active_gateway_session_ids:
        return False

    if _safe_first(session.get("end_reason")) in _STALE_MESSAGING_END_REASONS:
        return True

    if not _has_durable_messaging_identity(session):
        if _is_pre_compression_continuation_row(session):
            return False
        parent_sid = _safe_first(session.get("parent_session_id"))
        if parent_sid and parent_sid in active_gateway_session_ids:
            return True
        return True

    if session.get("parent_session_id") and not _is_pre_compression_continuation_row(session):
        return True

    message_count = _numeric_count(session.get("message_count"))
    actual_count = _numeric_count(session.get("actual_message_count"))
    if message_count <= 0 and actual_count <= 0:
        return True

    return False


def _is_messaging_session_record(session) -> bool:
    """Return true for sessions backed by external messaging channels."""
    if not session:
        return False
    if (
        (getattr(session, "session_source", None) if not isinstance(session, dict) else session.get("session_source")) == "messaging"
    ):
        return True
    raw = _safe_first(
        getattr(session, "raw_source", None) if not isinstance(session, dict) else session.get("raw_source"),
        getattr(session, "source_tag", None) if not isinstance(session, dict) else session.get("source_tag"),
        getattr(session, "source", None) if not isinstance(session, dict) else session.get("source"),
        session.get("source_label") if isinstance(session, dict) else None,
    )
    return _is_known_messaging_source(raw)


def _messages_include_tool_metadata(messages) -> bool:
    """Return true when returned messages can reconstruct their own tool cards."""
    if not isinstance(messages, list):
        return False
    for msg in messages:
        if not isinstance(msg, dict) or msg.get("role") != "assistant":
            continue
        if isinstance(msg.get("tool_calls"), list) and msg.get("tool_calls"):
            return True
        content = msg.get("content")
        if isinstance(content, list) and any(
            isinstance(part, dict) and part.get("type") == "tool_use"
            for part in content
        ):
            return True
    return False


def _tool_calls_for_message_window(tool_calls, start_idx: int, message_count: int) -> list:
    """Keep session-level tool calls that point into a returned message window.

    ``assistant_msg_idx`` is stored in the full transcript coordinate space, but
    the frontend renders the returned ``messages`` array from index 0. Rebase the
    index into the returned window so legacy session-level tool cards still
    anchor to their visible assistant turn after paginated loads.
    """
    if not isinstance(tool_calls, list) or message_count <= 0:
        return []
    end_idx = start_idx + message_count
    filtered = []
    for tool_call in tool_calls:
        if not isinstance(tool_call, dict):
            continue
        assistant_idx = tool_call.get("assistant_msg_idx")
        if isinstance(assistant_idx, bool) or not isinstance(assistant_idx, int):
            continue
        if start_idx <= assistant_idx < end_idx:
            rebased = dict(tool_call)
            rebased["assistant_msg_idx"] = assistant_idx - start_idx
            filtered.append(rebased)
    return filtered


def _message_counts_as_renderable_for_window(message) -> bool:
    """Return true when a paginated window should include this transcript row.

    Tool result rows are rendered through their assistant anchor or hidden as raw
    tool output. Empty partial activity rows can be preserved after cancellation
    to keep thinking/tool details inspectable, but they are not reply text. A
    tail page containing only transient metadata makes the frontend open to
    collapsed activity while newer real replies sit behind "load older messages".
    """
    if not isinstance(message, dict):
        return False
    if _is_empty_partial_activity_message(message):
        return False
    role = str(message.get("role") or "").strip().lower()
    return bool(role and role != "tool")


def _tool_call_ids_in_messages(messages) -> set:
    """Collect tool-call IDs declared on renderable rows (assistant tool_calls /
    partial tool_calls / Anthropic tool_use content blocks) so trailing
    tool-result rows can be matched back to a call present in the window."""
    ids = set()
    for msg in messages or []:
        if not isinstance(msg, dict):
            continue
        for key in ("tool_calls", "_partial_tool_calls"):
            for call in msg.get(key) or []:
                if isinstance(call, dict):
                    cid = call.get("id") or call.get("tool_call_id")
                    if cid:
                        ids.add(str(cid))
        content = msg.get("content")
        if isinstance(content, list):
            for part in content:
                if isinstance(part, dict) and part.get("type") == "tool_use":
                    cid = part.get("id")
                    if cid:
                        ids.add(str(cid))
    return ids


def _tool_result_matches_call_ids(message, call_ids) -> bool:
    """Return True if a role:tool row's tool_call_id/tool_use_id is in ``call_ids``."""
    if not call_ids or not isinstance(message, dict):
        return False
    if str(message.get("role") or "").lower() != "tool":
        return False
    tid = message.get("tool_call_id") or message.get("tool_use_id") or ""
    return bool(tid) and str(tid) in call_ids


def _message_window_for_display(messages, msg_limit=None, msg_before=None, expand_renderable=False) -> tuple[list, int]:
    """Return a paginated message window plus its offset in ``messages``.

    ``msg_limit`` is a visible transcript limit, not a raw storage-row cap.
    Tool result rows are hidden or folded into assistant tool cards, so they
    should not consume the user's "load N messages" budget. Return the smallest
    suffix containing the last ``msg_limit`` renderable user/assistant rows, plus
    any intervening tool rows needed for card snippets.

    ``expand_renderable`` is accepted for compatibility with older frontend
    callers. Visible-row expansion is now the default for every limited window.
    """
    _ = expand_renderable
    messages = list(messages or [])
    if msg_before is not None:
        before_idx = max(0, min(int(msg_before), len(messages)))
    else:
        before_idx = len(messages)
    source = messages[:before_idx]
    if not source:
        return [], 0
    if not msg_limit:
        return source, 0
    limit = max(1, int(msg_limit))
    end_idx = len(source)
    last_renderable_idx = None
    for idx in range(end_idx - 1, -1, -1):
        if _message_counts_as_renderable_for_window(source[idx]):
            last_renderable_idx = idx
            break
    if last_renderable_idx is None:
        start_idx = max(0, end_idx - limit)
        return source[start_idx:end_idx], start_idx
    # Keep the last renderable row, plus any immediately-following tool-result
    # rows whose tool_call_id matches a tool-call on a renderable row already in
    # the window. The renderer rebuilds tool cards (CLI-origin / empty
    # S.toolCalls path) from role:"tool" rows indexed by tool_call_id
    # (static/ui.js resultsByTid), so dropping the result row that follows the
    # newest assistant tool-call would leave that card without its snippet.
    # Orphan trailing tool-only rows (no matching call in the window) are still
    # skipped, preserving the visible-row budget. (#4070 ship-review)
    end_idx = last_renderable_idx + 1
    window_tool_call_ids = _tool_call_ids_in_messages(source[: last_renderable_idx + 1])
    while end_idx < len(source) and not _message_counts_as_renderable_for_window(
        source[end_idx]
    ):
        if _tool_result_matches_call_ids(source[end_idx], window_tool_call_ids):
            end_idx += 1
        else:
            break
    start_idx = 0
    renderable_count = 0
    for idx in range(last_renderable_idx, -1, -1):
        if not _message_counts_as_renderable_for_window(source[idx]):
            continue
        renderable_count += 1
        if renderable_count >= limit:
            start_idx = idx
            break
    window = source[start_idx:end_idx]
    return window, start_idx


_LIMITED_TOOL_CONTENT_MAX_CHARS = 4096
# Server-side ceiling on the ?msg_limit= tail-window size. A client could
# otherwise request msg_limit=1000000 and force the server to assemble and
# serialize an unbounded message payload (the frontend's own pagination grows
# by ~30 at a time, with one outline-jump path asking for 9999). The ceiling is
# generous — far above any legitimate visible-row window — so real pagination is
# unaffected; it only caps the pathological/oversized request. When the request
# exceeds the ceiling the response is silently clamped and _messages_truncated
# is set (the existing truncation signal already covers "more rows exist").
_MAX_MSG_LIMIT = 500


def _parse_msg_limit(raw):
    """Parse and clamp the ``?msg_limit=`` query value.

    Returns a positive int clamped to ``[1, _MAX_MSG_LIMIT]``, or ``None`` when
    the value is absent/empty/malformed.  ``?msg_limit=all`` also returns
    ``None`` — an explicit escape hatch for the full-transcript paths the
    frontend genuinely needs; the handler distinguishes it from a bare request
    via :func:`_resolve_effective_msg_limit`.  Extracted from the handler so
    the clamp expression has direct test coverage.
    """
    if not raw:
        return None
    try:
        value = int(raw)
    except (TypeError, ValueError):
        return None
    return max(1, min(value, _MAX_MSG_LIMIT))


def _resolve_effective_msg_limit(raw_limit):
    """Resolve the effective ``msg_limit`` for ``GET /api/session``.

    Returns ``(effective_limit, explicit_all)``.

    - numeric ``?msg_limit=N`` → clamped int (existing pagination).
    - ``?msg_limit=all`` → ``(None, True)``: explicit full-transcript escape
      hatch.  Frontend paths that address rows by absolute transcript index
      (outline jump, jump-to-start) genuinely need everything; they pass
      ``all`` instead of relying on the bare no-limit shape.
    - any other bare shape (no limit) → ``(None, False)``: the historical
      full-transcript contract is preserved (contract tests pin tool-row
      preservation and the runtime-journal snapshot on this shape), so the
      bounded-window fix is enforced at the frontend call sites instead.
    """
    explicit_all = str(raw_limit or "").strip().lower() == "all"
    limit = _parse_msg_limit(raw_limit)
    return limit, explicit_all


# If a sidecar JSON file exceeds this threshold, the display-path tail
# optimization fires regardless of message count.  Sessions with few messages
# but large tool outputs (multi-MB JSON) should not force a full-scan merge.
_SIDECAR_BYTE_TAIL_THRESHOLD = 500_000  # 500 KB
# Defensive row backstop for the GET /api/session display path's state.db read.
# This is NOT a semantic window (the display window counts visible rows
# post-reconciliation via _message_window_for_display); it is a safety net so a
# pathological/huge state.db cannot materialize unbounded rows into memory on the
# display path. Legitimate sessions stay far below this; the compressed-session
# case where _state_db_since_timestamp_for_limited_display bails (and would
# otherwise full-scan) is the main beneficiary. Generous on purpose: no real
# conversation approaches it, and the existing since_timestamp optimization
# already handles the common tail-load case. The full-history model-context
# callers (reconciliation, new-turn context) do NOT use this cap.
_STATE_DB_DISPLAY_ROW_BACKSTOP = 50000


def _state_db_backstop_limit_for_display(session, msg_before) -> int | None:
    """Return the row backstop to apply to the display path's state.db read, or
    ``None`` for an uncapped (full-history) read.

    The backstop is a defensive net against a pathological/huge state.db, NOT a
    semantic window. It is applied ONLY on provably-safe reads where no
    ``truncation_boundary`` prefix is required for the merge:
    ``merge_session_messages_append_only`` needs the rows at/around the session's
    ``truncation_boundary`` to reconcile correctly, and a newest-N-only SQL cap
    would drop those boundary rows for a >N-row session and corrupt the merge
    (silently losing the preserved prefix). So this mirrors the same conditions
    ``_state_db_since_timestamp_for_limited_display`` uses to decide a read is
    boundary-free: not ``msg_before`` paging, and no ``truncation_watermark`` /
    ``truncation_boundary``. Extracted for direct test coverage.
    """
    has_boundary_prefix = (
        msg_before is not None
        or getattr(session, "truncation_watermark", None) not in (None, "")
        or getattr(session, "truncation_boundary", None) not in (None, "")
    )
    return None if has_boundary_prefix else _STATE_DB_DISPLAY_ROW_BACKSTOP


_LIMITED_TOOL_CONTENT_NOTICE = (
    "\n\n[Tool output truncated in paginated session response; "
    "load the full transcript to inspect the complete result.]"
)


def _tool_message_for_limited_payload(message):
    """Return a bounded copy of large hidden tool-result rows for paginated loads."""
    if not isinstance(message, dict) or str(message.get("role") or "").lower() != "tool":
        return message
    content = message.get("content")
    if content in (None, ""):
        return message
    if isinstance(content, str):
        text = content
    else:
        try:
            text = json.dumps(content, ensure_ascii=False, default=str)
        except Exception:
            text = str(content)
    if len(text) <= _LIMITED_TOOL_CONTENT_MAX_CHARS:
        return message
    clipped = dict(message)
    preview = text[:_LIMITED_TOOL_CONTENT_MAX_CHARS] + _LIMITED_TOOL_CONTENT_NOTICE
    if isinstance(content, str):
        clipped["content"] = preview
    elif isinstance(content, list):
        clipped["content"] = [{"type": "text", "text": preview}]
    elif isinstance(content, dict):
        clipped["content"] = {"_truncated": True, "preview": preview}
    else:
        clipped["content"] = preview
    clipped["_content_truncated"] = True
    clipped["_content_original_chars"] = len(text)
    return clipped


def _messages_for_limited_payload(messages) -> list:
    """Bound hidden tool-result payloads before sending a msg_limit response."""
    return [_tool_message_for_limited_payload(msg) for msg in list(messages or [])]


def _limited_webui_messages_for_display(session, state_db_messages) -> list:
    """Return the display sidecar plus only necessary state.db rows for msg_limit.

    Paginated session loads are latency-sensitive and should not stitch every
    lineage segment before slicing the tail. Keep the lightweight
    pre-compression snapshot stitch so continuation sessions can still reveal
    archived history, then merge only newer state.db rows that have not reached
    the sidecar yet.
    """
    sidecar_messages = _webui_sidecar_lineage_messages_for_display(session)
    return _limited_webui_messages_for_display_with_sidecar(
        session,
        sidecar_messages,
        state_db_messages,
    )


def _display_merge_session_is_active(session) -> bool:
    """Return whether any canonical in-memory projection is active/pending."""
    if getattr(session, "active_stream_id", None) or getattr(
        session, "pending_user_message", None
    ):
        return True
    sid = str(getattr(session, "session_id", "") or "")
    if not sid:
        return True
    with LOCK:
        live = SESSIONS.get(sid)
    if live is None or live is session:
        return False
    if str(getattr(live, "session_id", "") or "") != sid:
        return True
    return bool(
        getattr(live, "active_stream_id", None)
        or getattr(live, "pending_user_message", None)
    )


def _display_merge_cached_messages(session, sidecar_messages, *, msg_before=None):
    """Return the memoized merged transcript, or None when it can't be reused.

    Lets GET /api/session skip loading the state.db rows entirely on a hit. That
    is only sound because the cache key can be built from the existing
    commit-reliable DB/WAL/SHM signature (`_state_db_session_signature`) rather
    than a fingerprint computed FROM the loaded rows -- otherwise the key could
    not be built without paying exactly the cost we are trying to avoid.

    Fail-closed by construction: returns None whenever the session is active,
    the key cannot be built, or the cached entry does not match, and the caller
    then performs the normal full load + merge.
    """
    if msg_before is not None or _display_merge_session_is_active(session):
        return None
    sid = str(getattr(session, "session_id", "") or "")
    if not sid:
        return None
    with _display_merge_cache_lock:
        entry = _display_merge_cache.get(sid)
        if entry is None:
            return None
    # Resolve the sidecar exactly like the merge helper does: it treats None as
    # "load the lineage myself", and the cache entry was keyed on that RESOLVED
    # list. Probing with a bare None would key on an empty sidecar and miss
    # every time -- silently reverting this optimisation.
    if sidecar_messages is None:
        sidecar_messages = _webui_sidecar_lineage_messages_for_display(session)
    else:
        sidecar_messages = list(sidecar_messages or [])
    # Building the key requires the sidecar rows (cheap: already in memory or
    # served from the lineage cache) but not the state.db rows -- that
    # asymmetry is the whole point.
    cache_key = _display_merge_cache_key(session, sidecar_messages, None)
    if cache_key is None:
        return None
    with _display_merge_cache_lock:
        entry = _display_merge_cache.get(sid)
        if not _display_merge_cache_entry_usable(entry, cache_key):
            return None
        _display_merge_cache.move_to_end(sid, last=True)
        return [dict(m) if isinstance(m, dict) else m for m in entry["messages"]]


_DISPLAY_STATE_SIGNATURE_UNSET = object()


def _limited_webui_messages_for_display_with_sidecar(
    session,
    sidecar_messages,
    state_db_messages,
    *,
    state_db_signature=_DISPLAY_STATE_SIGNATURE_UNSET,
    msg_before=None,
) -> list:
    if sidecar_messages is None:
        sidecar_messages = _webui_sidecar_lineage_messages_for_display(session)
    else:
        sidecar_messages = list(sidecar_messages or [])
    state_db_messages = list(state_db_messages or [])
    if not state_db_messages:
        return sidecar_messages
    state_db_messages = _suppress_native_image_display_mirrors(
        session,
        state_db_messages,
    )
    if not state_db_messages:
        return sidecar_messages

    # NOTE: do not short-circuit to the sidecar when state.db has no strictly
    # newer rows. A state.db row whose timestamp is at-or-before the sidecar's
    # newest (recovery / edited-in-place / missing-timestamp cases) is still
    # absent from the sidecar and must be reconciled — dropping it would render
    # a tail that differs from the full merge path (silent wrong-transcript on
    # the paginated load). The append-only merge is O(n) over already-bounded
    # in-memory lists; the real latency win here is skipping the lineage-parent
    # DISK load above, which we still skip. (#4070 ship-review)
    #
    # perf: the merge itself is still expensive for multi-thousand-message
    # historical transcripts (~2-3s per request: json.dumps merge keys +
    # loose-content probes per row), and GET /api/session re-runs it on every
    # open/poll. Memoize per session id. Validity is fail-closed:
    #   - only INACTIVE sessions (no active stream, no pending user message):
    #     an active session's in-memory tail can be ahead of its disk
    #     signature, so it always recomputes;
    #   - the child sidecar's exact stat signature plus every lineage parent
    #     signature recorded by _webui_sidecar_lineage_messages_for_display;
    #   - the sidecar row count and last timestamp (guards unsaved in-memory
    #     appends that have not reached disk yet);
    #   - a content fingerprint of the (bounded) state.db rows.
    # Any uncertainty (missing signature, fingerprint failure) skips caching.
    cache_key = None
    # A msg_before request deliberately reads a different (uncapped) state.db
    # scope than the initial tail request.  It must bypass both cache layers:
    # skipping only the pre-load probe still let this inner lookup reuse the
    # initial 50k-row backstop merge and made the oldest row unreachable.
    if msg_before is None and not _display_merge_session_is_active(session):
        if state_db_signature is _DISPLAY_STATE_SIGNATURE_UNSET:
            _state_key = _state_db_rows_fingerprint(state_db_messages)
        else:
            _state_key = state_db_signature
            if _state_key is not None:
                _current_key = _state_db_session_signature(
                    getattr(session, "session_id", None),
                    getattr(session, "profile", None) or None,
                )
                if _current_key != _state_key:
                    _state_key = None
        if _state_key is not None:
            cache_key = _display_merge_cache_key(
                session,
                sidecar_messages,
                state_db_messages,
                state_db_signature=_state_key,
            )
    if cache_key is not None:
        sid = str(getattr(session, "session_id", "") or "")
        with _display_merge_cache_lock:
            entry = _display_merge_cache.get(sid)
            if _display_merge_cache_entry_usable(entry, cache_key):
                _display_merge_cache.move_to_end(sid, last=True)
                return [dict(m) if isinstance(m, dict) else m for m in entry["messages"]]
    merged = merge_session_messages_append_only(
        sidecar_messages,
        state_db_messages,
        truncation_watermark=getattr(session, "truncation_watermark", None),
        truncation_boundary=getattr(session, "truncation_boundary", None),
        incoming_provenance="state_db",
    )
    merged = _project_native_image_payload_conflicts_for_display(
        sidecar_messages,
        state_db_messages,
        merged,
    )
    if cache_key is not None:
        _state_key = cache_key[4]
        _streaming_key = (
            isinstance(_state_key, (list, tuple))
            and bool(_state_key)
            and _state_key[0] == "streaming"
        )
        if (
            state_db_signature is not _DISPLAY_STATE_SIGNATURE_UNSET
            and not _streaming_key
            and _state_db_session_signature(
                getattr(session, "session_id", None),
                getattr(session, "profile", None) or None,
            )
            != state_db_signature
        ):
            cache_key = None
    if cache_key is not None:
        sid = str(getattr(session, "session_id", "") or "")
        with _display_merge_cache_lock:
            _display_merge_cache[sid] = {
                "key": cache_key,
                "messages": merged,
                "stored_at": time.monotonic(),
            }
            _display_merge_cache.move_to_end(sid, last=True)
            while len(_display_merge_cache) > _DISPLAY_MERGE_CACHE_MAX:
                _display_merge_cache.popitem(last=False)
        # Same shallow-copy contract as the cache-hit path (and as the lineage
        # cache): callers may attach display metadata to the returned rows.
        return [dict(m) if isinstance(m, dict) else m for m in merged]
    return merged


# perf: memoized sidecar↔state.db display merges for GET /api/session.
# See _limited_webui_messages_for_display_with_sidecar for the validity rules.
_DISPLAY_MERGE_CACHE_MAX = 16
# Legacy streaming-freeze keys are still accepted defensively and remain
# tightly bounded. Production streaming keys now carry an exact target-session
# digest, so unrelated deltas stay stable without hiding target mutations.
_DISPLAY_MERGE_STREAMING_TTL_SECONDS = 5.0
_display_merge_cache: "OrderedDict[str, dict]" = OrderedDict()
_display_merge_cache_lock = threading.Lock()


def _display_merge_cache_entry_usable(entry, cache_key) -> bool:
    if entry is None or entry.get("key") != cache_key:
        return False
    try:
        state_key = cache_key[4]
    except (IndexError, TypeError):
        return False
    is_streaming_key = (
        isinstance(state_key, (list, tuple))
        and bool(state_key)
        and state_key[0] == "streaming"
    )
    if not is_streaming_key:
        return True
    try:
        age = time.monotonic() - float(entry["stored_at"])
    except (KeyError, TypeError, ValueError):
        return False
    return 0.0 <= age <= _DISPLAY_MERGE_STREAMING_TTL_SECONDS


def _display_merge_requires_lineage_provenance(session) -> bool:
    """Return whether this sidecar view depends on a stitched snapshot parent."""
    parent_id = str(getattr(session, "parent_session_id", "") or "").strip()
    if not parent_id:
        return False
    if not is_safe_session_id(parent_id):
        return True
    try:
        parent = Session.load(parent_id)
    except Exception:
        return True
    if parent is None:
        return True
    if not getattr(parent, "pre_compression_snapshot", False):
        return False
    source = str(getattr(session, "session_source", "") or "").strip().lower()
    parent_source = str(
        getattr(parent, "session_source", "") or ""
    ).strip().lower()
    if source == "fork" and parent_source != "fork":
        return False
    return not _messages_start_with_visible_prefix(
        list(getattr(session, "messages", []) or []),
        list(getattr(parent, "messages", []) or []),
    )


def _evict_lineage_display_cache_entry(sid, expected_entry) -> None:
    """Evict only the lineage entry that this caller validated as stale."""
    with _lineage_display_cache_lock:
        if _lineage_display_cache.get(sid) is expected_entry:
            _lineage_display_cache.pop(sid, None)


def _display_merge_cache_key(
    session,
    sidecar_messages,
    state_db_messages,
    *,
    state_db_signature=_DISPLAY_STATE_SIGNATURE_UNSET,
):
    """Return a fail-closed validity key for the display-merge cache, or None.

    None means "do not cache": any component that cannot be resolved exactly
    (missing sidecar signature, unfingerprintable state rows) disables the
    cache for this request rather than risking a stale transcript.
    """
    from api.models import _sidecar_stat_signature

    sid = str(getattr(session, "session_id", "") or "")
    if not sid or not is_safe_session_id(sid):
        return None
    self_sig = _sidecar_stat_signature(SESSION_DIR / f"{sid}.json")
    if self_sig is None:
        return None
    # Lineage parents: reuse the signatures recorded by the (already memoized)
    # lineage stitch so a write to any parent snapshot invalidates this cache
    # too. A lineage without snapshot parents records no entry — empty tuple.
    parent_sigs = ()
    with _lineage_display_cache_lock:
        lineage_entry = _lineage_display_cache.get(sid)
    if lineage_entry is not None:
        if (
            lineage_entry.get("provenance_complete") is not True
            or lineage_entry.get("self_sig") != self_sig
        ):
            _evict_lineage_display_cache_entry(sid, lineage_entry)
            return None
        parent_sigs = tuple(
            (str(path), tuple(sig) if isinstance(sig, (list, tuple)) else sig)
            for path, sig in (lineage_entry.get("parent_sigs") or [])
        )
        for parent_path, parent_sig in parent_sigs:
            if _sidecar_stat_signature(Path(parent_path)) != parent_sig:
                _evict_lineage_display_cache_entry(sid, lineage_entry)
                return None
        with _lineage_display_cache_lock:
            if _lineage_display_cache.get(sid) is not lineage_entry:
                return None
    if not parent_sigs and _display_merge_requires_lineage_provenance(session):
        return None
    last_ts = None
    if sidecar_messages:
        last = sidecar_messages[-1]
        if isinstance(last, dict):
            last_ts = last.get("timestamp")
    # Prefer the existing commit-reliable DB/WAL/SHM signature outside streams.
    # While another turn streams, use an exact digest scoped to this target
    # session so unrelated per-delta commits do not churn the key. Fail closed
    # onto the exact row fingerprint whenever the signature cannot be read.
    if state_db_signature is _DISPLAY_STATE_SIGNATURE_UNSET:
        state_fp = _state_db_session_signature(
            sid, getattr(session, "profile", None) or None
        )
    else:
        state_fp = state_db_signature
    if state_fp is None:
        # state_db_messages is None on the cache-probe path, where the rows were
        # deliberately not loaded. Fingerprinting None would key on the empty
        # row set and could match an entry built from real rows, so fail closed.
        if state_db_messages is None:
            return None
        state_fp = _state_db_rows_fingerprint(state_db_messages)
    if state_fp is None:
        return None
    return (
        self_sig,
        parent_sigs,
        len(sidecar_messages),
        last_ts,
        state_fp,
        getattr(session, "truncation_watermark", None),
        getattr(session, "truncation_boundary", None),
    )


def _state_db_target_session_signature(db_path, session_id):
    """Hash every target-session row without materialising display dictionaries."""
    try:
        uri_path = quote(str(Path(db_path).resolve()), safe="/")
        with closing(
            sqlite3.connect(f"file:{uri_path}?mode=ro", uri=True, timeout=5.0)
        ) as conn:
            conn.execute("PRAGMA query_only=ON")
            columns = [str(row[1]) for row in conn.execute("PRAGMA table_info(messages)")]
            if "session_id" not in columns or "id" not in columns:
                return None
            quoted_columns = ", ".join(
                '"' + column.replace('"', '""') + '"' for column in columns
            )
            conn.text_factory = lambda raw: ("text", raw)
            rows = conn.execute(
                f'SELECT {quoted_columns} FROM messages '
                'WHERE session_id = ? ORDER BY id',
                (str(session_id),),
            )
            digest = hashlib.blake2b(digest_size=32)
            digest.update("\x1f".join(columns).encode("utf-8"))
            row_count = 0
            for row in rows:
                row_count += 1
                for value in row:
                    if value is None:
                        tag, payload = b"n", b""
                    elif isinstance(value, tuple) and value[:1] == ("text",):
                        tag, payload = b"t", value[1]
                    elif isinstance(value, bytes):
                        tag, payload = b"b", value
                    elif isinstance(value, int):
                        tag, payload = b"i", str(value).encode("ascii")
                    elif isinstance(value, float):
                        tag, payload = b"f", value.hex().encode("ascii")
                    else:
                        return None
                    digest.update(tag)
                    digest.update(len(payload).to_bytes(8, "big"))
                    digest.update(payload)
            return ("streaming-target", row_count, digest.hexdigest())
    except (OSError, sqlite3.Error, ValueError):
        return None


def _state_db_target_session_revision(db_path, session_id):
    """Return a bounded cross-process revision for one session's display rows.

    Supported writers update ``sessions.last_activity_at``/``message_count``.
    The indexed tail revision additionally catches direct appends, tail deletes,
    and raw changes to timestamp, row flags, or byte lengths on the newest row.
    Legacy stores without a sessions row use full numeric/length aggregates. A
    same-length raw SQL rewrite of an older row that bypasses session metadata
    is outside the state-store writer contract; detecting it exactly would
    require scanning and hashing every message payload (148 MB on the production
    long session), which would make the cache slower than the uncached path.
    """
    message_length_columns = (
        "role",
        "content",
        "tool_call_id",
        "tool_calls",
        "tool_name",
        "finish_reason",
        "reasoning",
        "reasoning_content",
        "reasoning_details",
        "codex_reasoning_items",
        "codex_message_items",
        "effect_disposition",
        "api_content",
        "display_kind",
        "display_metadata",
    )
    message_sum_columns = ("observed", "active", "compacted")
    session_revision_columns = (
        "message_count",
        "last_activity_at",
        "ended_at",
        "end_reason",
        "rewind_count",
        "archived",
    )
    try:
        uri_path = quote(str(Path(db_path).resolve()), safe="/")
        with closing(
            sqlite3.connect(f"file:{uri_path}?mode=ro", uri=True, timeout=0.25)
        ) as conn:
            conn.execute("PRAGMA query_only=ON")
            conn.execute("PRAGMA busy_timeout=250")
            message_columns = {
                str(row[1]) for row in conn.execute("PRAGMA table_info(messages)")
            }
            if "session_id" not in message_columns or "id" not in message_columns:
                return None
            session_columns = {
                str(row[1]) for row in conn.execute("PRAGMA table_info(sessions)")
            }
            available_session_columns = [
                column
                for column in session_revision_columns
                if column in session_columns
            ]
            session_revision = None
            if "id" in session_columns and available_session_columns:
                session_revision = conn.execute(
                    "SELECT "
                    + ", ".join(
                        f'"{column}"' for column in available_session_columns
                    )
                    + " FROM sessions WHERE id = ?",
                    (str(session_id),),
                ).fetchone()
            if session_revision is not None:
                latest_parts = ["id"]
                latest_parts.append(
                    "timestamp" if "timestamp" in message_columns else "NULL"
                )
                latest_parts.extend(
                    f'LENGTH(COALESCE("{column}", \'\'))'
                    for column in message_length_columns
                    if column in message_columns
                )
                latest_parts.extend(
                    f'COALESCE("{column}", 0)'
                    for column in message_sum_columns
                    if column in message_columns
                )
                latest_revision = conn.execute(
                    f"SELECT {', '.join(latest_parts)} FROM messages "
                    "WHERE session_id = ? ORDER BY id DESC LIMIT 1",
                    (str(session_id),),
                ).fetchone()
                return (
                    "target-session-revision-v2",
                    tuple(available_session_columns),
                    tuple(session_revision),
                    tuple(latest_revision) if latest_revision is not None else None,
                )

            # Legacy state stores without a sessions row have no supported
            # O(1) activity revision. Keep them conservative by scanning compact
            # numeric/length aggregates instead of trusting a global DB stamp.
            aggregate_parts = ["COUNT(*)", "MAX(id)"]
            aggregate_parts.append(
                "MAX(timestamp)" if "timestamp" in message_columns else "NULL"
            )
            aggregate_parts.extend(
                f'SUM(LENGTH(COALESCE("{column}", \'\')))'
                for column in message_length_columns
                if column in message_columns
            )
            aggregate_parts.extend(
                f'SUM(COALESCE("{column}", 0))'
                for column in message_sum_columns
                if column in message_columns
            )
            message_revision = conn.execute(
                f"SELECT {', '.join(aggregate_parts)} FROM messages "
                "WHERE session_id = ?",
                (str(session_id),),
            ).fetchone()
            return (
                "target-session-revision-v1-legacy",
                (),
                None,
                tuple(message_revision) if message_revision is not None else None,
            )
    except (OSError, sqlite3.Error, ValueError):
        return None


def _state_db_session_signature(session_id, profile=None):
    """Return a cross-process target-session cache revision, fail-closed.

    The session-scoped revision avoids DB/WAL false invalidations caused by a
    different conversation streaming in another WebUI process. If the schema
    cannot provide that revision, fall back to the existing global file key.
    """
    from api.models import _agent_state_db_path, _sqlite_file_stat_cache_key

    sid = str(session_id or "")
    if not sid or not is_safe_session_id(sid):
        return None
    try:
        db_path = _agent_state_db_path(profile=profile)
        if not db_path or not Path(db_path).exists():
            return None
    except Exception:
        return None
    target_revision = _state_db_target_session_revision(db_path, sid)
    if target_revision is not None:
        return target_revision
    try:
        signature = _sqlite_file_stat_cache_key(Path(db_path))
    except Exception:
        return None
    if signature is None:
        return None
    # ``_sqlite_file_stat_cache_key`` is a tuple of the content fingerprint and
    # DB/WAL/SHM stat stamps. A completely empty result is not a valid key.
    try:
        if not any(component is not None for component in signature):
            return None
    except TypeError:
        return None
    return signature


def _load_state_db_messages_with_stable_signature(session_id, profile, reader_kwargs):
    """Load rows and return the database signature that brackets that read."""
    before = _state_db_session_signature(session_id, profile)
    rows = get_state_db_session_messages(session_id, **dict(reader_kwargs or {}))
    after = _state_db_session_signature(session_id, profile)
    stable = before if before is not None and before == after else None
    return rows, stable


def _state_db_rows_fingerprint(rows) -> str | None:
    """Content fingerprint of the state.db display rows, or None on failure."""
    try:
        h = hashlib.sha256()
        h.update(str(len(rows)).encode("utf-8"))
        for row in rows:
            if isinstance(row, dict):
                h.update(json.dumps(row, sort_keys=True, default=str).encode("utf-8", "replace"))
            else:
                h.update(repr(row).encode("utf-8", "replace"))
        return h.hexdigest()
    except Exception:
        return None


def _sidecar_file_exceeds_threshold(session_id, threshold_bytes) -> bool:
    """Check if the sidecar JSON file for ``session_id`` exceeds ``threshold_bytes``."""
    from api.config import SESSION_DIR
    try:
        p = SESSION_DIR / f"{session_id}.json"
        return os.path.isfile(p) and os.path.getsize(p) > threshold_bytes
    except Exception:
        return False


def _state_db_since_timestamp_for_limited_display(session, msg_limit, msg_before=None):
    """Return (timestamp floor, sidecar messages) for bounded state.db tail reads.

    The display window limit counts visible transcript rows after WebUI sidecar
    and state.db reconciliation, so this deliberately does not SQL ``LIMIT`` raw
    rows.  Instead, for the common initial tail load, keep the full sidecar
    coordinate space and read a conservative recent state.db superset.  Older
    page loads and edit/truncation recovery shapes stay on the full state.db
    path because their correctness depends on older reconciliation rows.
    """
    if msg_limit is None or msg_before is not None:
        return None, None
    if getattr(session, "truncation_watermark", None) not in (None, ""):
        return None, None
    if getattr(session, "truncation_boundary", None) not in (None, ""):
        return None, None

    sidecar_messages = _webui_sidecar_lineage_messages_for_display(session)
    if not sidecar_messages:
        return None, sidecar_messages
    sidecar_timestamps = [_message_timestamp_as_float(msg) for msg in sidecar_messages]
    if any(ts is None for ts in sidecar_timestamps):
        return None, sidecar_messages

    try:
        limit = max(1, int(msg_limit))
    except (TypeError, ValueError):
        return None, sidecar_messages
    raw_budget = max(300, limit * 10)
    if len(sidecar_messages) <= raw_budget:
        _sid = getattr(session, "session_id", "") or ""
        if not _sid or not _sidecar_file_exceeds_threshold(_sid, _SIDECAR_BYTE_TAIL_THRESHOLD):
            return None, sidecar_messages

    floor = min(sidecar_timestamps[-raw_budget:])
    sidecar_before_count = sum(1 for ts in sidecar_timestamps if ts < floor)
    prefix_summary = get_state_db_session_message_prefix_summary(
        getattr(session, "session_id", None),
        floor,
        profile=getattr(session, "profile", None) or None,
    )
    if prefix_summary is None:
        return None, sidecar_messages
    try:
        state_before_count = int(prefix_summary["count"])
        null_timestamp_count = int(prefix_summary["null_timestamp_count"])
    except (KeyError, TypeError, ValueError):
        return None, sidecar_messages
    if null_timestamp_count or state_before_count != sidecar_before_count:
        return None, sidecar_messages
    if sidecar_before_count == 0:
        return floor, sidecar_messages

    sidecar_before_keys = [
        _session_message_visible_key(msg)
        for msg, ts in zip(sidecar_messages, sidecar_timestamps, strict=True)
        if ts < floor
    ]
    state_before_keys = get_state_db_session_message_keys_before_timestamp(
        getattr(session, "session_id", None),
        floor,
        profile=getattr(session, "profile", None) or None,
    )
    if state_before_keys is None or state_before_keys != sidecar_before_keys:
        return None, sidecar_messages
    return floor, sidecar_messages


def _messages_start_with_visible_prefix(messages, prefix) -> bool:
    """Return True when ``messages`` already replays ``prefix`` in display order."""
    messages = list(messages or [])
    prefix = list(prefix or [])
    if not prefix:
        return True
    if len(messages) < len(prefix):
        return False
    try:
        return all(
            _session_message_visible_key(messages[idx]) == _session_message_visible_key(prefix_msg)
            for idx, prefix_msg in enumerate(prefix)
        )
    except Exception:
        return False


# perf: memoized lineage-stitch results for GET /api/session. Keyed by session
# id; validity = exact stat signature of the child sidecar AND every snapshot
# parent involved in the stitch. Any write to any involved sidecar changes its
# signature and invalidates the entry. Bounded LRU — historical lineages are
# few but their merges cost seconds each.
_LINEAGE_DISPLAY_CACHE_MAX = 16
_lineage_display_cache: "OrderedDict[str, dict]" = OrderedDict()
_lineage_display_cache_lock = threading.Lock()


def _webui_sidecar_lineage_messages_for_display(session, *, max_hops: int = 20) -> list:
    """Return WebUI sidecar messages stitched across compression snapshots.

    WebUI compression continuations persist the archived transcript in a parent
    sidecar marked ``pre_compression_snapshot`` and keep subsequent turns in the
    child sidecar. Opening the child alone makes older turns look lost. Stitch
    only those snapshot parents for display; ordinary forks also carry
    ``parent_session_id`` but must remain independent conversations.

    perf: the stitched merge is O(total messages) with expensive per-row keys
    (json.dumps of tool_calls, loose-content regex). For multi-thousand-message
    lineages it costs seconds per request, and GET /api/session re-runs it on
    every open/poll. The result is cached per session id, keyed by the stat
    signature of every sidecar involved (child + each snapshot parent), so an
    idle historical lineage merges once and any write to any involved sidecar
    invalidates naturally. Cache hits return shallow-copied rows so callers can
    attach display metadata without corrupting the cache.
    """
    from api.models import _sidecar_stat_signature

    cache_allowed = not _display_merge_session_is_active(session)
    sid = str(getattr(session, "session_id", "") or "")
    self_sig = None
    if sid and is_safe_session_id(sid):
        self_sig = _sidecar_stat_signature(SESSION_DIR / f"{sid}.json")
    if cache_allowed and self_sig is not None:
        with _lineage_display_cache_lock:
            entry = _lineage_display_cache.get(sid)
        if (
            entry is not None
            and entry.get("provenance_complete") is True
            and entry.get("self_sig") == self_sig
        ):
            stale = False
            for parent_path, parent_sig in entry.get("parent_sigs") or []:
                if _sidecar_stat_signature(Path(parent_path)) != parent_sig:
                    stale = True
                    break
            if not stale:
                with _lineage_display_cache_lock:
                    current_entry = _lineage_display_cache.get(sid)
                    if current_entry is entry:
                        _lineage_display_cache.move_to_end(sid, last=True)
                        return [
                            dict(m) if isinstance(m, dict) else m
                            for m in entry["messages"]
                        ]
            else:
                _evict_lineage_display_cache_entry(sid, entry)

    segments = []
    current = session
    session_messages = list(getattr(session, "messages", []) or [])
    source = str(getattr(session, "session_source", "") or "").strip().lower()
    root_is_fork = source == "fork"
    seen = {str(getattr(session, "session_id", "") or "")}
    parent_sigs: list[tuple[str, tuple]] = []
    parent_signatures_complete = True
    for _ in range(max(0, int(max_hops))):
        parent_id = str(getattr(current, "parent_session_id", "") or "").strip()
        if not parent_id:
            break
        if parent_id in seen or not is_safe_session_id(parent_id):
            parent_signatures_complete = False
            break
        parent_path = SESSION_DIR / f"{parent_id}.json"
        parent_sig_before = _sidecar_stat_signature(parent_path)
        parent = Session.load(parent_id)
        if not parent:
            parent_signatures_complete = False
            break
        if not getattr(parent, "pre_compression_snapshot", False):
            break
        parent_sig = _sidecar_stat_signature(parent_path)
        if (
            parent_sig_before is None
            or parent_sig is None
            or parent_sig_before != parent_sig
        ):
            parent_signatures_complete = False
        else:
            parent_sigs.append((str(parent_path), parent_sig))
        parent_source = str(getattr(parent, "session_source", "") or "").strip().lower()
        if root_is_fork and parent_source != "fork":
            break
        if not segments and _messages_start_with_visible_prefix(
            session_messages,
            getattr(parent, "messages", []) or [],
        ):
            return session_messages
        segments.append(parent)
        seen.add(parent_id)
        current = parent
    else:
        # Exhausting max_hops means the declared ancestry may continue beyond
        # the signatures captured above. Never publish partial provenance.
        parent_signatures_complete = False

    if not segments:
        return list(getattr(session, "messages", []) or [])

    merged = []
    for segment in reversed(segments):
        merged = merge_session_messages_append_only(
            merged,
            getattr(segment, "messages", []) or [],
            truncation_watermark=getattr(segment, "truncation_watermark", None),
            truncation_boundary=getattr(segment, "truncation_boundary", None),
        )
    merged = merge_session_messages_append_only(
        merged,
        getattr(session, "messages", []) or [],
        truncation_watermark=None,
    )
    if (
        cache_allowed
        and self_sig is not None
        and parent_sigs
        and parent_signatures_complete
    ):
        with _lineage_display_cache_lock:
            _lineage_display_cache[sid] = {
                "self_sig": self_sig,
                "parent_sigs": parent_sigs,
                "provenance_complete": True,
                "messages": merged,
            }
            _lineage_display_cache.move_to_end(sid, last=True)
            while len(_lineage_display_cache) > _LINEAGE_DISPLAY_CACHE_MAX:
                _lineage_display_cache.popitem(last=False)
        # Hand out copies so caller-side metadata mutation cannot corrupt
        # the cached rows (same contract as the cache-hit path).
        return [dict(m) if isinstance(m, dict) else m for m in merged]
    return merged


def _merged_session_messages_for_display(session, cli_messages=None) -> list:
    """Return the message coordinate space exposed by ``GET /api/session``.

    Messaging sessions can have a WebUI sidecar transcript plus messages from
    the Agent/CLI store. WebUI compression continuations can have an archived
    snapshot parent plus a child continuation sidecar. The frontend computes
    fork keep-counts against this merged display list, so branch/fork must slice
    the same list rather than the sidecar-only ``session.messages`` array.
    """
    cli_messages = list(cli_messages or [])
    sidecar_messages = _webui_sidecar_lineage_messages_for_display(session)
    if cli_messages:
        if sidecar_messages and sidecar_messages != cli_messages:
            if len(sidecar_messages) >= len(cli_messages):
                return merge_session_messages_append_only(
                    sidecar_messages,
                    cli_messages,
                    truncation_watermark=getattr(session, "truncation_watermark", None),
                    truncation_boundary=getattr(session, "truncation_boundary", None),
                )
            merged_messages = []
            seen_message_keys = set()
            for msg in sorted(list(cli_messages) + list(sidecar_messages), key=lambda m: (
                float(m.get("timestamp") or 0),
                str(m.get("role") or ""),
                str(m.get("content") or ""),
            )):
                key = _session_message_merge_key(msg)
                if key in seen_message_keys:
                    continue
                seen_message_keys.add(key)
                merged_messages.append(msg)
            return merged_messages
        return sidecar_messages if len(sidecar_messages) > len(cli_messages) else cli_messages
    return sidecar_messages



_LINEAGE_PARENT_SESSION_UNSET = object()


def _webui_lineage_parent_session_for_display(session):
    """Load the immediate parent only for display-eligible continuations."""
    parent_id = str(getattr(session, "parent_session_id", "") or "").strip()
    if not parent_id:
        return None
    if (
        str(getattr(session, "compression_recovery_source_session_id", "") or "").strip()
        and str(getattr(session, "compression_recovery_action", "") or "").strip()
    ):
        return None
    source = str(getattr(session, "session_source", "") or "").strip().lower()
    relationship = str(getattr(session, "relationship_type", "") or "").strip().lower()
    if source == "fork" or relationship == "child_session":
        return None
    try:
        return get_session(parent_id, metadata_only=False)
    except Exception:
        return None


def _merged_webui_lineage_messages_for_display(
    session,
    messages=None,
    *,
    parent_session=_LINEAGE_PARENT_SESSION_UNSET,
) -> list:
    """Include immediate parent-only rows when a WebUI continuation sidecar is partial.

    Compression/continuation sessions should render as one conversation. Most
    child sidecars are cumulative, so this is usually a cheap no-op. If a child
    sidecar accidentally omits rows that still exist in the immediate parent,
    merge those parent-only rows into the display transcript. Explicit forks and
    generic child-session rows remain isolated; they intentionally start from a
    subset of their parent.
    """
    primary_messages = list(messages if messages is not None else (getattr(session, "messages", []) or []))
    if parent_session is _LINEAGE_PARENT_SESSION_UNSET:
        parent_session = _webui_lineage_parent_session_for_display(session)
    parent_messages = list(getattr(parent_session, "messages", []) or [])
    if not parent_messages:
        return primary_messages
    if _messages_start_with_visible_prefix(primary_messages, parent_messages):
        return primary_messages
    merged_messages = []
    seen_message_keys = set()
    seen_messages_by_key = {}
    for msg in sorted(list(parent_messages) + list(primary_messages), key=lambda m: (
        float(m.get("timestamp") or 0),
        str(m.get("role") or ""),
        str(m.get("content") or ""),
    )):
        key = _session_message_merge_key(msg)
        if key in seen_message_keys:
            _merge_session_display_metadata(seen_messages_by_key.get(key), msg)
            continue
        seen_message_keys.add(key)
        seen_messages_by_key[key] = msg
        merged_messages.append(msg)
    return merged_messages


def _message_summary(messages) -> dict:
    messages = list(messages or [])
    last_message_at = 0.0
    for msg in messages:
        if not isinstance(msg, dict):
            continue
        try:
            last_message_at = max(last_message_at, float(msg.get("timestamp") or 0))
        except (TypeError, ValueError):
            pass
    return {"message_count": len(messages), "last_message_at": last_message_at}


def _metadata_only_message_summary(sid: str, profile: str | None = None) -> dict:
    """Return the cheap message summary used by metadata-only session loads.

    Threads ``profile=`` through to ``get_state_db_session_summary`` so
    background-thread reads land on the correct profile's state.db (per the
    cookie-bound profile selector — fixes the same TLS-vs-thread race the
    #2762 fix addressed for write paths).

    This intentionally does not full-read or merge transcripts.  If state.db has
    grown beyond the sidecar count, report that growth so active-session polling
    can refresh.  If state.db only contains restamped replay rows at or below the
    sidecar count, keep the sidecar metadata so polling does not loop forever on
    a false "newer transcript" signal.
    """
    sidecar_session = Session.load_metadata_only(sid)
    sidecar_count = 0
    sidecar_last_message_at = 0.0
    if sidecar_session:
        sidecar_count = _numeric_count(getattr(sidecar_session, "_metadata_message_count", None))
        if sidecar_count <= 0:
            sidecar_count = _numeric_count(sidecar_session.compact().get("message_count"))
        try:
            sidecar_last_message_at = float(getattr(sidecar_session, "updated_at", 0) or 0)
        except (TypeError, ValueError):
            sidecar_last_message_at = 0.0
        if getattr(sidecar_session, "truncation_watermark", None) is not None:
            # Intentional: once the user has truncated this sidecar, metadata
            # polling must keep the sidecar as authoritative.  A full message
            # load can still apply the watermark-aware merge, but the cheap
            # metadata path should not treat later state.db rows as external
            # growth and resurrect turns the user deliberately cut away.
            return {
                "message_count": sidecar_count,
                "last_message_at": sidecar_last_message_at,
            }
    state_summary = get_state_db_session_summary(sid, profile=profile)
    state_count = _numeric_count(state_summary.get("message_count"))
    try:
        state_last_message_at = float(state_summary.get("last_message_at") or 0)
    except (TypeError, ValueError):
        state_last_message_at = 0.0
    if state_count > sidecar_count and state_last_message_at > sidecar_last_message_at:
        return {
            "message_count": state_count,
            "last_message_at": state_last_message_at,
        }
    return {
        "message_count": sidecar_count,
        "last_message_at": sidecar_last_message_at,
    }


def _session_requires_cli_metadata_lookup(session) -> bool:
    """Return True when a sidecar/session row still needs CLI metadata.

    Legacy imported sidecars may predate the ``read_only`` field and therefore
    load with ``read_only=False``. They still persist ``is_cli_session`` and/or
    source metadata from import time, so those markers intentionally keep them
    on the CLI lookup path while ordinary WebUI-native sessions take the fast
    path.

    Supersedes the simpler is-cli-or-messaging gate from PR #1822 — the new
    gate is strictly more inclusive (also covers ``read_only=True`` sidecars,
    ``session_source`` markers, and source_tag/raw_source/platform metadata)
    so all sessions that previously took the slow path still do, plus a few
    more legacy shapes.
    """
    if not session:
        return False

    def _field(name):
        return session.get(name) if isinstance(session, dict) else getattr(session, name, None)

    if _is_messaging_session_record(session):
        return True
    if bool(_field("is_cli_session")) or bool(_field("read_only")):
        return True
    session_source = _normalize_messaging_source(_safe_first(_field("session_source")))
    if session_source in {"messaging", "external_agent", "external-agent"}:
        return True
    return bool(_safe_first(
        _field("source_tag"),
        _field("raw_source"),
        _field("source"),
        _field("source_label"),
        _field("platform"),
    ))


def _is_messaging_session_id(sid: str) -> bool:
    """Detect messaging-backed sessions from WebUI metadata or Agent rows."""
    try:
        session = Session.load(sid)
        if _is_messaging_session_record(session):
            return True
    except Exception:
        pass
    return _is_messaging_session_record(_lookup_cli_session_metadata(sid))


def _session_sort_timestamp(session: dict) -> float:
    return float(
        _safe_first(
            session.get("last_message_at"),
            session.get("updated_at"),
            session.get("created_at"),
            session.get("started_at"),
            0,
        ) or 0
    ) or 0.0


def _is_cli_session_for_settings(session: dict) -> bool:
    """Return True for importable CLI sessions that are safe to classify for settings."""
    if not isinstance(session, dict):
        return False
    if is_cli_session_row(session):
        return True

    # Fallback for legacy local copies that had weak/empty metadata:
    # keep this conservative so messaging sessions do not collapse incorrectly.
    if not session.get("is_cli_session"):
        return False
    source = str(session.get("source") or "").strip().lower()
    if source in MESSAGING_SOURCES:
        return False
    title = str(session.get("title") or "").strip().lower()
    return title in ("", "untitled", "cli", "cli session") or title.endswith(" session") and (
        not source or source == "cli"
    )


def _normalize_sidebar_source_flags(session: dict) -> dict:
    """Return a sidebar row with the frontend CLI flag matching source metadata."""
    if not isinstance(session, dict):
        return session
    normalized = dict(session)
    normalized["is_cli_session"] = is_cli_session_row(normalized)
    return normalized


def _reconcile_session_detail_source_flags(session: dict, state_meta: dict) -> dict:
    """Return a /api/session payload whose source flags match state.db truth.

    WebUI-origin sidecars can carry stale CLI/import flags after older repair or
    import paths touched the JSON file. The sidebar projection already trusts the
    state.db source row for those sessions; the detail endpoint must do the same
    or the frontend opens a WebUI-native transcript as an external session and
    starts the destructive active-refresh reload loop.
    """
    if not isinstance(session, dict):
        return session
    if not _session_source_is_webui(state_meta):
        return dict(session)

    reconciled = dict(session)
    reconciled["is_cli_session"] = False
    reconciled["read_only"] = False
    reconciled["source_tag"] = _safe_first(state_meta.get("source_tag"), "webui")
    reconciled["raw_source"] = _safe_first(state_meta.get("raw_source"), "webui")
    reconciled["session_source"] = _safe_first(state_meta.get("session_source"), "webui")
    reconciled["source_label"] = _safe_first(state_meta.get("source_label"), "WebUI")
    if state_meta.get("source"):
        reconciled["source"] = state_meta["source"]

    for key in ("message_count", "actual_message_count"):
        if state_meta.get(key) is not None:
            reconciled[key] = max(
                _numeric_count(reconciled.get(key)),
                _numeric_count(state_meta.get(key)),
            )
    for key in ("created_at", "updated_at", "last_message_at"):
        if state_meta.get(key) is not None:
            current = reconciled.get(key)
            try:
                reconciled[key] = max(float(current or 0), float(state_meta.get(key) or 0))
            except (TypeError, ValueError):
                reconciled[key] = state_meta[key]
    return reconciled


def _session_source_is_webui(session: dict) -> bool:
    """Return True for state.db/sidebar rows that describe WebUI-origin sessions."""
    if not isinstance(session, dict):
        return False
    for key in ("source_tag", "raw_source", "session_source", "source"):
        if str(session.get(key) or "").strip().lower() == "webui":
            return True
    return False


def _normalized_source_marker(value) -> str:
    marker = str(value or "").strip().lower()
    if marker.endswith(" session"):
        marker = marker[:-len(" session")].strip()
    return marker.replace("-", "_").replace(" ", "_")


def _is_api_server_sidecar_row(session: dict) -> bool:
    """Return True for API-server imported sidecars that need orphan pruning."""
    if not isinstance(session, dict) or _session_source_is_webui(session):
        return False
    markers = {
        _normalized_source_marker(session.get(key))
        for key in ("source", "source_tag", "raw_source", "session_source", "source_label")
    }
    return bool(markers & {"api", "api_server"})


def _session_lineage_ids(session: dict) -> set[str]:
    """Return known ids that identify one logical sidebar lineage."""
    if not isinstance(session, dict):
        return set()
    ids: set[str] = set()
    for key in ("session_id", "_lineage_root_id", "_lineage_tip_id"):
        value = session.get(key)
        if value:
            ids.add(str(value))
    return ids


def _is_duplicate_webui_state_projection(session: dict, represented_webui_ids: set[str]) -> bool:
    """Return True when a state.db row is only a duplicate WebUI-origin projection.

    The "Show non-WebUI sessions" toggle should add external/agent-owned
    conversations, not make WebUI compression continuations appear only when the
    external-session bridge is enabled. WebUI-origin state.db rows are still
    useful metadata sidecars, but if any id in their compression lineage is
    already represented by WebUI session JSON, they should not be injected as an
    additive external row.
    """
    if not _session_source_is_webui(session):
        return False
    return bool(_session_lineage_ids(session) & represented_webui_ids)


def _dedupe_cli_sidebar_sessions_for_api(
    cli: list[dict],
    represented_webui_ids: set[str],
    *,
    show_cron_sessions: bool = False,
    show_webhook_sessions: bool = False,
    show_kanban_sessions: bool = False,
    source_filter: str | None = None,
) -> list[dict]:
    """Return state sidebar rows while preserving project-hidden background rows.

    Agent-side cron and webhook sessions come from state.db rather than the WebUI
    session store. They should stay hidden from the default sidebar, but
    project-assigned messageful rows must remain in the `/api/sessions` payload
    with `default_hidden` so the matching project chip can reveal them (#3134).

    An explicit ``source_filter`` for a background source (cron/webhook/kanban)
    is a deliberate request to view those rows, so it overrides the default
    hide for that source only — the user asked for them.
    """
    from api.models import (
        _hide_from_default_sidebar as _hide_background,
        _include_project_hidden_background_sidebar_sessions,
    )

    # An explicit background source filter reveals that source (override the hide).
    # Normalize to match how the loader canonicalizes source_filter (strip+lower).
    _sf = str(source_filter or '').strip().lower()
    if _sf == 'cron':
        show_cron_sessions = True
    elif _sf == 'webhook':
        show_webhook_sessions = True
    elif _sf == 'kanban':
        show_kanban_sessions = True

    candidates = [
        s for s in cli
        if s["session_id"] not in represented_webui_ids
        and not _is_duplicate_webui_state_projection(s, represented_webui_ids)
        and is_cli_session_row_visible(s)
    ]
    visible = [
        s for s in candidates
        if not _hide_background(
            s,
            show_cron=show_cron_sessions,
            show_webhook=show_webhook_sessions,
            show_kanban=show_kanban_sessions,
        )
    ]
    return _include_project_hidden_background_sidebar_sessions(candidates, visible)


def _cli_visible_session_cap() -> int:
    """Shared sidebar window, not a second hard-coded 20."""
    from api.config import CLI_VISIBLE_SESSION_LIMIT
    return CLI_VISIBLE_SESSION_LIMIT


# Bound on project-assigned CLI rows in the FINAL MERGED payload, across EVERY
# project. This is the only place that sees every source of assigned rows at
# once — state.db's own bounded passes plus imported WebUI sidecars from
# all_sessions(), which no model-side cap applies to — so it is the only place
# that can actually bound the assigned set (#6659 review finding 1).
#
# It has to bound the MERGED set, not one project: 200 rows x N projects grows
# with the project count, and 1,000 assigned conversations spread over 5 projects
# still returned all 1,000 — the exact reproduction from that finding. Pinned
# equal to models.PROJECT_ASSIGNED_CLI_LIMIT (an independent literal on the
# model side; the test is what keeps the two in lockstep) by
# test_route_merged_assigned_cap_is_the_existing_model_row_cap.
CLI_PROJECT_ASSIGNED_CAP = 200


def _draw_assigned_cli_rows_fairly(
    rows_by_project: dict[str, list[int]], budget: int
) -> set[int]:
    """Pick ``budget`` assigned row indices, spread fairly across the projects.

    ``rows_by_project`` maps project id -> that project's row indices, newest
    first, keyed in order of each project's most recent assigned conversation
    (``sessions`` is newest-first, so insertion order already is that order).

    Each round hands one slot to every project that still has history left, so:

    * the drawn set never exceeds ``budget`` — that is the whole point, a
      per-project bound does not bound the payload (#6659 review finding 1);
    * no single busy project can eat every slot, which a flat ``sessions[:200]``
      truncation would do to whichever project sorts first — the starvation
      greptile rejected as P1 on #6659;
    * every project keeps at least one row whenever
      ``budget >= len(rows_by_project)``. Past that the bound wins: with more
      assigned projects than slots, the ``budget`` most recently active projects
      get one row each, because the review's number is the hard constraint.

    Within a project the draw is newest-first, so what a chip loses is always the
    oldest end of its own history.
    """
    drawn: set[int] = set()
    if budget <= 0 or not rows_by_project:
        return drawn
    queues = list(rows_by_project.values())
    offsets = [0] * len(queues)
    remaining = budget
    while remaining > 0:
        progressed = False
        for position, project_rows in enumerate(queues):
            offset = offsets[position]
            if offset >= len(project_rows):
                continue
            drawn.add(project_rows[offset])
            offsets[position] = offset + 1
            remaining -= 1
            progressed = True
            if remaining <= 0:
                break
        if not progressed:
            # Every project is exhausted — the whole assigned set fits.
            break
    return drawn


def _cap_recent_cli_sessions(
    sessions: list[dict],
    cli_cap: int | None = None,
    project_cap: int = CLI_PROJECT_ASSIGNED_CAP,
) -> list[dict]:
    """Cap the default CLI list while retaining project-addressable rows.

    ``sessions`` is newest-first and already deduplicated (WebUI sidecars merged,
    lineages collapsed, messaging sources folded), so every row counted here is
    one logical conversation.

    Two independent budgets, because they answer to different users (#6659):

    * ``cli_cap`` unassigned conversations own the default sidebar window. An
      assigned row must not spend one of those slots, or assigning three sessions
      to a project silently shortens everyone's sidebar to 17 rows. Resolved
      lazily from the shared configurable window (HERMES_WEBUI_VISIBLE_SESSION_LIMIT),
      never a second hard-coded 20.
    * ``project_cap`` assigned conversations IN TOTAL, across every project, stay
      in the payload so the project chips can reveal them, marked
      ``default_hidden`` once the recent window is full. Past that bound they are
      dropped: keeping assigned rows past the *recent* cap is the fix, keeping
      them past *all* bounds just trades a vanishing session for a stalled
      sidebar.

    That assigned budget is spent by a fair round-robin draw across the projects
    (see ``_draw_assigned_cli_rows_fairly``) instead of by truncating the merged
    list, so bounding the payload cannot starve a quiet project (greptile P1 on
    #6659). ``project_cap <= 0`` disables the assigned bound entirely.
    """
    if cli_cap is None:
        cli_cap = _cli_visible_session_cap()
    if cli_cap <= 0:
        return sessions
    # Group the assigned rows per project first: the draw has to weigh the
    # projects against each other, which a single forward pass cannot do.
    rows_by_project: dict[str, list[int]] = {}
    for index, session in enumerate(sessions):
        if not _is_cli_session_for_settings(session):
            continue
        project_id = str(session.get("project_id") or "").strip()
        if project_id:
            rows_by_project.setdefault(project_id, []).append(index)
    if project_cap > 0 and rows_by_project:
        # Reserve the CLI rows the recent window already shows — the first
        # ``cli_cap`` CLI rows in sort order, assigned or not — before the fair
        # draw spreads the REST of the assigned budget across projects. Without
        # this, a project holding all the newest sessions can lose its newest
        # rows in the draw and the payload drops sessions the base displays
        # (2026-09-24 re-gate reproduction: 11 projects x 20 sessions, all 20
        # newest in one project, ``p0-19`` vanished from the payload).
        reserved: set[int] = set()
        seen_cli = 0
        for index, session in enumerate(sessions):
            if not _is_cli_session_for_settings(session):
                continue
            if seen_cli >= cli_cap:
                break
            seen_cli += 1
            reserved.add(index)
        # The reserved rows have already been paid for by the recent window;
        # draw the remaining budget over the queues MINUS those rows, so the
        # reservation cannot double-spend slots the draw would have granted.
        remaining_rows_by_project: dict[str, list[int]] = {
            project: [index for index in indices if index not in reserved]
            for project, indices in rows_by_project.items()
        }
        assigned_reserved = sum(
            1 for index in reserved
            if str(sessions[index].get("project_id") or "").strip()
        )
        drawn = reserved | _draw_assigned_cli_rows_fairly(
            remaining_rows_by_project, max(project_cap - assigned_reserved, 0)
        )
    else:
        drawn = None if project_cap <= 0 else set()
    kept = []
    recent_seen = 0
    unassigned_seen = 0
    for index, session in enumerate(sessions):
        if _is_cli_session_for_settings(session):
            project_id = str(session.get("project_id") or "").strip()
            if not project_id:
                unassigned_seen += 1
                if unassigned_seen > cli_cap:
                    continue
                recent_seen += 1
            else:
                if drawn is not None and index not in drawn:
                    continue
                if recent_seen >= cli_cap:
                    session = dict(session)
                    session["default_hidden"] = True
                else:
                    recent_seen += 1
        kept.append(session)
    return kept


def _merge_cli_sidebar_metadata(ui_session: dict, cli_meta: dict) -> dict:
    """Merge source-of-truth CLI metadata into a sidebar session row.

    Preserve UI-owned state (archived/pinned) while replacing metadata that can
    legitimately drift in WebUI snapshots.
    """
    if not ui_session:
        return ui_session
    if not cli_meta:
        return dict(ui_session)
    merged = dict(ui_session)
    # Only preserve the CLI flag when the imported metadata is actually a CLI
    # row. WebUI sessions are also mirrored into state.db; treating every
    # matching state row as CLI hides long WebUI continuations from the default
    # sidebar source tab.
    merged["is_cli_session"] = is_cli_session_row(cli_meta)
    for key in (
        "source_tag",
        "raw_source",
        "session_source",
        "source_label",
        "user_id",
        "chat_id",
        "chat_type",
        "thread_id",
        "session_key",
        "platform",
        "parent_session_id",
        "end_reason",
        "actual_message_count",
        "_lineage_root_id",
        "_lineage_tip_id",
        "_compression_segment_count",
    ):
        value = _safe_first(cli_meta.get(key))
        if value:
            merged[key] = value

    if cli_meta.get("created_at") is not None:
        merged["created_at"] = cli_meta["created_at"]
    if cli_meta.get("updated_at") is not None:
        merged["updated_at"] = cli_meta["updated_at"]
    if cli_meta.get("last_message_at") is not None:
        merged["last_message_at"] = cli_meta["last_message_at"]
    if cli_meta.get("message_count") is not None:
        merged["message_count"] = max(
            _numeric_count(merged.get("message_count")),
            _numeric_count(cli_meta.get("message_count")),
        )
    elif cli_meta.get("actual_message_count") is not None:
        merged["message_count"] = max(
            _numeric_count(merged.get("message_count")),
            _numeric_count(cli_meta.get("actual_message_count")),
        )

    if cli_meta.get("title"):
        current_title = merged.get("title")
        if not current_title or current_title == "Untitled":
            merged["title"] = cli_meta["title"]

    if cli_meta.get("model"):
        if not merged.get("model") or merged.get("model") == "unknown":
            merged["model"] = cli_meta["model"]
    return merged


def _messaging_source_key(session: dict) -> str | None:
    raw = _session_messaging_raw_source(session)
    if not _is_known_messaging_source(raw):
        return None
    return _messaging_session_identity(session, raw)


def _keep_latest_messaging_session_per_source(
    sessions: list[dict],
    *,
    show_previous_messaging_sessions: bool = False,
) -> list[dict]:
    """Keep only the newest sidebar row per messaging session identity."""
    if show_previous_messaging_sessions:
        return sorted(sessions, key=_session_sort_timestamp, reverse=True)

    gateway_metadata = _load_gateway_session_identity_map()
    active_gateway_session_ids = {str(sid) for sid in gateway_metadata.keys() if sid}
    session_ids = {
        _safe_first(session.get("session_id"))
        for session in sessions
        if isinstance(session, dict)
    }
    visible_active_gateway_session_ids = active_gateway_session_ids & session_ids
    active_gateway_sources = {
        _normalize_messaging_source(_safe_first(meta.get("raw_source"), meta.get("platform")))
        for sid, meta in gateway_metadata.items()
        if sid in visible_active_gateway_session_ids and isinstance(meta, dict)
    }
    active_gateway_sources = {source for source in active_gateway_sources if _is_known_messaging_source(source)}

    kept_sources: set[str] = set()
    best_by_source: dict[str, dict] = {}
    kept: list[dict] = []
    for session in sessions:
        key = _messaging_source_key(session)
        if not key:
            kept.append(session)
            continue
        if _should_hide_stale_messaging_session(session, visible_active_gateway_session_ids, active_gateway_sources):
            continue
        if key in kept_sources:
            kept_sources.add(key)
            current = best_by_source.get(key)
            if current is None or _session_sort_timestamp(session) > _session_sort_timestamp(current):
                best_by_source[key] = session
            continue
        kept_sources.add(key)
        best_by_source[key] = session

    kept.extend(best_by_source.values())
    kept.sort(key=_session_sort_timestamp, reverse=True)
    return kept


from api.models import (
    Session,
    get_session,
    get_session_for_scan,
    find_compression_recovery_session,
    get_session_for_file_ops,
    persist_recovered_workspace_binding,
    WorkspaceBindingPersistenceError,
    new_session,
    all_sessions,
    title_from,
    SESSION_INDEX_FILE,
    _active_state_db_path,
    load_projects,
    save_projects,
    import_cli_session,
    CLAUDE_CODE_SOURCE,
    get_cli_sessions,
    get_cli_session_messages,
    get_state_db_session_messages,
    get_state_db_session_message_prefix_summary,
    get_state_db_session_message_keys_before_timestamp,
    get_state_db_session_summary,
    merge_session_messages_append_only,
    _project_native_image_payload_conflicts_for_display,
    _suppress_native_image_display_mirrors,
    _reconcile_api_content_sidecars,
    _enrich_sidebar_lineage_metadata,
    _active_stream_ids,
    _evict_sessions_over_cap,
    _merge_session_display_metadata,
    _session_message_merge_key,
    _session_messages_have_prefix,
    _session_message_visible_key,
    _message_timestamp_as_float,
    _is_empty_partial_activity_message,
    _hide_from_default_sidebar,
    prune_session_from_index,
    agent_session_rows_existing,
    agent_session_zero_message_sids,
    _load_webui_zero_message_orphan_tombstone,
    _record_webui_zero_message_orphan_tombstone,
    _clear_webui_zero_message_orphan_tombstone,
    _load_webui_deleted_session_tombstone,
    _record_webui_deleted_session_tombstone,
    ensure_cron_project,
    _profile_has_user_projects,
    is_cron_session,
    is_safe_session_id,
    PROCESS_WAKEUP_PAUSE_ERROR,
    clear_process_wakeup_pause,
    clear_process_wakeup_pause_if_model_changed,
    process_wakeup_pause_matches,
    process_wakeup_credential_state_fingerprint,
    process_wakeup_pause_credential_state_changed,
    suppress_process_wakeup_for_provider_pause,
)


_COMPRESSION_RECOVERY_START_LOCK = threading.Lock()


def _pre_compression_continuation_session_id(session) -> str | None:
    """Return the newest visible descendant for a hidden compression snapshot.

    Mobile browsers can miss the final SSE `done` handoff while backgrounded.
    On reload they may request the archived pre-compression session id from the
    stale URL/localStorage. The old snapshot is intentionally hidden from the
    sidebar, so expose a lightweight recovery hint when a child continuation
    exists either in memory or on disk. Follow bounded snapshot-to-snapshot hops
    so repeated compression still lands on the latest visible continuation.
    """
    from api.compression_continuation import durable_compression_continuation

    sealed, tip = durable_compression_continuation(session)
    if sealed:
        return tip
    if not getattr(session, "pre_compression_snapshot", False):
        return None
    sid = _safe_first(getattr(session, "session_id", None))
    if not sid:
        return None
    # #2980 hardening: the resolved continuation is written to the client's
    # URL/localStorage, so it must stay within the requested snapshot's own
    # profile. Children are matched only by parent_session_id below; a
    # crafted/corrupt foreign-profile sidecar whose parent_session_id collided
    # with this snapshot's id would otherwise leak cross-profile. Pin the
    # snapshot's profile and reject any child that isn't profile-matched.
    snapshot_profile = getattr(session, "profile", None)

    def _child_rows_from_memory(seen_ids: set[str]) -> list:
        rows = []
        try:
            with LOCK:
                memory_sessions = list(SESSIONS.values())
            for child in memory_sessions:
                child_sid = _safe_first(getattr(child, "session_id", None))
                if not child_sid or child_sid in seen_ids:
                    continue
                seen_ids.add(child_sid)
                rows.append(child)
        except Exception:
            pass
        return rows

    def _child_rows_from_index(seen_ids: set[str]) -> list | None:
        if not SESSION_INDEX_FILE.exists():
            return None
        try:
            entries = json.loads(SESSION_INDEX_FILE.read_bytes())
        except Exception:
            return None
        if not isinstance(entries, list):
            return None
        try:
            persisted_sidecar_ids = {
                path.stem
                for path in SESSION_DIR.glob("*.json")
                if not path.name.startswith("_") and is_safe_session_id(path.stem)
            }
        except Exception:
            return None
        indexed_ids: set[str] = set()
        row_seen_ids = set(seen_ids)
        rows = []
        for entry in entries:
            if not isinstance(entry, dict):
                continue
            child_sid = _safe_first(entry.get("session_id"))
            if not child_sid or not is_safe_session_id(child_sid):
                continue
            indexed_ids.add(child_sid)
            if child_sid in row_seen_ids or not _safe_first(entry.get("parent_session_id")):
                continue
            row_seen_ids.add(child_sid)
            rows.append(entry)
        # Guarantee here is index MEMBERSHIP-completeness, not per-entry content
        # freshness: if any persisted continuation sidecar is absent from the index
        # we bail to the full scan. A sidecar that IS in the index but whose entry is
        # content-stale (mid-write) still yields a valid continuation of the same
        # snapshot; proving freshness would require reading every sidecar, defeating
        # the optimization, so membership-completeness is the intended bar.
        if persisted_sidecar_ids - indexed_ids - seen_ids:
            return None
        return rows

    def _child_rows_from_sidecars(seen_ids: set[str]) -> list:
        rows = []
        try:
            for path in SESSION_DIR.glob("*.json"):
                if path.name.startswith("_"):
                    continue
                child_sid = path.stem
                if not child_sid or child_sid in seen_ids:
                    continue
                child = Session.load_metadata_only(child_sid)
                if child:
                    seen_ids.add(child_sid)
                    rows.append(child)
        except Exception:
            pass
        return rows

    def _row_value(row, key, default=None):
        return row.get(key, default) if isinstance(row, dict) else getattr(row, key, default)

    def _row_has_backing_state(row) -> bool:
        child_sid = _safe_first(_row_value(row, "session_id"))
        if not child_sid or not is_safe_session_id(child_sid):
            return False
        if not isinstance(row, dict):
            return True
        return (SESSION_DIR / f"{child_sid}.json").exists()

    def _resolve_from_rows(rows: list) -> str | None:
        children_by_parent: dict[str, list] = {}
        for child in rows:
            parent_sid = _safe_first(_row_value(child, "parent_session_id"))
            child_sid = _safe_first(_row_value(child, "session_id"))
            if not parent_sid or not child_sid or child_sid == sid:
                continue
            # Cross-profile guard: only follow continuations within the snapshot's profile.
            if not _profiles_match(_row_value(child, "profile"), snapshot_profile):
                continue
            children_by_parent.setdefault(parent_sid, []).append(child)

        candidates = []
        frontier = [sid]
        seen = {sid}
        for _ in range(20):
            if not frontier:
                break
            parent_sid = frontier.pop(0)
            for child in children_by_parent.get(parent_sid, []):
                child_sid = _safe_first(_row_value(child, "session_id"))
                if not child_sid or child_sid in seen or not _row_has_backing_state(child):
                    continue
                seen.add(child_sid)
                if _row_value(child, "pre_compression_snapshot", False):
                    frontier.append(child_sid)
                else:
                    candidates.append(child)

        if not candidates:
            return None
        latest = max(
            candidates,
            key=lambda child: (
                float(
                    _safe_first(
                        _row_value(child, "updated_at"),
                        _row_value(child, "created_at"),
                        0,
                    ) or 0
                ),
                # Secondary tiebreaker so the index-fast-path and the sidecar-scan
                # path resolve byte-identically on an updated_at/created_at tie
                # (otherwise the chosen sid could differ by iteration order).
                str(_safe_first(_row_value(child, "session_id"), "") or ""),
            ),
        )
        latest_sid = _safe_first(_row_value(latest, "session_id", None)) or None
        # Only hand the client a well-formed session id (it gets written to URL/localStorage).
        if latest_sid and not is_safe_session_id(latest_sid):
            return None
        return latest_sid

    memory_seen_ids: set[str] = set()
    rows = _child_rows_from_memory(memory_seen_ids)
    index_rows = _child_rows_from_index(memory_seen_ids)
    if index_rows is not None:
        return _resolve_from_rows(rows + index_rows)

    rows.extend(_child_rows_from_sidecars(memory_seen_ids))
    return _resolve_from_rows(rows)

from api.workspace import (
    load_workspaces,
    save_workspaces,
    get_last_workspace,
    get_profile_default_workspace,
    set_last_workspace,
    git_info_for_workspace,
    authorize_escape_target,
    EscapeAuthorizationExpiredError,
    list_dir,
    list_authorized_escape_dir,
    serialize_workspace_entries_for_browser,
    dir_signature,
    list_workspace_suggestions,
    read_file_content,
    read_authorized_escape_file_content,
    safe_resolve_ws,
    raw_authorized_escape_target,
    resolve_trusted_workspace,
    _resolve_path,
    resolve_implicit_workspace_with_recovery,
    open_anchored_fd,
    open_anchored_create_fd,
    open_anchored_write_fd,
    unlink_anchored,
    rmtree_anchored,
    rename_anchored,
    make_anchored_dir,
    validate_workspace_to_add,
    _is_blocked_system_path,
    _home_path,
    _is_within,
    _strip_surrounding_quotes,
    _is_remote_terminal_backend,
    _workspace_blocked_roots,
)
from api.upload import (
    handle_upload,
    handle_upload_extract,
    handle_transcribe,
    handle_transcribe_capability,
    handle_workspace_upload,
)
from api.streaming import (
    _sse,
    _sse_set_write_deadline,
    _run_agent_streaming,
    cancel_stream,
    _materialize_pending_user_turn_before_error,
    generate_session_title_for_session,
    _compact_for_echo_compare,
    _CompactEchoIndex,
)
from api.gateway_chat import _run_gateway_chat_streaming, webui_gateway_chat_enabled
from api.run_journal import (
    runtime_model_from_events,
    _parse_run_journal_event_id as _shared_parse_run_journal_event_id,
    _summary_from_events,
    bound_run_journal_snapshot_args,
    find_run_file,
    find_run_summary,
    journal_replay_visible,
    read_run_events,
    read_session_run_events,
    session_journal_fingerprint,
    stale_interrupted_event,
    SSE_RELAY_CLOSE_EVENTS,
)
from api.todo_state import attach_todo_state
from api.providers import (
    get_providers,
    get_provider_quota,
    get_provider_cost_history,
    provider_has_process_wakeup_recovery_credential,
    set_provider_key,
    remove_provider_key,
)
from api.onboarding import (
    apply_onboarding_setup,
    get_onboarding_status,
    complete_onboarding,
    probe_provider_endpoint,
)
from api.oauth import (
    cancel_onboarding_oauth_flow,
    poll_onboarding_oauth_flow,
    start_onboarding_oauth_flow,
)

# Approval system -- state and helpers live in api.route_approvals; imported
# here for backward compatibility so existing call sites continue to resolve.
from api.route_approvals import (  # noqa: F401 — re-exports for backward compat
    _submit_pending_raw,
    approve_session,
    approve_permanent,
    save_permanent_allowlist,
    is_approved,
    _pending,
    _lock,
    _permanent_approved,
    _gateway_queues,
    resolve_gateway_approval,
    enable_session_yolo,
    disable_session_yolo,
    is_session_yolo_enabled,
    _approval_sse_subscribers,
    _approval_sse_subscribe,
    _approval_sse_unsubscribe,
    _approval_sse_notify_locked,
    _approval_sse_notify,
    _GATEWAY_AGENT_IDENTITY_V1,
    _GATEWAY_MIRROR_FLAG,
    _GATEWAY_MIRROR_TOKEN,
    _gateway_mirror_entry_token,
    gateway_yolo_handoff,
    begin_session_yolo_transition,
    claim_gateway_approval_relay_owner,
    finish_session_yolo_transition,
    gateway_pending_mirror,
    gateway_pending_mirrors,
    release_gateway_approval_relay_owner,
    retire_gateway_pending_mirror,
    settle_gateway_pending_run,
    reconcile_gateway_pending_mirror_locked,
    resolve_gateway_pending_local,
    resolve_gateway_pending_run,
    resolve_gateway_pending_local_all,
    resolve_gateway_pending_local_no_run_mirror,
    set_session_yolo_enabled,
    submit_gateway_pending_mirror,
    submit_pending,
)

# Clarify prompts (optional -- graceful fallback if agent not available)
try:
    from api.clarify import (
        submit_pending as submit_clarify_pending,
        get_pending as get_clarify_pending,
        pending_count as get_clarify_pending_count,
        resolve_clarify,
        resolve_clarify_by_id,
        sse_subscribe as clarify_sse_subscribe,
        sse_unsubscribe as clarify_sse_unsubscribe,
    )
except ImportError:
    submit_clarify_pending = lambda *a, **k: None
    get_clarify_pending = lambda *a, **k: None
    get_clarify_pending_count = lambda *a, **k: 0
    clarify_sse_subscribe = None
    resolve_clarify = lambda *a, **k: 0
    resolve_clarify_by_id = lambda *a, **k: False


def _session_attention_summary(session_id: str) -> dict | None:
    """Return sidebar attention metadata for pending approval/clarify work."""
    approval_count = 0
    with _lock:
        reconcile_gateway_pending_mirror_locked(session_id)
        queue_list = _pending.get(session_id)
        if isinstance(queue_list, list):
            approval_count = len(queue_list)
        elif queue_list:
            approval_count = 1
    if approval_count > 0:
        return {
            "kind": "approval",
            "count": approval_count,
            "severity": "critical",
        }

    clarify_count = int(get_clarify_pending_count(session_id) or 0)
    if clarify_count > 0:
        return {
            "kind": "clarify",
            "count": clarify_count,
            "severity": "question",
        }
    return None


# One canonical allowlist owns both cache projection and final serialization.
# Keeping it in the cache module avoids a circular-import fallback that could
# silently truncate otherwise valid sidebar fields.
_SIDEBAR_SESSION_RESPONSE_FIELDS = (
    _route_session_list_cache._SIDEBAR_SESSION_RESPONSE_FIELDS
)


def _sidebar_session_response_item(session: dict, *, redact_enabled: bool | None = None) -> dict:
    """Return the bounded /api/sessions row shape used by the sidebar.

    Full session/detail fields such as messages, tool calls, compression
    summaries, context-engine state, gateway routing history, drafts, and
    pending user text are intentionally excluded from the list endpoint. Large
    installs should not ship tens of KB of per-row detail just to render a
    conversation title.
    """
    item = {
        key: value
        for key, value in dict(session).items()
        if key in _SIDEBAR_SESSION_RESPONSE_FIELDS
    }
    if isinstance(item.get("title"), str):
        item["title"] = _redact_text(item["title"], _enabled=redact_enabled)
    _redact_sidebar_title_fields(item, redact_enabled)
    item["attention"] = _session_attention_summary(str(item.get("session_id") or ""))
    return item


def _redact_sidebar_title_fields(item: dict, redact_enabled: bool | None = None) -> None:
    """Redact every user-content-derived title field on a sidebar/search row in place.

    `title` is redacted by the callers directly (they special-case it), but
    `display_title`, `_state_db_title`, and `parent_title` can ALSO carry raw
    user-message-derived text — e.g. #6056 derives a delegated subagent's
    `display_title` from its first user message, and `parent_title` copies a
    parent session's (possibly derived) title. Without this a credential-shaped
    value in a delegated goal would surface in the sidebar / search results even
    with `api_redact_enabled=True`. Shared by `_sidebar_session_response_item`
    (`/api/sessions`) and every `/api/sessions/search` response branch so the two
    endpoints can never drift on which fields get redacted.
    """
    for field in ("display_title", "_state_db_title", "parent_title"):
        value = item.get(field)
        if isinstance(value, str):
            item[field] = _redact_text(value, _enabled=redact_enabled)


# ── Login page locale strings ─────────────────────────────────────────────────
# Add entries here to support more languages on the login page.
# The key must match the 'language' setting value (from static/i18n.js LOCALES).
_LOGIN_LOCALE = {
    "en": {
        "lang": "en",
        "title": "Sign in",
        "subtitle": "Enter your password to continue",
        "placeholder": "Password",
        "btn": "Sign in",
        "invalid_pw": "Invalid password",
        "conn_failed": "Connection failed",
    },
    "fr": {
        "lang": "fr-FR",
        "title": "Se connecter",
        "subtitle": "Entrez votre mot de passe pour continuer",
        "placeholder": "Mot de passe",
        "btn": "Se connecter",
        "invalid_pw": "Mot de passe invalide",
        "conn_failed": "\u00c9chec de la connexion",
    },
    "es": {
        "lang": "es-ES",
        "title": "Iniciar sesi\u00f3n",
        "subtitle": "Introduce tu contrase\u00f1a para continuar",
        "placeholder": "Contrase\u00f1a",
        "btn": "Entrar",
        "invalid_pw": "Contrase\u00f1a inv\u00e1lida",
        "conn_failed": "Error de conexi\u00f3n",
    },
    "de": {
        "lang": "de-DE",
        "title": "Anmelden",
        "subtitle": "Geben Sie Ihr Passwort ein, um fortzufahren",
        "placeholder": "Passwort",
        "btn": "Anmelden",
        "invalid_pw": "Ung\u00fcltiges Passwort",
        "conn_failed": "Verbindung fehlgeschlagen",
    },
    "ru": {
        "lang": "ru-RU",
        "title": "\u0412\u043e\u0439\u0442\u0438",
        "subtitle": "\u0412\u0432\u0435\u0434\u0438\u0442\u0435 \u043f\u0430\u0440\u043e\u043b\u044c, \u0447\u0442\u043e\u0431\u044b \u043f\u0440\u043e\u0434\u043e\u043b\u0436\u0438\u0442\u044c",
        "placeholder": "\u041f\u0430\u0440\u043e\u043b\u044c",
        "btn": "\u0412\u043e\u0439\u0442\u0438",
        "invalid_pw": "\u041d\u0435\u0432\u0435\u0440\u043d\u044b\u0439 \u043f\u0430\u0440\u043e\u043b\u044c",
        "conn_failed": "\u041d\u0435 \u0443\u0434\u0430\u043b\u043e\u0441\u044c \u043f\u043e\u0434\u043a\u043b\u044e\u0447\u0438\u0442\u044c\u0441\u044f",
    },
    "zh": {
        "lang": "zh-CN",
        "title": "\u767b\u5f55",
        "subtitle": "\u8f93\u5165\u5bc6\u7801\u7ee7\u7eed\u4f7f\u7528",
        "placeholder": "\u5bc6\u7801",
        "btn": "\u767b\u5f55",
        "invalid_pw": "\u5bc6\u7801\u9519\u8bef",
        "conn_failed": "\u8fde\u63a5\u5931\u8d25",
    },
    "zh-Hant": {
        "lang": "zh-TW",
        "title": "\u767b\u5f55",
        "subtitle": "\u8f38\u5165\u5bc6\u78bc\u7e7c\u7e8c\u4f7f\u7528",
        "placeholder": "\u5bc6\u78bc",
        "btn": "\u767b\u5f55",
        "invalid_pw": "\u5bc6\u78bc\u932f\u8aa4",
        "conn_failed": "\u9023\u63a5\u5931\u6557",
    },
    # Strings mirror static/i18n.js login_* keys for the corresponding locale.
    # See issue #1442. When adding a new locale to LOCALES in i18n.js, also add
    # the matching entry here — tests/test_login_locale_parity.py enforces this.
    "it": {
        "lang": "it-IT",
        "title": "Accedi",
        "subtitle": "Inserisci la password per continuare",
        "placeholder": "Password",
        "btn": "Accedi",
        "invalid_pw": "Password non valida",
        "conn_failed": "Connessione fallita",
    },
    "ja": {
        "lang": "ja-JP",
        "title": "\u30b5\u30a4\u30f3\u30a4\u30f3",
        "subtitle": "\u30d1\u30b9\u30ef\u30fc\u30c9\u3092\u5165\u529b\u3057\u3066\u7d9a\u884c",
        "placeholder": "\u30d1\u30b9\u30ef\u30fc\u30c9",
        "btn": "\u30b5\u30a4\u30f3\u30a4\u30f3",
        "invalid_pw": "\u30d1\u30b9\u30ef\u30fc\u30c9\u304c\u7121\u52b9\u3067\u3059",
        "conn_failed": "\u63a5\u7d9a\u5931\u6557",
    },
    "pt": {
        "lang": "pt-BR",
        "title": "Entrar",
        "subtitle": "Digite sua senha para continuar",
        "placeholder": "Senha",
        "btn": "Entrar",
        "invalid_pw": "Senha inv\u00e1lida",
        "conn_failed": "Falha na conex\u00e3o",
    },
    "ko": {
        "lang": "ko-KR",
        "title": "\ub85c\uadf8\uc778",
        "subtitle": "\uacc4\uc18d\ud558\ub824\uba74 \ube44\ubc00\ubc88\ud638\ub97c \uc785\ub825\ud558\uc138\uc694",
        "placeholder": "\ube44\ubc00\ubc88\ud638",
        "btn": "\ub85c\uadf8\uc778",
        "invalid_pw": "\ube44\ubc00\ubc88\ud638\uac00 \uc62c\ubc14\ub974\uc9c0 \uc54a\uc2b5\ub2c8\ub2e4",
        "conn_failed": "\uc5f0\uacb0 \uc2e4\ud328",
    },
    "tr": {
        "lang": "tr-TR",
        "title": "Oturum a\u00e7",
        "subtitle": "Devam etmek i\u00e7in \u015fifrenizi girin",
        "placeholder": "\u015eifre",
        "btn": "Oturum a\u00e7",
        "invalid_pw": "Ge\u00e7ersiz \u015fifre",
        "conn_failed": "Ba\u011flant\u0131 ba\u015far\u0131s\u0131z",
    },
    "pl": {
        "lang": "pl-PL",
        "title": "Zaloguj si\u0119",
        "subtitle": "Wpisz has\u0142o, aby kontynuowa\u0107",
        "placeholder": "Has\u0142o",
        "btn": "Zaloguj si\u0119",
        "invalid_pw": "Nieprawid\u0142owe has\u0142o",
        "conn_failed": "Po\u0142\u0105czenie nie powiod\u0142o si\u0119",
    },
    "vi": {
        "lang": "vi",
        "title": "\u0110\u0103ng nh\u1eadp",
        "subtitle": "Nh\u1eadp m\u1eadt kh\u1ea9u c\u1ee7a b\u1ea1n \u0111\u1ec3 ti\u1ebfp t\u1ee5c",
        "placeholder": "M\u1eadt kh\u1ea9u",
        "btn": "\u0110\u0103ng nh\u1eadp",
        "invalid_pw": "M\u1eadt kh\u1ea9u kh\u00f4ng h\u1ee3p l\u1ec7",
        "conn_failed": "K\u1ebft n\u1ed1i th\u1ea5t b\u1ea1i",
    },
    "cs": {
        "lang": "cs-CZ",
        "title": "P\u0159ihl\u00e1sit se",
        "subtitle": "Zadejte heslo pro pokra\u010dov\u00e1n\u00ed",
        "placeholder": "Heslo",
        "btn": "P\u0159ihl\u00e1sit se",
        "invalid_pw": "Neplatn\u00e9 heslo",
        "conn_failed": "P\u0159ipojen\u00ed selhalo",
    },
}


def _resolve_login_locale_key(raw_lang: str | None) -> str:
    """Resolve settings.language to a known _LOGIN_LOCALE key."""
    if not raw_lang:
        return "en"
    lang = str(raw_lang).strip()
    if not lang:
        return "en"
    if lang in _LOGIN_LOCALE:
        return lang

    normalized = lang.replace("_", "-")
    lower = normalized.lower()

    # Case-insensitive direct key match first.
    for key in _LOGIN_LOCALE:
        if key.lower() == lower:
            return key

    # Common Chinese aliases.
    if lower == "zh" or lower.startswith("zh-cn") or lower.startswith("zh-sg") or lower.startswith("zh-hans"):
        return "zh"
    if lower.startswith("zh-tw") or lower.startswith("zh-hk") or lower.startswith("zh-mo") or lower.startswith("zh-hant"):
        return "zh-Hant" if "zh-Hant" in _LOGIN_LOCALE else "zh"

    # Fallback to base language subtag (e.g. en-US -> en).
    base = lower.split("-", 1)[0]
    for key in _LOGIN_LOCALE:
        if key.lower() == base:
            return key
    return "en"

# ── Login page (self-contained, no external deps) ────────────────────────────
_LOGIN_PAGE_HTML = """<!doctype html>
<html lang="{{LANG}}"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>{{BOT_NAME}} — {{LOGIN_TITLE}}</title>
<style>
*{box-sizing:border-box;margin:0;padding:0}
body{background:#1a1a2e;color:#e8e8f0;font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",system-ui,sans-serif;
  height:100vh;display:flex;align-items:center;justify-content:center}
.card{background:#16213e;border:1px solid rgba(255,255,255,.08);border-radius:16px;padding:36px 32px;
  width:320px;text-align:center;box-shadow:0 8px 32px rgba(0,0,0,.3)}
.logo{width:48px;height:48px;border-radius:12px;background:linear-gradient(145deg,#e8a030,#e94560);
  display:flex;align-items:center;justify-content:center;font-weight:800;font-size:20px;color:#fff;
  margin:0 auto 12px;box-shadow:0 2px 12px rgba(233,69,96,.3)}
h1{font-size:18px;font-weight:600;margin-bottom:4px}
.sub{font-size:12px;color:#8888aa;margin-bottom:24px}
input{width:100%;padding:10px 14px;border-radius:10px;border:1px solid rgba(255,255,255,.1);
  background:rgba(255,255,255,.04);color:#e8e8f0;font-size:14px;outline:none;margin-bottom:14px;
  transition:border-color .15s}
input:focus{border-color:rgba(124,185,255,.5);box-shadow:0 0 0 3px rgba(124,185,255,.1)}
button{width:100%;padding:10px;border-radius:10px;border:none;background:rgba(124,185,255,.15);
  border:1px solid rgba(124,185,255,.3);color:#7cb9ff;font-size:14px;font-weight:600;cursor:pointer;
  transition:all .15s}
button:hover{background:rgba(124,185,255,.25)}
.oidc-login{display:block;margin-top:10px;padding:10px;border-radius:10px;text-decoration:none;
  background:rgba(255,255,255,.04);border:1px solid rgba(111,214,164,.35);color:#6fd6a4;
  font-size:14px;font-weight:600;cursor:pointer;transition:all .15s}
.oidc-login:hover{background:rgba(111,214,164,.12)}
.passkey-login{margin-top:10px;background:rgba(255,255,255,.04);border-color:rgba(232,160,48,.35);color:#e8a030}
.err{color:#e94560;font-size:12px;margin-top:10px;display:none}
</style></head><body>
<div class="card">
  <div class="logo">{{BOT_NAME_INITIAL}}</div>
  <h1>{{BOT_NAME}}</h1>
  <p class="sub">{{LOGIN_SUBTITLE}}</p>
  <form id="login-form" data-invalid-pw="{{LOGIN_INVALID_PW}}" data-conn-failed="{{LOGIN_CONN_FAILED}}">
    {{PASSWORD_FORM_HTML}}
    {{OIDC_LOGIN_HTML}}
  </form>
  <div class="err" id="err"></div>
</div>
<!-- Keep login.js relative so subpath mounts load it under the current scope. -->
<script src="static/login.js?v={{WEBUI_VERSION}}"></script>
</body></html>"""


def _safe_login_redirect_path(raw_path: str | None) -> str:
    path = str(raw_path or "").strip()
    if not path:
        return "/"
    if path[0] != "/":
        return "/"
    if path[1:2] in {"/", "\\"}:
        return "/"
    if re.search(r"[\x00-\x1f\x7f\s]", path):
        return "/"
    # #5578: reject a `next` that points back at the login page, so an
    # expired-auth bounce on the login page can't feed the redirect its own
    # address and grow the URL exponentially. Length cap is belt-and-suspenders:
    # a legitimate app path is never this long.
    if len(path) > 2048:
        return "/"
    # Detect a login-route target even through nested percent-encoding: a nested
    # login-redirect chain looks like `/session/login%3Fnext%3D...`, where the
    # `?` separating the path from the query is itself encoded, so a plain
    # split("?") wouldn't isolate the real path. Fully decode (bounded) and check
    # the leading PATH of EVERY decode level, including the final fully-decoded
    # form. Only collapse login-route chains — a legitimate non-login path that
    # merely carries its own `next=` query key (e.g. `/admin?action=foo&next=/x`)
    # must still round-trip (regression guarded by test_v050258_opus_followups.py).
    _probe = path
    for _ in range(8):
        _path_only = _probe.split("?", 1)[0].split("#", 1)[0].split("&", 1)[0].rstrip("/")
        if _path_only.endswith("/login") or _path_only == "/login":
            return "/"
        _decoded = unquote(_probe)
        if _decoded == _probe:
            break
        _probe = _decoded
    else:
        # Loop exhausted the cap while STILL decoding (pathologically deep
        # encoding): check the final decoded form too, then fail closed — an
        # 8-level-deep encoded value is never a legitimate redirect.
        _path_only = _probe.split("?", 1)[0].split("#", 1)[0].split("&", 1)[0].rstrip("/")
        if _path_only.endswith("/login") or _path_only == "/login":
            return "/"
        return "/"
    return path


def _request_base_url(handler) -> str:
    from api.auth import _is_secure_context

    scheme = "https" if _is_secure_context(handler) else "http"
    host = str(handler.headers.get("Host") or "").strip() or "127.0.0.1:8787"
    return f"{scheme}://{host}"


def _oidc_login_html(parsed) -> str:
    try:
        from api.auth_oidc import is_oidc_enabled
    except Exception:
        return ""
    if not is_oidc_enabled():
        return ""
    next_path = _safe_login_redirect_path(
        parse_qs(parsed.query or "").get("next", [""])[0]
    )
    href = "/api/auth/oidc/start"
    if next_path != "/":
        href += "?next=" + quote(next_path, safe="/")
    return (
        '<a id="oidc-login" class="oidc-login" '
        f'href="{_html.escape(href, quote=True)}">Continue with SSO</a>'
    )


# ── Logs endpoint ─────────────────────────────────────────────────────────────
_LOG_FILE_WHITELIST = {
    "agent": "agent.log",
    "errors": "errors.log",
    "gateway": "gateway.log",
}
_LOG_TAIL_VALUES = {100, 200, 500, 1000}
_LOG_DEFAULT_TAIL = 200
_LOG_MAX_BYTES = 4 * 1024 * 1024


def _normalize_logs_tail(raw_tail) -> int:
    try:
        tail = int(str(raw_tail or "").strip())
    except (TypeError, ValueError):
        return _LOG_DEFAULT_TAIL
    return tail if tail in _LOG_TAIL_VALUES else _LOG_DEFAULT_TAIL


def _handle_logs(handler, parsed) -> bool:
    """Return a bounded tail window for an active-profile Hermes log file."""
    query = parse_qs(parsed.query)
    file_key = (query.get("file", ["agent"])[0] or "agent").strip().lower()
    filename = _LOG_FILE_WHITELIST.get(file_key)
    if not filename:
        return bad(handler, "Unknown log file", status=400)

    tail = _normalize_logs_tail(query.get("tail", [None])[0])
    try:
        from api.profiles import get_active_hermes_home

        hermes_home = Path(get_active_hermes_home()).expanduser()
    except Exception:
        hermes_home = Path(os.environ.get("HERMES_HOME") or (Path.home() / ".hermes")).expanduser()

    log_dir = hermes_home / "logs"
    log_path = log_dir / filename
    try:
        # Defense in depth: the filename is hardcoded above, but keep the final
        # path anchored under the active profile's logs directory.
        if log_path.resolve(strict=False).parent != log_dir.resolve(strict=False):
            return bad(handler, "Invalid log file", status=400)
        if not log_path.exists() or not log_path.is_file():
            return j(handler, {
                "file": file_key,
                "tail": tail,
                "lines": [],
                "truncated": False,
                "total_bytes": 0,
                "mtime": None,
                "hint": f"Log file for {file_key} not found yet.",
            })
        st = log_path.stat()
        total_bytes = int(st.st_size)
        read_bytes = min(total_bytes, _LOG_MAX_BYTES)
        with log_path.open("rb") as fh:
            if total_bytes > read_bytes:
                fh.seek(total_bytes - read_bytes)
            raw = fh.read(read_bytes)
        text = raw.decode("utf-8", errors="replace")
        lines = text.splitlines()[-tail:]
        return j(handler, {
            "file": file_key,
            "tail": tail,
            "lines": lines,
            "truncated": total_bytes > read_bytes,
            "total_bytes": total_bytes,
            "mtime": st.st_mtime,
            "hint": "",
        })
    except Exception as exc:
        logger.exception("Failed to read whitelisted log file %s", file_key)
        return bad(handler, _sanitize_error(exc), status=500)

# ── Insights endpoint ──────────────────────────────────────────────────────────

_LLM_WIKI_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/user-guide/skills/bundled/research/research-llm-wiki"
_LLM_WIKI_PAGE_DIRS = ("entities", "concepts", "comparisons", "queries")


def _llm_wiki_active_hermes_home() -> Path:
    try:
        from api.profiles import get_active_hermes_home
        return Path(get_active_hermes_home()).expanduser()
    except Exception:
        return Path(os.getenv("HERMES_HOME", str(Path.home() / ".hermes"))).expanduser()


def _llm_wiki_env_file_path(hermes_home: Path) -> str | None:
    env_path = hermes_home / ".env"
    if not env_path.exists() or not env_path.is_file():
        return None
    try:
        for line in env_path.read_text(encoding="utf-8", errors="replace").splitlines():
            stripped = line.strip()
            if not stripped or stripped.startswith("#") or "=" not in stripped:
                continue
            key, value = stripped.split("=", 1)
            if key.strip() != "WIKI_PATH":
                continue
            value = value.strip().strip('"').strip("'")
            return value or None
    except Exception:
        return None
    return None


def _llm_wiki_get_config_path_value(config: dict, dotted_key: str) -> str | None:
    if not isinstance(config, dict):
        return None
    if dotted_key in config and config.get(dotted_key):
        return str(config.get(dotted_key))
    cur = config
    for part in dotted_key.split("."):
        if not isinstance(cur, dict) or part not in cur:
            return None
        cur = cur[part]
    return str(cur) if cur else None


def _llm_wiki_config_path() -> str | None:
    try:
        from api.config import get_config as _get_cfg
        cfg = _get_cfg()
    except Exception:
        return None
    return (
        _llm_wiki_get_config_path_value(cfg, "skills.config.wiki.path")
        or _llm_wiki_get_config_path_value(cfg, "wiki.path")
    )


# Cap WIKI walks to prevent self-DoS if WIKI_PATH points at /, /etc, /home, etc.
# Real LLM wikis have under a few thousand files; 10k is generous and catches misconfig.
_LLM_WIKI_MAX_FILES = 10000
# Cap a single served wiki page at 2 MiB so a huge/binary file can't be slurped
# wholesale into memory + a JSON response (DoS / memory-blowup guard).
_LLM_WIKI_MAX_PAGE_BYTES = 2 * 1024 * 1024
# Refuse to walk these system roots even if explicitly configured.
_LLM_WIKI_FORBIDDEN_ROOTS = frozenset(
    str(Path(p).expanduser().resolve()) for p in ("/", "/etc", "/usr", "/var", "/opt", "/sys", "/proc")
)
_WIKI_ALLOWLIST_TTL = 5.0  # seconds
_wiki_allowlist_cache: dict[str, dict[str, object]] = {}
_wiki_allowlist_cache_lock = threading.Lock()


def _llm_wiki_resolve_path() -> tuple[Path, str, bool]:
    hermes_home = _llm_wiki_active_hermes_home()
    raw = os.getenv("WIKI_PATH") or _llm_wiki_env_file_path(hermes_home)
    source = "WIKI_PATH" if raw else "default"
    configured = bool(raw)
    if not raw:
        raw = _llm_wiki_config_path()
        if raw:
            source = "skills.config.wiki.path"
            configured = True
    if not raw:
        raw = "~/wiki"
    return Path(os.path.expandvars(raw)).expanduser(), source, configured


def _llm_wiki_safe_iso(ts: float | None) -> str | None:
    if not ts:
        return None
    try:
        from datetime import datetime, timezone
        return datetime.fromtimestamp(ts, tz=timezone.utc).isoformat().replace("+00:00", "Z")
    except Exception:
        return None


def _llm_wiki_count_files(root: Path) -> int:
    if not root.exists() or not root.is_dir():
        return 0
    # Defense in depth: refuse to walk forbidden system roots even if WIKI_PATH
    # was set to one. The endpoint is auth-gated but a misconfigured server
    # shouldn't self-DoS by rglob'ing all of /etc on every Insights load.
    try:
        if str(root.resolve()) in _LLM_WIKI_FORBIDDEN_ROOTS:
            return 0
    except Exception:
        return 0
    count = 0
    iterated = 0
    for item in root.rglob("*"):
        iterated += 1
        if iterated > _LLM_WIKI_MAX_FILES:
            break  # bounded — prevents hangs on symlink loops or huge trees
        try:
            if item.is_file() and not any(part.startswith(".") for part in item.relative_to(root).parts):
                count += 1
        except Exception:
            continue
    return count


def _llm_wiki_page_files_cache_signature(wiki_path: Path) -> tuple:
    """Return a change-detection signature over the configured wiki sections.

    Reads only the section-level directories (not every page), using
    st_mtime_ns, st_dev, and st_ino so that quick section replacements and
    mtime-ns-resolution changes are both caught.  Missing or inaccessible
    section dirs are represented as ``(section, None)`` so their appearance
    or disappearance also invalidates the cache.
    """
    sig = []
    for section in _LLM_WIKI_PAGE_DIRS:
        section_dir = wiki_path / section
        try:
            st = section_dir.lstat()
        except OSError:
            sig.append((section, None))
            continue
        sig.append((section, st.st_dev, st.st_ino, st.st_mtime_ns))
    return tuple(sig)


def _llm_wiki_page_files_uncached(wiki_path: Path) -> list[Path]:
    pages: list[Path] = []
    # Defense in depth: refuse forbidden system roots, and resolve the wiki root
    # ONCE as the single trust base for all containment checks below.
    try:
        wiki_real = wiki_path.resolve()
        if str(wiki_real) in _LLM_WIKI_FORBIDDEN_ROOTS:
            return pages
    except Exception:
        return pages

    def _is_clean_relpath(rel: Path) -> bool:
        # No dot-prefixed segment (dotfile/dotdir) anywhere in the path.
        return not any(part.startswith(".") for part in rel.parts)

    iterated = 0
    for dirname in _LLM_WIKI_PAGE_DIRS:
        section = wiki_path / dirname
        if not section.exists() or not section.is_dir():
            continue
        # The section itself must resolve UNDER the real wiki root — guards a
        # symlinked section (e.g. concepts -> /tmp/outside) from exposing files
        # outside the wiki tree entirely.
        try:
            section_real = section.resolve()
            section_real.relative_to(wiki_real)
        except (OSError, ValueError):
            continue
        for item in section.rglob("*.md"):
            iterated += 1
            if iterated > _LLM_WIKI_MAX_FILES:
                return pages  # bounded
            try:
                rel = item.relative_to(section)
                if not item.is_file() or not _is_clean_relpath(rel):
                    continue
                # Reject multi-link (hardlinked) page files. A hardlink at a
                # clean *.md name can point at an arbitrary inode (incl. one
                # outside the wiki); O_NOFOLLOW + inode-identity at read time
                # cannot tell such a hardlink apart from the real page, so
                # exclude any file with st_nlink > 1 from the allowlist. (#4375)
                try:
                    if item.lstat().st_nlink > 1:
                        continue
                except OSError:
                    continue
                # Resolve the real target and require it to live under BOTH the
                # real wiki root and the real section, with no dot-prefixed
                # segment on the resolved-relative path. This closes symlink
                # escapes whose link name looks like a clean *.md page but whose
                # target is an arbitrary / hidden / out-of-tree file (the read
                # endpoint would otherwise serve it).
                item_real = item.resolve()
                item_real.relative_to(section_real)
                rel_real = item_real.relative_to(wiki_real)
                if not _is_clean_relpath(rel_real):
                    continue
                pages.append(item)
            except (OSError, ValueError):
                continue
    return pages


def _llm_wiki_page_files(wiki_path: Path) -> list[Path]:
    """Return all allowlisted wiki page paths under *wiki_path*.

    Trust boundary: the allowlist uses ``(st_dev, st_ino)`` inode identity.
    A hardlink created at a listed page name *before* this snapshot is taken
    would carry that page's identity through to the caller's fstat check.
    This is only exploitable with write access to the wiki directory, which
    is outside the realistic threat model (the wiki directory is
    operator-controlled).  Defense-in-depth via an ``openat``-chain with
    no-follow directory fds would close the gap but is deferred — the inode
    match already defeats path-swap and symlink races at open time.
    """
    try:
        wiki_root = wiki_path.resolve()
    except OSError:
        wiki_root = wiki_path

    key = str(wiki_root)
    sig = _llm_wiki_page_files_cache_signature(wiki_root)
    now = time.monotonic()

    with _wiki_allowlist_cache_lock:
        cached = _wiki_allowlist_cache.get(key)
        if (
            cached is not None
            and cached.get("signature") == sig
            and now < cached.get("expires_at", 0)
        ):
            return list(cached["files"])

    pages = _llm_wiki_page_files_uncached(wiki_root)

    with _wiki_allowlist_cache_lock:
        _wiki_allowlist_cache[key] = {
            "signature": sig,
            "expires_at": now + _WIKI_ALLOWLIST_TTL,
            "files": tuple(pages),
        }
    return list(pages)


def _llm_wiki_clear_page_files_cache() -> None:
    """Clear the allowlist cache; intended for tests only."""
    with _wiki_allowlist_cache_lock:
        _wiki_allowlist_cache.clear()


def _llm_wiki_allowlisted_entries(wiki_path: Path) -> dict[str, tuple[Path, tuple[int, int]]]:
    """Return listed relpaths mapped to resolved targets plus stable identity."""
    try:
        wiki_real = wiki_path.resolve()
    except OSError:
        return {}

    def _wiki_read_relpath_is_clean(rel: Path) -> bool:
        rel_text = rel.as_posix()
        return bool(rel.parts) and "\\" not in rel_text and not any(part.startswith(".") for part in rel.parts)

    entries: dict[str, tuple[Path, tuple[int, int]]] = {}
    try:
        for listed_path in _llm_wiki_page_files(wiki_real):
            try:
                rel_listed = listed_path.relative_to(wiki_real)
                if not _wiki_read_relpath_is_clean(rel_listed):
                    continue
                section_real = (wiki_real / rel_listed.parts[0]).resolve()
                section_real.relative_to(wiki_real)
                resolved_target = listed_path.resolve()
                resolved_target.relative_to(section_real)
                if not resolved_target.is_file():
                    continue
                # Reject hardlinked targets (st_nlink > 1): a multi-link file at
                # a clean page name can carry an arbitrary inode through the
                # O_NOFOLLOW + identity read check. Defense in depth alongside
                # the same rejection in the allowlist walk. (#4375)
                if resolved_target.stat().st_nlink > 1:
                    continue
                rel_resolved = resolved_target.relative_to(wiki_real)
                if not _wiki_read_relpath_is_clean(rel_resolved):
                    continue
                st0 = resolved_target.stat()
                if listed_path.resolve() != resolved_target:
                    continue
                entries[rel_listed.as_posix()] = (resolved_target, (st0.st_dev, st0.st_ino))
            except (OSError, ValueError):
                continue
    except Exception:
        return {}
    return entries


def _llm_wiki_status_file_entry_stat(wiki_path: Path, path: Path) -> os.stat_result | None:
    """Return the current top-level status-file entry metadata when it is safe to trust."""
    try:
        wiki_root = wiki_path.resolve()
    except OSError:
        wiki_root = wiki_path
    try:
        path.resolve().relative_to(wiki_root)
        st_entry = path.lstat()
    except (OSError, ValueError):
        return None
    if not _stat.S_ISREG(st_entry.st_mode):
        return None
    return st_entry


def _llm_wiki_verified_status_file_stat(wiki_path: Path, path: Path) -> os.stat_result | None:
    """Return identity-checked metadata for a top-level wiki status file."""
    st_entry = _llm_wiki_status_file_entry_stat(wiki_path, path)
    if st_entry is None:
        return None
    fd = None
    try:
        fd = os.open(str(path), os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
        st_open = os.fstat(fd)
        if (st_open.st_dev, st_open.st_ino) != (st_entry.st_dev, st_entry.st_ino):
            return None
        return st_open
    except OSError:
        return None
    finally:
        if fd is not None:
            os.close(fd)


def _llm_wiki_last_writer(
    wiki_path: Path,
    page_files: list[Path] | list[tuple[Path, tuple[int, int]]],
) -> str:
    """Best-effort last-writer detection for the LLM Wiki status card.

    Closes the gap left by the original panel (commit 2684d6fa, Issue #1257):
    the field was reserved as ``"last_writer": None`` with no reader wired up,
    so the UI always rendered "Not available". This helper makes the field
    useful without breaking the private-safe contract (reads only one line
    of frontmatter and one line of log.md headings, never page bodies).

    Priority:
      1. Most-recently-modified page frontmatter ``updated_by`` / ``writer`` /
         ``author`` (case-insensitive).
      2. Most recent ``log.md`` heading of the form
         ``## [YYYY-MM-DD] <action> | subject`` — returns
         ``"ai-agent (<action>)"`` so the user can see ingest vs update.
      3. Static fallback ``"ai-agent"`` so the UI never shows "Not available"
         for a configured wiki.
    """
    # #3455 review (Codex): resolve the wiki root once and require every file we
    # read to stay under it, so a symlinked .md page can't leak frontmatter from
    # outside the wiki. Also read bounded line-by-line (frontmatter only / log
    # headings only), never full page bodies, per the private-safe status contract.
    try:
        wiki_root = wiki_path.resolve()
    except Exception:
        wiki_root = wiki_path

    def _within_wiki(p: Path) -> bool:
        try:
            return p.resolve().is_relative_to(wiki_root)
        except Exception:
            return False

    # Priority 1: most recent page frontmatter (resolved-path must stay in-wiki)
    latest_page: Path | None = None
    latest_identity: tuple[int, int] | None = None
    latest_mtime = -1.0
    for item in page_files:
        candidate, identity = item if isinstance(item, tuple) else (item, None)
        if not _within_wiki(candidate):
            continue  # skip symlinks resolving outside the wiki
        try:
            st_candidate = candidate.stat()
        except Exception:
            continue
        if identity is not None and (st_candidate.st_dev, st_candidate.st_ino) != identity:
            continue
        mtime = st_candidate.st_mtime
        if mtime > latest_mtime:
            latest_mtime = mtime
            latest_page = candidate
            latest_identity = identity
    if latest_page is not None:
        try:
            fd = os.open(str(latest_page), os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
            try:
                if latest_identity is not None:
                    st_open = os.fstat(fd)
                    if (st_open.st_dev, st_open.st_ino) != latest_identity:
                        raise FileNotFoundError("wiki page changed after allowlist snapshot")
                with os.fdopen(fd, encoding="utf-8", errors="replace") as fh:
                    fd = None
                    first = fh.readline()
                    if first.strip() == "---":
                        # Read only the frontmatter block, bounded to a small line cap.
                        for _ in range(200):
                            line = fh.readline()
                            if line == "" or line.strip() == "---":
                                break  # EOF or end of frontmatter — never touch the body
                            stripped = line.strip()
                            lower = stripped.lower()
                            for key in ("updated_by", "writer", "author"):
                                if lower.startswith(f"{key}:"):
                                    value = stripped.split(":", 1)[1].strip()
                                    if value:
                                        return value
            finally:
                if fd is not None:
                    os.close(fd)
        except Exception:
            pass

    # Priority 2: log.md last entry action verb (heading lines only, bounded)
    log_path = wiki_path / "log.md"
    log_entry = _llm_wiki_status_file_entry_stat(wiki_path, log_path)
    if log_entry is not None:
        try:
            fd = os.open(str(log_path), os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
            try:
                st_open = os.fstat(fd)
                if (st_open.st_dev, st_open.st_ino) != (log_entry.st_dev, log_entry.st_ino):
                    raise FileNotFoundError("wiki log changed before open")
                with os.fdopen(fd, encoding="utf-8", errors="replace") as fh:
                    fd = None
                    for _ in range(5000):  # cap the heading scan
                        line = fh.readline()
                        if line == "":
                            break
                        stripped = line.strip()
                        if not stripped.startswith("## [") or "|" not in stripped:
                            continue
                        tail = stripped.split("]", 1)[1].strip() if "]" in stripped else ""
                        action = tail.split()[0] if tail else "update"
                        return f"ai-agent ({action})"
            finally:
                if fd is not None:
                    os.close(fd)
        except Exception:
            pass

    # Priority 3: never return None / "Not available" for a configured wiki
    return "ai-agent"


def _build_llm_wiki_status() -> dict:
    """Return private-safe LLM Wiki status metadata without reading page bodies."""
    try:
        wiki_path, path_source, path_configured = _llm_wiki_resolve_path()
        base = {
            "available": False,
            "enabled": False,
            "status": "missing",
            "entry_count": 0,
            "page_count": 0,
            "raw_source_count": 0,
            "last_updated": None,
            "last_writer": "ai-agent",
            "path_configured": path_configured,
            "path_source": path_source,
            "toggle_available": False,
            "toggle_reason": "Hermes Agent exposes WIKI_PATH/wiki.path for location, but no stable on/off config flag is currently available.",
            "docs_url": _LLM_WIKI_DOCS_URL,
        }
        if not wiki_path.exists():
            return base
        if not wiki_path.is_dir():
            base["status"] = "not_directory"
            return base

        allowlisted_entries = _llm_wiki_allowlisted_entries(wiki_path)
        verified_page_entries: list[tuple[Path, tuple[int, int], os.stat_result]] = []
        for target, identity in allowlisted_entries.values():
            try:
                st = target.stat()
            except Exception:
                continue
            if (st.st_dev, st.st_ino) != identity:
                continue
            verified_page_entries.append((target, identity, st))
        page_entries = [(target, identity) for target, identity, _ in verified_page_entries]
        page_files = [target for target, _, _ in verified_page_entries]
        status_files: list[tuple[Path, float]] = []
        for path in (wiki_path / "SCHEMA.md", wiki_path / "index.md", wiki_path / "log.md"):
            st_status = _llm_wiki_verified_status_file_stat(wiki_path, path)
            if st_status is None:
                continue
            status_files.append((path, st_status.st_mtime))
        status_files.extend((target, st.st_mtime) for target, _, st in verified_page_entries)
        latest = None
        for _, mtime in status_files:
            latest = mtime if latest is None else max(latest, mtime)

        base.update({
            "available": True,
            "enabled": True,
            "status": "ready" if page_files else "empty",
            "entry_count": len(page_files),
            "page_count": len(page_entries),
            "raw_source_count": _llm_wiki_count_files(wiki_path / "raw"),
            "last_updated": _llm_wiki_safe_iso(latest),
            "last_writer": _llm_wiki_last_writer(wiki_path, page_entries),
        })
        return base
    except Exception as exc:
        return {
            "available": False,
            "enabled": False,
            "status": "error",
            "entry_count": 0,
            "page_count": 0,
            "raw_source_count": 0,
            "last_updated": None,
            "last_writer": "ai-agent",
            "path_configured": False,
            "path_source": "unknown",
            "toggle_available": False,
            "toggle_reason": "Unable to inspect LLM Wiki status safely.",
            "docs_url": _LLM_WIKI_DOCS_URL,
            "error": type(exc).__name__,
        }


def _handle_llm_wiki_status(handler, parsed) -> bool:
    j(handler, _build_llm_wiki_status())
    return True


def _handle_insights(handler, parsed) -> bool:
    """Return usage analytics from local WebUI session data."""
    import collections
    import time as _time

    from api.usage import prompt_cache_hit_percent

    query = parse_qs(parsed.query)
    try:
        days = min(max(int(query.get("days", ["30"])[0]), 1), 365)
    except (ValueError, TypeError):
        days = 30

    now = _time.time()
    today = _time.localtime(now)
    today_midnight = _time.mktime((today.tm_year, today.tm_mon, today.tm_mday, 0, 0, 0, today.tm_wday, today.tm_yday, today.tm_isdst))
    day_secs = 86400
    first_day_ts = today_midnight - ((days - 1) * day_secs)
    cutoff = first_day_ts

    def _safe_usage_int(value) -> int:
        try:
            return max(int(float(value or 0)), 0)
        except (TypeError, ValueError):
            return 0

    def _safe_cost_float(value) -> float:
        if value is None:
            return 0.0
        try:
            if isinstance(value, str):
                value = value.strip().replace("$", "").replace(",", "")
                if not value:
                    return 0.0
            return max(float(value), 0.0)
        except (TypeError, ValueError):
            return 0.0

    def _session_usage_ts(session: dict) -> float:
        return session.get("updated_at", session.get("created_at", 0)) or session.get("created_at", 0) or 0

    # Walk session index (fast, no full JSON parse)
    sessions_data = []
    idx_path = SESSION_DIR / "_index.json"
    if idx_path.exists():
        try:
            idx = json.loads(idx_path.read_text(encoding="utf-8"))
        except Exception:
            idx = []
    else:
        idx = []

    for entry in idx:
        created = entry.get("created_at", 0) or 0
        updated = entry.get("updated_at", 0) or 0
        # Session is relevant if it was created or updated within the calendar window.
        if max(created, updated) < cutoff:
            continue
        sessions_data.append(entry)

    # Aggregate
    total_sessions = len(sessions_data)
    total_messages = 0
    total_input_tokens = 0
    total_output_tokens = 0
    total_cache_read_tokens = 0
    total_cost = 0.0
    model_stats: dict[str, dict] = {}
    daily_tokens: dict[str, dict] = {}
    # Activity by day of week (0=Mon .. 6=Sun)
    dow_activity = collections.Counter()
    # Activity by hour of day (0-23)
    hod_activity = collections.Counter()

    for s in sessions_data:
        input_tokens = _safe_usage_int(s.get("input_tokens"))
        output_tokens = _safe_usage_int(s.get("output_tokens"))
        cache_read_tokens = _safe_usage_int(s.get("cache_read_tokens"))
        cost_value = _safe_cost_float(s.get("estimated_cost"))
        total_messages += _safe_usage_int(s.get("message_count"))
        total_input_tokens += input_tokens
        total_output_tokens += output_tokens
        total_cache_read_tokens += cache_read_tokens
        total_cost += cost_value

        model = s.get("model") or "unknown"
        bucket = model_stats.setdefault(model, {
            "sessions": 0,
            "input_tokens": 0,
            "output_tokens": 0,
            "cache_read_tokens": 0,
            "cost": 0.0,
        })
        bucket["sessions"] += 1
        bucket["input_tokens"] += input_tokens
        bucket["output_tokens"] += output_tokens
        bucket["cache_read_tokens"] += cache_read_tokens
        bucket["cost"] += cost_value

        # Activity patterns
        ts = _session_usage_ts(s)
        if ts:
            try:
                dt = _time.localtime(ts)
                day_key = _time.strftime("%Y-%m-%d", dt)
                daily_bucket = daily_tokens.setdefault(day_key, {
                    "input_tokens": 0,
                    "output_tokens": 0,
                    "cache_read_tokens": 0,
                    "sessions": 0,
                    "cost": 0.0,
                })
                daily_bucket["input_tokens"] += input_tokens
                daily_bucket["output_tokens"] += output_tokens
                daily_bucket["cache_read_tokens"] += cache_read_tokens
                daily_bucket["sessions"] += 1
                daily_bucket["cost"] += cost_value
                dow_activity[dt.tm_wday] += 1
                hod_activity[dt.tm_hour] += 1
            except Exception:
                pass

    # ── Also include CLI sessions from Hermes state.db ─────────────────────
    try:
        from api.models import _active_state_db_path
        db_path = _active_state_db_path()
        if db_path and db_path.exists():
            with closing(open_state_db_readonly(db_path)) as conn:
                conn.row_factory = sqlite3.Row
                cur = conn.cursor()
                # cache_read_tokens may not exist on older agent state DBs;
                # fall back to a query without it if the column is missing.
                try:
                    cur.execute("""
                        SELECT id, model, message_count, input_tokens, output_tokens,
                               estimated_cost_usd,
                               COALESCE(cache_read_tokens, 0) AS cache_read_tokens,
                               started_at, ended_at
                        FROM sessions
                        WHERE (started_at >= ? OR ended_at >= ?)
                          AND COALESCE(source, '') != 'webui'
                    """, (cutoff, cutoff))
                except sqlite3.OperationalError:
                    cur.execute("""
                        SELECT id, model, message_count, input_tokens, output_tokens,
                               estimated_cost_usd,
                               0 AS cache_read_tokens,
                               started_at, ended_at
                        FROM sessions
                        WHERE (started_at >= ? OR ended_at >= ?)
                          AND COALESCE(source, '') != 'webui'
                    """, (cutoff, cutoff))
                for row in cur.fetchall():
                    _input = _safe_usage_int(row["input_tokens"])
                    _output = _safe_usage_int(row["output_tokens"])
                    _cache_read = _safe_usage_int(row["cache_read_tokens"])
                    _cost = _safe_cost_float(row["estimated_cost_usd"])
                    _msgs = _safe_usage_int(row["message_count"])
                    total_sessions += 1
                    total_messages += _msgs
                    total_input_tokens += _input
                    total_output_tokens += _output
                    total_cache_read_tokens += _cache_read
                    total_cost += _cost

                    _model = row["model"] or "unknown"
                    bucket = model_stats.setdefault(_model, {
                        "sessions": 0,
                        "input_tokens": 0,
                        "output_tokens": 0,
                        "cache_read_tokens": 0,
                        "cost": 0.0,
                    })
                    bucket["sessions"] += 1
                    bucket["input_tokens"] += _input
                    bucket["output_tokens"] += _output
                    bucket["cache_read_tokens"] += _cache_read
                    bucket["cost"] += _cost

                    _ts = row["started_at"] or row["ended_at"] or 0
                    if _ts:
                        _dt = _time.localtime(_ts)
                        _day_key = _time.strftime("%Y-%m-%d", _dt)
                        _daily = daily_tokens.setdefault(_day_key, {
                            "input_tokens": 0,
                            "output_tokens": 0,
                            "cache_read_tokens": 0,
                            "sessions": 0,
                            "cost": 0.0,
                        })
                        _daily["input_tokens"] += _input
                        _daily["output_tokens"] += _output
                        _daily["cache_read_tokens"] += _cache_read
                        _daily["sessions"] += 1
                        _daily["cost"] += _cost
                        dow_activity[_dt.tm_wday] += 1
                        hod_activity[_dt.tm_hour] += 1
    except Exception:
        logger.debug("Failed to include CLI sessions in insights", exc_info=True)

    # Build model breakdown
    total_tokens = total_input_tokens + total_output_tokens
    models_breakdown = []
    for model, stats in model_stats.items():
        row_total_tokens = stats["input_tokens"] + stats["output_tokens"]
        row_cost = round(stats["cost"], 6)
        row_cache_read = stats["cache_read_tokens"]
        # Bounded prompt-cache hit rate: cached reads over the FULL prompt total
        # (ordinary input + cache reads), so it can never exceed 100%. Computing
        # cache_read / input_tokens alone would overshoot 100% on cache-heavy
        # sessions. prompt_cache_hit_percent clamps to [0,100] and returns None
        # when there is nothing meaningful to display.
        row_cache_hit_percent = prompt_cache_hit_percent(
            row_cache_read, stats["input_tokens"] + row_cache_read
        )
        models_breakdown.append({
            "model": model,
            "sessions": stats["sessions"],
            "input_tokens": stats["input_tokens"],
            "output_tokens": stats["output_tokens"],
            "cache_read_tokens": row_cache_read,
            "cache_hit_percent": row_cache_hit_percent,
            "total_tokens": row_total_tokens,
            "cost": row_cost,
            "session_share": int(round((stats["sessions"] / total_sessions) * 100)) if total_sessions else 0,
            "token_share": int(round((row_total_tokens / total_tokens) * 100)) if total_tokens else 0,
            "cost_share": int(round((row_cost / total_cost) * 100)) if total_cost else 0,
        })
    models_breakdown.sort(key=lambda r: (-r["cost"], -r["sessions"], r["model"]))

    daily_series = []
    for i in range(days):
        day_ts = first_day_ts + (i * day_secs)
        day_key = _time.strftime("%Y-%m-%d", _time.localtime(day_ts))
        bucket = daily_tokens.get(day_key, {
            "input_tokens": 0,
            "output_tokens": 0,
            "cache_read_tokens": 0,
            "sessions": 0,
            "cost": 0.0,
        })
        daily_series.append({
            "date": day_key,
            "input_tokens": bucket["input_tokens"],
            "output_tokens": bucket["output_tokens"],
            "cache_read_tokens": bucket.get("cache_read_tokens", 0),
            "sessions": bucket["sessions"],
            "cost": round(bucket["cost"], 6),
        })

    # Day-of-week labels
    dow_labels = ["Mon", "Tue", "Wed", "Thu", "Fri", "Sat", "Sun"]
    dow_data = [{"day": dow_labels[i], "sessions": dow_activity.get(i, 0)} for i in range(7)]

    # Hour-of-day data
    hod_data = [{"hour": h, "sessions": hod_activity.get(h, 0)} for h in range(24)]

    return j(handler, {
        "period_days": days,
        "total_sessions": total_sessions,
        "total_messages": total_messages,
        "total_input_tokens": total_input_tokens,
        "total_output_tokens": total_output_tokens,
        "total_cache_read_tokens": total_cache_read_tokens,
        # Aggregate prompt-cache hit rate, bounded 0-100% via the shared helper
        # (cache_read over input + cache_read).
        "total_cache_hit_percent": prompt_cache_hit_percent(
            total_cache_read_tokens, total_input_tokens + total_cache_read_tokens
        ),
        "total_tokens": total_tokens,
        "total_cost": round(total_cost, 6),
        "models": models_breakdown,
        "daily_tokens": daily_series,
        "activity_by_day": dow_data,
        "activity_by_hour": hod_data,
    })


def _project_os_workspace_read(repo_root: Path, rel: str) -> dict | None:
    try:
        return read_file_content(repo_root, rel)
    except Exception:
        return None


def _project_os_workspace_json(repo_root: Path, rel: str) -> dict | None:
    payload = _project_os_workspace_read(repo_root, rel)
    if not payload or not isinstance(payload.get("content"), str):
        return None
    try:
        parsed = json.loads(payload.get("content") or "")
        return parsed if isinstance(parsed, dict) else None
    except Exception:
        return None


def _project_os_truth_board_slugs(repo_root: Path) -> set[str]:
    slugs: set[str] = set()

    def add(value) -> None:
        text = str(value or "").strip()
        if text:
            slugs.add(text)

    for rel in (
        ".ax/handoff/current.json",
        ".ax/status/active.json",
        ".ax/status/heartbeat.json",
    ):
        truth = _project_os_workspace_json(repo_root, rel)
        if not isinstance(truth, dict):
            continue
        raw_board = truth.get("board")
        if isinstance(raw_board, dict):
            add(raw_board.get("slug"))
            add(raw_board.get("id"))
            add(raw_board.get("name"))
            add(raw_board.get("display_name"))
        else:
            add(raw_board)
        for key in (
            "selected_board_slug",
            "canonical_backlog_board_id",
            "current_browser_board_id",
            "active_proof_board_id",
            "recover_board_id",
        ):
            add(truth.get(key))
    return slugs


def _project_os_repo_matches_board(repo_root: Path, board_slug: str | None) -> bool:
    slug = str(board_slug or "").strip()
    return bool(slug and slug in _project_os_truth_board_slugs(repo_root))


def _project_os_candidate_repo_roots(workspace_root: Path | None) -> list[Path]:
    candidates: list[Path] = []
    seen: set[str] = set()

    def add(path: Path | None) -> None:
        if path is None:
            return
        try:
            resolved = path.expanduser().resolve()
        except Exception:
            return
        if not resolved.is_dir():
            return
        key = str(resolved)
        if key not in seen:
            seen.add(key)
            candidates.append(resolved)

    add(workspace_root)

    scan_root = None
    if workspace_root is not None:
        try:
            scan_root = workspace_root.expanduser().resolve()
        except Exception:
            scan_root = None
    if not scan_root or not scan_root.is_dir():
        return candidates

    skip_names = {
        ".git",
        ".hg",
        ".svn",
        ".venv",
        "__pycache__",
        "node_modules",
        "vendor",
        "dist",
        "build",
    }
    queue_dirs: list[tuple[Path, int]] = [(scan_root, 0)]
    inspected = 0
    while queue_dirs and inspected < 300:
        current, depth = queue_dirs.pop(0)
        inspected += 1
        if (current / ".ax").is_dir() or (current / "docs" / "project-os").is_dir():
            add(current)
        if depth >= 3:
            continue
        try:
            children = sorted(
                (child for child in current.iterdir() if child.is_dir()),
                key=lambda child: child.name,
            )
        except Exception:
            continue
        for child in children:
            name = child.name
            if name in skip_names or (name.startswith(".") and name != ".ax"):
                continue
            queue_dirs.append((child, depth + 1))
    add(Path.cwd())
    return candidates


def _project_os_resolve_repo_root_for_board(repo_root: Path | None, board_slug: str | None) -> Path | None:
    slug = str(board_slug or "").strip()
    if not slug:
        return repo_root if repo_root and repo_root.exists() else None
    if repo_root and repo_root.exists() and _project_os_repo_matches_board(repo_root, slug):
        return repo_root
    for candidate in _project_os_candidate_repo_roots(repo_root):
        if _project_os_repo_matches_board(candidate, slug):
            return candidate
    return repo_root if repo_root and repo_root.exists() else None


def _project_os_goal_summary(project_md: dict | None, handoff: dict | None, status_md: dict | None, board_name: str | None = None, board_desc: str | None = None) -> str:
    project_text = str((project_md or {}).get("content") or "")
    for line in project_text.splitlines():
        text = line.strip().lstrip("- ").strip()
        if not text or text.startswith("#"):
            continue
        return text[:220]
    handoff_summary = str((handoff or {}).get("goal_summary") or "").strip()
    if handoff_summary:
        return handoff_summary[:220]
    desc = str(board_desc or "").strip()
    if desc:
        return desc[:220]
    status_text = str((status_md or {}).get("content") or "")
    for line in status_text.splitlines():
        text = line.strip().lstrip("- ").strip()
        if not text or text.startswith("#"):
            continue
        return text[:220]
    return str(board_name or "Project OS").strip()[:220]


def _project_os_onboarding_context(repo_root: Path, project_md: dict | None, plan_md: dict | None, status_md: dict | None) -> dict:
    project_text = str((project_md or {}).get("content") or "")
    plan_text = str((plan_md or {}).get("content") or "")
    status_text = str((status_md or {}).get("content") or "")
    merged = "\n".join([project_text, plan_text, status_text])
    is_non_git_workspace = not (repo_root / ".git").exists()
    has_boundary_hold = "TO_BE_VALIDATED_BY_HERMES" in merged
    child_repo_blocked = (
        "auto-promoted" in merged
        or "auto-adopted" in merged
        or "auto-adoption | `금지`" in merged
        or "자동 승격 금지" in merged
        or "canonical repo continuity로 승격하지 않습니다" in merged
    )
    workspace_root_confirmed = str(repo_root) in merged
    if not (is_non_git_workspace and (project_text or plan_text or status_text)):
        return {
            "active": False,
            "doc_source": "project-os",
        }
    status_label = "보류(안전)" if has_boundary_hold else "확인됨"
    summary = "workspace root onboarding 진행 중 · 저장소 경계는 아직 미확정이며 자동 승격은 금지됩니다."
    next_safe_action = "workspace-root 기준으로 경계만 좁게 검증"
    if has_boundary_hold:
        summary = "workspace root onboarding 진행 중 · 저장소 경계는 아직 미확정이며 TO_BE_VALIDATED_BY_HERMES 상태를 유지합니다."
    return {
        "active": True,
        "doc_source": "root",
        "status_label": status_label,
        "summary": summary,
        "next_safe_action": next_safe_action,
        "workspace_root_confirmed": workspace_root_confirmed,
        "repo_boundary_status": "TO_BE_VALIDATED_BY_HERMES" if has_boundary_hold else "confirmed",
        "child_repo_auto_promotion_blocked": bool(child_repo_blocked),
        "guardrails": [
            "workspace root 확인됨" if workspace_root_confirmed else "workspace root 확인 필요",
            "child repo 자동 승격 금지 유지" if child_repo_blocked else "child repo guardrail 확인 필요",
            "repo boundary 미확정 유지" if has_boundary_hold else "repo boundary confirmed",
        ],
    }


def _handle_project_os_dashboard(handler, parsed) -> bool:
    qs = parse_qs(parsed.query or "")
    requested_board = str((qs.get("board") or [""])[0] or "").strip()
    workspace_raw = str(get_last_workspace() or "").strip()
    repo_root = Path(workspace_raw).expanduser() if workspace_raw else None
    selected_board_meta = None
    if requested_board:
        try:
            from api.kanban_bridge import _kb, _board_meta_dict
            kb = _kb()
            for meta in kb.list_boards(include_archived=True) or []:
                board = _board_meta_dict(meta)
                if str(board.get("slug") or "") == requested_board:
                    selected_board_meta = board
                    workdir = str(board.get("default_workdir") or "").strip()
                    if workdir:
                        candidate = Path(workdir).expanduser()
                        if candidate.exists():
                            repo_root = candidate
                    break
        except Exception:
            selected_board_meta = None
    repo_root = _project_os_resolve_repo_root_for_board(repo_root, requested_board)
    if not repo_root or not repo_root.exists():
        j(handler, {
            "workspace": None,
            "repo_root": None,
            "git": None,
            "docs": {},
            "handoff": None,
            "active": None,
            "heartbeat": None,
            "goal_summary": "",
        })
        return True

    handoff = _project_os_workspace_json(repo_root, ".ax/handoff/current.json")
    active = _project_os_workspace_json(repo_root, ".ax/status/active.json")
    heartbeat = _project_os_workspace_json(repo_root, ".ax/status/heartbeat.json")
    project_md = _project_os_workspace_read(repo_root, "docs/project-os/PROJECT.md")
    plan_md = _project_os_workspace_read(repo_root, "docs/project-os/PLAN.md")
    status_md = _project_os_workspace_read(repo_root, "docs/project-os/STATUS.md")
    blocker_md = _project_os_workspace_read(repo_root, "docs/project-os/BLOCKER-RESOLVER.md")
    root_project_md = _project_os_workspace_read(repo_root, "PROJECT.md")
    root_plan_md = _project_os_workspace_read(repo_root, "PLAN.md")
    root_status_md = _project_os_workspace_read(repo_root, "STATUS.md")
    onboarding = _project_os_onboarding_context(repo_root, root_project_md, root_plan_md, root_status_md)
    if onboarding.get("active"):
        project_md = root_project_md or project_md
        plan_md = root_plan_md or plan_md
        status_md = root_status_md or status_md

    if isinstance(active, dict):
        original_repo_root = repo_root
        active_repo_root = str(active.get("repo_root") or "").strip()
        if active_repo_root:
            candidate = Path(active_repo_root).expanduser()
            if candidate.exists():
                try:
                    repo_root = candidate.resolve()
                except Exception:
                    repo_root = candidate
        if repo_root != original_repo_root:
            handoff = _project_os_workspace_json(repo_root, ".ax/handoff/current.json")
            active = _project_os_workspace_json(repo_root, ".ax/status/active.json")
            heartbeat = _project_os_workspace_json(repo_root, ".ax/status/heartbeat.json")
            project_md = _project_os_workspace_read(repo_root, "docs/project-os/PROJECT.md")
            plan_md = _project_os_workspace_read(repo_root, "docs/project-os/PLAN.md")
            status_md = _project_os_workspace_read(repo_root, "docs/project-os/STATUS.md")
            blocker_md = _project_os_workspace_read(repo_root, "docs/project-os/BLOCKER-RESOLVER.md")
            root_project_md = _project_os_workspace_read(repo_root, "PROJECT.md")
            root_plan_md = _project_os_workspace_read(repo_root, "PLAN.md")
            root_status_md = _project_os_workspace_read(repo_root, "STATUS.md")
            onboarding = _project_os_onboarding_context(repo_root, root_project_md, root_plan_md, root_status_md)
            if onboarding.get("active"):
                project_md = root_project_md or project_md
                plan_md = root_plan_md or plan_md
                status_md = root_status_md or status_md

    try:
        git = git_info_for_workspace(repo_root)
    except Exception:
        git = None

    board_name = None
    board_desc = None
    if isinstance(handoff, dict):
        board_dict: dict = {}
        raw_board = handoff.get("board")
        if isinstance(raw_board, dict):
            board_dict = raw_board
        board_name = board_dict.get("display_name") or board_dict.get("name") or board_dict.get("slug")
        board_desc = handoff.get("goal_summary") or board_dict.get("repo_corroboration")
    if selected_board_meta:
        board_name = board_name or selected_board_meta.get("name") or selected_board_meta.get("slug")
        board_desc = board_desc or selected_board_meta.get("description")

    j(handler, {
        "workspace": str(repo_root),
        "repo_root": str(repo_root),
        "selected_board_slug": requested_board or (selected_board_meta or {}).get("slug"),
        "git": git,
        "docs": {
            "project": project_md,
            "plan": plan_md,
            "status": status_md,
            "blocker_resolver": blocker_md,
        },
        "handoff": handoff,
        "active": active,
        "heartbeat": heartbeat,
        "onboarding": onboarding,
        "goal_summary": _project_os_goal_summary(project_md, handoff, status_md, board_name, board_desc),
    })
    return True


# ── GET routes ────────────────────────────────────────────────────────────────


def _accept_loop_health(handler) -> dict:
    server = getattr(handler, "server", None)
    return {
        "requests_total": int(getattr(server, "accept_loop_requests_total", 0) or 0),
        "last_request_at": round(float(getattr(server, "accept_loop_last_request_at", 0.0) or 0.0), 3),
    }


def _streams_lock_health(timeout_seconds: float = 0.5) -> dict:
    t0 = time.time()
    acquired = STREAMS_LOCK.acquire(timeout=timeout_seconds)
    elapsed_ms = round((time.time() - t0) * 1000, 1)
    if not acquired:
        return {
            "status": "blocked",
            "timeout_seconds": timeout_seconds,
            "ms": elapsed_ms,
        }
    try:
        return {
            "status": "ok",
            "active_streams": len(STREAMS),
            "ms": elapsed_ms,
        }
    finally:
        STREAMS_LOCK.release()


def _stream_runtime_diagnostics() -> dict:
    """Return non-sensitive SSE stream diagnostics for health/deep status.

    The WebUI chat path can feel slow or stuck when streams are alive but no
    browser is attached, or when many events are buffering offline. This helper
    exposes counts only — stream ids plus subscriber/buffer sizes — and avoids
    event payloads, prompts, tool arguments, or paths.
    """
    streams = []
    total_subscribers = 0
    total_offline_buffered_events = 0
    with STREAMS_LOCK:
        items = list(STREAMS.items())
    for stream_id, stream in items:
        snapshot = {}
        diagnostic_snapshot = getattr(stream, "diagnostic_snapshot", None)
        if callable(diagnostic_snapshot):
            try:
                raw_snapshot = diagnostic_snapshot()
                if isinstance(raw_snapshot, dict):
                    snapshot = raw_snapshot
            except Exception:
                snapshot = {}
        subscriber_count = int(snapshot.get("subscriber_count") or 0)
        offline_buffered_events = int(snapshot.get("offline_buffered_events") or 0)
        total_subscribers += subscriber_count
        total_offline_buffered_events += offline_buffered_events
        streams.append({
            "stream_id": str(stream_id),
            "subscriber_count": subscriber_count,
            "offline_buffered_events": offline_buffered_events,
        })
    streams.sort(key=lambda item: item["stream_id"])
    return {
        "active_streams": len(streams),
        "total_subscribers": total_subscribers,
        "total_offline_buffered_events": total_offline_buffered_events,
        "streams": streams,
    }


def _run_lifecycle_health() -> dict:
    """Return active worker-run state independent of SSE stream presence."""
    # Import the module rather than relying only on imported scalar aliases so
    # LAST_RUN_FINISHED_AT stays fresh after unregister_active_run() updates it.
    from api import config as _live_config

    now = time.time()
    with _live_config.ACTIVE_RUNS_LOCK:
        runs = []
        for _stream_id, raw in (_live_config.ACTIVE_RUNS or {}).items():
            item = dict(raw or {})
            item.pop("session_id", None)
            item.pop("stream_id", None)
            item.pop("workspace", None)
            started_at = item.get("started_at")
            try:
                age = max(0.0, now - float(started_at))
            except Exception:
                age = 0.0
            item["age_seconds"] = round(age, 1)
            runs.append(item)
        last_finished = _live_config.LAST_RUN_FINISHED_AT
    runs.sort(key=lambda item: float(item.get("started_at") or 0.0))
    payload = {
        "active_runs": len(runs),
        "runs": runs,
        "last_run_finished_at": last_finished,
    }
    if runs:
        payload["oldest_run_age_seconds"] = runs[0].get("age_seconds", 0.0)
    elif last_finished:
        payload["idle_seconds_since_last_run"] = round(max(0.0, now - float(last_finished)), 1)
    return payload


def _deep_health_checks(stream_check: dict | None = None) -> tuple[dict, bool]:
    """Run cheap probes that exercise the state paths used by the UI shell.

    Plain /health intentionally stays tiny. /health?deep=1 is for supervisors
    and watchdogs that need to know whether the process can still touch the
    shared stream map, sidebar/session path, project state, and Hermes state.db
    without hitting the RST-before-write failure mode from #1458.

    `stream_check` is the result from a prior `_streams_lock_health()` call;
    if provided, it's reused so we don't acquire `STREAMS_LOCK` twice on the
    same /health?deep=1 request (per Opus advisor on stage-297).
    """
    checks: dict[str, dict] = {}

    checks["streams_lock"] = stream_check if stream_check is not None else _streams_lock_health()
    checks["stream_runtime"] = {
        "status": "ok",
        **_stream_runtime_diagnostics(),
    }
    if checks["streams_lock"].get("status") != "ok":
        return checks, False

    t0 = time.time()
    try:
        sessions = all_sessions()
        checks["sessions"] = {
            "status": "ok",
            "count": len(sessions),
            "ms": round((time.time() - t0) * 1000, 1),
        }
    except Exception as exc:
        checks["sessions"] = {
            "status": "error",
            "error": type(exc).__name__,
            "ms": round((time.time() - t0) * 1000, 1),
        }

    t0 = time.time()
    try:
        projects = load_projects(_migrate=False)
        checks["projects"] = {
            "status": "ok",
            "count": len(projects),
            "ms": round((time.time() - t0) * 1000, 1),
        }
    except Exception as exc:
        checks["projects"] = {
            "status": "error",
            "error": type(exc).__name__,
            "ms": round((time.time() - t0) * 1000, 1),
        }

    t0 = time.time()
    try:
        db_path = _active_state_db_path()
        if not db_path.exists():
            checks["state_db"] = {
                "status": "missing",
                "ms": round((time.time() - t0) * 1000, 1),
            }
        else:
            with closing(open_state_db_readonly(db_path)) as conn:
                conn.execute("PRAGMA schema_version").fetchone()
            checks["state_db"] = {
                "status": "ok",
                "ms": round((time.time() - t0) * 1000, 1),
            }
    except Exception as exc:
        checks["state_db"] = {
            "status": "error",
            "error": type(exc).__name__,
            "ms": round((time.time() - t0) * 1000, 1),
        }

    healthy = all(
        check.get("status") in {"ok", "missing"}
        for check in checks.values()
    )
    return checks, healthy


def _handle_health(handler, parsed):
    deep = parse_qs(parsed.query or "").get("deep", [""])[0].lower() in {"1", "true", "yes", "on"}
    stream_check = _streams_lock_health()
    run_check = _run_lifecycle_health()
    payload = {
        "status": "ok" if stream_check.get("status") == "ok" else "degraded",
        "sessions": len(SESSIONS),
        "active_streams": int(stream_check.get("active_streams") or 0),
        "active_runs": int(run_check.get("active_runs") or 0),
        "runs": run_check.get("runs", []),
        "last_run_finished_at": run_check.get("last_run_finished_at"),
        "server_started_at": SERVER_START_TIME,
        "uptime_seconds": round(time.time() - SERVER_START_TIME, 1),
        "accept_loop": _accept_loop_health(handler),
    }
    if "oldest_run_age_seconds" in run_check:
        payload["oldest_run_age_seconds"] = run_check["oldest_run_age_seconds"]
    if "idle_seconds_since_last_run" in run_check:
        payload["idle_seconds_since_last_run"] = run_check["idle_seconds_since_last_run"]
    if deep:
        if stream_check.get("status") != "ok":
            payload["checks"] = {"streams_lock": stream_check}
            return j(handler, payload, status=503)
        checks, healthy = _deep_health_checks(stream_check=stream_check)
        payload["checks"] = checks
        if not healthy:
            payload["status"] = "degraded"
            return j(handler, payload, status=503)
    if payload["status"] != "ok":
        return j(handler, payload, status=503)
    return j(handler, payload)


# ── Plugin visibility endpoint (#539) ───────────────────────────────────────
_PLUGIN_VISIBILITY_HOOKS = (
    "pre_tool_call",
    "post_tool_call",
    "pre_llm_call",
    "post_llm_call",
)
_PLUGIN_VISIBILITY_HOOK_SET = set(_PLUGIN_VISIBILITY_HOOKS)


def _get_plugin_manager_for_visibility():
    """Return Hermes Agent's plugin manager for read-only WebUI visibility."""
    from hermes_cli.plugins import get_plugin_manager

    return get_plugin_manager()


def _clean_plugin_visibility_text(value, *, limit=240) -> str:
    """Return bounded display text without path/callback-like internals."""
    if value is None:
        return ""
    text = str(value).replace("\x00", "").strip()
    # Display metadata should be plain labels/descriptions. Drop multiline text
    # and common path separators rather than risk leaking local plugin paths.
    text = " ".join(text.split())
    if len(text) > limit:
        text = text[: limit - 1].rstrip() + "…"
    return text


def _plugin_visibility_category_from_key(key: str) -> str:
    """Return the config category prefix for nested plugin keys."""
    raw = str(key or "").strip().replace("\\", "/")
    if "/" not in raw:
        return ""
    category = raw.split("/", 1)[0].strip()
    return category if category and category not in {".", ".."} else ""


def _plugin_visibility_selected_provider(category: str) -> str:
    """Read ``<category>.provider`` without surfacing config errors."""
    if not category:
        return ""
    try:
        from api.config import get_config as _get_cfg

        cfg = _get_cfg() or {}
    except Exception:
        return ""
    category_cfg = cfg.get(category, {}) if isinstance(cfg, dict) else {}
    if not isinstance(category_cfg, dict):
        return ""
    return str(category_cfg.get("provider") or "").strip().lower()


def _plugin_visibility_payload(manager=None) -> dict:
    """Build a sanitized plugin/hook visibility payload for Settings.

    The Hermes Agent manager stores manifests and callback objects internally.
    This endpoint intentionally exposes only safe, user-facing metadata and the
    four lifecycle hook names called out by the Settings visibility MVP. It
    never includes plugin source paths, callback names, callback reprs, or raw
    load errors because those can contain private filesystem details.

    Exclusive plugins (e.g. memory providers) are activated through their
    category's ``<category>.provider`` config, not through ``plugins.enabled``.
    Their ``loaded.enabled`` stays False by design and they register hooks
    outside the four visibility hooks below. The payload surfaces ``kind``
    and ``activation`` plus ``is_active_provider`` when the plugin key carries a
    category prefix so the panel can render the selected provider distinctly
    instead of mislabeling it as "Disabled" with no hooks (issue #2659), while
    still showing unselected exclusive providers as disabled. Flat-key exclusive
    plugins omit the new field so older activation-based badge semantics remain
    intact when the category cannot be inferred.
    """
    manager = manager or _get_plugin_manager_for_visibility()
    manager.discover_and_load(force=False)

    plugins = []

    # Hermes Agent lifecycle-hook plugins
    raw_plugins = getattr(manager, "_plugins", {}) or {}
    for key, loaded in sorted(raw_plugins.items(), key=lambda item: str(item[0])):
        manifest = getattr(loaded, "manifest", None)
        if manifest is None:
            continue
        plugin_key = _clean_plugin_visibility_text(
            getattr(manifest, "key", None) or key or getattr(manifest, "name", ""),
            limit=120,
        )
        name = _clean_plugin_visibility_text(getattr(manifest, "name", "") or plugin_key, limit=120)
        version = _clean_plugin_visibility_text(getattr(manifest, "version", ""), limit=80)
        description = _clean_plugin_visibility_text(getattr(manifest, "description", ""), limit=280)
        kind = _clean_plugin_visibility_text(getattr(manifest, "kind", "") or "standalone", limit=40)
        enabled_flag = bool(getattr(loaded, "enabled", False))
        category = _plugin_visibility_category_from_key(plugin_key)
        selected_provider = _plugin_visibility_selected_provider(category)
        plugin_slug = plugin_key.rsplit("/", 1)[-1].strip().lower()
        if kind == "exclusive":
            activation = "exclusive"
        elif kind == "model-provider" and enabled_flag:
            activation = "provider"
        else:
            activation = "enabled" if enabled_flag else "disabled"
        include_active_provider = True
        if kind == "exclusive":
            if category:
                is_active_provider = bool(selected_provider) and plugin_slug == selected_provider
            else:
                include_active_provider = False
                is_active_provider = False
        else:
            is_active_provider = kind == "model-provider" and enabled_flag
        registered = []
        for hook in list(getattr(manifest, "provides_hooks", []) or []) + list(getattr(loaded, "hooks_registered", []) or []):
            hook_name = str(hook or "").strip()
            if hook_name in _PLUGIN_VISIBILITY_HOOK_SET and hook_name not in registered:
                registered.append(hook_name)
        registered.sort(key=_PLUGIN_VISIBILITY_HOOKS.index)
        plugin_payload = {
            "name": name,
            "key": plugin_key or name,
            "version": version,
            "description": description,
            # `enabled` is preserved for back-compat with older WebUI clients
            # that key off it directly. New clients should prefer `activation`.
            "enabled": enabled_flag,
            "kind": kind,
            "activation": activation,
            "hooks": registered,
        }
        if include_active_provider:
            plugin_payload["is_active_provider"] = bool(is_active_provider)
        plugins.append(plugin_payload)

    return {
        "plugins": plugins,
        "empty": not bool(plugins),
        "supported_hooks": list(_PLUGIN_VISIBILITY_HOOKS),
        "read_only": True,
    }


# WebUI dashboard plugins (from manifest.json discovery)
def _dashboard_plugin_enabled(plugin_name: str) -> bool:
    """True if a dashboard plugin is enabled in settings.

    Dashboard plugins are opt-in (default off). Enforced server-side so a
    disabled plugin's page + asset URLs are fully 404'd, not merely hidden in
    the Settings UI.
    """
    try:
        from api.config import load_settings
        prefs = (load_settings() or {}).get("dashboard_plugins", {}) or {}
        return bool(prefs.get(plugin_name, False))
    except Exception:
        return False


def _webui_plugin_payload() -> list[dict]:
    try:
        from api.plugins import get_plugin_metadata
        return get_plugin_metadata()
    except Exception:
        return []


def _handle_plugins(handler, parsed) -> bool:
    try:
        hermes_plugins = _plugin_visibility_payload()
        webui = _webui_plugin_payload()
        all_plugins = hermes_plugins["plugins"] + webui
        return j(handler, {
            "plugins": all_plugins,
            "empty": not bool(all_plugins),
            "supported_hooks": hermes_plugins["supported_hooks"],
            "read_only": True,
        })
    except Exception as exc:
        logger.warning("Failed to build plugin visibility payload: %s", exc)
        return j(
            handler,
            {
                "plugins": [],
                "empty": True,
                "supported_hooks": list(_PLUGIN_VISIBILITY_HOOKS),
                "read_only": True,
                "unavailable": True,
            },
        )


_SHELL_ERROR_HTML = """<!doctype html>
<html lang=\"en\">
<head>
  <meta charset=\"utf-8\">
  <meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">
  <title>Hermes is restarting</title>
</head>
<body style=\"margin:0;padding:2rem;font-family:-apple-system,BlinkMacSystemFont,'Segoe UI',sans-serif;background:#111827;color:#e5e7eb;\">
  <main style=\"max-width:40rem;margin:10vh auto;line-height:1.5;\">
    <h1 style=\"font-size:1.5rem;margin:0 0 0.75rem;\">Hermes is restarting…</h1>
    <p style=\"margin:0;color:#cbd5e1;\">The WebUI shell could not load cleanly. Refresh in a moment if this page does not update automatically.</p>
  </main>
</body>
</html>"""


def _serve_shell_unavailable(handler, exc: Exception) -> bool:
    """Return HTML for shell-route failures so `/` never renders JSON."""
    logger.warning("Failed to serve WebUI shell route: %s", exc)
    t(
        handler,
        _SHELL_ERROR_HTML,
        status=503,
        content_type="text/html; charset=utf-8",
    )
    return True


_SHUTDOWN_LOG_VALUE_RE = re.compile(r"[\x00-\x1f\x7f]+")


def _shutdown_log_value(value, *, default: str = "unknown", max_len: int = 160) -> str:
    """Return a bounded single-line value safe for shutdown diagnostics."""
    if value is None:
        return default
    try:
        text = str(value)
    except Exception:
        return default
    text = _SHUTDOWN_LOG_VALUE_RE.sub("?", text).strip()
    if not text:
        return default
    if len(text) > max_len:
        text = f"{text[:max_len]}…"
    return text


def _handle_shutdown(handler) -> bool:
    """Shut down the WebUI server process."""
    headers = getattr(handler, "headers", {})
    ua = headers.get("User-Agent", "no-ua") if hasattr(headers, "get") else "no-ua"
    remote = "unknown"
    if getattr(handler, "client_address", None):
        remote = getattr(handler, "client_address", ("unknown",))[0]
    logger.info(
        "[shutdown-request] remote=%s method=%s path=%s ua=%s",
        _shutdown_log_value(remote),
        _shutdown_log_value(getattr(handler, "command", None)),
        _shutdown_log_value(getattr(handler, "path", None), max_len=240),
        _shutdown_log_value(ua, default="no-ua", max_len=240),
    )
    j(handler, {"status": "shutting_down"})
    import signal
    import threading

    def _do_shutdown():
        import time
        time.sleep(0.3)
        os.kill(os.getpid(), signal.SIGINT)

    threading.Thread(target=_do_shutdown, daemon=True).start()
    return True


def _handle_health_restart(handler) -> bool:
    """Restart the Hermes messaging gateway service."""
    # This endpoint never consumes its request body on any outcome, so close when
    # one was DECLARED -- and only then. Arming unconditionally closed the socket
    # on every call including the successful, body-less one the WebUI actually
    # makes: verified on the wire, `POST /api/health/restart` with no
    # `Content-Length` answered with `Connection: close` and the pipelined
    # `GET /api/health/agent` was never served. The single arming covers every
    # outcome below (completed / in_progress / busy / error) because the framing,
    # not the result, decides.
    arm_connection_close_if_body_pending(handler)
    outcome = restart_active_profile_gateway()

    if outcome.get("status") == "completed":
        return j(handler, {"ok": True, "message": "Gateway service restarted successfully"})

    if outcome.get("status") == "in_progress":
        return j(handler, {"ok": True, "message": "Gateway service restart initiated (in progress)"})

    if outcome.get("status") == "busy":
        return j(
            handler,
            {"ok": False, "error": outcome.get("message", "Restart already in progress. Please wait a moment and try again.")},
            status=429,
        )

    return j(
        handler,
        {"ok": False, "error": outcome.get("message", "Internal error running restart")},
        status=500,
    )


def _serve_manifest(handler) -> bool:
    """Serve static/manifest.json with the correct PWA Content-Type.

    Shared by the root (/manifest.json, /manifest.webmanifest) and
    session-prefixed (/session/manifest.json, /session/manifest.webmanifest)
    routes so Firefox Android can fetch the manifest when installing from
    a /session/<id> page.  See #2226.
    """
    static_root = api_config.get_static_root()
    manifest_path = (static_root / "manifest.json").resolve()
    if manifest_path.exists():
        data = manifest_path.read_bytes()
        handler.send_response(200)
        handler.send_header("Content-Type", "application/manifest+json; charset=utf-8")
        handler.send_header("Cache-Control", "no-store")
        handler.send_header("Content-Length", str(len(data)))
        handler.end_headers()
        handler.wfile.write(data)
        return True
    return j(handler, {"error": "not found"}, status=404)


def _saved_prompts_path() -> "Path":
    try:
        from api.profiles import get_active_hermes_home
        return Path(get_active_hermes_home()).expanduser() / "webui" / "saved_prompts.json"
    except Exception:
        return Path(os.getenv("HERMES_HOME", str(Path.home() / ".hermes"))).expanduser() / "webui" / "saved_prompts.json"


def _load_saved_prompts() -> list:
    p = _saved_prompts_path()
    if not p.exists():
        return []
    try:
        return json.loads(p.read_text(encoding="utf-8"))
    except Exception:
        return []


def _save_saved_prompts(prompts: list) -> None:
    p = _saved_prompts_path()
    p.parent.mkdir(parents=True, exist_ok=True)
    p.write_text(json.dumps(prompts, ensure_ascii=False, indent=2), encoding="utf-8")


# In-process cache for the app-shell template. The `/`, `/index.html`, and
# `/session/<id>` routes are the hottest navigations and each re-read the
# ~190 KB static/index.html from disk and re-ran the two process-constant
# substitutions (__WEBUI_VERSION__, __MAX_UPLOAD_BYTES__) on every request.
# Those values are fixed for the process lifetime, so we cache the partially
# rendered template here, keyed by (size, nanosecond mtime) exactly like
# _STATIC_CACHE so a redeploy is picked up without a restart. The two values
# that genuinely vary per request — the per-session CSRF token and the runtime
# extension tags (inject_extension_tags) — are still applied on each request
# against the cached base, so caching changes no observable output.
_INDEX_SHELL_CACHE: dict = {}
_INDEX_SHELL_CACHE_LOCK = threading.Lock()


def _render_index_shell_base() -> str:
    """Return static/index.html with the process-constant tokens substituted.

    Cached and invalidated on (size, mtime_ns) change. The CSRF token and
    extension-tag injection are intentionally NOT applied here — they vary per
    request and are applied by the caller against this base string.
    """
    from api.updates import WEBUI_VERSION

    index_path = api_config.get_index_html_path()
    st = index_path.stat()
    sig = (index_path, st.st_size, st.st_mtime_ns)
    with _INDEX_SHELL_CACHE_LOCK:
        cached = _INDEX_SHELL_CACHE.get("base")
        if cached and cached[0] == sig:
            return cached[1]
    from urllib.parse import quote

    version_token = quote(WEBUI_VERSION, safe="")
    base = (
        index_path.read_text(encoding="utf-8")
        .replace("__WEBUI_VERSION__", version_token)
        .replace("__MAX_UPLOAD_BYTES__", str(MAX_UPLOAD_BYTES))
    )
    with _INDEX_SHELL_CACHE_LOCK:
        _INDEX_SHELL_CACHE["base"] = (sig, base)
    return base


def _handle_session_get(handler, parsed) -> bool:
    """GET /api/session — full session payload (messages, tool calls, lineage...). Extracted verbatim from handle_get; every early-return path calls _diag.finish() (see the tier2c note inside)."""
    import time as _time
    _t0 = _time.monotonic()
    _debug_slow = os.environ.get("HERMES_DEBUG_SLOW", "")
    # perf(webui/session-load-latency) tier2c: per-stage breakdown via
    # RequestDiagnostics. maybe_start() returns None for paths not in
    # the allowlist, in which case the existing _tN-driven [SLOW] log
    # is the only signal — same as before.
    _diag = RequestDiagnostics.maybe_start("GET", parsed.path, logger=logger, print_fn=getattr(handler, '_safe_webui_print', None))
    # perf(webui/session-load-latency) tier2c-followup: every early-return
    # in this handler calls `_diag.finish()` before returning so the
    # watchdog's _watchdog_pending dict stays bounded to in-flight requests.
    # Greptile flagged this in PR review — finish() unregisters the
    # pending watchdog entry; without it the entry stays for the full
    # 5s slow-request timeout and emits a spurious "Slow WebUI request
    # still running" log. Idempotent — finish() no-ops if already called.
    query = parse_qs(parsed.query)
    sid = query.get("session_id", [""])[0]
    if not sid:
        if _diag: _diag.finish()
        return j(handler, {"error": "session_id is required"}, status=400)
    # ?messages=0 skips the message payload for fast session switching.
    # The frontend uses this when switching conversations in the sidebar
    # (only needs metadata). The full message array is loaded lazily
    # via ?messages=1 when the message panel opens.
    load_messages = query.get("messages", ["1"])[0] != "0"
    resolve_model_default = "1" if load_messages else "0"
    resolve_model = query.get("resolve_model", [resolve_model_default])[0] != "0"
    # ?msg_limit=N returns a tail window containing the last N visible
    # transcript rows. Hidden tool-result rows do not consume the budget;
    # they are included only when they sit inside the selected window and
    # are bounded before serialization. Older rows load on-demand.
    # Clamp to _MAX_MSG_LIMIT so an oversized request (e.g. msg_limit=9999
    # from an outline jump, or a hostile value) can't force an unbounded
    # payload; the existing _messages_truncated signal covers the clamped
    # case (the client sees there are more rows than returned). Parsing +
    # clamping live in _parse_msg_limit so the expression has direct test
    # coverage.  The frontend recovery paths request a bounded tail
    # explicitly (msg_limit=30), and the two absolute-index paths opt in to
    # the full transcript via msg_limit=all (#7310/#7625).
    _raw_msg_limit = query.get("msg_limit", [None])[0]
    # ?msg_before=N — 0-based index into the full message array.
    # Returns messages before this index (for scroll-to-top lazy loading).
    # Combined with msg_limit for paging.
    _msg_before = query.get("msg_before", [None])[0]
    try:
        msg_before = int(_msg_before) if _msg_before else None
    except (ValueError, TypeError):
        msg_before = None
    msg_limit, _msg_limit_explicit_all = _resolve_effective_msg_limit(
        _raw_msg_limit,
    )
    # ?expand_renderable=1 is retained for compatibility with older
    # frontends. msg_limit now counts visible transcript rows by default, so
    # the flag no longer changes the server-side pagination semantics.
    _expand_renderable = query.get("expand_renderable", [None])[0]
    expand_renderable = str(_expand_renderable).strip() in ("1", "true", "True")
    try:
        _t1 = _time.monotonic()
        if _diag: _diag.stage("t1_after_get_session_check")
        s = get_session(sid, metadata_only=(not load_messages))
        _session_profile = getattr(s, 'profile', None) or None
        if not _session_visible_to_active_profile(_session_profile, handler):
            if _session_profile:
                # Valid session owned by a KNOWN other profile: 409 so the
                # client can offer to switch to it (#5419).
                if _diag: _diag.finish()
                return j(handler, {
                    "error": "Session belongs to a different profile",
                    "code": "session_profile_mismatch",
                    "session_id": sid,
                    "profile": _session_profile,
                }, status=409)
            # Unknown/legacy None-profile sidecar: keep the original 404 so
            # the frontend's self-heal (clear stale URL + localStorage) still
            # fires. _profiles_match coerces None->'default', so a truly
            # missing/legacy session under a non-default active profile would
            # otherwise emit a useless 409 with profile=null.
            if _diag: _diag.finish()
            return bad(handler, "Session not found", 404)
        original_stream_id = getattr(s, "active_stream_id", None)
        _clear_stale_stream_state(s)
        cli_meta = _lookup_cli_session_metadata(sid) if _session_requires_cli_metadata_lookup(s) else {}
        is_messaging_session = _is_messaging_session_record(s) or _is_messaging_session_record(cli_meta)
        cli_messages = []
        state_db_messages = []
        metadata_summary = None
        limited_sidecar_messages = None
        state_db_since_timestamp = None
        # Set by the limited-display path when the memoized merge can be
        # reused without loading the state.db rows; must exist for every
        # branch below, including the ones that never probe the cache.
        _display_cache_hit = None
        _display_state_db_signature = None
        if is_messaging_session:
            cli_messages = get_cli_session_messages(sid)
        elif load_messages:
            if msg_limit is not None:
                (
                    state_db_since_timestamp,
                    limited_sidecar_messages,
                ) = _state_db_since_timestamp_for_limited_display(
                    s,
                    msg_limit,
                    msg_before=msg_before,
                )
            _state_db_reader_kwargs = {"profile": _session_profile}
            if state_db_since_timestamp is not None:
                _state_db_reader_kwargs["since_timestamp"] = state_db_since_timestamp
            # Apply the display-path row backstop ONLY on provably-safe
            # reads where no truncation_boundary prefix is required for the
            # merge — see _state_db_backstop_limit_for_display. Compressed
            # sessions and msg_before paging need their full prefix rows for
            # correct reconciliation, so those stay uncapped.
            _backstop = _state_db_backstop_limit_for_display(s, msg_before)
            if _backstop is not None:
                _state_db_reader_kwargs["limit"] = _backstop
            # perf: on the limited-display path the state.db rows are only
            # consumed by the memoized merge below. Now that the cache key
            # is a bounded SQL signature rather than a fingerprint OF these
            # rows, a hit no longer needs them -- and materialising tens of
            # thousands of dicts was the dominant remaining cost (~2.3s on a
            # 36k-row session) even when the merge itself was served from
            # cache. Probe the cache first and skip the load on a hit.
            #
            # Deliberately narrow: only when msg_limit is set (the merge
            # helper below is the sole consumer) and only for inactive
            # sessions, matching the cache's own validity rule. Any miss
            # falls through to the normal full load, so this can only skip
            # work that would have produced an identical merged result.
            _display_cache_hit = None
            if (
                msg_limit is not None
                and not getattr(s, "active_stream_id", None)
                and not getattr(s, "pending_user_message", None)
            ):
                _display_cache_hit = _display_merge_cached_messages(
                    s,
                    limited_sidecar_messages,
                    msg_before=msg_before,
                )
            if _display_cache_hit is not None:
                state_db_messages = []
            else:
                if (
                    msg_limit is not None
                    and not getattr(s, "active_stream_id", None)
                    and not getattr(s, "pending_user_message", None)
                ):
                    (
                        state_db_messages,
                        _display_state_db_signature,
                    ) = _load_state_db_messages_with_stable_signature(
                        sid,
                        _session_profile,
                        _state_db_reader_kwargs,
                    )
                else:
                    state_db_messages = get_state_db_session_messages(
                        sid,
                        **_state_db_reader_kwargs,
                    )
        elif not is_messaging_session:
            # Metadata-only callers still need the same append-only
            # reconciliation contract as full loads so stale/replayed
            # state.db rows do not make sidebar polling think the
            # transcript is always newer. Helper threads profile= to
            # honor #2827's TLS-vs-thread fix.
            metadata_summary = _metadata_only_message_summary(sid, profile=_session_profile)
        _t2 = _time.monotonic()
        if _diag: _diag.stage("t2_after_state_db_load")
        effective_model = (
            _resolve_effective_session_model_for_display(s)
            if resolve_model
            else None
        )
        effective_provider = (
            _resolve_effective_session_model_provider_for_display(s)
            if resolve_model
            else None
        )
        _t3 = _time.monotonic()
        if _diag: _diag.stage("t3_after_model_resolve")
        if load_messages:
            if is_messaging_session and cli_messages:
                # Recovery/aggregate sidecars can intentionally contain a
                # longer visible conversation than the single state.db
                # segment for this messaging session id. Prefer the longer
                # sidecar so repaired WebUI history is not hidden behind the
                # canonical per-segment transcript. When both sources carry
                # different slices of the same stitched conversation, merge
                # them chronologically and dedupe exact repeats.
                _all_msgs = _merged_session_messages_for_display(s, cli_messages)
            elif msg_limit is not None:
                if _display_cache_hit is not None:
                    _all_msgs = _display_cache_hit
                else:
                    _all_msgs = _limited_webui_messages_for_display_with_sidecar(
                        s,
                        limited_sidecar_messages,
                        state_db_messages,
                        state_db_signature=_display_state_db_signature,
                        msg_before=msg_before,
                    )
            else:
                state_db_messages = _suppress_native_image_display_mirrors(
                    s,
                    state_db_messages,
                )
                sidecar_messages = _webui_sidecar_lineage_messages_for_display(s)
                lineage_parent = _webui_lineage_parent_session_for_display(s)
                projection_sidecar_messages = _merged_webui_lineage_messages_for_display(
                    s,
                    sidecar_messages,
                    parent_session=lineage_parent,
                )
                _all_msgs = merge_session_messages_append_only(
                    sidecar_messages,
                    state_db_messages,
                    truncation_watermark=getattr(s, "truncation_watermark", None),
                    truncation_boundary=getattr(s, "truncation_boundary", None),
                )
                _all_msgs = _merged_webui_lineage_messages_for_display(
                    s,
                    _all_msgs,
                    parent_session=lineage_parent,
                )
                _all_msgs = _project_native_image_payload_conflicts_for_display(
                    projection_sidecar_messages,
                    state_db_messages,
                    _all_msgs,
                )
        else:
            if is_messaging_session and cli_messages:
                _all_msgs = _merged_session_messages_for_display(s, cli_messages)
            else:
                if metadata_summary is None:
                    metadata_summary = _message_summary(getattr(s, "messages", []) or [])
                _summary_message_count = metadata_summary["message_count"]
                _summary_last_message_at = metadata_summary["last_message_at"]
                _all_msgs = []
        if not load_messages:
            if metadata_summary is None:
                metadata_summary = _message_summary(_all_msgs)
                _summary_message_count = metadata_summary["message_count"]
                _summary_last_message_at = metadata_summary["last_message_at"]
            if _summary_message_count == 0:
                # Legacy session with no loaded sidecar and no state.db summary —
                # fall back to the persisted metadata count from session JSON.
                # See PR #2605 (LumenYoung): without this, the metadata poll
                # returns 0 and the active-session external-refresh signal
                # never trips on legacy sessions.
                try:
                    metadata_count = getattr(s, "_metadata_message_count", None)
                    if metadata_count is not None:
                        _summary_message_count = max(0, int(metadata_count))
                except (TypeError, ValueError):
                    pass
        else:
            _summary_message_count = None
            _summary_last_message_at = None
        if load_messages:
            _truncated_msgs, _messages_offset = _message_window_for_display(
                _all_msgs,
                msg_limit=msg_limit,
                msg_before=msg_before,
                expand_renderable=expand_renderable,
            )
            if msg_limit is not None:
                _truncated_msgs = _messages_for_limited_payload(_truncated_msgs)
            _truncated_msgs = _hydrate_anchor_activity_scenes(
                _truncated_msgs,
                getattr(s, "anchor_activity_scenes", None),
                message_offset=_messages_offset,
                tool_calls=getattr(s, "tool_calls", None),
            )
        else:
            _truncated_msgs = []
            _messages_offset = 0
        # Index of the first returned message in the full message array.
        # Frontend uses this as cursor for scroll-to-top paging.
        # Session-level tool_calls windowing keys off whether the returned
        # message array was actually truncated (msg_before paging, any
        # effective msg_limit) rather than whether a limit parameter was
        # present — the full-transcript shape returns everything, so the
        # length comparison alone decides (#7310/#7625).
        _windowed_messages = (
            load_messages
            and (msg_before is not None or len(_truncated_msgs) < len(_all_msgs))
        )
        # Resolve effective context_length with model-metadata fallback so
        # older sessions (pre-#1318) that have context_length=0 persisted
        # still render a meaningful indicator on load.  Mirrors the
        # SSE-path fallback in api/streaming.py:2333-2342.  Fixes #1436.
        #
        # #1896: pass config_context_length, provider, and custom_providers
        # so explicit config overrides win over the 256K default fallback.
        # Without these, an old session loaded after a user upgraded to a
        # 1M-context model with `model.context_length: 1048576` in
        # config.yaml gets a 256K window in the initial UI indicator and
        # /api/session/get response — the same wrong-window display this
        # fix addresses on the streaming side.
        _persisted_cl = getattr(s, "context_length", 0) or 0
        _threshold_tokens = getattr(s, "threshold_tokens", 0) or 0
        if (not _persisted_cl) or resolve_model:
            _stored_model_for_lookup = getattr(s, "model", "") or ""
            _stored_provider_for_lookup = getattr(s, "model_provider", None) or ""
            _model_for_lookup = (
                effective_model or _stored_model_for_lookup
            ).strip()
            (
                _model_for_lookup,
                _provider_for_lookup,
                _base_url_for_lookup,
                _api_key_for_lookup,
            ) = _session_context_length_lookup_state(
                _model_for_lookup,
                effective_provider or getattr(s, "model_provider", None) or "",
            )
            _fb_cl = _resolve_context_length_for_session_model(
                _model_for_lookup,
                _provider_for_lookup,
                base_url=_base_url_for_lookup,
                api_key=_api_key_for_lookup,
            )
            _model_changed_for_context = not _session_model_identity_matches(
                _stored_model_for_lookup,
                _stored_provider_for_lookup,
                _model_for_lookup,
                _provider_for_lookup,
            )
            if _should_accept_session_context_length_refresh(
                _persisted_cl,
                _fb_cl,
                model_changed=_model_changed_for_context,
            ):
                if _persisted_cl and _fb_cl != _persisted_cl:
                    # The old threshold belongs to the old window. Hiding it
                    # is less useful than keeping the same compression ratio
                    # against the freshly resolved context length.
                    _threshold_tokens = _rescale_threshold_tokens_for_context_window(
                        _threshold_tokens,
                        _persisted_cl,
                        _fb_cl,
                    )
                _persisted_cl = _fb_cl
        _session_tool_calls = getattr(s, "tool_calls", []) if load_messages else []
        # Always include session-level tool_calls so the browser can merge
        # them with per-message tool_calls for messages that lack the
        # per-message variant (older messages whose tool_calls live only
        # in the session-level list).  The browser-side
        # _syncToolCallsForLoadedMessages handles deduplication by tid.
        if _windowed_messages:
            _session_tool_calls = _tool_calls_for_message_window(
                _session_tool_calls,
                _messages_offset,
                len(_truncated_msgs),
            )
        _merged_message_count = _summary_message_count if _summary_message_count is not None else len(_all_msgs)
        _merged_last_message_at = _summary_last_message_at if _summary_last_message_at is not None else 0
        if _summary_last_message_at is None and _all_msgs:
            try:
                _merged_last_message_at = max(
                    float((m or {}).get("timestamp") or 0)
                    for m in _all_msgs
                    if isinstance(m, dict)
                )
            except (TypeError, ValueError):
                _merged_last_message_at = 0
        active_stream_ids = _active_stream_ids()
        try:
            compact_session = s.compact(
                include_runtime=True,
                active_stream_ids=active_stream_ids,
            )
        except TypeError:
            compact_session = s.compact()
        raw = compact_session | {
            "messages": _truncated_msgs,
            "message_count": _merged_message_count,
            "tool_calls": _session_tool_calls,
            "active_stream_id": getattr(s, "active_stream_id", None),
            "pending_user_message": getattr(s, "pending_user_message", None),
            "pending_attachments": getattr(s, "pending_attachments", []) if (load_messages or getattr(s, "pending_user_message", None)) else [],
            "pending_started_at": getattr(s, "pending_started_at", None),
            "pending_user_source": getattr(s, "pending_user_source", None),
            "context_length": _persisted_cl,
            "threshold_tokens": _threshold_tokens,
            "last_prompt_tokens": getattr(s, "last_prompt_tokens", 0) or 0,
        }
        if original_stream_id:
            try:
                journal = find_run_summary(original_stream_id)
            except Exception:
                journal = None
            if journal:
                journal_active = bool(original_stream_id in active_stream_ids)
                raw["runtime_journal"] = _run_journal_status_payload(
                    journal,
                    active=journal_active,
                )
                if journal_active and (not load_messages or msg_limit is None):
                    try:
                        snapshot = _run_journal_live_snapshot(original_stream_id, handler=handler)
                    except Exception:
                        logger.debug(
                            "Failed to build runtime journal snapshot for %s",
                            original_stream_id,
                            exc_info=True,
                        )
                        snapshot = None
                    if snapshot:
                        raw["runtime_journal_snapshot"] = _runtime_journal_snapshot_for_session_payload(snapshot)
                        raw["pending_attachments"] = getattr(s, "pending_attachments", []) or []
        # Cold-load: derive the latest settled todo snapshot from the full
        # merged transcript, not the truncated display window. This keeps
        # the Todos panel correct after refresh even when the latest todo
        # tool result is outside msg_limit, and treats an explicit empty
        # todo list as the current state instead of falling through to an
        # older non-empty write.
        if load_messages and _all_msgs:
            attach_todo_state(raw, _all_msgs)
        if _merged_last_message_at:
            raw["last_message_at"] = max(
                float(raw.get("last_message_at") or 0),
                _merged_last_message_at,
            )
            raw["updated_at"] = max(
                float(raw.get("updated_at") or 0),
                _merged_last_message_at,
            )
        # #2980: surface the visible continuation for a hidden pre-compression
        # snapshot so a mobile reload mid-compression can recover to it.
        continuation_sid = _pre_compression_continuation_session_id(s)
        if continuation_sid:
            raw["continuation_session_id"] = continuation_sid
        if cli_meta and _session_source_is_webui(cli_meta):
            raw = _reconcile_session_detail_source_flags(raw, cli_meta)
        elif cli_meta and _is_messaging_session_record(cli_meta):
            raw = _merge_cli_sidebar_metadata(raw, cli_meta)
            # ``message_count`` in /api/session is the display coordinate
            # space used for pagination and the header badge. Messaging
            # state.db metadata can include raw duplicate transport rows that
            # _merged_session_messages_for_display() intentionally dedupes;
            # keep the raw count available as ``actual_message_count`` but
            # do not let it make the frontend expect phantom messages.
            raw["message_count"] = _merged_message_count
        # Signal to the frontend that older messages were omitted. The
        # message window cursor already reflects visible-row pagination and
        # avoids false positives when raw hidden tool rows exceed msg_limit.
        _truncated = load_messages and msg_limit is not None and _messages_offset > 0
        raw["_messages_truncated"] = _truncated
        raw["_messages_offset"] = _messages_offset
        raw["_msg_limit_max"] = _MAX_MSG_LIMIT
        _t4 = _time.monotonic()
        if _diag: _diag.stage("t4_after_compact_and_merge")
        if effective_model:
            raw["model"] = effective_model
        if effective_provider:
            raw["model_provider"] = effective_provider
        # A subagent child (#5307) is view-only regardless of what a stale
        # sidecar stored: coerce the serialized flags so the browser never
        # treats an existing subagent sidecar as writable / CLI-classified.
        if (
            (str(raw.get("source_tag") or raw.get("raw_source") or raw.get("session_source") or "").strip().lower() == "subagent")
            or _is_subagent_child_session_id(sid)
        ):
            raw["is_cli_session"] = False
            raw["read_only"] = True
        imported_turn_marker = any(
            isinstance(row, dict) and row.get("_active_turn_token")
            for row in _all_msgs
        )
        if (
            not raw.get("read_only")
            and not _truncated
            and (not raw.get("is_cli_session") or imported_turn_marker)
        ):
            from api.session_ops import regeneration_authority, regeneration_state
            canonical_state = regeneration_state(s)
            revision = regeneration_authority(
                s,
                rows=canonical_state[0],
                context=canonical_state[1],
                full_transcript=True,
                canonical_state=canonical_state,
            )
            if revision:
                raw["regeneration_revision"] = revision
        redact = redact_session_data(raw)
        _t5 = _time.monotonic()
        if _diag: _diag.stage("t5_after_redact")
        resp = j(handler, {"session": redact})
        _t6 = _time.monotonic()
        if _diag: _diag.stage("t6_after_json_write")
        _total_ms = (_t6 - _t0) * 1000
        # Always log when slow (>2s) so we don't need HERMES_DEBUG_SLOW env var
        # to diagnose latency regressions. Opt-in env var still forces
        # logging on every request for development.
        if _debug_slow or _total_ms >= 2000:
            # perf(webui/session-load-latency) tier2c: route the [SLOW] line
            # through handler._safe_webui_print() rather than logger.warning().
            # The WebUI process starts the root logger without any handler, so
            # logger.warning() calls are silently dropped (the [SLOW] line
            # previously worked only on PIDs that happened to have a logger
            # handler set up by an earlier run; today the line is invisible).
            # _safe_webui_print writes to the systemd journal socket directly,
            # same as the per-request ms line — which is why THAT line keeps
            # working.
            handler._safe_webui_print(
                "[SLOW] session_id=%s get_session=%.1fms model_resolve=%.1fms "
                "compact=%.1fms redact=%.1fms json_write=%.1fms total=%.1fms" % (
                    sid,
                    (_t2-_t1)*1000, (_t3-_t2)*1000, (_t4-_t3)*1000,
                    (_t5-_t4)*1000, (_t6-_t5)*1000, _total_ms,
                )
            )
        if _diag: _diag.finish()
        return resp
    except KeyError:
        # perf(webui/session-load-latency) tier2c-followup: fire
        # _diag.finish() in the exception branch too. Greptile flagged
        # this in PR review — finish() unregisters the pending watchdog
        # entry; without it the entry stays for the full 5s slow-request
        # timeout and emits a spurious "Slow WebUI request still
        # running" log. Idempotent — finish() no-ops if already called.
        if _diag: _diag.finish()
        # No WebUI sidecar. Delegate to the shared foreign-session
        # synthesizer so GET and POST have symmetric writeable/read-only
        # behaviour for CLI/TUI/Desktop sessions. The helper enforces the
        # #2782 deleted-WebUI-session 404 contract (via
        # _session_index_marks_was_webui) and the #4911 source ownership
        # gate (via _is_claimable_cli_source) so the two endpoints can't
        # drift on foreign-session semantics.
        cli_meta = _lookup_cli_session_metadata(sid)
        _session_profile = (cli_meta or {}).get("profile") or None
        # Claude Code rows are profile-less by construction (they come from
        # ~/.claude/projects, not from any profile's state.db), so the gate
        # below would 404 every one of them under a named active profile
        # even though /api/sessions happily lists them. Exempt them.
        _profile_agnostic = _is_profile_agnostic_foreign_session(cli_meta)
        if not _profile_agnostic and not _session_visible_to_active_profile(_session_profile, handler):
            if _session_profile:
                # Valid CLI/foreign session owned by a KNOWN other profile:
                # 409 so the client can offer to switch to it (#5419).
                return j(handler, {
                    "error": "Session belongs to a different profile",
                    "code": "session_profile_mismatch",
                    "session_id": sid,
                    "profile": _session_profile,
                }, status=409)
            # Missing session (cli_meta={} -> profile=None): keep the 404
            # self-heal path. _profiles_match coerces None->'default', so a
            # truly-missing session under a non-default active profile would
            # otherwise emit a useless 409 with profile=null and skip the
            # frontend self-heal + spin the SSE reconnect against a dead sid.
            return bad(handler, "Session not found", 404)
        synth, reason = _claim_or_synthesize_cli_session(sid, cli_meta=cli_meta or {})
        if reason == "was_webui":
            # Deleted WebUI session: 404 so the client self-heals
            # (clears stale /session/<id> URL and localStorage, #2782).
            return bad(handler, "Session not found", 404)
        if synth is None:
            # 'no_foreign_state' / 'invalid_sid' — nothing to render.
            return bad(handler, "Session not found", 404)
        # Build the legacy dict response from the synthesized Session so
        # the wire shape stays byte-equivalent to the previous inline
        # synthesis (the frontend has been reading these exact keys).
        msgs = list(synth.messages or [])
        sess = {
            "session_id": synth.session_id,
            "title": synth.title,
            "workspace": synth.workspace,
            "model": synth.model,
            "message_count": len(msgs),
            "created_at": synth.created_at,
            "updated_at": synth.updated_at,
            "last_message_at": (
                (cli_meta or {}).get("last_message_at")
                or (cli_meta or {}).get("updated_at", 0)
                or ((msgs or [{}])[-1].get("timestamp", 0))
            ),
            "pinned": bool(getattr(synth, "pinned", False)),
            "archived": bool(getattr(synth, "archived", False)),
            "project_id": getattr(synth, "project_id", None),
            "profile": synth.profile,
            # Read is_cli_session from the synthesized Session, not a
            # hardcoded True: delegated subagent children (#5307) are
            # recovered read-only with is_cli_session=False so they don't
            # pass the frontend _isExternalSession poll-skip / active-refresh
            # gates (#3603). Every other synthesized foreign session keeps
            # is_cli_session=True so its source badge renders.
            "is_cli_session": bool(getattr(synth, "is_cli_session", False)),
            "source_tag": synth.source_tag,
            "raw_source": synth.raw_source,
            "session_source": synth.session_source,
            "source_label": synth.source_label,
            # Greptile #4911 follow-up: read read_only from the
            # synthesized Session, NOT from cli_meta directly.
            # The helper sets synth.read_only=True for BOTH
            # explicit read_only=True cli_meta AND source-refused
            # sessions (messaging / claude_code / external_agent).
            # cli_meta.get("read_only") is only populated for the
            # explicit case, so reading it from there causes the
            # frontend to render the composer for source-refused
            # sessions and the user only discovers the block at
            # POST time with a confusing 403.
            "read_only": bool(getattr(synth, "read_only", False)),
            "messages": msgs,
            "tool_calls": [],
        }
        attach_todo_state(sess, msgs)
        sess = _merge_cli_sidebar_metadata(sess, cli_meta)
        return j(handler, {"session": public_session_projection(sess)})


def handle_get(handler, parsed) -> bool:
    """Handle all GET routes. Returns True if handled, False for 404."""
    proxy_result = _handle_extension_sidecar_proxy(handler, parsed, "GET")
    if proxy_result is not False:
        return proxy_result

    if parsed.path.startswith("/session/static/"):
        # Strip the leading "/session" so _serve_static() sees a path that
        # starts with "/static/" (its required prefix). _serve_static enforces
        # its own path-traversal sandbox via Path.resolve()+relative_to().
        stripped = parsed._replace(path=parsed.path[len("/session"):])
        return _serve_static(handler, stripped)

    # Firefox Android resolves <link rel="manifest"> against the page URL
    # before the dynamic <base href> script runs when installing from
    # /session/<id>, producing requests like /session/manifest.json.
    # Without this guard the catch-all below returns index.html instead of
    # the manifest, and Firefox falls back to a generated letter icon.
    # See #2226.
    if parsed.path in ("/session/manifest.json", "/session/manifest.webmanifest"):
        return _serve_manifest(handler)

    if parsed.path in ("/", "/index.html", "/sessions") or parsed.path.startswith("/session/"):
        try:
            from api.extensions import inject_extension_tags

            csrf_token = ""
            try:
                from api.auth import csrf_token_for_session, is_auth_enabled, parse_cookie, verify_session

                if is_auth_enabled():
                    cookie_val = parse_cookie(handler)
                    if not cookie_val:
                        cookie_val = getattr(handler, "_trusted_auth_session_cookie_value", None)
                    if cookie_val and verify_session(cookie_val):
                        csrf_token = csrf_token_for_session(cookie_val) or ""
            except Exception:
                csrf_token = ""

            # The disk read + process-constant token substitutions are cached;
            # only the per-session CSRF token and per-request extension tags are
            # applied here (see _render_index_shell_base).
            html = _render_index_shell_base().replace(
                "__CSRF_TOKEN_JSON__", json.dumps(csrf_token)
            )
            return t(
                handler,
                inject_extension_tags(html),
                content_type="text/html; charset=utf-8",
            )
        except Exception as exc:
            return _serve_shell_unavailable(handler, exc)

    if parsed.path == "/share" or parsed.path.startswith("/share/"):
        share_path = (Path(__file__).parent.parent / "static" / "share.html").resolve()
        return t(
            handler,
            share_path.read_text(encoding="utf-8"),
            content_type="text/html; charset=utf-8",
            extra_headers={
                "X-Robots-Tag": "noindex, nofollow",
            },
        )

    if parsed.path == "/login":
        _settings = load_settings()
        _bn = _html.escape(_settings.get("bot_name") or "Hermes")
        _lang = _settings.get("language", "en")
        _login_strings = _LOGIN_LOCALE[
            _resolve_login_locale_key(_lang)
        ]
        from urllib.parse import quote
        from api.updates import WEBUI_VERSION
        # #7056: only render the password input / submit / passkey controls
        # when password auth is actually enabled. With native OIDC configured
        # and ``HERMES_WEBUI_PASSWORD`` unset, the form previously still
        # displayed the password prompt and accepted — silently 401-ing at
        # the server — every submit. The OIDC SSO entry point stays the
        # sole path. ``is_password_auth_enabled`` is the same predicate
        # ``/api/auth/status`` reports as ``password_auth_enabled``.
        from api.auth import are_passkeys_enabled, is_password_auth_enabled

        # The password INPUT is gated on a configured password, but the passkey
        # button must survive a passwordless-passkey deployment: settings expose
        # ``passwordless_enabled = passkeys registered AND not password_auth_enabled``
        # (routes.py ~14059) and ``is_auth_enabled()`` counts passkeys as an
        # independent auth method, so hiding the button when no password is set
        # would remove the ONLY working login affordance for those instances.
        _passkey_button_html = (
            '<button type="button" id="passkey-login" class="passkey-login" '
            'style="display:none">Sign in with passkey</button>'
        )
        if is_password_auth_enabled():
            _password_form_html = (
                f'<input type="password" id="pw" '
                f'placeholder="{_html.escape(_login_strings["placeholder"])}" autofocus>'
                f'<button type="submit">{_html.escape(_login_strings["btn"])}</button>'
                f'{_passkey_button_html}'
            )
        elif are_passkeys_enabled():
            _password_form_html = _passkey_button_html
        else:
            _password_form_html = ""
        version_token = quote(WEBUI_VERSION, safe="")
        _page = (
            _LOGIN_PAGE_HTML.replace("{{BOT_NAME}}", _bn)
            .replace("{{BOT_NAME_INITIAL}}", _bn[0].upper())
            .replace("{{WEBUI_VERSION}}", version_token)
            .replace("{{LANG}}", _html.escape(_login_strings["lang"]))
            .replace("{{LOGIN_TITLE}}", _html.escape(_login_strings["title"]))
            .replace("{{LOGIN_SUBTITLE}}", _html.escape(_login_strings["subtitle"]))
            .replace("{{PASSWORD_FORM_HTML}}", _password_form_html)
            .replace("{{LOGIN_INVALID_PW}}", _html.escape(_login_strings["invalid_pw"]))
            .replace(
                "{{LOGIN_CONN_FAILED}}", _html.escape(_login_strings["conn_failed"])
            )
            .replace("{{OIDC_LOGIN_HTML}}", _oidc_login_html(parsed))
        )
        return t(handler, _page, content_type="text/html; charset=utf-8")

    if parsed.path == "/api/auth/oidc/start":
        from api.auth_oidc import OIDCAuthError, OIDCConfigError, build_authorization_redirect

        next_path = _safe_login_redirect_path(
            parse_qs(parsed.query or "").get("next", [""])[0]
        )
        try:
            location = build_authorization_redirect(
                _request_base_url(handler), next_path
            )
        except OIDCConfigError as exc:
            return j(handler, {"error": str(exc)}, status=404)
        except OIDCAuthError as exc:
            return j(handler, {"error": str(exc)}, status=exc.status_code)
        handler.send_response(302)
        handler.send_header("Location", location)
        handler.send_header("Cache-Control", "no-store")
        handler.send_header("Content-Length", "0")
        _security_headers(handler)
        handler.end_headers()
        return True

    if parsed.path == "/api/auth/oidc/callback":
        from api.auth import create_session, set_auth_cookie
        from api.auth_oidc import OIDCAuthError, OIDCConfigError, complete_authorization_code_flow

        query = parse_qs(parsed.query or "")
        error = str(query.get("error", [""])[0] or "").strip()
        if error:
            description = str(query.get("error_description", [""])[0] or "").strip()
            return j(handler, {"error": description or error}, status=401)
        state = str(query.get("state", [""])[0] or "").strip()
        code = str(query.get("code", [""])[0] or "").strip()
        if not state or not code:
            return j(handler, {"error": "Missing OIDC callback state or code"}, status=400)
        try:
            result = complete_authorization_code_flow(
                _request_base_url(handler), state, code
            )
        except OIDCConfigError as exc:
            return j(handler, {"error": str(exc)}, status=404)
        except OIDCAuthError as exc:
            return j(handler, {"error": str(exc)}, status=exc.status_code)
        cookie_val = create_session()
        handler.send_response(302)
        handler.send_header(
            "Location",
            _safe_login_redirect_path(result.get("next_path")),
        )
        handler.send_header("Cache-Control", "no-store")
        _security_headers(handler)
        set_auth_cookie(handler, cookie_val)
        handler.send_header("Content-Length", "0")
        handler.end_headers()
        return True

    if parsed.path == "/api/auth/status":
        from api.auth import (
            _passkey_feature_flag_enabled,
            ensure_trusted_auth_session,
            get_password_hash,
            is_auth_enabled,
            is_oidc_auth_enabled,
            is_trusted_auth_enabled,
        )
        from api.passkeys import registered_credentials

        logged_in = False
        session_info = None
        auth_enabled = is_auth_enabled()
        oidc_enabled = is_oidc_auth_enabled()
        if auth_enabled:
            session_info = ensure_trusted_auth_session(handler)
            logged_in = bool(session_info)
        passkey_flag = _passkey_feature_flag_enabled()
        passkeys = registered_credentials() if passkey_flag else []
        password_auth_enabled = get_password_hash() is not None
        payload = {
            "auth_enabled": auth_enabled,
            "logged_in": logged_in,
            "oidc_enabled": oidc_enabled,
            "password_auth_enabled": password_auth_enabled,
            "passwordless_enabled": bool(passkeys) and not password_auth_enabled,
            "passkeys_enabled": bool(passkeys),
            "passkeys_count": len(passkeys),
            "passkey_feature_flag": passkey_flag,
            "auth_disabled_acknowledged": bool(load_settings().get("auth_disabled_acknowledged")) if not auth_enabled else False,
        }
        if is_trusted_auth_enabled() or (session_info and session_info.get("auth_type") == "trusted"):
            payload["trusted_auth_enabled"] = True
        if session_info and session_info.get("auth_type") == "trusted":
            payload["auth_type"] = session_info.get("auth_type")
            payload["user"] = session_info.get("username")
            payload["bound_profile"] = session_info.get("bound_profile")
        return j(handler, payload)

    if parsed.path.startswith("/api/share/"):
        token = parsed.path[len("/api/share/"):].strip()
        share = load_share(token)
        if not share:
            return bad(handler, "Shared conversation not found", 404)
        return j(
            handler,
            {"share": share},
            extra_headers={
                "Cache-Control": "no-store",
                "X-Robots-Tag": "noindex, nofollow",
            },
        )

    if parsed.path in ("/manifest.json", "/manifest.webmanifest"):
        return _serve_manifest(handler)

    if parsed.path == "/sw.js":
        static_root = api_config.get_static_root()
        sw_path = (static_root / "sw.js").resolve()
        if sw_path.exists():
            # Inject the current git-derived version as the cache name so the
            # service worker cache busts automatically on every new deploy.
            from urllib.parse import quote
            from api.updates import WEBUI_VERSION
            version_token = quote(WEBUI_VERSION, safe="")
            text = sw_path.read_text(encoding="utf-8").replace(
                "__WEBUI_VERSION__", version_token
            )
            data = text.encode("utf-8")
            handler.send_response(200)
            handler.send_header("Content-Type", "application/javascript; charset=utf-8")
            handler.send_header("Cache-Control", "no-store")
            handler.send_header("Service-Worker-Allowed", "/")
            handler.send_header("Content-Length", str(len(data)))
            handler.end_headers()
            handler.wfile.write(data)
            return True
        return j(handler, {"error": "not found"}, status=404)

    if parsed.path == "/favicon.ico":
        static_root = api_config.get_static_root()
        ico_path = (static_root / "favicon.ico").resolve()
        if ico_path.exists() and ico_path.is_file():
            data = ico_path.read_bytes()
            handler.send_response(200)
            handler.send_header("Content-Type", "image/x-icon")
            handler.send_header("Content-Length", str(len(data)))
            handler.send_header("Cache-Control", "public, max-age=86400")
            handler.end_headers()
            handler.wfile.write(data)
        else:
            handler.send_response(204)
            handler.end_headers()
        return True

    if parsed.path.startswith("/api/") and not _guard_request_session_visibility(handler, parsed, method="GET"):
        return True

    # ── Insights / knowledge status ──
    if parsed.path == "/api/insights":
        return _handle_insights(handler, parsed)
    if parsed.path == "/api/project-os/dashboard":
        return _handle_project_os_dashboard(handler, parsed)

    if parsed.path.startswith("/api/kanban/"):
        from api.kanban_bridge import handle_kanban_get

        # Only treat an explicit False as "no route matched". None means the
        # bridge already sent a response via bad()/j() — emitting our own 404
        # on top of that produces concatenated JSON bodies on the wire.
        result = handle_kanban_get(handler, parsed)
        if result is False:
            return _kanban_unknown_endpoint(handler, parsed, "GET")
        return True
    if parsed.path == "/api/wiki/status":
        return _handle_llm_wiki_status(handler, parsed)
    if parsed.path == "/api/wiki/browse":
        wiki_root, _, _ = _llm_wiki_resolve_path()
        if not wiki_root or not os.path.isdir(wiki_root):
            return bad(handler, "Wiki not configured or directory not found", status=404)
        allowlisted_entries = _llm_wiki_allowlisted_entries(Path(wiki_root))
        pages = []
        for rel_path, (fp, identity) in sorted(allowlisted_entries.items(), key=lambda item: item[0].lower()):
            try:
                st = fp.stat()
            except OSError:
                continue
            if (st.st_dev, st.st_ino) != identity:
                continue
            pages.append({"name": Path(rel_path).name, "path": rel_path, "size": st.st_size, "mtime": int(st.st_mtime)})
        return j(handler, {"pages": pages})
    if parsed.path == "/api/wiki/page":
        wiki_root, _, _ = _llm_wiki_resolve_path()
        page_path = parse_qs(parsed.query or "").get("path", [""])[0]
        if not wiki_root or not page_path:
            return bad(handler, "Wiki not configured or path not provided", status=400)
        if "\\" in page_path:
            return bad(handler, "Invalid path", status=400)
        # Reject a real `..` path SEGMENT (or absolute path), not the bare
        # substring — a legitimate listed filename like `v1..v2.md` contains
        # ".." without being traversal. Containment + the resolved-allowlist
        # membership check below are the actual security boundary.
        requested_key = page_path.replace("\\", "/")
        _page_parts = requested_key.split("/")
        if os.path.isabs(page_path) or any(part == ".." for part in _page_parts):
            return bad(handler, "Invalid path", status=400)
        if any(part in ("", ".") for part in _page_parts):
            return bad(handler, "Invalid path", status=400)
        full_path = Path(os.path.join(wiki_root, page_path))
        if not _skill_path_within(Path(wiki_root), full_path):
            return bad(handler, "Invalid path", status=400)
        try:
            wiki_real = Path(wiki_root).resolve()
        except OSError:
            return bad(handler, "Page not found", status=404)
        # Only serve files the browse/list path would surface (same allowlist:
        # *.md under the wiki page-dirs, no dotfiles, forbidden-roots guard).
        # Without this the read endpoint could return ANY file inside the wiki
        # root (e.g. .env / .git/config / non-.md), since containment alone
        # doesn't constrain which files are readable (Opus review finding).
        # Capture each allowlisted page's STABLE IDENTITY (st_dev, st_ino) so the
        # post-open fstat below can detect a file/parent-dir swapped in after the
        # allowlist check (TOCTOU write-race, Codex finding) — a pathname re-open
        # alone can't, since O_NOFOLLOW only guards the final component, not a
        # swapped parent directory.
        allowed_identity = _llm_wiki_allowlisted_entries(wiki_real)
        try:
            resolved_target = full_path.resolve()
        except OSError:
            return bad(handler, "Page not found", status=404)
        requested_entry = allowed_identity.get(requested_key)
        if requested_entry is None:
            return bad(handler, "Page not found", status=404)
        allowlisted_target, allowlisted_identity = requested_entry
        if resolved_target != allowlisted_target:
            return bad(handler, "Page not found", status=404)
        # Read the ALREADY-RESOLVED, allowlisted real path with O_NOFOLLOW so a
        # symlink swapped in for the final component between the allowlist check
        # and the read is refused rather than followed. Then fstat the open fd
        # and require its (st_dev, st_ino) to match the identity captured during
        # allowlisting — this closes a parent-directory swap that O_NOFOLLOW
        # would otherwise follow. Any mismatch / vanished / swapped page returns
        # a clean 404, never a 500.
        try:
            fd = os.open(str(resolved_target), os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
            try:
                st_open = os.fstat(fd)
                if (st_open.st_dev, st_open.st_ino) != allowlisted_identity:
                    return bad(handler, "Page not found", status=404)
                raw = os.read(fd, _LLM_WIKI_MAX_PAGE_BYTES + 1)
            finally:
                os.close(fd)
            if len(raw) > _LLM_WIKI_MAX_PAGE_BYTES:
                raw = raw[:_LLM_WIKI_MAX_PAGE_BYTES]
            content = raw.decode("utf-8", errors="replace")
        except (FileNotFoundError, IsADirectoryError):
            return bad(handler, "Page not found", status=404)
        except OSError:
            # ELOOP (symlink swapped in under O_NOFOLLOW) or any other read
            # failure → clean 404, never a 500.
            return bad(handler, "Could not read page", status=404)
        return j(handler, {"content": content, "path": page_path})
    if parsed.path == "/api/logs":
        return _handle_logs(handler, parsed)

    if parsed.path == "/health":
        return _handle_health(handler, parsed)

    if parsed.path == "/api/health/agent":
        payload = build_agent_health_payload()
        payload["gateway_chat"] = gateway_chat_config_status()
        j(handler, payload)
        return True

    if parsed.path == "/api/system/health":
        j(handler, build_system_health_payload())
        return True

    if parsed.path == "/api/models":
        # Profile-scoping for non-default profiles (#3957) is handled INSIDE
        # get_available_models() — it binds the active profile's env + TLS on
        # the detached rebuild worker (and the legacy synchronous rebuild),
        # which the request-thread wrapper could not reach. See
        # api.config.get_available_models cold path + profile_scope_for_detached_worker.
        freshness = parse_qs(parsed.query or "").get("freshness", [""])[0].strip().lower()
        diag = RequestDiagnostics.maybe_start("GET", parsed.path, logger=logger, print_fn=getattr(handler, '_safe_webui_print', None))
        try:
            diag.stage(f"enter:freshness={freshness or 'default'}") if diag else None
            if freshness == "session_visit":
                result = get_available_models_for_session_visit()
                diag.stage("response_serialize") if diag else None
                return j(handler, result)
            if freshness:
                return bad(handler, f"unknown models freshness: {freshness}", status=400)
            return j(handler, get_available_models())
        finally:
            if diag:
                diag.finish()

    if parsed.path == "/api/models/live":
        from api.profiles import profile_env_for_active_request
        with profile_env_for_active_request("/api/models/live", logger_override=logger):
            return _handle_live_models(handler, parsed)

    # ── Auxiliary models (GET/POST) ──
    if parsed.path == "/api/model/auxiliary":
        from api.config import get_auxiliary_models
        return j(handler, get_auxiliary_models())

    if parsed.path == "/api/dashboard/status":
        from api import dashboard_probe

        j(handler, dashboard_probe.get_dashboard_status())
        return True

    if parsed.path == "/api/dashboard/config":
        from api import dashboard_probe

        try:
            j(handler, dashboard_probe.get_dashboard_config())
        except ValueError as exc:
            bad(handler, str(exc), status=400)
        return True

    # ── Providers (GET) ──
    if parsed.path == "/api/providers":
        # Apply the active per-request profile's env so provider auth probes
        # resolve against that profile's credentials, not the process-default
        # profile's (#3957). Without this, get_auth_status() probes on a
        # non-default profile resolve the wrong/empty creds and can stall past
        # the 30s frontend timeout. No-op for the default profile.
        from api.profiles import profile_env_for_active_request_readonly
        with profile_env_for_active_request_readonly("/api/providers", logger_override=logger):
            return j(handler, get_providers())

    # ── Plugins/hooks visibility (read-only, no callback/source internals) ──
    if parsed.path == "/api/plugins":
        return _handle_plugins(handler, parsed)
    if parsed.path == "/api/provider/quota":
        query = parse_qs(parsed.query)
        provider_id = (query.get("provider", [""])[0] or None)
        refresh = (query.get("refresh", [""])[0] or "").strip().lower() in {"1", "true", "yes", "on"}
        # Bind the active request's profile env (matches /api/providers and
        # /api/models/live). #4365 added a credential_pool.load_pool() path in
        # get_provider_quota for all pooled providers; without this wrapper that
        # read/write runs under the process-default profile, so a multi-profile
        # client would see (and seed) the default profile's pool instead of its
        # own (#4247/#4067 profile-isolation class).
        from api.profiles import profile_env_for_active_request_readonly
        with profile_env_for_active_request_readonly("/api/provider/quota", logger_override=logger):
            return j(handler, get_provider_quota(provider_id, refresh=refresh))

    if parsed.path == "/api/provider/cost-history":
        query = parse_qs(parsed.query)
        provider_id = (query.get("provider", [""])[0] or None)
        days_raw = (query.get("days", ["7"])[0] or "7").strip()
        try:
            days = max(1, min(int(days_raw), 365))
        except (ValueError, TypeError):
            days = 7
        return j(handler, get_provider_cost_history(provider_id, days))

    if parsed.path == "/api/settings":
        settings = load_settings()
        settings["persisted_speech_keys"] = persisted_speech_settings_keys()
        # Never expose the stored password hash to clients
        settings.pop("password_hash", None)
        settings.setdefault("max_tokens", None)
        settings.setdefault("max_tokens_effective", None)
        settings.setdefault("max_tokens_fallback", None)
        try:
            from api.config import get_max_tokens_status
            settings.update(get_max_tokens_status())
        except Exception:
            settings["max_tokens"] = None
            settings["max_tokens_effective"] = None
            settings["max_tokens_fallback"] = None
        # Surface env-var precedence so the UI can disable the password field
        # instead of silently no-oping the save (#1560). The setting takes
        # precedence in api.auth.get_password_hash(), but until now the UI
        # had no way to know — see issue #1139 / #1560.
        settings["password_env_var"] = bool(
            os.getenv("HERMES_WEBUI_PASSWORD", "").strip()
        )
        # Auth-state fields for frontend safety badge / confirmation flows
        from api.auth import get_password_hash, is_auth_enabled
        settings["auth_enabled"] = is_auth_enabled()
        settings["password_auth_enabled"] = get_password_hash() is not None
        try:
            from api.auth import _passkey_feature_flag_enabled as _pffe
            from api.passkeys import registered_credentials as _rc
            if _pffe():
                settings["passkeys_enabled"] = bool(_rc())
                settings["passwordless_enabled"] = bool(_rc()) and not settings["password_auth_enabled"]
            else:
                settings["passkeys_enabled"] = False
                settings["passwordless_enabled"] = False
        except Exception:
            pass
        # Inject the running version so the UI badge stays in sync with git tags
        # without any manual release step.
        try:
            from api.updates import AGENT_VERSION, WEBUI_VERSION
            settings["webui_version"] = WEBUI_VERSION
            settings["agent_version"] = AGENT_VERSION
        except Exception:
            pass
        # Channel-scoped display badge — SEPARATE from webui_version (which is
        # load-bearing for asset cache-busting / SW cache / skew detection and
        # must stay channel-neutral). update_channel_version is display-only.
        try:
            from api.updates import channel_version_badge, _read_update_channel
            channel = _read_update_channel()
            settings["update_channel"] = channel
            settings["update_channel_version"] = channel_version_badge(channel)
        except Exception:
            pass
        return j(handler, settings)

    if parsed.path == "/api/transcribe/capability":
        return handle_transcribe_capability(handler)

    if parsed.path == "/api/reasoning":
        # Current reasoning config (shared source of truth with the CLI —
        # reads display.show_reasoning and agent.reasoning_effort from
        # the active profile's config.yaml).
        query = parse_qs(parsed.query)
        model_id = (query.get("model", [""])[0] or "").strip() or None
        provider_id = (query.get("provider", [""])[0] or "").strip() or None
        base_url = (query.get("base_url", [""])[0] or "").strip() or None
        return j(
            handler,
            get_reasoning_status(
                model_id=model_id,
                provider_id=provider_id,
                base_url=base_url,
            ),
        )

    if parsed.path == "/api/onboarding/status":
        return j(handler, get_onboarding_status())

    if parsed.path == "/api/extensions/status":
        from api.extensions import get_extension_status

        return j(handler, get_extension_status())

    if parsed.path == "/api/extensions/registry":
        from api.extensions import get_extension_registry

        return j(handler, get_extension_registry())

    if parsed.path.startswith("/extensions/"):
        from api.extensions import serve_extension_static

        return serve_extension_static(handler, parsed)

    if parsed.path.startswith("/static/"):
        return _serve_static(handler, parsed)


    if parsed.path == "/api/session/worktree/status":
        query = parse_qs(parsed.query)
        sid = query.get("session_id", [""])[0]
        if not sid:
            return bad(handler, "session_id is required", status=400)
        try:
            s = get_session(sid, metadata_only=True)
        except KeyError:
            return bad(handler, "Session not found", status=404)
        try:
            from api.worktrees import worktree_status_for_session

            return j(handler, {"status": worktree_status_for_session(s)})
        except ValueError as exc:
            return bad(handler, str(exc), status=400)
        except Exception as exc:
            logger.exception("failed to read worktree status for session %s", sid)
            return bad(handler, _sanitize_error(exc), status=500)

    if parsed.path == "/api/session/compress/status":
        query = parse_qs(parsed.query)
        _handle_session_compress_status(handler, query.get("session_id", [""])[0])
        return True

    if parsed.path == "/api/session":
        return _handle_session_get(handler, parsed)

    if parsed.path == "/api/session/lineage/report":
        sid = parse_qs(parsed.query).get("session_id", [""])[0]
        if not sid:
            return bad(handler, "session_id required", 400)
        report = read_session_lineage_report(_active_state_db_path(), sid)
        if not report.get("found"):
            return bad(handler, "Session not found", 404)
        return j(handler, report)

    if parsed.path == "/api/session/recovery/audit":
        from api.session_recovery import audit_session_recovery
        return j(handler, audit_session_recovery(SESSION_DIR, state_db_path=_active_state_db_path()))

    if parsed.path == "/api/session/status":
        sid = parse_qs(parsed.query).get("session_id", [""])[0]
        if not sid:
            return bad(handler, "Missing session_id")
        try:
            from api.session_ops import session_status
            _clear_stale_stream_state(get_session(sid, metadata_only=True))
            return j(handler, session_status(sid))
        except KeyError:
            return bad(handler, "Session not found", 404)

    if parsed.path == "/api/session/yolo":
        sid = parse_qs(parsed.query).get("session_id", [""])[0]
        if not sid:
            return bad(handler, "Missing session_id")
        return j(handler, {"yolo_enabled": is_session_yolo_enabled(sid)})

    if parsed.path == "/api/session/usage":
        sid = parse_qs(parsed.query).get("session_id", [""])[0]
        if not sid:
            return bad(handler, "Missing session_id")
        try:
            from api.session_ops import session_usage
            return j(handler, session_usage(sid))
        except KeyError:
            return bad(handler, "Session not found", 404)

    if parsed.path == "/api/background/status":
        sid = parse_qs(parsed.query).get("session_id", [""])[0]
        if not sid:
            return bad(handler, "Missing session_id")
        from api.background import get_results
        return j(handler, {"results": get_results(sid)})

    if parsed.path == "/api/sessions":
        diag = RequestDiagnostics.maybe_start("GET", parsed.path, logger=logger, print_fn=getattr(handler, '_safe_webui_print', None))
        try:
            from api import profiles as profiles_api

            diag.stage("load_settings")
            settings = load_settings()
            show_cli_sessions = bool(settings.get("show_cli_sessions"))
            show_claude_code_sessions = bool(settings.get("show_claude_code_sessions"))
            show_previous_messaging_sessions = bool(
                settings.get("show_previous_messaging_sessions")
            )
            show_cron_sessions = bool(settings.get("show_cron_sessions"))
            show_webhook_sessions = bool(settings.get("show_webhook_sessions"))
            show_kanban_sessions = bool(settings.get("show_kanban_sessions"))
            agent_session_source_filter = settings.get("agent_session_source_filter")
            active_profile = profiles_api.get_active_profile_name()
            all_profiles = _all_profiles_enabled(parsed)
            include_archived = _query_flag(parsed, "include_archived")
            exclude_hidden = _query_flag(parsed, "exclude_hidden")
            archived_limit = _query_positive_int(parsed, "archived_limit", default=None, maximum=2000)
            archived_offset = _query_positive_int(parsed, "archived_offset", default=0, maximum=200000)
            sidebar_source = parse_qs(parsed.query).get("sidebar_source", [""])[0].strip().lower() or None
            if sidebar_source not in ("webui", "cli"):
                sidebar_source = None
            # /api/sessions is the default sidebar contract, so keep the route-owned
            # visible-row filter in the shared cache builder for both cache hits and misses.
            key = _session_list_cache_key(
                active_profile=active_profile,
                all_profiles=all_profiles,
                show_cli_sessions=show_cli_sessions,
                show_claude_code_sessions=show_claude_code_sessions,
                show_previous_messaging_sessions=show_previous_messaging_sessions,
                show_cron_sessions=show_cron_sessions,
                include_archived=include_archived,
                exclude_hidden=exclude_hidden,
                visible_only=True,
                show_webhook_sessions=show_webhook_sessions,
                show_kanban_sessions=show_kanban_sessions,
                source_filter=agent_session_source_filter,
                sidebar_source=sidebar_source,
                archived_limit=archived_limit,
                archived_offset=archived_offset,
            )
            # Keep the visible /api/sessions contract unchanged even though the
            # heavy lifting now lives in the cache builder: profile scoping via
            # `_profiles_match(s.get("profile"), active_profile)` still happens
            # before `_keep_latest_messaging_session_per_source(`.
            payload = _get_cached_session_list_payload(
                key=key,
                builder=lambda: _build_session_list_cache_payload(
                    active_profile=active_profile,
                    all_profiles=all_profiles,
                    show_cli_sessions=show_cli_sessions,
                    show_claude_code_sessions=show_claude_code_sessions,
                    show_previous_messaging_sessions=show_previous_messaging_sessions,
                    show_cron_sessions=show_cron_sessions,
                    include_archived=include_archived,
                    exclude_hidden=exclude_hidden,
                    visible_only=True,
                    show_webhook_sessions=show_webhook_sessions,
                    show_kanban_sessions=show_kanban_sessions,
                    source_filter=agent_session_source_filter,
                    sidebar_source=sidebar_source,
                    archived_limit=archived_limit,
                    archived_offset=archived_offset,
                    diag=diag,
                ),
                diag=diag,
            )
            diag.stage("response_write")
            return j(handler, _session_list_payload_to_response(payload), pretty=False)
        finally:
            diag.finish()

    if parsed.path == "/api/projects":
        # ── Profile scoping (#1614) ────────────────────────────────────────
        # Default: filter to the active profile. ?all_profiles=1 returns the
        # aggregate list so settings/admin UIs can still see everything.
        from api import profiles as profiles_api

        active_profile = profiles_api.get_active_profile_name()
        all_projects = load_projects()
        isolated_profile_mode = _is_isolated_profile_mode()
        all_profiles = _all_profiles_enabled(parsed)
        if all_profiles:
            scoped = all_projects
            other_profile_count = 0
        else:
            scoped = [p for p in all_projects
                      if _profiles_match(p.get("profile"), active_profile)]
            other_profile_count = 0 if isolated_profile_mode else len(all_projects) - len(scoped)
        return j(handler, {
            "projects": scoped,
            "all_profiles": all_profiles,
            "active_profile": active_profile,
            "other_profile_count": other_profile_count,
        })

    if parsed.path == "/api/prompts":
        return j(handler, {"prompts": _load_saved_prompts()})

    if parsed.path == "/api/session/export":
        return _handle_session_export(handler, parsed)

    if parsed.path == "/api/workspaces":
        from api.profiles import get_active_profile_name
        active_profile = get_active_profile_name()
        try:
            wss = load_workspaces(profile=active_profile)
        except TypeError:
            wss = load_workspaces()
        try:
            lw = get_last_workspace(profile=active_profile)
        except TypeError:
            lw = get_last_workspace()
        return j(
            handler,
            {
                "workspaces": wss,
                "last": lw,
                "terminal_remote_backend": _terminal_remote_backend_enabled(),
            },
        )

    if parsed.path == "/api/workspaces/suggest":
        from api.profiles import get_active_profile_name

        qs = parse_qs(parsed.query)
        prefix = qs.get("prefix", [""])[0]
        active_profile = get_active_profile_name()
        try:
            suggestions = list_workspace_suggestions(prefix, profile=active_profile)
        except TypeError:
            suggestions = list_workspace_suggestions(prefix)
        return j(
            handler,
            {
                "suggestions": suggestions,
                "prefix": prefix,
            },
        )

    if parsed.path == "/api/sessions/search":
        return _handle_sessions_search(handler, parsed)

    if parsed.path == "/api/list":
        return _handle_list_dir(handler, parsed)

    if parsed.path == "/api/escape/list":
        return _handle_escape_list_dir(handler, parsed)

    if parsed.path == "/api/git/status":
        return _handle_git_status(handler, parsed)

    if parsed.path == "/api/git/branches":
        return _handle_git_branches(handler, parsed)

    if parsed.path == "/api/git/diff":
        return _handle_git_diff(handler, parsed)

    if parsed.path == "/api/personalities":
        # Read personalities from config.yaml agent.personalities section
        # (matches hermes-agent CLI behavior, not filesystem SOUL.md approach)
        from api.config import reload_config as _reload_cfg

        _reload_cfg()  # pick up config.yaml changes without server restart
        from api.config import get_config as _get_cfg

        _cfg = _get_cfg()
        agent_cfg = _cfg.get("agent", {})
        raw_personalities = agent_cfg.get("personalities", {})
        personalities = []
        if isinstance(raw_personalities, dict):
            for name, value in raw_personalities.items():
                desc = ""
                if isinstance(value, dict):
                    desc = value.get("description", "")
                elif isinstance(value, str):
                    desc = value[:80] + ("..." if len(value) > 80 else "")
                personalities.append({"name": name, "description": desc})
        return j(handler, {"personalities": personalities})

    if parsed.path == "/api/git-info":
        qs = parse_qs(parsed.query)
        sid = qs.get("session_id", [""])[0]
        if not sid:
            return bad(handler, "session_id required")
        try:
            workspace = get_session(sid).workspace
        except KeyError:
            # state.db-only sessions (CLI, delegated subagents): same fallback as /api/list.
            cli_meta = _lookup_cli_session_metadata(sid)
            if not cli_meta:
                return bad(handler, "Session not found", 404)
            if not cli_meta.get("workspace"):
                return j(handler, {"git": None})
            try:
                workspace = resolve_trusted_workspace(cli_meta["workspace"])
            except (FileNotFoundError, ValueError):
                return j(handler, {"git": None})
        from api.workspace_git import GitWorkspaceError, git_status

        try:
            status = git_status(Path(workspace))
        except GitWorkspaceError as e:
            return _git_bad(handler, e)
        totals = status.get("totals") or {}
        info = None if not status.get("is_git") else {
            "branch": status.get("branch"),
            "dirty": totals.get("changed", 0),
            "modified": (totals.get("staged", 0) or 0) + (totals.get("unstaged", 0) or 0),
            "untracked": totals.get("untracked", 0),
            "ahead": status.get("ahead", 0),
            "behind": status.get("behind", 0),
            "is_git": True,
        }
        return j(handler, {"git": info})

    if parsed.path == "/api/commands":
        from api.commands import list_commands
        return j(handler, {"commands": list_commands()})

    if parsed.path == "/api/commands/bundles":
        from api.commands import list_command_bundles
        return j(handler, {"bundles": list_command_bundles()})

    if parsed.path == "/api/commands/moa/resolve":
        from api.commands import resolve_moa_config
        try:
            return j(handler, resolve_moa_config())
        except RuntimeError as e:
            return bad(handler, str(e), 503)

    if parsed.path == "/api/updates/check":
        settings = load_settings()
        if not settings.get("check_for_updates", True):
            return j(handler, {"disabled": True})
        include_agent_updates = not bool(settings.get("ignore_agent_updates"))
        qs = parse_qs(parsed.query)
        # ?simulate=1 returns fake behind counts for UI testing (localhost only)
        if (
            qs.get("simulate", ["0"])[0] == "1"
            and handler.client_address[0] == "127.0.0.1"
        ):
            return j(
                handler,
                {
                    "webui": {
                        "name": "webui",
                        "behind": 3,
                        "current_sha": "abc1234",
                        "latest_sha": "def5678",
                        "branch": "master",
                        "repo_url": "https://github.com/nesquena/hermes-webui",
                        "compare_url": "https://github.com/nesquena/hermes-webui/compare/abc1234...def5678",
                    },
                    "agent": {
                        "name": "agent",
                        "behind": 1 if include_agent_updates else 0,
                        "ignored": not include_agent_updates,
                        "current_sha": "aaa0001",
                        "latest_sha": "bbb0002",
                        "branch": "master",
                        "repo_url": "https://github.com/NousResearch/hermes-agent",
                        "compare_url": "https://github.com/NousResearch/hermes-agent/compare/aaa0001...bbb0002",
                    },
                    "checked_at": 0,
                },
            )
        from api.updates import cached_update_status

        return j(handler, cached_update_status(include_agent=include_agent_updates))

    if parsed.path == "/api/chat/stream/status":
        stream_id = parse_qs(parsed.query).get("stream_id", [""])[0]
        if not _stream_id_visible_to_request_profile(handler, stream_id):
            return True
        active = stream_id in STREAMS
        payload = {"active": active, "stream_id": stream_id, "replay_available": False}
        try:
            journal = find_run_summary(stream_id) if stream_id else None
        except Exception:
            journal = None
        if journal:
            payload["replay_available"] = True
            payload["journal"] = _run_journal_status_payload(journal, active=active)
        return j(handler, payload)

    if parsed.path == "/api/chat/cancel":
        stream_id = parse_qs(parsed.query).get("stream_id", [""])[0]
        if not stream_id:
            return bad(handler, "stream_id required")
        if not _stream_id_visible_to_request_profile(handler, stream_id):
            return True
        gateway_stop_blocked = False
        try:
            from api.gateway_chat import (
                GATEWAY_RUN_ID_WAIT_TIMEOUT,
                stop_gateway_run,
                wait_for_gateway_run_id,
            )

            structured_gateway, run_id = wait_for_gateway_run_id(stream_id, GATEWAY_RUN_ID_WAIT_TIMEOUT)
            if not run_id and structured_gateway:
                gateway_stop_blocked = True
            if run_id:
                if stop_gateway_run(run_id):
                    owner_sid = stream_owner_session_id(stream_id)
                    if owner_sid:
                        settle_gateway_pending_run(
                            owner_sid,
                            run_id,
                            reason="Gateway run was cancelled before approval resolution",
                        )
                else:
                    gateway_stop_blocked = True
        except Exception:
            logger.debug("Failed to stop gateway run during chat cancellation", exc_info=True)
            gateway_stop_blocked = True
        if gateway_stop_blocked:
            return j(
                handler,
                {
                    "ok": False,
                    "cancelled": False,
                    "stream_id": stream_id,
                    "error": "Gateway stop failed",
                },
                status=502,
            )

        from api.runtime_adapter import LegacyJournalRuntimeAdapter, runtime_adapter_enabled

        if runtime_adapter_enabled():
            adapter = LegacyJournalRuntimeAdapter(cancel_delegate=cancel_stream)
            cancelled = adapter.cancel_run(stream_id).accepted
        else:
            cancelled = cancel_stream(stream_id)
        return j(handler, {"ok": True, "cancelled": cancelled, "stream_id": stream_id})

    if parsed.path == "/api/chat/stream":
        return _handle_sse_stream(handler, parsed)

    if parsed.path == "/api/terminal/output":
        return _handle_terminal_output(handler, parsed)

    if parsed.path == '/api/sessions/gateway/stream':
        return _handle_gateway_sse_stream(handler, parsed)

    if parsed.path == '/api/sessions/events':
        return _handle_session_events_stream(handler)

    session_events_session_id = _session_events_path_session_id(parsed.path)
    if session_events_session_id is not None:
        return _handle_session_sse_stream_for_session(handler, parsed, session_events_session_id)

    if parsed.path == "/api/media":
        return _handle_media(handler, parsed)

    if parsed.path == "/api/file/raw":
        return _handle_file_raw(handler, parsed)

    if parsed.path == "/api/escape/file/raw":
        return _handle_escape_file_raw(handler, parsed)

    if parsed.path == "/api/folder/download":
        return _handle_folder_download(handler, parsed)

    if parsed.path == "/api/file":
        return _handle_file_read(handler, parsed)

    if parsed.path == "/api/escape/file/read":
        return _handle_escape_file_read(handler, parsed)

    if parsed.path == "/api/approval/pending":
        return _handle_approval_pending(handler, parsed)

    if parsed.path == "/api/approval/stream":
        return _handle_approval_sse_stream(handler, parsed)

    if parsed.path == "/api/approval/inject_test":
        # Loopback-only: used by automated tests; blocked from any remote client
        if handler.client_address[0] != "127.0.0.1":
            return j(handler, {"error": "not found"}, status=404)
        return _handle_approval_inject(handler, parsed)

    if parsed.path == "/api/clarify/pending":
        return _handle_clarify_pending(handler, parsed)

    if parsed.path == "/api/clarify/stream":
        return _handle_clarify_sse_stream(handler, parsed)

    if parsed.path == "/api/session/stream":
        return _handle_session_sse_stream(handler, parsed)

    if parsed.path == "/api/clarify/inject_test":
        # Loopback-only: used by automated tests; blocked from any remote client
        if handler.client_address[0] != "127.0.0.1":
            return j(handler, {"error": "not found"}, status=404)
        return _handle_clarify_inject(handler, parsed)

    if parsed.path == "/api/onboarding/oauth/poll":
        qs = parse_qs(parsed.query)
        flow_id = qs.get("flow_id", [""])[0]
        try:
            return j(
                handler,
                poll_onboarding_oauth_flow(flow_id),
                extra_headers={"Cache-Control": "no-store"},
            )
        except ValueError as e:
            return bad(handler, str(e))
        except KeyError as e:
            return bad(handler, str(e), 404)

    # ── Cron API (GET) ──
    # Cron reads are active-profile-scoped by default. The list route now
    # aggregates per visible profile home so the UI can surface hidden-row
    # counts and, when opted in, read-only foreign rows.
    if parsed.path == "/api/crons":
        # #4768: in split-container / minimal Docker deployments the WebUI image may
        # not ship the agent's `cron` package on its import path. Degrade gracefully
        # (empty list + cron_unavailable flag) instead of 500ing the whole Task tab.
        # Only treat a genuinely-absent cron package as "unavailable"; a
        # ModuleNotFoundError whose missing module is an internal dependency of an
        # existing cron/jobs.py is a real bug and must still surface.
        _ensure_agent_cron_import_path()
        active_profile = _get_active_profile_name() or "default"
        try:
            active_jobs, other_jobs = _cron_jobs_cross_profile(active_profile)
        except ModuleNotFoundError as exc:
            if exc.name in ("cron", "cron.jobs"):
                return j(handler, {"jobs": [], "cron_unavailable": True})
            raise
        all_profiles = _all_profiles_enabled(parsed)
        jobs = active_jobs + other_jobs if all_profiles else active_jobs
        hidden_other_count = 0 if all_profiles else len(other_jobs)
        return j(handler, {
            "jobs": jobs,
            "all_profiles": all_profiles,
            "active_profile": active_profile,
            "other_profile_count": hidden_other_count,
        })

    if parsed.path == "/api/crons/output":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_output(handler, parsed)

    if parsed.path == "/api/crons/history":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_history(handler, parsed)

    if parsed.path == "/api/crons/run":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_run_detail(handler, parsed)

    if parsed.path == "/api/crons/recent":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_recent(handler, parsed)

    if parsed.path == "/api/crons/status":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            return _handle_cron_status(handler, parsed)

    if parsed.path == "/api/crons/delivery-options":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_delivery_options(handler)

    # ── Skills API (GET) ──
    if parsed.path == "/api/skills":
        qs = parse_qs(parsed.query)
        category = qs.get("category", [None])[0]
        data = _skills_list_from_dir(_active_skills_dir(), category=category)
        return j(handler, {"skills": data.get("skills", [])})

    if parsed.path == "/api/skills/usage":
        from api.skill_usage import read_skill_usage
        raw = read_skill_usage(_active_skills_dir())
        # Pass through agent's format as-is; defensive coercion for fields
        usage = {}
        if isinstance(raw, dict):
            for k, v in raw.items():
                if not isinstance(v, dict):
                    usage[k] = {"use_count": 0, "view_count": 0, "patch_count": 0}
                    continue
                usage[k] = {
                    "use_count": (int(v["use_count"]) if v.get("use_count") is not None else 0),
                    "view_count": (int(v["view_count"]) if v.get("view_count") is not None else 0),
                    "patch_count": (int(v["patch_count"]) if v.get("patch_count") is not None else 0),
                }
                # Preserve agent's metadata (timestamps, state, etc.)
                for meta_key in v:
                    if meta_key not in usage[k]:
                        usage[k][meta_key] = v[meta_key]
        skills_data = _skills_list_from_dir(_active_skills_dir()).get("skills", [])
        skill_names = sorted({s["name"] for s in skills_data})
        total = sum(
            e.get("use_count", 0) + e.get("view_count", 0) + e.get("patch_count", 0)
            for e in usage.values()
        )
        unique = sum(
            1 for e in usage.values()
            if e.get("use_count", 0) > 0 or e.get("view_count", 0) > 0 or e.get("patch_count", 0) > 0
        )
        return j(handler, {
            "usage": usage,
            "skill_names": skill_names,
            "total_invocations": total,
            "unique_skills_used": unique,
        })

    if parsed.path == "/api/skills/content":
        qs = parse_qs(parsed.query)
        name = qs.get("name", [""])[0]
        if not name:
            return j(handler, {"error": "name required"}, status=400)
        file_path = qs.get("file", [""])[0]
        if file_path:
            # Serve a linked file from the skill directory
            import re as _re

            if _re.search(r"[*?\[\]]", name):
                return bad(handler, "Invalid skill name", 400)
            skills_dir = _active_skills_dir()
            skill_dir, _skill_md = _find_skill_in_dirs(
                name, _active_skill_search_dirs(skills_dir)
            )
            if not skill_dir:
                return bad(handler, "Skill not found", 404)
            target = (skill_dir / file_path).resolve()
            try:
                target.relative_to(skill_dir.resolve())
            except ValueError:
                return bad(handler, "Invalid file path", 400)
            if not target.exists() or not target.is_file():
                return bad(handler, "File not found", 404)
            return j(
                handler,
                {"content": target.read_text(encoding="utf-8"), "path": file_path},
            )
        data = _skill_view_from_active_dir(name)
        if not isinstance(data.get("linked_files"), dict):
            data["linked_files"] = {}
        return j(handler, data)

    # ── Memory API (GET) ──
    if parsed.path == "/api/memory":
        return _handle_memory_read(handler, parsed)

    # ── Profile API (GET) ──
    if parsed.path == "/api/profiles":
        from api import profiles as profiles_api
        diag = RequestDiagnostics.maybe_start("GET", parsed.path, logger=logger, print_fn=getattr(handler, '_safe_webui_print', None))
        try:
            diag.stage("list_profiles_api") if diag else None
            profiles_payload = profiles_api.list_profiles_api()
            diag.stage("active_profile_lookup") if diag else None
            active = profiles_api.get_active_profile_name()
            diag.stage("isolated_mode_check") if diag else None
            return j(
                handler,
                {
                    "profiles": profiles_payload,
                    "active": active,
                    "single_profile_mode": _is_isolated_profile_mode(),
                },
            )
        finally:
            if diag:
                diag.finish()

    if parsed.path == "/api/profile/active":
        from api import profiles as profiles_api

        active_profile_name = profiles_api.get_active_profile_name()
        # Resolve the ACTIVE PROFILE's configured workspace so a cold boot with a
        # profile cookie shows the right composer workspace chip on a blank
        # new-chat page (#5169). Use get_profile_default_workspace() (NOT
        # get_last_workspace) so a named profile without its own last_workspace.txt
        # resolves to its config.yaml workspace/terminal.cwd rather than leaking the
        # GLOBAL last-workspace file (the #5169 regression Codex flagged). It is
        # profile-scoped via the per-request hermes_profile cookie set in server.py.
        # Fail open: a resolution error must never 500 this boot-critical endpoint.
        try:
            try:
                _profile_default_workspace = get_profile_default_workspace(profile=active_profile_name)
            except TypeError:
                _profile_default_workspace = get_profile_default_workspace()
        except Exception:
            logger.debug("Failed to resolve profile default workspace for /api/profile/active", exc_info=True)
            _profile_default_workspace = None
        return j(
            handler,
            {
                "name": active_profile_name,
                "path": str(profiles_api.get_active_hermes_home()),
                "is_default": profiles_api._is_root_profile(active_profile_name),
                "default_workspace": _profile_default_workspace,
            },
        )

    # ── Gateway Status (GET) ──
    if parsed.path == "/api/gateway/status":
        return j(handler, _gateway_status_payload())

    # ── MCP Servers (GET) ──
    if parsed.path == "/api/mcp/servers":
        return _handle_mcp_servers_list(handler)

    # ── MCP Tools (GET) ──
    if parsed.path == "/api/mcp/tools":
        return _handle_mcp_tools_list(handler)

    if parsed.path == "/api/notes/sources":
        return _handle_notes_sources_list(handler)
    if parsed.path == "/api/notes/search":
        return _handle_notes_search(handler, parsed)
    if parsed.path == "/api/notes/item":
        return _handle_notes_item(handler, parsed)

    # ── Checkpoints / Rollback (GET) ──
    if parsed.path == "/api/rollback/list":
        qs = parse_qs(parsed.query)
        workspace = qs.get("workspace", [""])[0]
        if not workspace:
            return bad(handler, "workspace query parameter is required")
        try:
            from api.rollback import list_checkpoints
            return j(handler, list_checkpoints(workspace))
        except ValueError as e:
            return bad(handler, str(e))
        except Exception as e:
            logger.exception("rollback/list failed")
            return bad(handler, str(e), status=500)

    if parsed.path == "/api/rollback/diff":
        qs = parse_qs(parsed.query)
        workspace = qs.get("workspace", [""])[0]
        checkpoint = qs.get("checkpoint", [""])[0]
        if not workspace or not checkpoint:
            return bad(handler, "workspace and checkpoint query parameters are required")
        try:
            from api.rollback import get_checkpoint_diff
            return j(handler, get_checkpoint_diff(workspace, checkpoint))
        except ValueError as e:
            return bad(handler, str(e))
        except Exception as e:
            logger.exception("rollback/diff failed")
            return bad(handler, str(e), status=500)

    # ── Plugin shared assets (e.g. /plugins/plugin.css) ──
    # Restricted to shared plugin assets only — no cross-plugin file access.
    if parsed.path.startswith("/plugins/"):
        from api.plugins import _get_plugin_base
        plugin_base = _get_plugin_base()
        rel = parsed.path[len("/plugins/"):]
        allowed = {"plugin.css"}
        if rel not in allowed:
            return False  # 404
        safe = (plugin_base / rel).resolve()
        try:
            safe.relative_to(plugin_base.resolve())
        except ValueError:
            return False  # path traversal — 404
        if safe.is_file():
            import os as _os
            data = safe.read_bytes()
            ext = _os.path.splitext(rel.lower())[1]
            ct = {
                ".css": "text/css; charset=utf-8",
                ".js": "application/javascript; charset=utf-8",
                ".json": "application/json; charset=utf-8",
                ".png": "image/png",
                ".svg": "image/svg+xml",
            }.get(ext, "application/octet-stream")
            handler.send_response(200)
            handler.send_header("Content-Type", ct)
            handler.send_header("Content-Length", str(len(data)))
            handler.end_headers()
            handler.wfile.write(data)
            return True

    # ── Plugin static assets ──
    if parsed.path.startswith("/dashboard-plugins/"):
        parts = parsed.path.split("/", 3)
        if len(parts) >= 3:
            plugin_name = parts[2]
            rel_path = parts[3] if len(parts) > 3 else ""
            # Server-side enable-gate: a plugin disabled in Settings must have its
            # entire URL surface shut off, not merely hidden in the UI.
            if not _dashboard_plugin_enabled(plugin_name):
                return False  # 404 — disabled plugins serve nothing
            from api.plugins import serve_plugin_static
            result = serve_plugin_static(plugin_name, rel_path)
            if result:
                data, content_type = result
                handler.send_response(200)
                handler.send_header("Content-Type", content_type)
                # Defense-in-depth: plugin-controlled assets are served from the
                # WebUI's own origin. Sandbox them (null origin) so a plugin's
                # .html/.svg can't run privileged same-origin script if navigated
                # to directly (the in-panel iframe sandbox doesn't cover direct
                # navigation). nosniff prevents content-type confusion.
                handler.send_header("Content-Security-Policy", "sandbox allow-scripts allow-forms allow-popups")
                handler.send_header("X-Content-Type-Options", "nosniff")
                handler.send_header("Content-Length", str(len(data)))
                handler.end_headers()
                handler.wfile.write(data)
                return True

    # ── Plugin pages (HTML shell) ──
    from api.plugins import PLUGIN_MANIFESTS, _PLUGIN_STATIC_ROOTS
    for name, manifest in PLUGIN_MANIFESTS.items():
        tab = manifest.get("tab", {})
        tab_path = tab.get("path", f"/{name}")
        if parsed.path == tab_path:
            # Server-side enable-gate (opt-in): a disabled plugin's page 404s.
            if not _dashboard_plugin_enabled(name):
                return False
            dashboard_dir = _PLUGIN_STATIC_ROOTS.get(name)
            if dashboard_dir:
                # 1) dashboard/dist/index.html (full SPA build)
                index_html = dashboard_dir / "dist" / "index.html"
                if index_html.is_file():
                    data = index_html.read_bytes()
                    handler.send_response(200)
                    handler.send_header("Content-Type", "text/html; charset=utf-8")
                    handler.send_header("Content-Security-Policy", "sandbox allow-scripts allow-forms allow-popups")
                    handler.send_header("Content-Length", str(len(data)))
                    handler.end_headers()
                    handler.wfile.write(data)
                    return True
                # 2) static/index.html in plugin root (content page for IIFE loader)
                plugin_root = dashboard_dir.parent
                static_html = plugin_root / "static" / "index.html"
                if static_html.is_file():
                    data = static_html.read_bytes()
                    handler.send_response(200)
                    handler.send_header("Content-Type", "text/html; charset=utf-8")
                    handler.send_header("Content-Security-Policy", "sandbox allow-scripts allow-forms allow-popups")
                    handler.send_header("Content-Length", str(len(data)))
                    handler.end_headers()
                    handler.wfile.write(data)
                    return True
                # 3) Fallback: generate shell that loads the IIFE bundle
                index_js = dashboard_dir / "dist" / "index.js"
                if index_js.is_file():
                    import html
                    label = html.escape(manifest.get("label") or name)
                    css = html.escape(manifest.get("css", ""))
                    name_escaped = html.escape(name)
                    css_tag = f'<link rel="stylesheet" href="/dashboard-plugins/{name_escaped}/{css}">' if css else ""
                    html_content = (
                        f"<!doctype html>\n"
                        f"<html lang=\"en\">\n"
                        f"<head>\n"
                        f"  <meta charset=\"utf-8\">\n"
                        f"  <title>{label}</title>\n"
                        f"  {css_tag}\n"
                        f"</head>\n"
                        f"<body>\n"
                        f'  <div id="pluginPageContainer"></div>\n'
                        f'  <script src="/dashboard-plugins/{name_escaped}/dist/index.js"></script>\n'
                        f"</body>\n"
                        f"</html>\n"
                    ).encode("utf-8")
                    handler.send_response(200)
                    handler.send_header("Content-Type", "text/html; charset=utf-8")
                    handler.send_header("Content-Security-Policy", "sandbox allow-scripts allow-forms allow-popups")
                    handler.send_header("Content-Length", str(len(html_content)))
                    handler.end_headers()
                    handler.wfile.write(html_content)
                    return True

    return False  # 404


# ── POST auth helpers

def _require_passkey_registration_auth(handler) -> tuple[bool, str, int]:
    """Require auth, or the existing local-only first-run bootstrap gate.

    Registering additional passkeys is an auth-factor enrollment action and
    requires a valid WebUI session.  The first passkey can still bootstrap a
    passkey-only instance, but only through the same local/private-network
    onboarding gate used for first password setup.
    """
    from api.auth import is_auth_enabled, parse_cookie, verify_session

    auth_enabled = is_auth_enabled()
    if not auth_enabled:
        if _onboarding_gate_allows(handler, auth_enabled):
            return True, "", 200
        return False, "Authentication required", 401
    cookie_val = parse_cookie(handler)
    if not cookie_val or not verify_session(cookie_val):
        return False, "Authentication required", 401
    return True, "", 200

def _validate_session_toolsets_shape(toolsets):
    """Validate per-session toolset override shape without catalog lookup."""
    if toolsets is None:
        return None
    if not isinstance(toolsets, list) or not toolsets:
        raise ValueError("toolsets must be a non-empty list or null")
    if not all(isinstance(t, str) and t for t in toolsets):
        raise ValueError("each toolset must be a non-empty string")
    return toolsets


def _resolve_new_session_workspace(body, visible_prev_session_id, profile=None):
    """Resolve a new-session workspace, recovering only verified inheritance."""
    candidate = body.get("workspace")
    if not candidate:
        return None

    def _rtw(value):
        # Legacy test doubles may predate the profile kwarg.
        try:
            return resolve_trusted_workspace(value, profile=profile)
        except TypeError:
            return resolve_trusted_workspace(value)

    if (
        body.get("workspace_inherited_from_prev_session") is not True
        or not visible_prev_session_id
    ):
        return str(_rtw(candidate))
    try:
        previous_session = get_session(visible_prev_session_id, metadata_only=True)
    except KeyError:
        return str(_rtw(candidate))
    if str(getattr(previous_session, "workspace", None) or "") != str(candidate):
        return str(_rtw(candidate))
    try:
        workspace, _recovered = resolve_implicit_workspace_with_recovery(
            candidate,
            get_last_workspace,
            profile=profile,
        )
    except TypeError:
        workspace, _recovered = resolve_implicit_workspace_with_recovery(
            candidate,
            get_last_workspace,
        )
    return str(workspace)


def _llm_update_summary(system_prompt: str, user_prompt: str, active_profile: str | None = None) -> str:
    from api import profiles as profiles_api

    profile = active_profile or profiles_api.get_active_profile_name() or "default"

    with profiles_api.profile_env_for_background_worker(
        profile,
        "update summary",
        logger_override=logger,
    ):
        from api.config import (
            get_effective_default_model,
            resolve_model_provider,
        )

        messages = [
            {"role": "system", "content": system_prompt},
            {"role": "user", "content": user_prompt},
        ]

        _main_model, _main_provider, _main_base_url = resolve_model_provider(get_effective_default_model())
        _main_api_key = None
        _rt = None
        try:
            from api.oauth import resolve_runtime_provider_with_anthropic_env_lock
            from hermes_cli.runtime_provider import resolve_runtime_provider

            _rt = resolve_runtime_provider_with_anthropic_env_lock(
                resolve_runtime_provider,
                requested=_main_provider,
            )
            _main_api_key = _rt.get("api_key")
            if not _main_provider:
                _main_provider = _rt.get("provider")
            if not _main_base_url:
                _main_base_url = _rt.get("base_url")
        except Exception as _e:
            logger.debug("update summary runtime provider resolution failed: %s", _e)
        # Atomic custom-provider authority (see the /api/chat note): the record
        # that supplies the endpoint must also supply the credential — and the
        # wire protocol, credential pool and ACP transport that go with it.
        _bundle = _resolve_agent_connection_bundle(
            _main_provider, _main_api_key, _main_base_url, _rt
        )
        _main_provider = _bundle["provider"]
        _main_api_key = _bundle["api_key"]
        _main_base_url = _bundle["base_url"]

        main_runtime = _auxiliary_main_runtime(_bundle, _main_model)

        ensure_agent_runtime_current()
        try:
            from agent.auxiliary_client import get_text_auxiliary_client

            aux_client, aux_model = get_text_auxiliary_client(
                "compression",
                main_runtime=main_runtime,
            )
            if aux_client is not None and aux_model:
                response = aux_client.chat.completions.create(
                    model=aux_model,
                    messages=messages,
                )
                return str(response.choices[0].message.content or "").strip()
        except Exception as _e:
            logger.debug("update summary auxiliary model failed; falling back to main model: %s", _e)

        AIAgent = require_ai_agent_class()

        agent = AIAgent(
            model=_main_model,
            provider=_main_provider,
            base_url=_main_base_url,
            api_key=_main_api_key,
            platform="webui",
            quiet_mode=True,
            enabled_toolsets=[],
            session_id=f"updates-summary-{uuid.uuid4().hex[:8]}",
            **_agent_bundle_kwargs(AIAgent, _bundle),
        )
        result = agent.run_conversation(
            user_message=user_prompt,
            system_message=system_prompt,
            conversation_history=[],
            task_id=f"updates-summary-{uuid.uuid4().hex[:8]}",
        )
        return str(result.get("final_response") or "").strip()


def handle_post(handler, parsed) -> bool:
    """Handle all POST routes. Returns True if handled, False for 404."""
    diag = RequestDiagnostics.maybe_start("POST", parsed.path, logger=logger, print_fn=getattr(handler, '_safe_webui_print', None))
    if parsed.path == "/api/csp-report":
        if diag:
            diag.stage("csp_report")
        try:
            return _handle_csp_report(handler)
        finally:
            if diag:
                diag.finish()
    # T1 deprecation alias for the legacy ack endpoint that the pre-rename
    # WebUI used to POST to after handling ``process_complete``. The new
    # canonical SSE event is ``bg_task_complete`` and the new ack endpoint
    # will be ``/api/bg-task-complete-ack`` (introduced by PR (b), the WebUI
    # half of the split). Until PR (b) lands we keep the old path responding
    # with HTTP 410 Gone + ``X-Replaced-By`` so any stale tab posting under
    # the old name fails loudly with a discoverable hint. The handler runs
    # BEFORE the CSRF gate on purpose: an old tab will not carry a CSRF token
    # for the deprecated path, and surfacing 410 (not 403) is the correct
    # contract here.
    if parsed.path == "/api/process-complete-ack":
        if diag:
            diag.stage("process_complete_ack_deprecated")
        try:
            # The 410 runs before the stale tab's JSON body is read;
            # close-and-advertise so those unread bytes can't corrupt the next
            # pooled request -- but a body-less ack has none, and closing then
            # just kills keep-alive (on the wire: 410 + `Connection: close` with
            # no `Content-Length`, pipelined follow-up dropped).
            arm_connection_close_if_body_pending(handler)
            j(
                handler,
                {
                    "error": (
                        "gone: /api/process-complete-ack was replaced by "
                        "/api/bg-task-complete-ack as part of the "
                        "process_complete -> bg_task_complete event rename"
                    ),
                    "replaced_by": "/api/bg-task-complete-ack",
                },
                status=410,
                extra_headers={"X-Replaced-By": "/api/bg-task-complete-ack"},
            )
            return True
        finally:
            if diag:
                diag.finish()
    # CSRF: reject cross-origin or tokenless authenticated browser requests.
    # /api/auth/login has no authenticated session token yet, and /api/csp-report
    # is intentionally unauthenticated for browser-generated violation reports.
    if diag:
        diag.stage("csrf")
    if not _csrf_exempt_path(parsed.path) and not _check_csrf(handler):
        try:
            return j(handler, {"error": _csrf_rejection_error(handler)}, status=403)
        finally:
            if diag:
                diag.finish()
    proxy_result = _handle_extension_sidecar_proxy(
        handler,
        parsed,
        "POST",
        read_request_body=True,
    )
    if proxy_result is not False:
        if diag:
            diag.finish()
        return proxy_result

    if parsed.path == "/api/shutdown":
        return _handle_shutdown(handler)

    if parsed.path == "/api/health/restart":
        return _handle_health_restart(handler)

    if parsed.path == "/api/upload":
        return handle_upload(handler)
    if parsed.path == "/api/upload/extract":
        return handle_upload_extract(handler)
    if parsed.path == "/api/workspace/upload":
        return handle_workspace_upload(handler)

    if parsed.path == "/api/transcribe":
        return handle_transcribe(handler)

    if parsed.path == "/api/tts":
        return _handle_tts(handler, parsed)

    if parsed.path == "/api/client-events/log":
        if diag:
            diag.stage("read_client_event_body")
        return _handle_client_event_log(handler, _read_client_event_payload(handler))

    if diag:
        diag.stage("read_body")
    try:
        body = read_body(handler)
    except ValueError as exc:
        if diag:
            diag.finish()
        status = 413 if "too large" in str(exc).lower() else 400
        return bad(handler, str(exc), status=status)
    except Exception:
        if diag:
            diag.finish()
        raise
    if not _guard_request_session_visibility(handler, parsed, body=body, method="POST"):
        if diag:
            diag.finish()
        return True

    if parsed.path == "/api/escape/authorize":
        return _handle_escape_authorize(handler, parsed, body)

    if parsed.path == "/api/updates/check":
        settings = load_settings()
        if not settings.get("check_for_updates", True):
            force = bool(body.get("force", False)) if isinstance(body, dict) else False
            if force:
                # Manual force-check bypasses auto-check toggle (#6082)
                pass
            else:
                return j(handler, {"disabled": True})
        include_agent_updates = not bool(settings.get("ignore_agent_updates"))
        force = bool(body.get("force", False))
        # Allow the client to pass the channel explicitly in the POST body. This
        # avoids a race on channel switch: the Settings dropdown re-checks
        # immediately, but its autosave PUT (debounced) may not have landed
        # server-side yet, so reading the saved setting here could answer for the
        # OLD channel. An explicit body channel (validated against the enum) wins;
        # otherwise fall back to the saved setting. (Fable UX gate.)
        channel = body.get("channel") if isinstance(body, dict) else None
        if channel not in ("stable", "experimental"):
            channel = settings.get("update_channel")
        from api.updates import check_for_updates

        logger.info("checking for updates (force=%s, include_agent=%s, channel=%s)", force, include_agent_updates, channel)
        # Defensive-only guard: wrap check_for_updates() for consistent
        # exception protection across all route handlers. Does NOT fix #6086
        # (root cause is likely signal/process-group reaping, per maintainer analysis).
        try:
            payload = check_for_updates(force=force, include_agent=include_agent_updates, channel=channel)
        except Exception:
            logger.exception("update check failed unexpectedly (defensive guard caught exception)")
            return bad(handler, "Update check failed, see server log for details", status=500)
        logger.info("update check completed")
        return j(handler, payload)

    if parsed.path == "/api/extensions/toggle":
        from api.extensions import ExtensionToggleError, set_extension_user_enabled

        try:
            return j(
                handler,
                set_extension_user_enabled(body.get("id"), body.get("enabled")),
            )
        except ExtensionToggleError as exc:
            return bad(handler, str(exc), status=exc.status)
        except Exception:
            logger.exception("extension toggle failed")
            return bad(handler, "Failed to update extension state", status=500)

    if parsed.path == "/api/extensions/sidecar-proxy-consent":
        from api.extensions import (
            ExtensionSidecarProxyError,
            set_extension_sidecar_proxy_consent,
        )

        try:
            return j(
                handler,
                set_extension_sidecar_proxy_consent(
                    body.get("id"),
                    body.get("approved"),
                ),
            )
        except ExtensionSidecarProxyError as exc:
            return bad(handler, str(exc), status=exc.status)
        except Exception:
            logger.exception("extension sidecar proxy consent update failed")
            return bad(handler, "Failed to update extension state", status=500)

    if parsed.path == "/api/extensions/install":
        from api.extensions import ExtensionInstallError, install_extension

        try:
            return j(
                handler,
                install_extension(body.get("id"), body.get("download_url"), body.get("sha256")),
            )
        except ExtensionInstallError as exc:
            return bad(handler, str(exc), status=exc.status)
        except Exception:
            logger.exception("extension install failed")
            return bad(handler, "Failed to install extension", status=500)

    if parsed.path == "/api/extensions/uninstall":
        from api.extensions import ExtensionInstallError, uninstall_extension

        try:
            return j(
                handler,
                uninstall_extension(body.get("id")),
            )
        except ExtensionInstallError as exc:
            return bad(handler, str(exc), status=exc.status)
        except Exception:
            logger.exception("extension uninstall failed")
            return bad(handler, "Failed to uninstall extension", status=500)

    if parsed.path == "/api/session/recovery/repair-safe":
        from api.session_recovery import repair_safe_session_recovery
        result = repair_safe_session_recovery(SESSION_DIR, state_db_path=_active_state_db_path())
        return j(handler, result, status=200 if result.get("clean") else 409)

    if parsed.path.startswith("/api/kanban/"):
        from api.kanban_bridge import handle_kanban_post

        result = handle_kanban_post(handler, parsed, body)
        if result is False:
            return _kanban_unknown_endpoint(handler, parsed, "POST")
        return True
    if parsed.path == "/api/dashboard/config":
        from api import dashboard_probe

        try:
            j(handler, dashboard_probe.save_dashboard_config(body))
        except ValueError as exc:
            bad(handler, str(exc), status=400)
        except Exception as exc:
            logger.exception("dashboard config save failed")
            bad(handler, str(exc), status=500)
        return True

    if parsed.path == "/api/prompts":
        text = str(body.get("text") or "").strip()
        label = str(body.get("label") or "").strip()
        if not text:
            return bad(handler, "text is required")
        if len(text) > 8000:
            return bad(handler, "text too long (max 8000 chars)")
        prompts = _load_saved_prompts()
        if len(prompts) >= 200:
            return bad(handler, "saved prompts limit reached (max 200)")
        new_prompt = {"id": uuid.uuid4().hex[:12], "label": label or text[:60], "text": text, "created_at": time.time()}
        prompts.append(new_prompt)
        _save_saved_prompts(prompts)
        return j(handler, {"ok": True, "prompt": new_prompt})

    if parsed.path == "/api/share/create":
        sid = str(body.get("session_id") or "").strip()
        if not sid:
            return bad(handler, "session_id is required", 400)
        try:
            snapshot_session, stored_session, cli_meta = _resolve_share_session_pair(sid, handler)
        except KeyError:
            return bad(handler, "Session not found", 404)
        try:
            share_meta = create_or_refresh_share(snapshot_session)
        except ValueError as exc:
            return bad(handler, str(exc), 400)
        persisted_session = stored_session
        if persisted_session is None:
            persisted_session = _build_share_metadata_sidecar(
                sid,
                snapshot_session,
                cli_meta=cli_meta,
            )
        persisted_session.share_token = share_meta["share_token"]
        persisted_session.share_created_at = share_meta["share_created_at"]
        persisted_session.save(touch_updated_at=False)
        _publish_session_list_changed(
            "session_share_create",
            profile=getattr(persisted_session, "profile", None),
            session_id=sid,
        )
        response_session = copy.copy(persisted_session)
        response_session.messages = list(getattr(snapshot_session, "messages", None) or [])
        return j(
            handler,
            {
                "ok": True,
                "share": {
                    "token": share_meta["share_token"],
                    "url": f"/share/{share_meta['share_token']}",
                    "title": share_meta["share_title"],
                    "message_count": share_meta["share_message_count"],
                    "created_at": share_meta["share_created_at"],
                    "updated_at": share_meta["share_updated_at"],
                },
                "session": public_session_projection(
                    response_session.compact() | {"messages": response_session.messages}
                ),
            },
        )

    if parsed.path == "/api/share/revoke":
        sid = str(body.get("session_id") or "").strip()
        if not sid:
            return bad(handler, "session_id is required", 400)
        try:
            snapshot_session, stored_session, cli_meta = _resolve_share_session_pair(sid, handler)
        except KeyError:
            return bad(handler, "Session not found", 404)
        target_session = stored_session
        if target_session is None:
            token = str(getattr(snapshot_session, "share_token", "") or "").strip()
            if not token:
                return bad(handler, "Session not found", 404)
            target_session = _build_share_metadata_sidecar(
                sid,
                snapshot_session,
                cli_meta=cli_meta,
            )
            target_session.share_token = token
            target_session.share_created_at = getattr(snapshot_session, "share_created_at", None)
        revoke_share(target_session)
        target_session.share_token = None
        target_session.share_created_at = None
        target_session.save(touch_updated_at=False)
        _publish_session_list_changed(
            "session_share_revoke",
            profile=getattr(target_session, "profile", None),
            session_id=sid,
        )
        response_session = copy.copy(target_session)
        response_session.messages = list(getattr(snapshot_session, "messages", None) or [])
        return j(
            handler,
            {
                "ok": True,
                "session": public_session_projection(
                    response_session.compact() | {"messages": response_session.messages}
                ),
            },
        )

    if parsed.path == "/api/session/new":
        workspace_prev_session_id = body.get("prev_session_id")
        if workspace_prev_session_id and not _session_id_visible_to_request_profile(
            handler, workspace_prev_session_id, emit_error=False
        ):
            workspace_prev_session_id = None
        try:
            workspace = _resolve_new_session_workspace(
                body, workspace_prev_session_id, profile=body.get("profile") or None
            )
        except (TypeError, ValueError) as e:
            return bad(handler, str(e))
        worktree_info = None
        worktree_skipped = None
        # Three-value worktree model (#6022): an explicit body value always
        # wins; an ABSENT key falls back to the agent's config-level
        # ``worktree:`` default so WebUI sessions and CLI sessions agree on
        # isolation for the same repo.  Clients that must never create a
        # worktree (e.g. the boot-time auto-bind) send ``worktree: false``
        # explicitly.
        raw_worktree = body.get("worktree")
        # Presence-based, not truthiness-based: a client that sends the key at
        # all (even ``worktree: null``) has spoken explicitly and never falls
        # through to the config default.  ``null`` parses as non-true below,
        # i.e. an explicit opt-out — only a genuinely ABSENT key inherits.
        worktree_explicit = "worktree" in body
        if worktree_explicit:
            worktree_requested = (
                raw_worktree is True
                or str(raw_worktree).strip().lower() in {"1", "true", "yes", "on"}
            )
        else:
            worktree_requested = _worktree_default_from_config(body.get("profile") or None)
        if worktree_requested:
            try:
                from api.worktrees import create_worktree_for_workspace
                base_workspace = workspace
                if not base_workspace:
                    _new_profile = body.get("profile") or None
                    try:
                        _lw = get_last_workspace(profile=_new_profile)
                    except TypeError:
                        _lw = get_last_workspace()
                    try:
                        base_workspace = str(
                            resolve_trusted_workspace(_lw, profile=_new_profile)
                        )
                    except TypeError:
                        base_workspace = str(resolve_trusted_workspace(_lw))
                worktree_info = create_worktree_for_workspace(base_workspace)
                workspace = worktree_info["path"]
            except (TypeError, ValueError) as e:
                # Explicit requests keep the hard failure.  A config-default
                # request on a non-git workspace (create_worktree_for_workspace
                # raises ValueError) degrades to a plain session instead —
                # otherwise `worktree: true` in config.yaml would 400 every
                # session in every non-git directory.
                if worktree_explicit:
                    return bad(handler, str(e), status=400)
                worktree_info = None
                worktree_skipped = str(e)
            except Exception as e:
                logger.exception("failed to create worktree-backed session")
                return bad(handler, f"Failed to create worktree: {e}", status=500)
        model, model_provider = _session_model_state_from_request(
            body.get("model"),
            body.get("model_provider"),
        )
        try:
            enabled_toolsets = _validate_session_toolsets_shape(body.get("enabled_toolsets"))
        except ValueError as e:
            return bad(handler, str(e), status=400)
        # Use the profile sent by the client tab (if any) so that two tabs on
        # different profiles never clobber each other via the process-level global.
        # ── Memory lifecycle: commit the previous session before starting a new one ──
        prev_session_id = body.get("prev_session_id")
        if prev_session_id:
            if not _session_id_visible_to_request_profile(
                handler, prev_session_id, emit_error=False
            ):
                # Cross-profile hand-off after a profile switch: skip memory
                # commit for the previous profile's session, but still create
                # the new session (#5420).
                prev_session_id = None
            if prev_session_id:
                # Fire-and-forget: commit_memory_session() can take 1-5+ seconds
                # (extraction call to the memory provider), and blocking the
                # response here made "+ New Chat" feel slow/unresponsive.
                # commit_session_memory() already serialises overlapping commits
                # for a session via its own in-flight guard, so running it off
                # the request thread is safe.
                def _commit_prev_session_memory(_sid=prev_session_id):
                    try:
                        from api.session_lifecycle import commit_session_memory
                        from api.config import SESSION_AGENT_CACHE, SESSION_AGENT_CACHE_LOCK
                        prev_agent = None
                        with SESSION_AGENT_CACHE_LOCK:
                            _cached = SESSION_AGENT_CACHE.get(_sid)
                            if _cached:
                                prev_agent = _cached[0]
                        commit_session_memory(_sid, agent=prev_agent)
                    except Exception:
                        logger.warning(
                            "Lifecycle commit for prev_session %s failed",
                            _sid,
                            exc_info=True,
                        )
                    finally:
                        # Self-unregister so the background-commit registry does
                        # not leak completed threads; drain only tracks live ones.
                        try:
                            from api.session_lifecycle import _unregister_background_commit_thread
                            _unregister_background_commit_thread(threading.current_thread())
                        except Exception:
                            pass

                t = threading.Thread(
                    target=_commit_prev_session_memory,
                    daemon=True,
                    name=f"commit-memory-{prev_session_id}",
                )
                from api.session_lifecycle import _register_background_commit_thread
                # Refused only if shutdown draining has already begun; in that
                # window the inline drain commits the pending generation instead,
                # so skipping the worker start is safe (avoids a late daemon
                # thread the drain snapshot already missed).
                if _register_background_commit_thread(t):
                    t.start()
        s = new_session(
            workspace=workspace,
            model=model,
            model_provider=model_provider,
            profile=body.get("profile") or None,
            project_id=body.get("project_id") or None,
            worktree_info=worktree_info,
            enabled_toolsets=enabled_toolsets,
        )
        if worktree_info:
            publish_session_list_changed(
                "session_new",
                profile=getattr(s, "profile", None),
                session_id=getattr(s, "session_id", None),
            )
        payload = {
            "session": public_session_projection(s.compact() | {"messages": s.messages})
        }
        if worktree_skipped:
            # Config-default worktree was skipped (non-git workspace); tell the
            # client the session is plain so the UI doesn't assume isolation.
            payload["worktree_skipped"] = worktree_skipped
        return j(handler, payload)

    if parsed.path == "/api/session/compression-recovery/start":
        return _handle_session_compression_recovery_start(handler, body)

    if parsed.path == "/api/session/duplicate":
        try:
            sid = body.get("session_id")
            if not sid:
                return bad(handler, "session_id is required")
            if _session_is_subagent_view_only(sid):
                return bad(handler, "Subagent sessions are view-only and cannot be duplicated from WebUI", 400)

            session = Session.load(sid)
            if not session:
                # 404, not 400 — missing resource, not a malformed request.
                return bad(handler, "Session not found", status=404)

            # Deep-copy mutable lists so the duplicate is *actually* independent.
            # `Session.__init__` does `self.messages = messages or []` — plain
            # assignment, no copy. Without deepcopy, both sessions share the same
            # list object in memory; appending to one mutates the other.
            # Items inside `messages` are dicts with mutable values (tool_calls,
            # content arrays), so a shallow `list(...)` is not enough.
            copied_session = Session(
                session_id=uuid.uuid4().hex[:12],
                # Defensive: legacy sessions may have title=None on disk; fall back to 'Untitled'
                # so `+ " (copy)"` doesn't TypeError.
                title=(session.title or "Untitled") + " (copy)",
                workspace=session.workspace,
                model=session.model,
                model_provider=session.model_provider,
                messages=copy.deepcopy(session.messages),
                tool_calls=copy.deepcopy(session.tool_calls),
                # Reset ephemeral / per-session-instance flags. Duplicating an
                # archived conversation should produce a visible (un-archived)
                # copy; pinned status doesn't transfer either.
                pinned=False,
                archived=False,
                project_id=session.project_id,
                profile=session.profile,
                input_tokens=session.input_tokens,
                output_tokens=session.output_tokens,
                estimated_cost=session.estimated_cost,
                cache_read_tokens=getattr(session, "cache_read_tokens", 0),
                cache_write_tokens=getattr(session, "cache_write_tokens", 0),
                # Per-session settings the user may have customized — carry them over
                # so the duplicate behaves identically until further edits. Compression
                # anchor + last_prompt_tokens are intentionally NOT carried — those
                # re-derive on the next turn.
                personality=session.personality,
                enabled_toolsets=getattr(session, "enabled_toolsets", None),
                context_length=getattr(session, "context_length", None),
                threshold_tokens=getattr(session, "threshold_tokens", None),
                truncation_watermark=getattr(session, "truncation_watermark", None),
                truncation_boundary=getattr(session, "truncation_boundary", None),
                # context_messages is the authoritative model-facing prefix — must be
                # deepcopied so the duplicate has its own independent context that won't
                # be mutated when the original session's context changes (#2914).
                context_messages=copy.deepcopy(getattr(session, "context_messages", None) or []),
                # Gateway routing — if the user customized routing for this session,
                # the duplicate should behave identically.
                gateway_routing=copy.deepcopy(getattr(session, "gateway_routing", None)),
                gateway_routing_history=copy.deepcopy(getattr(session, "gateway_routing_history", None) or []),
                # Preserve LLM-generated title flag so we don't regenerate title on duplicate.
                llm_title_generated=getattr(session, "llm_title_generated", False),
                manual_title=getattr(session, "manual_title", False),
                # Composer draft — preserve per-session draft state.
                composer_draft=copy.deepcopy(getattr(session, "composer_draft", None) or {}),
                # Context engine state — preserve so the duplicate's context engine
                # starts from the same point as the original.
                context_engine=getattr(session, "context_engine", None),
                context_engine_state=copy.deepcopy(getattr(session, "context_engine_state", None) or {}),
                created_at=time.time(),
                updated_at=time.time(),
            )

            with LOCK:
                SESSIONS[copied_session.session_id] = copied_session
                SESSIONS.move_to_end(copied_session.session_id)
                _evict_sessions_over_cap()  # #4765: safe LRU eviction (never active/unsaved)
            # Persist immediately. The pre-PR flow (/api/session/new + /api/session/rename)
            # accidentally avoided this because `/api/session/rename` calls `s.save()`.
            # Without this explicit save, the duplicate is in-memory only — if the user
            # refreshes before sending a turn, the duplicate vanishes.
            copied_session.save()
            publish_session_list_changed(
                "session_duplicate",
                profile=getattr(copied_session, "profile", None),
                session_id=getattr(copied_session, "session_id", None),
            )

            return j(
                handler,
                {
                    "session": public_session_projection(
                        copied_session.compact() | {"messages": copied_session.messages}
                    )
                },
            )
        except Exception as e:
            return bad(handler, str(e))

    if parsed.path == "/api/default-model":
        try:
            advanced = body.get("advanced") if isinstance(body, dict) else None
            provider = body.get("provider") if isinstance(body, dict) else None
            if str(provider or "").strip().lower() == "auto":
                provider = None
            return j(handler, set_hermes_default_model(body.get("model"), provider=provider, advanced=advanced))
        except ValueError as e:
            return bad(handler, str(e))
        except RuntimeError as e:
            return bad(handler, str(e), 500)

    # ── Auxiliary model set (POST) ──
    if parsed.path == "/api/model/set":
        scope = str(body.get("scope") or "").strip()
        task = str(body.get("task") or "").strip()
        provider = str(body.get("provider") or "auto").strip()
        model = str(body.get("model") or "").strip()
        advanced = body.get("advanced") if isinstance(body, dict) else None
        if scope == "auxiliary":
            from api.config import set_auxiliary_model
            try:
                return j(handler, set_auxiliary_model(task, provider, model, advanced=advanced))
            except Exception as exc:
                return bad(handler, str(exc), status=400)
        if scope == "main":
            try:
                main_provider = provider if provider != "auto" else None
                return j(handler, set_hermes_default_model(model, provider=main_provider, advanced=advanced))
            except ValueError as exc:
                return bad(handler, str(exc), status=400)
        return bad(handler, f"unknown scope: {scope}", status=400)

    # ── Providers (POST) ──
    if parsed.path == "/api/providers":
        provider_id = (body.get("provider") or "").strip().lower()
        api_key = body.get("api_key")
        if not provider_id:
            return bad(handler, "provider is required")
        if api_key is not None:
            api_key = str(api_key).strip() or None
        result = set_provider_key(provider_id, api_key)
        if not result.get("ok"):
            return bad(handler, result.get("error", "Unknown error"))
        return j(handler, result)

    if parsed.path == "/api/providers/delete":
        provider_id = (body.get("provider") or "").strip().lower()
        if not provider_id:
            return bad(handler, "provider is required")
        result = remove_provider_key(provider_id)
        if not result.get("ok"):
            return bad(handler, result.get("error", "Unknown error"))
        return j(handler, result)

    if parsed.path == "/api/providers/self-hosted":
        try:
            from api.onboarding import apply_self_hosted_provider_setup
            return j(handler, apply_self_hosted_provider_setup(body))
        except ValueError as exc:
            return bad(handler, str(exc), 400)

    if parsed.path == "/api/models/refresh":
        provider_id = (body.get("provider") or "").strip().lower()
        if not provider_id:
            return bad(handler, "provider is required")
        from api.config import invalidate_provider_models_cache
        invalidate_provider_models_cache(provider_id)
        return j(handler, {"ok": True, "provider": provider_id})

    if parsed.path == "/api/reasoning":
        # CLI-parity /reasoning handler — writes to the same config.yaml keys
        # the CLI uses (display.show_reasoning, agent.reasoning_effort) so a
        # preference set via WebUI is honoured in the terminal REPL and vice
        # versa.  Body is one of:
        #   {"display": "show"|"hide"|"on"|"off"}   → display.show_reasoning
        #   {"effort":  "none"|"minimal"|"low"|"medium"|"high"|"xhigh"}
        #                                            → agent.reasoning_effort
        try:
            display = body.get("display")
            effort = body.get("effort")
            if display is not None:
                flag = str(display).strip().lower()
                if flag in ("show", "on", "true", "1"):
                    return j(handler, set_reasoning_display(True))
                if flag in ("hide", "off", "false", "0"):
                    return j(handler, set_reasoning_display(False))
                return bad(handler, f"display must be show|hide|on|off (got '{display}')")
            if effort is not None:
                model_id = str(body.get("model") or "").strip() or None
                provider_id = str(body.get("provider") or "").strip() or None
                base_url = str(body.get("base_url") or "").strip() or None
                return j(
                    handler,
                    set_reasoning_effort(
                        effort,
                        model_id=model_id,
                        provider_id=provider_id,
                        base_url=base_url,
                    ),
                )
            return bad(handler, "reasoning: must supply 'display' or 'effort'")
        except ValueError as e:
            return bad(handler, str(e))
        except RuntimeError as e:
            return bad(handler, str(e), 500)

    if parsed.path == "/api/admin/reload":
        # Hot-reload api.models module to pick up code changes without restart.
        import importlib
        from api import models as _models
        importlib.reload(_models)
        # Also re-expose get_session from the reloaded module so routes.py
        # continues to work (routes.py imported it at module level).
        import api.routes as _routes
        _routes.get_session = _models.get_session
        _routes.Session = _models.Session
        return j(handler, {"status": "ok", "reloaded": "api.models"})

    if parsed.path == "/api/sessions/cleanup":
        return _handle_sessions_cleanup(handler, body, zero_only=False)

    if parsed.path == "/api/sessions/cleanup_zero_message":
        return _handle_sessions_cleanup(handler, body, zero_only=True)

    if parsed.path == "/api/session/anchor-scene":
        return _handle_session_anchor_scene(handler, body)

    if parsed.path == "/api/session/rename":
        try:
            require(body, "session_id", "title")
        except ValueError as e:
            return bad(handler, str(e))
        try:
            s = _get_or_materialize_session(body["session_id"])
        except KeyError:
            return bad(handler, "Session not found", 404)
        except PermissionError:
            return bad(handler, "Read-only imported sessions cannot be renamed from WebUI", 403)
        with _get_session_agent_lock(body["session_id"]):
            from api.session_ops import apply_session_title_rename
            apply_session_title_rename(s, body["title"])
            s.save()
        _sync_session_title_to_insights(s)
        publish_session_list_changed(
            "session_rename",
            profile=getattr(s, "profile", None),
            session_id=getattr(s, "session_id", body["session_id"]),
        )
        return j(handler, {"session": s.compact()})


    if parsed.path == "/api/session/title/regenerate":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        sid = body["session_id"]
        prefer_latest = bool(body.get("prefer_latest", False))
        try:
            s = _get_or_materialize_session(sid)
        except KeyError:
            return bad(handler, "Session not found", 404)
        except PermissionError:
            return bad(handler, "Read-only imported sessions cannot regenerate titles", 403)
        next_title, reason, raw_preview = generate_session_title_for_session(s, prefer_latest=prefer_latest)
        if not next_title:
            return bad(handler, f"Could not generate a better title ({reason or 'empty'})", 422)
        _persist_generated_session_title(s, next_title, event_reason="session_title_regenerate")
        return j(handler, {
            "session": s.compact(),
            "title": s.title,
            "status": reason,
            "raw_preview": (raw_preview or "")[:240],
        })

    if parsed.path == "/api/personality/set":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if "name" not in body:
            return bad(handler, "Missing required field: name")
        sid = body["session_id"]
        if _session_is_subagent_view_only(sid):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        name = body["name"].strip()
        try:
            s = get_session(sid)
            s = _ensure_full_session_before_mutation(sid, s)
        except KeyError:
            return bad(handler, "Session not found", 404)
        # Resolve personality from config.yaml agent.personalities section
        # (matches hermes-agent CLI behavior)
        prompt = ""
        if name:
            from api.config import reload_config as _reload_cfg2

            _reload_cfg2()  # pick up config changes without restart
            from api.config import get_config as _get_cfg2

            _cfg2 = _get_cfg2()
            agent_cfg = _cfg2.get("agent", {})
            raw_personalities = agent_cfg.get("personalities", {})
            if not isinstance(raw_personalities, dict) or name not in raw_personalities:
                return bad(
                    handler, f'Personality "{name}" not found in config.yaml', 404
                )
            value = raw_personalities[name]
            # Resolve prompt using the same logic as hermes-agent cli.py
            if isinstance(value, dict):
                parts = [value.get("system_prompt", "") or value.get("prompt", "")]
                if value.get("tone"):
                    parts.append(f"Tone: {value['tone']}")
                if value.get("style"):
                    parts.append(f"Style: {value['style']}")
                prompt = "\n".join(p for p in parts if p)
            else:
                prompt = str(value)
        with _get_session_agent_lock(sid):
            s.personality = name if name else None
            s.save()
        return j(handler, {"ok": True, "personality": s.personality, "prompt": prompt})

    if parsed.path == "/api/session/toolsets":
        """Set or clear per-session toolset override (#493).

        POST body: { session_id, toolsets: [...] | null }
        - toolsets: list of toolset names to restrict the session to, or null to clear.
        """
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        sid = body["session_id"]
        if _session_is_subagent_view_only(sid):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        toolsets = body.get("toolsets")
        try:
            toolsets = _validate_session_toolsets_shape(toolsets)
        except ValueError as e:
            return bad(handler, str(e), status=400)
        try:
            s = get_session(sid)
        except KeyError:
            return bad(handler, "Session not found", 404)
        with _get_session_agent_lock(sid):
            s.enabled_toolsets = toolsets
            s.save()
        return j(handler, {"ok": True, "enabled_toolsets": s.enabled_toolsets})

    if parsed.path == "/api/session/draft":
        # GET ?session_id=X  → return current draft
        # POST body          → save draft { session_id, text?, files? }
        # HTTP method is in handler.command (e.g. "POST", "GET"), parsed has no .method
        import time as _draft_time
        _draft_t0 = _draft_time.monotonic()
        _draft_stages = []

        def _draft_mark(name):
            _draft_stages.append((name, _draft_time.monotonic()))
        _draft_mark("enter")
        if handler.command == "GET":
            query = parse_qs(parsed.query)
            sid = query.get("session_id", [""])[0] if parsed.query else ""
            if not sid:
                return bad(handler, "session_id is required", 400)
            try:
                s = get_session(sid)
            except KeyError:
                return bad(handler, "Session not found", 404)
            draft = getattr(s, "composer_draft", {}) or {}
            return j(handler, {"draft": draft})
        # POST
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        sid = body["session_id"]
        if _session_is_subagent_view_only(sid):
            return bad(handler, "Subagent sessions are view-only and cannot store a draft from WebUI", 400)
        text = body.get("text")
        files = body.get("files")
        # Stage-326 hardening (per Opus advisor): size + type validation on
        # the draft inputs. Without this, a misbehaving or malicious client
        # can persist multi-MB strings into the session JSON on every keystroke
        # via the 400ms debounced auto-save.
        _MAX_DRAFT_TEXT = 50_000  # 50 KB cap on textarea content
        _MAX_DRAFT_FILES = 50  # max number of attached file references
        if text is not None and not isinstance(text, str):
            text = ""
        if isinstance(text, str) and len(text) > _MAX_DRAFT_TEXT:
            text = text[:_MAX_DRAFT_TEXT]
        if files is not None and not isinstance(files, list):
            files = []
        if isinstance(files, list) and len(files) > _MAX_DRAFT_FILES:
            files = files[:_MAX_DRAFT_FILES]
        try:
            s = get_session(sid)
        except KeyError:
            return bad(handler, "Session not found", 404)
        _draft_mark("after_get_session")
        unchanged = False
        with _get_session_agent_lock(sid):
            _draft_mark("acquired_lock")
            current_draft = dict(getattr(s, "composer_draft", {}) or {})
            next_draft = dict(current_draft)
            if text is not None:
                next_draft["text"] = text
            if files is not None:
                next_draft["files"] = files
            if next_draft == current_draft:
                unchanged = True
                saved_draft = current_draft
            else:
                s.composer_draft = next_draft
                # Draft persistence is not conversation activity. Touching updated_at
                # here makes the active-session external-refresh poll force-reload the
                # current chat every few seconds while the user is typing, and that
                # delayed reload can restore an older draft over newer local input.
                _draft_mark("before_save")
                s.save(touch_updated_at=False, skip_index=True)
                _draft_mark("after_save")
                saved_draft = s.composer_draft
        _draft_mark("released_lock")
        payload = {"ok": True, "draft": saved_draft}
        if unchanged:
            payload["unchanged"] = True
        _draft_mark("before_json")
        j(handler, payload)
        _draft_mark("after_json")
        _draft_stages.append(("end", _draft_time.monotonic()))
        if _draft_stages[-1][1] - _draft_t0 > 0.2:
            parts = " ".join(
                f"{n}={((t - prev[1]) * 1000):.1f}ms"
                for (n, t), prev in zip(_draft_stages[1:], _draft_stages[:-1], strict=True)
            )
            handler._safe_webui_print(
                "[SLOW] /api/session/draft total=%.1fms stages: %s" % (
                    (_draft_stages[-1][1] - _draft_t0) * 1000,
                    parts,
                )
            )
        return True

    if parsed.path == "/api/session/update":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        try:
            s = _get_or_materialize_session(body["session_id"])
        except KeyError:
            return bad(handler, "Session not found", 404)
        except PermissionError:
            return bad(handler, "Read-only imported sessions cannot be updated from WebUI", 403)
        old_ws = getattr(s, "workspace", "")
        old_model = getattr(s, "model", None)
        old_provider = getattr(s, "model_provider", None)
        try:
            new_ws = str(resolve_trusted_workspace(body.get("workspace", s.workspace), profile=getattr(s, "profile", None)))
        except ValueError as e:
            return bad(handler, str(e))
        with _get_session_agent_lock(body["session_id"]):
            s.workspace = new_ws
            if "model" in body or "model_provider" in body:
                model, provider = _session_model_state_from_request(
                    body.get("model", s.model),
                    body.get("model_provider") if "model_provider" in body else None,
                    getattr(s, "model_provider", None),
                )
                if model is not None:
                    s.model = model
                s.model_provider = provider
                if (
                    str(old_model or "") != str(getattr(s, "model", "") or "")
                    or str(old_provider or "") != str(getattr(s, "model_provider", "") or "")
                ):
                    s.context_length = _resolve_context_length_for_session_model(
                        getattr(s, "model", None),
                        getattr(s, "model_provider", None),
                    )
                    s.threshold_tokens = 0
                    s.last_prompt_tokens = 0
                    from api.config import _evict_session_agent

                    _evict_session_agent(body["session_id"])
            s.save()
        if str(old_ws or "") != str(new_ws or ""):
            try:
                from api.terminal import close_terminal
                close_terminal(body["session_id"])
            except Exception:
                logger.debug("Failed to close workspace terminal after workspace update")
        set_last_workspace(new_ws, profile=getattr(s, "profile", None))
        return j(
            handler,
            {"session": public_session_projection(s.compact() | {"messages": s.messages})},
        )
    if parsed.path == "/api/session/worktree/remove":
        sid = body.get("session_id", "")
        if not sid or not isinstance(sid, str) or not sid.strip():
            return bad(handler, "session_id must be a non-empty string", status=400)
        sid = sid.strip()
        if not is_safe_session_id(sid):
            return bad(handler, "Invalid session_id", 400)
        try:
            s = get_session(sid, metadata_only=True)
        except KeyError:
            return bad(handler, "Session not found", status=404)
        force = bool(body.get("force", False))
        try:
            from api.worktrees import remove_worktree_for_session

            result = remove_worktree_for_session(s, force=force)
            return j(handler, result)
        except ValueError as exc:
            return bad(handler, str(exc), status=400)
        except Exception as exc:
            logger.exception("failed to remove worktree for session %s", sid)
            return bad(handler, _sanitize_error(exc), status=500)

    if parsed.path == "/api/session/delete":
        sid = body.get("session_id", "")
        if not sid:
            return bad(handler, "session_id is required")
        if not is_safe_session_id(sid):
            return bad(handler, "Invalid session_id", 400)
        cli_meta_for_delete = _lookup_cli_session_metadata(sid)
        if cli_meta_for_delete.get("read_only"):
            return bad(handler, "Read-only imported sessions cannot be deleted from WebUI", 400)
        # A delegated subagent child (#5307) is view-only and owned by the
        # delegate runner. Deleting it here would call delete_cli_session() and
        # erase the child's state.db transcript — refuse it.
        if _session_is_subagent_view_only(sid):
            return bad(handler, "Subagent sessions are view-only and cannot be deleted from WebUI", 400)
        is_messaging_session = _is_messaging_session_id(sid)
        worktree_retained = _worktree_retained_payload_for_session_id(sid)
        try:
            event_profile = getattr(get_session(sid, metadata_only=True), "profile", None)
        except KeyError:
            event_profile = None
        except Exception:
            logger.debug("Failed to resolve profile for deleted session %s", sid, exc_info=True)
            event_profile = None
        # Serialize with recovery, but bound contention so a browser timeout
        # cannot be followed by a delayed server-side delete.
        session_lock = _get_session_agent_lock(sid)
        if not session_lock.acquire(timeout=5):
            return bad(handler, "Session busy, try again", 503)
        try:
            with LOCK:
                SESSIONS.pop(sid, None)
            try:
                p = (SESSION_DIR / f"{sid}.json").resolve()
                p.relative_to(SESSION_DIR.resolve())
            except Exception:
                return bad(handler, "Invalid session_id", 400)
            sidecar_deleted = False
            try:
                p.unlink(missing_ok=True)
            except Exception:
                logger.debug("Failed to unlink session file %s", p)
            sidecar_deleted = not p.exists()
            try:
                prune_session_from_index(sid)
            except Exception:
                logger.debug("Failed to prune deleted session from index: %s", sid, exc_info=True)
            try:
                p.with_suffix('.json.bak').unlink(missing_ok=True)
            except Exception:
                logger.debug("Failed to unlink session backup file %s", p.with_suffix('.json.bak'))
            if sidecar_deleted and not is_messaging_session:
                try:
                    _record_webui_deleted_session_tombstone(sid)
                except Exception:
                    logger.debug("Failed to tombstone deleted WebUI session %s", sid, exc_info=True)
        finally:
            session_lock.release()
        # Evict outside the mutation lock: lifecycle commit may perform provider
        # I/O and must not hold a per-session Session lock.
        from api.config import _evict_session_agent
        _evict_session_agent(sid)
        try:
            from api.upload import _session_attachment_dir

            shutil.rmtree(_session_attachment_dir(sid), ignore_errors=True)
        except Exception:
            logger.debug("Failed to clean attachment dir for deleted session %s", sid)
        # Remove the turn-journal shards and the run-journal directory so a
        # deleted conversation is not recoverable from disk. The session JSON +
        # state.db rows are cleared above, but these journals retain the user's
        # messages (turn journal) and the full request/response payloads (run
        # journal) in plaintext. (#3802)
        try:
            from api.turn_journal import delete_turn_journal

            delete_turn_journal(sid)
        except Exception:
            logger.debug("Failed to delete turn journal for deleted session %s", sid)
        try:
            from api.run_journal import delete_run_journal

            delete_run_journal(sid)
        except Exception:
            logger.debug("Failed to delete run journal for deleted session %s", sid)
        # The weak lock registry releases this entry automatically after all
        # holders and waiters drop their strong references.
        # Prune the completion-dedup entry too. The reaper sweeps it once the
        # completion is delivered (drained from PENDING); a session deleted
        # while a completion is still pending would otherwise keep its entry.
        try:
            from api.background_process import forget_bg_task_completion_dedup

            forget_bg_task_completion_dedup(sid)
        except Exception:
            logger.debug("Failed to prune bg-task dedup entry for deleted session %s", sid)
        try:
            from api.terminal import close_terminal
            close_terminal(sid)
        except Exception:
            logger.debug("Failed to close workspace terminal for deleted session %s", sid)
        # Also delete from CLI state.db for CLI sessions shown in sidebar,
        # but never erase external messaging channel memory via WebUI delete.
        state_db_cleanup_failed = False
        if not is_messaging_session:
            try:
                from api.models import delete_cli_session

                state_db_cleanup_failed = not delete_cli_session(sid)
            except Exception:
                state_db_cleanup_failed = True
                logger.warning("Failed to delete CLI session %s", sid, exc_info=True)
        _publish_session_list_changed("session_delete", profile=event_profile)
        return j(
            handler,
            {
                "ok": True,
                "state_db_cleanup_failed": state_db_cleanup_failed,
                **worktree_retained,
            },
        )

    if parsed.path == "/api/session/clear":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if _session_is_subagent_view_only(body["session_id"]):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        try:
            s = get_session(body["session_id"])
        except KeyError:
            return bad(handler, "Session not found", 404)
        sid = body["session_id"]
        with _get_session_agent_lock(sid):
            had_sidecar_messages = bool(s.messages or [])
            # Clear is a full truncate-to-empty: route through the SAME helper the
            # /api/session/truncate handler uses (single source of truth) so the
            # display + context arrays are emptied AND the truncation watermark is
            # set via _truncation_watermark_for([]) == 0.0 — the #2914
            # truncate-to-empty sentinel that blocks state.db replay. Before this,
            # /clear wiped s.messages but left the watermark unset, so the
            # append-only state.db merge treated it as "keep everything" and the
            # cleared history resurrected on the next /api/session read (#5532).
            from api.session_ops import truncate_session_at_keep
            truncate_session_at_keep(s, 0)
            s.tool_calls = []
            # A compressed-continuation child keeps its archived transcript in a
            # parent sidecar marked pre_compression_snapshot;
            # _webui_sidecar_lineage_messages_for_display() stitches that parent
            # back with truncation_watermark=None, so the 0.0 sentinel on the
            # CHILD does NOT stop the parent from resurrecting the cleared history
            # on refresh. Detach the compression lineage (#5532/#5553) — but ONLY
            # when the parent is actually a pre_compression_snapshot; a genuine
            # fork parent (session_source="fork" from /api/session/branch) must
            # keep its link so the child still nests + shows "Forked from"
            # (sessions.js:5720/5964/7105). Dropping every parent broke that
            # (#5532 Codex gate).
            _parent_sid = getattr(s, "parent_session_id", None)
            if _parent_sid:
                _parent_is_compression_snapshot = False
                try:
                    _parent = get_session(_parent_sid, metadata_only=True)
                    _parent_is_compression_snapshot = bool(
                        getattr(_parent, "pre_compression_snapshot", False)
                    )
                except Exception:
                    _parent_is_compression_snapshot = False
                if _parent_is_compression_snapshot:
                    s.parent_session_id = None
                    s.compression_anchor_visible_idx = None
                    s.compression_anchor_message_key = None
            s.active_stream_id = None
            s.pending_user_message = None
            s.pending_attachments = []
            s.pending_started_at = None
            s.pending_user_source = None
            s.clear_generation = uuid.uuid4().hex if had_sidecar_messages else None
            # Reset the title via the rename helper so clearing a manually-named
            # session also clears manual_title/llm_title_generated — otherwise the
            # reused session keeps its manual-title protection and never auto-names
            # again (#3542 lifecycle gap).
            from api.session_ops import apply_session_title_rename
            apply_session_title_rename(s, "Untitled")
            s.save()
            persisted_clear = False
            try:
                persisted = json.loads(s.path.read_text(encoding="utf-8"))
                persisted_clear = (
                    persisted.get("messages") == []
                    and persisted.get("context_messages") == []
                    and persisted.get("truncation_watermark") == 0.0
                    and persisted.get("truncation_boundary") == 0.0
                    and persisted.get("active_stream_id") is None
                    and persisted.get("pending_user_message") is None
                    and persisted.get("pending_attachments") == []
                    and persisted.get("pending_started_at") is None
                    and persisted.get("pending_user_source") is None
                    and persisted.get("clear_generation") == s.clear_generation
                )
            except (OSError, json.JSONDecodeError, ValueError):
                logger.warning("session clear could not verify persisted empty state for %s", sid, exc_info=True)
            if had_sidecar_messages and persisted_clear:
                try:
                    s.path.with_suffix('.json.bak').unlink(missing_ok=True)
                except OSError:
                    logger.warning("session clear could not remove stale backup for %s", sid, exc_info=True)
        # Evict cached agent outside the per-session lock.  Eviction may run a
        # boundary memory commit for batch-extraction providers, and provider
        # I/O must not hold the session mutation lock.
        from api.config import _evict_session_agent
        _evict_session_agent(sid)
        return j(handler, {"ok": True, "session": s.compact()})

    if parsed.path == "/api/session/truncate":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if _session_is_subagent_view_only(body["session_id"]):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        if body.get("keep_count") is None:
            return bad(handler, "Missing required field(s): keep_count")
        try:
            s = get_session(body["session_id"])
        except KeyError:
            return bad(handler, "Session not found", 404)
        # Validate keep_count before it reaches the destructive `messages[:keep]`
        # slice. A non-numeric value would raise ValueError and surface as a
        # confusing 500; a NEGATIVE value slices as `messages[:-N]`, which
        # silently DELETES the most recent N messages (e.g. keep_count=-5 on a
        # 3-message session wipes the whole transcript) and then persists it via
        # save(). Mirror the explicit guard the /api/session/branch handler
        # already applies to its own keep_count. (Opus pre-release follow-up.)
        try:
            keep = int(body["keep_count"])
        except (ValueError, TypeError):
            return bad(handler, "keep_count must be an integer")
        if keep < 0:
            return bad(handler, "keep_count must be non-negative")
        with _get_session_agent_lock(body["session_id"]):
            from api.session_ops import truncate_session_at_keep

            old_msg_count, old_ctx_count = truncate_session_at_keep(s, keep)
            s.save()
            logger.info(
                "truncate %s: messages %d→%d, context_messages %d→%d, watermark=%.2f",
                body["session_id"], old_msg_count, len(s.messages or []),
                old_ctx_count, len(getattr(s, 'context_messages', None) or []),
                s.truncation_watermark or 0,
            )
        from api.config import _evict_session_agent
        _evict_session_agent(body["session_id"])
        return j(
            handler,
            {
                "ok": True,
                "session": public_session_projection(
                    s.compact() | {"messages": s.messages}
                ),
            },
        )

    if parsed.path == "/api/session/branch":
        # Fork a conversation from any message point (#465).
        # Accepts: {session_id, keep_count?, title?}
        #   keep_count: number of messages to copy (0=empty, undefined=full history)
        #   title: custom title (defaults to "<original title> (fork)")
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        # Reject non-string session_id explicitly so the failure surfaces as a
        # 400 instead of a generic 500 from get_session() raising TypeError.
        # (Opus pre-release follow-up.)
        if not isinstance(body["session_id"], str):
            return bad(handler, "session_id must be a string")
        source = _load_branch_source_or_refuse(handler, body["session_id"])
        if source is None:
            return True

        keep_count = body.get("keep_count")
        if keep_count is not None:
            try:
                keep_count = int(keep_count)
            except (ValueError, TypeError):
                return bad(handler, "keep_count must be an integer")
            # Negative slice (`messages[:-N]`) returns "all but last N", which
            # is a confusing fork semantic. Reject explicitly so the user
            # doesn't accidentally fork a session with the tail truncated when
            # they meant to copy the prefix. (Opus pre-release follow-up.)
            if keep_count < 0:
                return bad(handler, "keep_count must be non-negative")

        custom_title = body.get("title")
        if custom_title:
            custom_title = str(custom_title).strip()[:80] or None

        # Build messages slice in the same coordinate space exposed by GET
        # /api/session so frontend keep_count values from merged messaging
        # transcripts do not silently become full sidecar copies.
        try:
            if not getattr(source, "_branch_source_readonly", False): source.save()
        except Exception:
            pass
        cli_meta = _lookup_cli_session_metadata(source.session_id) if _session_requires_cli_metadata_lookup(source) else {}
        is_messaging_session = _is_messaging_session_record(source) or _is_messaging_session_record(cli_meta)
        cli_messages = get_cli_session_messages(source.session_id) if is_messaging_session else []
        if is_messaging_session:
            if cli_messages:
                source_messages = _merged_session_messages_for_display(source, cli_messages)
            else:
                # Match GET /api/session: a messaging session with no CLI
                # transcript does not fall back to state.db rows.
                source_messages = merge_session_messages_append_only(
                    _webui_sidecar_lineage_messages_for_display(source),
                    [],
                    truncation_watermark=getattr(source, "truncation_watermark", None),
                    truncation_boundary=getattr(source, "truncation_boundary", None),
                )
                source_messages = _merged_webui_lineage_messages_for_display(source, source_messages)
        else:
            # Match GET /api/session's full-transcript display path exactly:
            # sidecar lineage stitched across compression snapshots, merged
            # append-only with state.db rows, then parent-row backfill for
            # partial continuations. The frontend's keep_count is an index
            # into THAT merged list; slicing the raw sidecar instead landed
            # the cut too early whenever the merged view deduplicates rows
            # (replayed sidecar/state.db doubles, filtered prefixes), so the
            # fork stopped mid tool-run and dropped the final conclusion.
            _state_db_reader_kwargs = {
                "profile": getattr(source, "profile", None) or None,
            }
            _backstop = _state_db_backstop_limit_for_display(source, None)
            if _backstop is not None:
                _state_db_reader_kwargs["limit"] = _backstop
            source_messages = merge_session_messages_append_only(
                _webui_sidecar_lineage_messages_for_display(source),
                get_state_db_session_messages(
                    source.session_id,
                    **_state_db_reader_kwargs,
                ),
                truncation_watermark=getattr(source, "truncation_watermark", None),
                truncation_boundary=getattr(source, "truncation_boundary", None),
            )
            source_messages = _merged_webui_lineage_messages_for_display(source, source_messages)
        if keep_count is not None:
            forked_messages = source_messages[:keep_count]
        else:
            forked_messages = list(source_messages)

        # Derive title
        if custom_title:
            branch_title = custom_title
        else:
            source_title = source.title or "Untitled"
            branch_title = f"{source_title} (fork)"

        # Create new session inheriting workspace/model/profile
        from api.session_ops import truncate_context_for_display_keep

        fork_keep = keep_count if keep_count is not None else len(source_messages)
        forked_context = copy.deepcopy(
            truncate_context_for_display_keep(
                getattr(source, "context_messages", None),
                source_messages,
                fork_keep,
            )
        )
        # `truncate_context_for_display_keep` aligns rows by visible display
        # identity but intentionally does not carry provider replay sidecars.
        # Reconcile only the retained prefix before constructing the branch so
        # its persisted model context receives the same conservative
        # api_content bytes as the forked display messages.  The helper mutates
        # the freshly aligned context copy; the source session and
        # forked_messages ownership remain untouched.
        _reconcile_api_content_sidecars(forked_context, forked_messages)
        branch = Session(
            workspace=source.workspace,
            model=source.model,
            model_provider=getattr(source, "model_provider", None),
            profile=getattr(source, "profile", None),
            title=branch_title,
            messages=forked_messages,
            project_id=getattr(source, "project_id", None),
            personality=getattr(source, "personality", None),
            enabled_toolsets=getattr(source, "enabled_toolsets", None),
            context_length=getattr(source, "context_length", None),
            threshold_tokens=getattr(source, "threshold_tokens", None),
            # context_messages — truncated to fork prefix (not full parent copy)
            context_messages=copy.deepcopy(forked_context),
            # Gateway routing — inherit from source
            gateway_routing=copy.deepcopy(getattr(source, "gateway_routing", None)),
            # Context engine — inherit state so branch's context engine starts correctly
            context_engine=getattr(source, "context_engine", None),
            context_engine_state=copy.deepcopy(getattr(source, "context_engine_state", None) or {}),
            parent_session_id=source.session_id,
            session_source="fork",
        )
        with LOCK:
            SESSIONS[branch.session_id] = branch
            SESSIONS.move_to_end(branch.session_id)
            _evict_sessions_over_cap()  # #4765: safe LRU eviction (never active/unsaved)

        # Persist only if there are messages (matches new_session pattern)
        if forked_messages:
            branch.save()
            publish_session_list_changed(
                "session_branch",
                profile=getattr(branch, "profile", None),
                session_id=getattr(branch, "session_id", None),
            )

        return j(handler, {
            "session_id": branch.session_id,
            "title": branch_title,
            "parent_session_id": source.session_id,
        })

    if parsed.path == "/api/session/compress/start":
        return _handle_session_compress_start(handler, body)

    if parsed.path == "/api/session/compress":
        return _handle_session_compress(handler, body)

    if parsed.path == "/api/session/conversation-rounds":
        return _handle_conversation_rounds(handler, body)

    if parsed.path == "/api/session/handoff-summary":
        return _handle_handoff_summary(handler, body)

    if parsed.path == "/api/session/retry":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if _session_is_subagent_view_only(body["session_id"]):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        try:
            from api.session_ops import retry_last
            result = retry_last(body["session_id"])
            return j(handler, {"ok": True, **result})
        except KeyError:
            return bad(handler, "Session not found", 404)
        except ValueError as e:
            return j(handler, {"error": str(e)})

    if parsed.path == "/api/session/undo":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if _session_is_subagent_view_only(body["session_id"]):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        try:
            from api.session_ops import undo_last
            result = undo_last(body["session_id"])
            return j(handler, {"ok": True, **result})
        except KeyError:
            return bad(handler, "Session not found", 404)
        except ValueError as e:
            return j(handler, {"error": str(e)})

    # ── YOLO mode toggle (POST) ──
    # Session-scoped only — stored in-memory on the server side.
    # Important lifecycle notes:
    #   • Page reload: state PERSISTS (frontend re-fetches via GET endpoint)
    #   • Cross-tab: state is SHARED (same server-side flag per session)
    #   • Server restart: state is LOST (in-memory only)
    #   • Cross-session: isolated (each session has its own flag)
    # Fixes #467
    if parsed.path == "/api/session/yolo":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        sid = str(body["session_id"] or "").strip()
        enabled = bool(body.get("enabled", True))
        if not enabled:
            with gateway_yolo_handoff(sid):
                set_session_yolo_enabled(sid, False)
                return j(handler, {"ok": True, "yolo_enabled": bool(is_session_yolo_enabled(sid))})

        payload, status = _enable_session_yolo_and_release_pending(sid, choice="once")
        return j(handler, payload, status=status)

    if parsed.path == "/api/btw":
        return _handle_btw(handler, body)

    if parsed.path == "/api/background":
        return _handle_background(handler, body)

    if parsed.path == "/api/goal":
        return _handle_goal_command(handler, body)

    if parsed.path == "/api/bg-task-complete-ack":
        return _handle_bg_task_complete_ack(handler, body)

    if parsed.path == "/api/chat/start":
        return _handle_chat_start(handler, body, diag=diag)

    if parsed.path == "/api/chat":
        return _handle_chat_sync(handler, body)

    if parsed.path == "/api/chat/steer":
        from api.streaming import _handle_chat_steer
        return _handle_chat_steer(handler, body)

    if parsed.path == "/api/terminal/start":
        return _handle_terminal_start(handler, body)

    if parsed.path == "/api/terminal/input":
        return _handle_terminal_input(handler, body)

    if parsed.path == "/api/terminal/resize":
        return _handle_terminal_resize(handler, body)

    if parsed.path == "/api/terminal/close":
        return _handle_terminal_close(handler, body)

    # ── Cron API (POST) ──
    # See GET-side comment above: wrap in cron_profile_context so writes go
    # to the TLS-active profile's jobs.json instead of the process default.
    if parsed.path == "/api/crons/create":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_create(handler, body)

    if parsed.path == "/api/crons/update":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_update(handler, body)

    if parsed.path == "/api/crons/delete":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_delete(handler, body)

    if parsed.path == "/api/crons/run":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_run(handler, body)

    if parsed.path == "/api/crons/pause":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_pause(handler, body)

    if parsed.path == "/api/crons/resume":
        from api.profiles import cron_profile_context

        with cron_profile_context():
            _ensure_agent_cron_import_path()
            return _handle_cron_resume(handler, body)

    # ── Git workspace ops (POST) ──
    if parsed.path == "/api/git/stage":
        return _handle_git_stage(handler, body)

    if parsed.path == "/api/git/unstage":
        return _handle_git_unstage(handler, body)

    if parsed.path == "/api/git/discard":
        return _handle_git_discard(handler, body)

    if parsed.path == "/api/git/commit-message":
        return _handle_git_commit_message(handler, body)

    if parsed.path == "/api/git/commit-message-selected":
        return _handle_git_commit_message_selected(handler, body)

    if parsed.path == "/api/git/commit":
        return _handle_git_commit(handler, body)

    if parsed.path == "/api/git/commit-selected":
        return _handle_git_commit_selected(handler, body)

    if parsed.path == "/api/git/fetch":
        return _handle_git_remote_action(handler, body, "fetch")

    if parsed.path == "/api/git/pull":
        return _handle_git_remote_action(handler, body, "pull")

    if parsed.path == "/api/git/push":
        return _handle_git_remote_action(handler, body, "push")

    if parsed.path == "/api/git/checkout":
        return _handle_git_checkout(handler, body)

    if parsed.path == "/api/git/stash-checkout":
        return _handle_git_stash_checkout(handler, body)

    # ── File ops (POST) ──
    if parsed.path == "/api/file/delete":
        return _handle_file_delete(handler, body)

    if parsed.path == "/api/file/save":
        return _handle_file_save(handler, body)

    if parsed.path == "/api/file/office-save":
        return _handle_office_file_save(handler, body)

    if parsed.path == "/api/file/create":
        return _handle_file_create(handler, body)

    if parsed.path == "/api/file/rename":
        return _handle_file_rename(handler, body)

    if parsed.path == "/api/file/move":
        return _handle_file_move(handler, body)

    if parsed.path == "/api/file/create-dir":
        return _handle_create_dir(handler, body)

    if parsed.path == "/api/file/reveal":
        return _handle_file_reveal(handler, body)

    if parsed.path == "/api/file/path":
        return _handle_file_path(handler, body)

    if parsed.path == "/api/file/open-vscode":
        return _handle_file_open_vscode(handler, body)

    # ── Workspace management (POST) ──
    if parsed.path == "/api/workspaces/add":
        return _handle_workspace_add(handler, body)

    if parsed.path == "/api/workspaces/remove":
        return _handle_workspace_remove(handler, body)

    if parsed.path == "/api/workspaces/rename":
        return _handle_workspace_rename(handler, body)

    if parsed.path == "/api/workspaces/reorder":
        return _handle_workspace_reorder(handler, body)

    # ── Approval (POST) ──
    if parsed.path == "/api/approval/respond":
        return _handle_approval_respond(handler, body)

    # ── Clarify (POST) ──
    if parsed.path == "/api/clarify/respond":
        return _handle_clarify_respond(handler, body)

    # ── Commands (POST) ──
    if parsed.path == "/api/commands/bundles/resolve":
        from api.commands import resolve_bundle_command

        command = str(body.get("command", "") or "").strip()
        if not command:
            return bad(handler, "command is required")

        try:
            return j(handler, resolve_bundle_command(command))
        except KeyError:
            return bad(handler, "Bundle command not found", 404)
        except ValueError as e:
            return bad(handler, str(e), 400)
        except RuntimeError as e:
            return bad(handler, _sanitize_error(e), 500)

    if parsed.path == "/api/commands/exec":
        from api.commands import execute_agent_command, execute_plugin_command

        command = str(body.get("command", "") or "").strip()
        if not command:
            return bad(handler, "command is required")

        try:
            return j(handler, {"output": execute_agent_command(command)})
        except KeyError:
            pass
        except ValueError as e:
            return bad(handler, str(e), 400)
        except RuntimeError as e:
            return bad(handler, _sanitize_error(e), 500)

        try:
            return j(handler, {"output": execute_plugin_command(command)})
        except ValueError as e:
            return bad(handler, str(e), 400)
        except KeyError:
            return bad(handler, "Plugin command not found", 404)
        except RuntimeError as e:
            return bad(handler, _sanitize_error(e), 500)

    # ── Skills (POST) ──
    if parsed.path == "/api/skills/save":
        return _handle_skill_save(handler, body)

    if parsed.path == "/api/skills/delete":
        return _handle_skill_delete(handler, body)

    if parsed.path == "/api/skills/toggle":
        return _handle_skill_toggle(handler, body)

    # ── Memory (POST) ──
    if parsed.path == "/api/memory/write":
        return _handle_memory_write(handler, body)

    if parsed.path in {"/api/gateway/start", "/api/gateway/stop", "/api/gateway/restart"}:
        return _handle_gateway_lifecycle(handler, parsed.path.rsplit("/", 1)[-1], body)

    # ── Profile API (POST) ──
    if parsed.path == "/api/profile/switch":
        name = body.get("name", "").strip()
        if not name:
            return bad(handler, "name is required")
        try:
            from api.auth import ensure_trusted_auth_session
            from api.profiles import switch_profile, _validate_profile_name
            from api.helpers import build_profile_cookie
            if name != 'default':
                _validate_profile_name(name)
            session_info = ensure_trusted_auth_session(handler)
            if getattr(handler, '_trusted_auth_session_rejected', False):
                return bad(handler, 'Authentication required', 401)
            bound_profile = str((session_info or {}).get("bound_profile") or "").strip() or None
            if bound_profile and name != bound_profile:
                return bad(handler, "Profile is bound to the current session", 403)
            # process_wide=False: don't mutate the process-global _active_profile.
            # Per-client profile is managed via cookie + thread-local (#798).
            result = switch_profile(name, process_wide=False)
            # Invalidate the models cache so the very next /api/models request
            # rebuilds from the new profile's config.yaml rather than returning
            # the old profile's cached model list (#1200 — profile-switch model bug).
            # The per-profile disk snapshot is fingerprint-guarded, so keep it.
            from api.config import invalidate_models_cache
            invalidate_models_cache(delete_disk=False)
            try:
                from api.gateway_watcher import restart_watcher_for_profile
                restart_watcher_for_profile(name)
            except Exception as exc:
                logger.warning("Failed to restart gateway watcher for profile %s: %s", name, exc)
            session_cookie_value = getattr(handler, '_trusted_auth_session_cookie_value', None)
            if session_cookie_value:
                if bound_profile and name == bound_profile:
                    return j(handler, result)
                extra_header = build_profile_cookie(name, session_cookie_value=session_cookie_value)
            else:
                extra_header = build_profile_cookie(name, handler)
            return j(handler, result, extra_headers={
                'Set-Cookie': extra_header,
            })
        except PermissionError as e:
            return bad(handler, _sanitize_error(e), 403)
        except (ValueError, FileNotFoundError) as e:
            return bad(handler, _sanitize_error(e), 404)
        except RuntimeError as e:
            return bad(handler, str(e), 409)

    if parsed.path == "/api/profile/create":
        name = body.get("name", "").strip()
        if not name:
            return bad(handler, "name is required")
        import re as _re

        if not _re.match(r"^[a-z0-9][a-z0-9_-]{0,63}$", name):
            return bad(
                handler,
                "Invalid profile name: lowercase letters, numbers, hyphens, underscores only",
            )
        clone_from = body.get("clone_from")
        if clone_from is not None:
            clone_from = str(clone_from).strip()
            if not _re.match(r"^[a-z0-9][a-z0-9_-]{0,63}$", clone_from):
                return bad(handler, "Invalid clone_from name")
        base_url = body.get("base_url", "").strip() if body.get("base_url") else None
        api_key = body.get("api_key", "").strip() if body.get("api_key") else None
        default_model = body.get("default_model", "").strip() if body.get("default_model") else None
        model_provider = body.get("model_provider", "").strip() if body.get("model_provider") else None
        if base_url and not base_url.startswith(("http://", "https://")):
            return bad(handler, "base_url must start with http:// or https://")
        try:
            from api.profiles import create_profile_api

            result = create_profile_api(
                name,
                clone_from=clone_from,
                clone_config=bool(body.get("clone_config", False)),
                base_url=base_url,
                api_key=api_key,
                default_model=default_model,
                model_provider=model_provider,
            )
            return j(handler, {"ok": True, "profile": result})
        except PermissionError as e:
            return bad(handler, _sanitize_error(e), 403)
        except (ValueError, FileExistsError, RuntimeError) as e:
            return bad(handler, str(e))

    if parsed.path == "/api/profile/delete":
        name = body.get("name", "").strip()
        if not name:
            return bad(handler, "name is required")
        try:
            from api.profiles import delete_profile_api, _validate_profile_name

            _validate_profile_name(name)
            result = delete_profile_api(name)
            return j(handler, result)
        except PermissionError as e:
            return bad(handler, _sanitize_error(e), 403)
        except (ValueError, FileNotFoundError) as e:
            return bad(handler, _sanitize_error(e))
        except RuntimeError as e:
            return bad(handler, str(e), 409)

    # ── Settings (POST) ──
    if parsed.path == "/api/settings":
        from api.auth import (
            create_session,
            get_password_hash,
            is_auth_enabled,
            parse_cookie,
            set_auth_cookie,
            verify_password,
            verify_session,
        )

        if "bot_name" in body:
            body["bot_name"] = (str(body["bot_name"]) or "").strip() or "Hermes"

        auth_enabled_before = is_auth_enabled()
        password_auth_enabled_before = auth_enabled_before and get_password_hash() is not None
        current_cookie = parse_cookie(handler)
        logged_in_before = bool(current_cookie and verify_session(current_cookie))
        requested_password = bool(
            isinstance(body.get("_set_password"), str)
            and body.get("_set_password", "").strip()
        )
        requested_passwordless = bool(body.pop("_passwordless", False))
        requested_clear_password = bool(body.get("_clear_password") or requested_passwordless)
        if requested_passwordless:
            body["_clear_password"] = True

        current_password = body.pop("_current_password", None)

        # #1560: HERMES_WEBUI_PASSWORD env var takes precedence in
        # api.auth.get_password_hash(), so writing password_hash to settings.json
        # has no effect on auth. Refuse loudly with 409 instead of silently
        # succeeding — the previous behaviour returned 200 + a green save toast
        # while every subsequent login still required the env-var password.
        if requested_password or requested_clear_password:
            if os.getenv("HERMES_WEBUI_PASSWORD", "").strip():
                return bad(
                    handler,
                    "HERMES_WEBUI_PASSWORD env var is set — it overrides the settings password. "
                    "Unset the env var and restart the server before changing the password here.",
                    409,
                )

        max_tokens_provided = "max_tokens" in body
        max_tokens_status = None
        max_tokens_value = body.pop("max_tokens", None) if max_tokens_provided else None

        # First password creation decides who owns a previously passwordless
        # WebUI. While auth is disabled, the generic /api/settings route is also
        # unauthenticated, so gate bootstrap password setup the same way as
        # onboarding setup: local/private networks only, unless the operator
        # explicitly opts into remote bootstrap with HERMES_WEBUI_ONBOARDING_OPEN.
        if requested_password and not auth_enabled_before:
            if not _onboarding_gate_allows(handler, auth_enabled_before):
                return bad(
                    handler,
                    "First password setup is only available from local networks when auth is not enabled. "
                    "To bootstrap this on a remote server, set HERMES_WEBUI_ONBOARDING_OPEN=1.",
                    403,
                )

        # Auth-disable safety: when password auth is currently enabled, require
        # the current password to change, clear, or switch to passwordless.
        if auth_enabled_before and password_auth_enabled_before and (requested_password or requested_clear_password):
            if not isinstance(current_password, str) or not current_password:
                return bad(
                    handler,
                    "Current password is required to change or disable authentication.",
                    403,
                )
            if not verify_password(current_password):
                return bad(
                    handler,
                    "Current password is incorrect.",
                    403,
                )

        if requested_passwordless:
            from api.auth import _passkey_feature_flag_enabled
            from api.passkeys import registered_credentials

            if not _passkey_feature_flag_enabled():
                return bad(handler, "Passkey support is disabled. Enable HERMES_WEBUI_PASSKEY before going passwordless.", 409)
            if not registered_credentials():
                return bad(handler, "Register a passkey before going passwordless.", 409)
        elif requested_clear_password:
            from api.passkeys import clear_credentials

            clear_credentials()

        # Handle auth_disabled_acknowledged setting
        ack = body.pop("_auth_disabled_acknowledged", None)
        if ack is not None and not is_auth_enabled():
            body["auth_disabled_acknowledged"] = bool(ack)
        elif is_auth_enabled() or requested_password:
            body["auth_disabled_acknowledged"] = False

        from api.config import get_max_tokens_status, set_max_tokens

        saved = save_settings(body)
        saved["persisted_speech_keys"] = persisted_speech_settings_keys()
        if max_tokens_provided:
            max_tokens_status = set_max_tokens(max_tokens_value)
        saved.pop("password_hash", None)  # never expose hash to client
        saved.update(max_tokens_status if max_tokens_provided else get_max_tokens_status())

        # Settings that change which sessions appear in the sidebar must
        # invalidate the session-list cache directly. Relying on the cache's
        # settings-file mtime stamp is fragile: a toggle that writes the
        # settings file within the same mtime granularity as a cached entry (and
        # produces the default-valued key, e.g. show_cli_sessions back to its
        # True default) can leave a stale row set served for up to the cache TTL.
        # This is the root cause of the intermittent gateway_sync test flake
        # (a freshly-inserted CLI/gateway session occasionally absent from
        # /api/sessions right after the visibility toggle). Invalidate explicitly.
        if any(
            k in body
            for k in (
                "show_cli_sessions",
                "show_claude_code_sessions",
                "show_cron_sessions",
                "show_webhook_sessions",
                "show_kanban_sessions",
                "show_previous_messaging_sessions",
            )
        ):
            try:
                _clear_session_list_cache()
            except Exception:
                pass
            try:
                from api.models import clear_cli_sessions_cache
                clear_cli_sessions_cache()
            except Exception:
                pass

        auth_enabled_after = is_auth_enabled()
        auth_just_enabled = bool(
            requested_password and auth_enabled_after and not auth_enabled_before
        )
        logged_in_after = logged_in_before
        new_cookie = None

        if auth_just_enabled and not logged_in_before:
            new_cookie = create_session()
            logged_in_after = True

        saved["auth_enabled"] = auth_enabled_after
        saved["password_auth_enabled"] = get_password_hash() is not None
        saved["logged_in"] = logged_in_after
        saved["auth_just_enabled"] = auth_just_enabled
        try:
            from api.auth import _passkey_feature_flag_enabled as _pffe
            from api.passkeys import registered_credentials as _rc
            if _pffe():
                saved["passkeys_enabled"] = bool(_rc())
                saved["passwordless_enabled"] = bool(_rc()) and not saved["password_auth_enabled"]
            else:
                saved["passkeys_enabled"] = False
                saved["passwordless_enabled"] = False
        except Exception:
            pass

        if not new_cookie:
            return j(handler, saved)

        response_body = json.dumps(saved, ensure_ascii=False, indent=2).encode("utf-8")
        handler.send_response(200)
        handler.send_header("Content-Type", "application/json; charset=utf-8")
        handler.send_header("Content-Length", str(len(response_body)))
        handler.send_header("Cache-Control", "no-store")
        set_auth_cookie(handler, new_cookie)
        _security_headers(handler)
        handler.end_headers()
        handler.wfile.write(response_body)
        return True

    if parsed.path == "/api/onboarding/oauth/start":
        if not _onboarding_gate_allows(handler):
            return bad(handler, "Onboarding OAuth is only available from local networks when auth is not enabled. To bypass this on a remote server, set HERMES_WEBUI_ONBOARDING_OPEN=1.", 403)
        try:
            return j(handler, start_onboarding_oauth_flow(body), extra_headers={"Cache-Control": "no-store"})
        except ValueError as e:
            return bad(handler, str(e))
        except RuntimeError as e:
            return bad(handler, str(e), 500)

    if parsed.path == "/api/onboarding/oauth/cancel":
        try:
            return j(handler, cancel_onboarding_oauth_flow(body), extra_headers={"Cache-Control": "no-store"})
        except ValueError as e:
            return bad(handler, str(e))

    if parsed.path == "/api/onboarding/setup":
        # Writing API keys to disk - restrict to local/private networks unless auth is active.
        # In Docker, requests arrive from the bridge network (172.x.x.x), not 127.0.0.1,
        # even when the user accesses via localhost:8787 on the host.
        # Behind a reverse proxy (nginx/Caddy/Traefik) or SSH tunnel, X-Forwarded-For
        # carries the real origin IP — read it first before falling back to the raw socket addr.
        # HERMES_WEBUI_ONBOARDING_OPEN=1 lets operators on remote servers explicitly bypass
        # the check when they control network access themselves (e.g. firewall + VPN).
        if not _onboarding_gate_allows(handler):
            return bad(handler, "Onboarding setup is only available from local networks when auth is not enabled. To bypass this on a remote server, set HERMES_WEBUI_ONBOARDING_OPEN=1.", 403)
        try:
            return j(handler, apply_onboarding_setup(body))
        except ValueError as e:
            return bad(handler, str(e))
        except RuntimeError as e:
            return bad(handler, str(e), 500)

    if parsed.path == "/api/onboarding/complete":
        # Marking onboarding complete flips the first-run wizard off (persists
        # onboarding_completed=True). Gate it on the same local-network check as
        # the other onboarding mutators so an unauthenticated public client on a
        # passwordless bind can't hide the first-run wizard. (#3765)
        if not _onboarding_gate_allows(handler):
            return bad(handler, "Onboarding is only available from local networks when auth is not enabled. To bypass this on a remote server, set HERMES_WEBUI_ONBOARDING_OPEN=1.", 403)
        return j(handler, complete_onboarding())

    if parsed.path == "/api/onboarding/probe":
        # Probe a self-hosted provider endpoint (#1499).  Validates the
        # configured base URL is reachable + parses /models, returns the
        # model catalog so the wizard can populate its dropdown.
        # Read-only: no config.yaml or .env writes happen here.  Same local-
        # network gate as /api/onboarding/setup (also writing-adjacent in
        # spirit because it carries an api_key the user typed).
        if not _onboarding_gate_allows(handler):
            return bad(handler, "Onboarding probe is only available from local networks when auth is not enabled. To bypass this on a remote server, set HERMES_WEBUI_ONBOARDING_OPEN=1.", 403)
        provider = str((body or {}).get("provider") or "").strip().lower()
        base_url = str((body or {}).get("base_url") or "")
        api_key = str((body or {}).get("api_key") or "").strip() or None
        try:
            return j(handler, probe_provider_endpoint(provider, base_url, api_key))
        except Exception as e:
            return bad(handler, f"probe failed: {e}", 500)

    # ── Session pin (POST) ──
    if parsed.path == "/api/session/pin":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if _session_is_subagent_view_only(body["session_id"]):
            return bad(handler, "Subagent sessions are view-only and cannot be modified from WebUI", 400)
        try:
            s = get_session(body["session_id"])
            s = _ensure_full_session_before_mutation(body["session_id"], s)
        except KeyError:
            return bad(handler, "Session not found", 404)
        pin_requested = bool(body.get("pinned", True))
        # TOCTOU guard (Opus stage-389): the count check and the pin write
        # must happen under the same lock, otherwise two parallel pin
        # requests can both pass `len(pinned_ids) >= 3` against the same
        # snapshot and both succeed, leaving the user with 4 pins. The check
        # must be careful not to nest `all_sessions()` (which acquires LOCK
        # internally) inside a `with LOCK:` block — that's a deadlock since
        # LOCK is a non-reentrant `threading.Lock`. We snapshot the
        # persisted index outside the lock, then re-check the in-memory
        # mutation set inside the lock and commit the pin atomically.
        if pin_requested and not getattr(s, "pinned", False):
            # Pre-snapshot from persisted index (acquires LOCK internally,
            # so must run outside our own LOCK acquire below).
            persisted_rows = [
                existing for existing in all_sessions()
                if _session_counts_toward_pin_quota(existing)
            ]
            with LOCK:
                # Final authoritative count: merge persisted pinned rows with the
                # in-memory SESSIONS snapshot. Count logical sidebar-visible pin
                # lineages rather than raw session rows so continuation siblings
                # in the same visible lineage do not consume extra pin quota.
                candidate_rows = list(persisted_rows)
                candidate_rows.extend(
                    existing.compact() for existing in SESSIONS.values()
                    if _session_counts_toward_pin_quota(existing)
                )
                target_row = s.compact()
                candidate_rows.append(target_row)
                pinned_lineage_ids = _visible_pinned_lineage_ids(candidate_rows)
                target_lineage = _session_row_lineage_root_id(
                    target_row,
                    {
                        str(_session_field(row, "session_id", "") or ""): row
                        for row in candidate_rows
                        if _session_field(row, "session_id", None)
                    },
                )
                pinned_lineage_ids.discard(target_lineage)
                pinned_sessions_limit = int(load_settings().get("pinned_sessions_limit", 3) or 3)
                if len(pinned_lineage_ids) >= pinned_sessions_limit:
                    return bad(handler, f"Up to {pinned_sessions_limit} sessions can be pinned. Unpin one before pinning another.", 400)
                # Mark in-memory pin state under LOCK so concurrent pin
                # requests see the increment immediately, even before
                # save() finishes flushing to disk.
                s.pinned = True
            with _get_session_agent_lock(body["session_id"]):
                s.save()
        else:
            with _get_session_agent_lock(body["session_id"]):
                s.pinned = pin_requested
                s.save()
        publish_session_list_changed(
            "session_pin",
            profile=getattr(s, "profile", None),
            session_id=getattr(s, "session_id", body["session_id"]),
        )
        return j(handler, {"ok": True, "session": s.compact()})

    # ── Session archive (POST) ──
    if parsed.path == "/api/session/archive":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        sid = body["session_id"]
        if _session_is_subagent_view_only(sid):
            return bad(handler, "Subagent sessions are view-only and cannot be archived from WebUI", 400)
        try:
            s = get_session(sid)
            # #1558: save() refuses metadata-only session stubs because their
            # messages list is intentionally empty. If a sidebar/status preload
            # left one in the LRU cache, upgrade to a full disk load before
            # mutating archived state so the guard stays intact.
            if getattr(s, "_loaded_metadata_only", False):
                s = Session.load(sid)
                if s is None:
                    raise KeyError(sid)
                with LOCK:
                    SESSIONS[sid] = s
        except KeyError:
            cli_meta = _lookup_cli_session_metadata(sid)
            if not cli_meta:
                return bad(handler, "Session not found", 404)
            if cli_meta.get("read_only"):
                return bad(handler, "Read-only imported sessions cannot be archived from WebUI", 400)
            # Delegated subagent children (#5307) are view-only and owned by the
            # delegate runner — never materialize one into a writable WebUI
            # sidecar via the archive fallback (the 3rd of the shared
            # import_cli_session write paths).
            _arch_source_tag = (cli_meta.get("source_tag") or cli_meta.get("raw_source") or "").strip().lower()
            if _arch_source_tag == "subagent" or _is_subagent_child_session_id(sid):
                return bad(handler, "Subagent sessions cannot be archived from WebUI", 400)
            if _is_messaging_session_record(cli_meta):
                _arch_profile = cli_meta.get("profile") or None
                s = Session(
                    session_id=sid,
                    title=cli_meta.get("title") or title_from(get_cli_session_messages(sid), "CLI Session"),
                    workspace=get_last_workspace(profile=_arch_profile),
                    messages=[],
                    model=cli_meta.get("model") or "unknown",
                    created_at=cli_meta.get("created_at"),
                    updated_at=cli_meta.get("updated_at"),
                    profile=_arch_profile,
                )
                s.is_cli_session = is_cli_session_row(cli_meta)
                s.source_tag = cli_meta.get("source_tag")
                s.raw_source = cli_meta.get("raw_source") or cli_meta.get("source_tag")
                s.session_source = cli_meta.get("session_source")
                s.source_label = cli_meta.get("source_label")
                s.user_id = cli_meta.get("user_id")
                s.chat_id = cli_meta.get("chat_id")
                s.chat_type = cli_meta.get("chat_type")
                s.thread_id = cli_meta.get("thread_id")
                s.session_key = cli_meta.get("session_key")
                s.platform = cli_meta.get("platform")
                s.save(touch_updated_at=False)
            else:
                msgs = get_cli_session_messages(sid)
                if not msgs:
                    return bad(handler, "Session not found", 404)
                s = import_cli_session(
                    sid,
                    cli_meta.get("title") or title_from(msgs, "CLI Session"),
                    msgs,
                    cli_meta.get("model") or "unknown",
                    profile=cli_meta.get("profile"),
                    created_at=cli_meta.get("created_at"),
                    updated_at=cli_meta.get("updated_at"),
                )
                s.is_cli_session = is_cli_session_row(cli_meta)
                s.source_tag = cli_meta.get("source_tag")
                s.raw_source = cli_meta.get("raw_source") or cli_meta.get("source_tag")
                s.session_source = cli_meta.get("session_source")
                s.source_label = cli_meta.get("source_label")
                s.user_id = cli_meta.get("user_id")
                s.chat_id = cli_meta.get("chat_id")
                s.chat_type = cli_meta.get("chat_type")
                s.thread_id = cli_meta.get("thread_id")
                s.session_key = cli_meta.get("session_key")
                s.platform = cli_meta.get("platform")
        with _get_session_agent_lock(sid):
            s.archived = bool(body.get("archived", True))
            s.save(touch_updated_at=False)
        publish_session_list_changed(
            "session_archive",
            profile=getattr(s, "profile", None),
            session_id=getattr(s, "session_id", sid),
        )
        return j(handler, {"ok": True, "session": s.compact(), **_worktree_retained_payload(s)})

    # ── Session move to project (POST) ──
    if parsed.path == "/api/session/move":
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        try:
            s = _get_or_materialize_session(body["session_id"])
        except KeyError:
            return bad(handler, "Session not found", 404)
        except PermissionError:
            return bad(handler, "Read-only imported sessions cannot be moved from WebUI", 403)
        # #1614: refuse moves into a project owned by another profile.
        target_pid = body.get("project_id") or None
        if target_pid:
            # Use the session's own profile for authorization, not the global
            # active profile. A session belongs to a specific profile set at
            # creation; projects from that profile should always be assignable,
            # regardless of which profile is "active" at the process level.
            # Matches the same principle as the profile chip fix — prefer
            # session-scoped state over global active profile. (#3325 follow-up)
            _session_profile = getattr(s, 'profile', None) or get_active_profile_name()
            target = next(
                (p for p in load_projects() if p["project_id"] == target_pid),
                None,
            )
            if not target:
                return bad(handler, "Project not found", 404)
            if not _profiles_match(target.get("profile"), _session_profile):
                return bad(handler, "Project not found", 404)
        # #3746: acquire the per-session agent lock with a bounded timeout
        # instead of blocking indefinitely. The streaming thread holds this same
        # lock during checkpoint saves; on slow file I/O (e.g. WSL/DrvFs) a bare
        # blocking acquire could outlast the client's 30s abort and surface as a
        # silent "Request timed out" toast with no server-side signal. Bounding
        # the wait converts that into an actionable HTTP 503 the client can retry.
        # We keep the lock (rather than dropping it for this metadata-only write)
        # because s.save() still races the streaming thread's atomic writer.
        _move_lock = _get_session_agent_lock(body["session_id"])
        if not _move_lock.acquire(timeout=5):
            return j(
                handler,
                {"error": "Session is busy (streaming). Please try again in a moment."},
                status=503,
            )
        try:
            s.project_id = target_pid
            s.save()
        finally:
            _move_lock.release()
        publish_session_list_changed(
            "session_move",
            profile=getattr(s, "profile", None),
            session_id=getattr(s, "session_id", body["session_id"]),
        )
        return j(handler, {"ok": True, "session": s.compact()})

    # ── Project CRUD (POST) ──
    if parsed.path == "/api/projects/create":
        try:
            require(body, "name")
        except ValueError as e:
            return bad(handler, str(e))
        import re as _re

        name = body["name"].strip()[:128]
        if not name:
            return bad(handler, "name required")
        color = body.get("color")
        if color and not _re.match(r"^#[0-9a-fA-F]{3,8}$", color):
            return bad(handler, "Invalid color format")
        projects = load_projects()
        # #3331 follow-up (Codex+Opus gate): validate the optional client-supplied
        # `profile` before stamping it, mirroring /api/profile/switch — otherwise a
        # client could create a project tagged with an arbitrary/unknown profile,
        # producing hidden cross-profile rows that can't be managed normally.
        _requested_profile = str(body.get('profile') or "").strip()
        if _requested_profile and _requested_profile != "default":
            from api.profiles import _PROFILE_ID_RE
            if not _PROFILE_ID_RE.fullmatch(_requested_profile):
                return bad(handler, "invalid profile")
        proj = {
            "project_id": uuid.uuid4().hex[:12],
            "name": name,
            "color": color,
            "profile": _requested_profile or get_active_profile_name() or 'default',
            "created_at": time.time(),
        }
        projects.append(proj)
        save_projects(projects)
        return j(handler, {"ok": True, "project": proj})

    if parsed.path == "/api/projects/rename":
        try:
            require(body, "project_id", "name")
        except ValueError as e:
            return bad(handler, str(e))
        import re as _re

        projects = load_projects()
        proj = next(
            (p for p in projects if p["project_id"] == body["project_id"]), None
        )
        if not proj:
            return bad(handler, "Project not found", 404)
        # #1614: a project can only be renamed by the profile that owns it.
        active_profile = get_active_profile_name()
        if not _profiles_match(proj.get("profile"), active_profile):
            return bad(handler, "Project not found", 404)
        proj["name"] = body["name"].strip()[:128]
        if "color" in body:
            color = body["color"]
            if color and not _re.match(r"^#[0-9a-fA-F]{3,8}$", color):
                return bad(handler, "Invalid color format")
            proj["color"] = color
        save_projects(projects)
        return j(handler, {"ok": True, "project": proj})

    if parsed.path == "/api/projects/delete":
        try:
            require(body, "project_id")
        except ValueError as e:
            return bad(handler, str(e))
        projects = load_projects()
        proj = next(
            (p for p in projects if p["project_id"] == body["project_id"]), None
        )
        if not proj:
            return bad(handler, "Project not found", 404)
        # #1614: a project can only be deleted by the profile that owns it.
        active_profile = get_active_profile_name()
        if not _profiles_match(proj.get("profile"), active_profile):
            return bad(handler, "Project not found", 404)
        projects = [p for p in projects if p["project_id"] != body["project_id"]]
        save_projects(projects)
        # Unassign all sessions that belonged to this project.
        # #3746: this loop is O(N) full-JSON read+save per session, and each
        # save() reserializes the entire messages array. For a project with many
        # messageful sessions that throughput alone can blow past the client's
        # 30s timeout. For an actively-streaming session we must NOT issue our own
        # s.save() — it would race the streaming thread's atomic writer and it
        # carries the largest in-memory message array. Instead we clear project_id
        # on the live cached Session object (under LOCK); the streaming thread owns
        # that object and persists it on its next checkpoint/final save (the worker
        # always does a final s.save() at turn completion), so the unlink still
        # lands without a competing write. (If the streaming session isn't in the
        # cache for some reason, fall back to a direct save.) Guard each per-session
        # update so one slow/failing session can't abort the whole request.
        if SESSION_INDEX_FILE.exists():
            try:
                index = json.loads(SESSION_INDEX_FILE.read_bytes())
                active_ids = _active_stream_ids()
                deferred_to_stream = []
                for entry in index:
                    if entry.get("project_id") != body["project_id"]:
                        continue
                    sid = entry.get("session_id")
                    try:
                        if entry.get("active_stream_id") in active_ids:
                            # Clear on the live cached object so the streaming
                            # thread's own next save persists project_id=None.
                            cleared_in_cache = False
                            with LOCK:
                                cached = SESSIONS.get(sid)
                                if cached is not None:
                                    cached.project_id = None
                                    cleared_in_cache = True
                            if cleared_in_cache:
                                deferred_to_stream.append(sid)
                                continue
                            # Not cached — fall through to a direct save.
                        s = get_session(sid)
                        s.project_id = None
                        s.save()
                    except Exception:
                        logger.debug("Failed to update session %s", sid)
                if deferred_to_stream:
                    logger.info(
                        "projects/delete: cleared project_id on %d streaming session(s) "
                        "in-cache; streaming thread will persist: %s",
                        len(deferred_to_stream), deferred_to_stream,
                    )
            except Exception:
                logger.debug("Failed to load session index for project unlink")
        return j(handler, {"ok": True})

    # ── Session import from JSON (POST) ──
    if parsed.path == "/api/session/import":
        return _handle_session_import(handler, body)

    # ── Self-update (POST) ──
    if parsed.path == "/api/updates/apply":
        target = body.get("target", "")
        if target not in ("webui", "agent"):
            return bad(handler, 'target must be "webui" or "agent"')
        # Honor an explicit validated body channel (the client sends the channel
        # the banner was offering) so a channel switch whose debounced autosave
        # hasn't landed can't make apply read the OLD saved channel (Codex gate).
        # Fall back to the saved setting when absent/invalid.
        _apply_channel = body.get("channel") if isinstance(body, dict) else None
        if _apply_channel not in ("stable", "experimental"):
            _apply_channel = None
        from api.updates import apply_update

        return j(handler, apply_update(target, _apply_channel))

    if parsed.path == "/api/updates/force":
        target = body.get("target", "")
        if target not in ("webui", "agent"):
            return bad(handler, 'target must be "webui" or "agent"')
        _force_channel = body.get("channel") if isinstance(body, dict) else None
        if _force_channel not in ("stable", "experimental"):
            _force_channel = None
        from api.updates import apply_force_update

        return j(handler, apply_force_update(target, _force_channel))

    if parsed.path == "/api/updates/clear_lock":
        # Manual-instruction recovery for the .git/index.lock case. The
        # endpoint NEVER removes a lock file from the server -- it returns
        # the diagnostic + the exact 'rm' command for the operator, and on
        # a re-click with the lock already gone, it re-runs the normal
        # non-destructive apply path. See apply_clear_lock for the v2.2
        # design rationale (round-2 gate cert: fcntl-flock cannot detect
        # git's O_CREAT|O_EXCL locks, so any auto-delete path races).
        target = body.get("target", "")
        if target not in ("webui", "agent"):
            return bad(handler, 'target must be "webui" or "agent"')
        from api.updates import apply_clear_lock

        return j(handler, apply_clear_lock(target))

    if parsed.path == "/api/updates/summary":
        from api.updates import summarize_update_payload

        updates = body.get("updates") if isinstance(body, dict) else {}
        target = body.get("target") if isinstance(body, dict) else None

        return j(handler, summarize_update_payload(updates, llm_callback=_llm_update_summary, target=target))

    # ── CLI session import (POST) ──
    if parsed.path == "/api/session/import_cli":
        return _handle_session_import_cli(handler, body)

    # ── Auth endpoints (POST) ──
    if parsed.path == "/api/auth/login":
        from api.auth import (
            verify_password,
            create_session,
            set_auth_cookie,
            is_auth_enabled,
        )
        from api.auth import _check_login_rate, _record_login_attempt, _clear_login_attempts

        if not is_auth_enabled():
            return j(handler, {"ok": True, "message": "Auth not enabled"})
        client_ip = handler.client_address[0]
        if not _check_login_rate(client_ip):
            return j(
                handler,
                {"error": "Too many attempts. Try again in a minute."},
                status=429,
            )
        password = body.get("password", "")
        if not verify_password(password):
            _record_login_attempt(client_ip)
            return bad(handler, "Invalid password", 401)
        _clear_login_attempts(client_ip)
        cookie_val = create_session()
        body = json.dumps({"ok": True}).encode()
        handler.send_response(200)
        handler.send_header("Content-Type", "application/json")
        handler.send_header("Content-Length", str(len(body)))
        handler.send_header("Cache-Control", "no-store")
        _security_headers(handler)
        set_auth_cookie(handler, cookie_val)
        handler.end_headers()
        handler.wfile.write(body)
        return True

    if parsed.path == "/api/auth/passkey/options":
        from api.auth import _passkey_feature_flag_enabled, is_auth_enabled
        from api.passkeys import PasskeyError, PasskeyRateLimitError, authentication_options

        if not _passkey_feature_flag_enabled():
            return j(handler, {"error": "Passkey support is disabled. Set HERMES_WEBUI_PASSKEY=1 or webui_passkey_enabled: true to enable."}, status=404)
        if not is_auth_enabled():
            return j(handler, {"error": "Auth not enabled"}, status=400)
        try:
            return j(handler, {"ok": True, "publicKey": authentication_options(handler)})
        except PasskeyRateLimitError as e:
            return bad(handler, str(e), status=429)
        except PasskeyError as e:
            return bad(handler, str(e), status=400)

    if parsed.path == "/api/auth/passkey/login":
        from api.auth import _passkey_feature_flag_enabled, create_session, is_auth_enabled, set_auth_cookie
        from api.auth import _check_login_rate, _record_login_attempt
        from api.passkeys import PasskeyError, finish_login

        if not _passkey_feature_flag_enabled():
            return j(handler, {"error": "Passkey support is disabled."}, status=404)
        if not is_auth_enabled():
            return j(handler, {"error": "Auth not enabled"}, status=400)
        client_ip = handler.client_address[0]
        if not _check_login_rate(client_ip):
            return j(handler, {"error": "Too many attempts. Try again in a minute."}, status=429)
        try:
            finish_login(body, handler)
        except PasskeyError as e:
            _record_login_attempt(client_ip)
            return bad(handler, str(e), status=401)
        cookie_val = create_session()
        body = json.dumps({"ok": True}).encode()
        handler.send_response(200)
        handler.send_header("Content-Type", "application/json")
        handler.send_header("Content-Length", str(len(body)))
        handler.send_header("Cache-Control", "no-store")
        _security_headers(handler)
        set_auth_cookie(handler, cookie_val)
        handler.end_headers()
        handler.wfile.write(body)
        return True

    if parsed.path == "/api/auth/passkey/register/options":
        from api.auth import _passkey_feature_flag_enabled
        from api.passkeys import PasskeyError, PasskeyRateLimitError, registration_options

        if not _passkey_feature_flag_enabled():
            return j(handler, {"error": "Passkey support is disabled."}, status=404)
        ok, error, status = _require_passkey_registration_auth(handler)
        if not ok:
            return j(handler, {"error": error}, status=status)
        try:
            return j(handler, {"ok": True, "publicKey": registration_options(handler)})
        except PasskeyRateLimitError as e:
            return bad(handler, str(e), status=429)
        except PasskeyError as e:
            return bad(handler, str(e), status=400)

    if parsed.path == "/api/auth/passkey/register":
        from api.auth import _passkey_feature_flag_enabled
        from api.passkeys import PasskeyError, finish_registration, registered_credentials

        if not _passkey_feature_flag_enabled():
            return j(handler, {"error": "Passkey support is disabled."}, status=404)
        ok, error, status = _require_passkey_registration_auth(handler)
        if not ok:
            return j(handler, {"error": error}, status=status)
        try:
            result = finish_registration(body, handler)
            result["credentials"] = registered_credentials()
            return j(handler, result)
        except PasskeyError as e:
            return bad(handler, str(e), status=400)

    if parsed.path == "/api/auth/passkey/delete":
        from api.auth import _passkey_feature_flag_enabled, get_password_hash
        from api.passkeys import PasskeyError, delete_credential, registered_credentials

        if not _passkey_feature_flag_enabled():
            return j(handler, {"error": "Passkey support is disabled."}, status=404)
        try:
            credential_id = str(body.get("id") or "")
            creds = registered_credentials()
            if get_password_hash() is None and len(creds) <= 1 and any(c.get("id") == credential_id for c in creds):
                return bad(handler, "Set a password or disable auth before removing the last passkey.", 409)
            return j(handler, delete_credential(credential_id))
        except PasskeyError as e:
            return bad(handler, str(e), status=404)

    if parsed.path == "/api/auth/passkeys":
        from api.auth import _passkey_feature_flag_enabled
        from api.passkeys import registered_credentials

        if not _passkey_feature_flag_enabled():
            return j(handler, {"credentials": [], "disabled": True})
        return j(handler, {"credentials": registered_credentials()})

    if parsed.path == "/api/auth/logout":
        from api.auth import clear_auth_cookie, ensure_trusted_auth_session, get_trusted_auth_logout_url, invalidate_session, parse_cookie
        from api.helpers import clear_profile_cookie

        session_info = ensure_trusted_auth_session(handler)
        cookie_val = getattr(handler, '_trusted_auth_session_cookie_value', None) or parse_cookie(handler)
        if cookie_val:
            invalidate_session(cookie_val)
        payload = {"ok": True}
        if session_info and session_info.get("auth_type") == "trusted":
            logout_url = get_trusted_auth_logout_url()
            if logout_url:
                payload["trusted_logout_url"] = logout_url
        body = json.dumps(payload).encode()
        handler.send_response(200)
        handler.send_header("Content-Type", "application/json")
        handler.send_header("Content-Length", str(len(body)))
        handler.send_header("Cache-Control", "no-store")
        _security_headers(handler)
        clear_auth_cookie(handler)
        clear_profile_cookie(handler)
        handler.end_headers()
        handler.wfile.write(body)
        return True

    # ── Checkpoints / Rollback (POST) ──
    if parsed.path == "/api/rollback/restore":
        if not body:
            return bad(handler, "request body is required")
        workspace = body.get("workspace", "")
        checkpoint = body.get("checkpoint", "")
        if not workspace or not checkpoint:
            return bad(handler, "workspace and checkpoint are required")
        try:
            from api.rollback import restore_checkpoint
            return j(handler, restore_checkpoint(workspace, checkpoint))
        except ValueError as e:
            return bad(handler, str(e))
        except Exception as e:
            logger.exception("rollback/restore failed")
            return bad(handler, str(e), status=500)

    return False  # 404


def handle_patch(handler, parsed) -> bool:
    """Handle all PATCH routes. Returns True if handled, False for 404."""
    if not _check_csrf(handler):
        return j(handler, {"error": _csrf_rejection_error(handler)}, status=403)
    proxy_result = _handle_extension_sidecar_proxy(
        handler,
        parsed,
        "PATCH",
        read_request_body=True,
    )
    if proxy_result is not False:
        return proxy_result
    try:
        body = read_body(handler)
    except ValueError as exc:
        status = 413 if "too large" in str(exc).lower() else 400
        return bad(handler, str(exc), status=status)
    if not _guard_request_session_visibility(handler, parsed, body=body, method="PATCH"):
        return True
    if parsed.path.startswith("/api/mcp/servers/"):
        name = parsed.path[len("/api/mcp/servers/"):]
        return _handle_mcp_server_toggle(handler, name, body)
    if parsed.path.startswith("/api/kanban/"):
        from api.kanban_bridge import handle_kanban_patch

        result = handle_kanban_patch(handler, parsed, body)
        if result is False:
            return _kanban_unknown_endpoint(handler, parsed, "PATCH")
        return True
    return False


def handle_delete(handler, parsed) -> bool:
    """Handle all DELETE routes. Returns True if handled, False for 404."""
    if not _check_csrf(handler):
        return j(handler, {"error": _csrf_rejection_error(handler)}, status=403)
    proxy_result = _handle_extension_sidecar_proxy(
        handler,
        parsed,
        "DELETE",
        read_request_body=True,
    )
    if proxy_result is not False:
        return proxy_result
    try:
        body = read_body(handler)
    except ValueError as exc:
        status = 413 if "too large" in str(exc).lower() else 400
        return bad(handler, str(exc), status=status)
    if not _guard_request_session_visibility(handler, parsed, body=body, method="DELETE"):
        return True
    if parsed.path.startswith("/api/mcp/servers/"):
        name = parsed.path[len("/api/mcp/servers/"):]
        return _handle_mcp_server_delete(handler, name)
    if parsed.path == "/api/prompts":
        pid = str(body.get("id") or "").strip()
        if not pid:
            return bad(handler, "id is required")
        prompts = [p for p in _load_saved_prompts() if p.get("id") != pid]
        _save_saved_prompts(prompts)
        return j(handler, {"ok": True})

    if parsed.path.startswith("/api/kanban/"):
        from api.kanban_bridge import handle_kanban_delete

        result = handle_kanban_delete(handler, parsed, body)
        if result is False:
            return _kanban_unknown_endpoint(handler, parsed, "DELETE")
        return True
    return False


def handle_put(handler, parsed) -> bool:
    """Handle all PUT routes. Returns True if handled, False for 404."""
    if not _check_csrf(handler):
        return j(handler, {"error": "Cross-origin request rejected"}, status=403)
    proxy_result = _handle_extension_sidecar_proxy(
        handler,
        parsed,
        "PUT",
        read_request_body=True,
    )
    if proxy_result is not False:
        return proxy_result
    try:
        body = read_body(handler)
    except ValueError as exc:
        status = 413 if "too large" in str(exc).lower() else 400
        return bad(handler, str(exc), status=status)
    if not _guard_request_session_visibility(handler, parsed, body=body, method="PUT"):
        return True
    if parsed.path.startswith("/api/mcp/servers/"):
        name = parsed.path[len("/api/mcp/servers/"):]
        return _handle_mcp_server_update(handler, name, body)
    return False

# ── GET route helpers ─────────────────────────────────────────────────────────

# MIME types for static file serving. Hoisted to module scope to avoid
# rebuilding the dict on every request.
_STATIC_MIME = {
    "css": "text/css",
    "js": "application/javascript",
    "html": "text/html",
    "svg": "image/svg+xml",
    "png": "image/png",
    "jpg": "image/jpeg",
    "jpeg": "image/jpeg",
    "ico": "image/x-icon",
    "gif": "image/gif",
    "webp": "image/webp",
    "woff": "font/woff",
    "woff2": "font/woff2",
    # Python's built-in MIME table does not include APK, and platform MIME
    # databases are not consistent across Linux, macOS, and Windows.
    "apk": "application/vnd.android.package-archive",
}
# MIME types that are text-based and should carry charset=utf-8
_TEXT_MIME_TYPES = {"text/css", "application/javascript", "text/html", "image/svg+xml", "text/plain"}

# MIME types worth gzipping. Image and font formats (png/jpg/webp/woff2) are
# already compressed; gzip would only add CPU and a few bytes of framing.
_COMPRESSIBLE_MIME = {
    "text/css", "application/javascript", "text/html", "image/svg+xml",
    "application/json", "text/plain",
}

# In-process cache for raw bytes, compressed bytes, and ETag. The cache is keyed
# by absolute path and invalidated on (size, high-precision mtime) change, so a
# redeploy is picked up without a process restart. Missing/random paths never
# enter the cache; memory cost is bounded by the static/ tree's served files.
_STATIC_CACHE: dict = {}
_STATIC_CACHE_LOCK = threading.Lock()


def _serve_static(handler, parsed):
    static_root = api_config.get_static_root().resolve()
    # Strip the leading '/static/' prefix, then resolve and sandbox
    rel = parsed.path[len("/static/") :]
    static_file = (static_root / rel).resolve()
    try:
        static_file.relative_to(static_root)
    except ValueError:
        return j(handler, {"error": "not found"}, status=404)
    if not static_file.exists() or not static_file.is_file():
        return j(handler, {"error": "not found"}, status=404)
    ext = static_file.suffix.lower()
    ct = _STATIC_MIME.get(ext.lstrip("."))
    if ct is None:
        guessed_type, content_encoding = mimetypes.guess_type(static_file.name)
        # Encoded suffixes (for example .svgz/.tgz) need Content-Encoding
        # semantics this route does not implement. Fail closed instead of
        # advertising the decoded media type for still-compressed bytes.
        ct = guessed_type if guessed_type and not content_encoding else "application/octet-stream"
    ct_header = f"{ct}; charset=utf-8" if ct in _TEXT_MIME_TYPES else ct

    # Look up or populate the per-file cache (raw, optional gzip, ETag).
    # Keyed by absolute path; invalidated by (size, nanosecond mtime).
    st = static_file.stat()
    sig = (st.st_size, st.st_mtime_ns)
    cache_key = str(static_file)
    raw = gz = etag = None
    with _STATIC_CACHE_LOCK:
        cached = _STATIC_CACHE.get(cache_key)
        if cached and cached[0] == sig:
            _, raw, gz, etag = cached
    if raw is None:
        raw = static_file.read_bytes()
        # Weak ETag: equality semantics, derived from filesystem identity.
        etag = f'W/"{sig[0]:x}-{sig[1]:x}"'
        gz = (gzip.compress(raw, compresslevel=6)
              if ct in _COMPRESSIBLE_MIME and len(raw) > 1024
              else None)
        with _STATIC_CACHE_LOCK:
            _STATIC_CACHE[cache_key] = (sig, raw, gz, etag)

    # The page template substitutes __WEBUI_VERSION__ at request time (see the
    # `/`/`/index.html`/`/session/` branch above), and static/sw.js's
    # SHELL_ASSETS list relies on the same convention. So a fingerprinted URL
    # is safe to cache aggressively: any redeploy changes the URL.
    version_values = parse_qs(parsed.query, keep_blank_values=True).get("v", [""])
    has_fingerprint = bool(version_values[0])
    cache_control = (
        "public, max-age=31536000, immutable" if has_fingerprint
        else "public, max-age=300"
    )

    # 304 short-circuit on conditional GET.
    if handler.headers.get("If-None-Match") == etag:
        handler.send_response(304)
        handler.send_header("ETag", etag)
        handler.send_header("Cache-Control", cache_control)
        if gz is not None:
            handler.send_header("Vary", "Accept-Encoding")
        handler.end_headers()
        return True

    accept_enc = (handler.headers.get("Accept-Encoding") or "").lower()
    use_gzip = gz is not None and "gzip" in accept_enc
    body = gz if use_gzip else raw

    handler.send_response(200)
    handler.send_header("Content-Type", ct_header)
    handler.send_header("Content-Length", str(len(body)))
    handler.send_header("ETag", etag)
    handler.send_header("Cache-Control", cache_control)
    if gz is not None:
        handler.send_header("Vary", "Accept-Encoding")
    if use_gzip:
        handler.send_header("Content-Encoding", "gzip")
    handler.end_headers()
    handler.wfile.write(body)
    return True


def _handle_session_export(handler, parsed):
    sid = parse_qs(parsed.query).get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    try:
        s = get_session(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    active_profile = get_active_profile_name()
    if not _profiles_match(getattr(s, "profile", None), active_profile):
        return bad(handler, "Session not found", 404)
    # ``public_session_projection`` supersedes the narrower
    # ``redact_session_data`` path so export context_messages uses the same
    # alias-stripping boundary as the visible transcript.
    safe = public_session_projection(s.__dict__)
    qs = parse_qs(parsed.query)
    fmt = qs.get("format", ["json"])[0].lower()
    if fmt == "html":
        from api.session_export_html import render_session_html
        theme = qs.get("theme", ["dark"])[0].lower()
        palette: dict | None = None
        raw_palette = qs.get("palette", [""])[0]
        if raw_palette:
            try:
                import base64 as _b64
                decoded = _b64.b64decode(raw_palette, validate=False).decode("utf-8")
                parsed_palette = json.loads(decoded)
                if isinstance(parsed_palette, dict):
                    # Cap payload so a hostile client can't blow up the response.
                    if len(parsed_palette) <= 64:
                        palette = parsed_palette
            except Exception:
                palette = None
        payload = render_session_html(safe, theme=theme, palette=palette)
        content_type = "text/html; charset=utf-8"
        ext = "html"
    else:
        payload = json.dumps(safe, ensure_ascii=False, indent=2)
        content_type = "application/json; charset=utf-8"
        ext = "json"
    handler.send_response(200)
    handler.send_header("Content-Type", content_type)
    handler.send_header(
        "Content-Disposition", f'attachment; filename="hermes-{sid}.{ext}"'
    )
    handler.send_header("Content-Length", str(len(payload.encode("utf-8"))))
    handler.send_header("Cache-Control", "no-store")
    handler.end_headers()
    handler.wfile.write(payload.encode("utf-8"))
    return True


def _session_search_message_text(message):
    content = message.get("content") if isinstance(message, dict) else ""
    if isinstance(content, list):
        return " ".join(
            str(part.get("text", ""))
            for part in content
            if isinstance(part, dict) and part.get("type") == "text"
        )
    return str(content or "")


def _session_search_preview(text, query, max_len=124):
    normalized = re.sub(r"\s+", " ", str(text or "")).strip()
    q = re.sub(r"\s+", " ", str(query or "")).strip()
    if not normalized or not q:
        return ""
    idx = normalized.lower().find(q.lower())
    if idx < 0:
        return ""

    max_len = max(32, int(max_len or 124))
    if len(normalized) <= max_len:
        return normalized

    context = max(12, (max_len - len(q)) // 2)
    start = max(0, idx - context)
    end = min(len(normalized), idx + len(q) + context)
    if start > 0:
        while start < idx and normalized[start] != " ":
            start += 1
        if start >= idx:
            start = max(0, idx - context)
    if end < len(normalized):
        while end > idx + len(q) and normalized[end - 1] != " ":
            end -= 1
        if end <= idx + len(q):
            end = min(len(normalized), idx + len(q) + context)
    excerpt = normalized[start:end].strip()
    if start > 0:
        excerpt = "..." + excerpt
    if end < len(normalized):
        excerpt = excerpt + "..."
    return excerpt


def _handle_sessions_search(handler, parsed):
    qs = parse_qs(parsed.query)
    q = qs.get("q", [""])[0].lower().strip()
    content_search = qs.get("content", ["1"])[0] == "1"
    from api.profiles import get_active_profile_name
    active_profile = get_active_profile_name()
    all_profiles = _all_profiles_enabled(parsed)
    sessions = all_sessions()
    if not all_profiles:
        sessions = [
            s for s in sessions
            if _profiles_match(s.get("profile"), active_profile)
        ]
    # Reject a malformed depth instead of letting int() raise ValueError and
    # surface as a confusing 500. Clamp to >= 0 so a negative value can't reach
    # the messages[:depth] slice below — messages[:-n] would silently exclude
    # the most recent messages from the content search instead of capping it.
    # (depth == 0 keeps its existing meaning: search the full transcript.)
    try:
        depth = max(0, int(qs.get("depth", ["5"])[0]))
    except (ValueError, TypeError):
        depth = 5
    # Read the redaction setting ONCE for the whole response (mirrors the
    # /api/sessions read-once optimization, #4662) and thread it through every
    # branch + the shared title-field redactor so search rows redact the same
    # fields as the sidebar list.
    try:
        _search_redact_enabled = bool(load_settings().get("api_redact_enabled", True))
    except Exception:
        _search_redact_enabled = True  # fail safe: redact when settings unreadable
    if not q:
        safe_sessions = []
        for s in sessions:
            item = dict(s)
            if isinstance(item.get("title"), str):
                item["title"] = _redact_text(item["title"], _enabled=_search_redact_enabled)
            _redact_sidebar_title_fields(item, _search_redact_enabled)
            safe_sessions.append(item)
        return j(handler, {
            "sessions": safe_sessions,
            "all_profiles": all_profiles,
            "active_profile": active_profile,
        })
    results = []
    for s in sessions:
        title_match = q in (s.get("title") or "").lower()
        if title_match:
            item = dict(s, match_type="title")
            if isinstance(item.get("title"), str):
                item["title"] = _redact_text(item["title"], _enabled=_search_redact_enabled)
            _redact_sidebar_title_fields(item, _search_redact_enabled)
            results.append(item)
            continue
        if content_search:
            try:
                # Scan accessor, not get_session(): a content search walks every
                # session, and routing that through the LRU would evict the
                # user's working set on every keystroke-debounced search.
                sess = get_session_for_scan(s["session_id"])
                if sess is None:
                    continue
                msgs = sess.messages[:depth] if depth else sess.messages
                for m in msgs:
                    c = _session_search_message_text(m)
                    if q in str(c).lower():
                        item = dict(s, match_type="content")
                        preview = _session_search_preview(c, q)
                        if preview:
                            item["match_preview"] = _redact_text(preview, _enabled=_search_redact_enabled)
                        if isinstance(item.get("title"), str):
                            item["title"] = _redact_text(item["title"], _enabled=_search_redact_enabled)
                        _redact_sidebar_title_fields(item, _search_redact_enabled)
                        results.append(item)
                        break
            except (KeyError, Exception):
                pass
    return j(handler, {
        "sessions": results,
        "query": q,
        "count": len(results),
        "all_profiles": all_profiles,
        "active_profile": active_profile,
    })


def _handle_list_dir(handler, parsed):
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    webui_session = None
    try:
        s = get_session(sid)
        webui_session = s
        workspace = s.workspace
    except KeyError:
        # Fallback for CLI sessions not loaded in WebUI memory
        try:
            cli_meta = None
            for cs in get_cli_sessions():
                if cs["session_id"] == sid:
                    cli_meta = cs
                    break
            if not cli_meta:
                return bad(handler, "Session not found", 404)
            workspace = cli_meta.get("workspace", "")
        except Exception:
            return bad(handler, "Session not found", 404)
    try:
        _list_profile = getattr(webui_session, "profile", None)
        if webui_session is None:
            try:
                workspace = resolve_trusted_workspace(workspace, profile=_list_profile)
            except TypeError:
                workspace = resolve_trusted_workspace(workspace)
            recovered = False
        else:
            stored_workspace = workspace
            try:
                workspace, recovered = resolve_implicit_workspace_with_recovery(
                    stored_workspace,
                    get_last_workspace,
                    profile=_list_profile,
                )
            except TypeError:
                workspace, recovered = resolve_implicit_workspace_with_recovery(
                    stored_workspace,
                    get_last_workspace,
                )
            if recovered:
                persisted = persist_recovered_workspace_binding(
                    webui_session,
                    workspace,
                    expected_workspace=stored_workspace,
                )
                workspace = Path(persisted.workspace)
        rel_path = qs.get("path", ["."])[0]
        entries = list_dir(Path(workspace), rel_path)
        return j(
            handler,
            {
                "entries": serialize_workspace_entries_for_browser(entries),
                "signature": dir_signature(Path(workspace), rel_path, entries),
                "path": rel_path,
                "workspace": str(workspace),
                "workspace_recovered": recovered,
            },
        )
    except WorkspaceBindingPersistenceError as e:
        return bad(handler, _sanitize_error(e), 500)
    except (FileNotFoundError, ValueError) as e:
        return bad(handler, _sanitize_error(e), 404)


def _read_json_request_body(handler, *, max_bytes: int = 4096) -> dict:
    try:
        length = _safe_content_length(handler, max_bytes)
    except (ValueError, OverflowError) as exc:
        raise ValueError(_sanitize_error(exc)) from exc
    raw = handler.rfile.read(length) if length else b"{}"
    try:
        payload = json.loads(raw.decode("utf-8"))
    except Exception as exc:
        raise ValueError("invalid JSON body") from exc
    return payload if isinstance(payload, dict) else {}


def _handle_escape_authorize(handler, parsed, body: dict | None = None):
    if handler.command != "POST":
        return bad(handler, "method not allowed", 405)
    if not handler.headers.get("Origin"):
        return bad(handler, "browser origin required", 403)
    if not _check_csrf(handler):
        return bad(handler, _csrf_rejection_error(handler), 403)
    if body is None:
        try:
            body = _read_json_request_body(handler)
        except ValueError as exc:
            return bad(handler, _sanitize_error(exc), 400)
    qs = parse_qs(parsed.query)
    sid = str(body.get("session_id") or qs.get("session_id", [""])[0] or "").strip()
    rel = str(body.get("path") or qs.get("path", [""])[0] or "").strip()
    token = str(body.get("token") or qs.get("token", [""])[0] or "").strip()
    if token:
        return bad(handler, "token must not be provided", 400)
    if not sid:
        return bad(handler, "session_id is required")
    if not rel:
        return bad(handler, "path is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        payload = authorize_escape_target(Path(s.workspace), sid, rel)
    except ValueError as exc:
        return bad(handler, _sanitize_error(exc), 404)
    return j(handler, payload)


def _handle_escape_list_dir(handler, parsed):
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    token = qs.get("token", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    if not token:
        return bad(handler, "token is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    rel_path = qs.get("path", ["."])[0]
    try:
        payload = list_authorized_escape_dir(Path(s.workspace), sid, token, rel_path)
        payload["entries"] = serialize_workspace_entries_for_browser(payload.get("entries"))
        return j(handler, payload)
    except FileNotFoundError as exc:
        return bad(handler, _sanitize_error(exc), 404)
    except EscapeAuthorizationExpiredError as exc:
        return bad(handler, _sanitize_error(exc), 403)
    except ValueError as exc:
        return bad(handler, _sanitize_error(exc), 404)


def _handle_escape_file_read(handler, parsed):
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    token = qs.get("token", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    if not token:
        return bad(handler, "token is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    rel = qs.get("path", [""])[0]
    try:
        return j(handler, read_authorized_escape_file_content(Path(s.workspace), sid, token, rel))
    except FileNotFoundError as exc:
        return bad(handler, _sanitize_error(exc), 404)
    except EscapeAuthorizationExpiredError as exc:
        return bad(handler, _sanitize_error(exc), 403)
    except ImportError as exc:
        # Optional Office parsers absent on a lean install — mirror
        # _handle_file_read: a 503 with the install hint, not a 500 traceback.
        return bad(handler, _sanitize_error(exc), 503)
    except ValueError as exc:
        return bad(handler, _sanitize_error(exc), 404)


def _handle_escape_file_raw(handler, parsed):
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    token = qs.get("token", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    if not token:
        return bad(handler, "token is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    rel = qs.get("path", [""])[0]
    force_download = qs.get("download", [""])[0] == "1"
    try:
        anchor_root, target = raw_authorized_escape_target(Path(s.workspace), sid, token, rel)
    except FileNotFoundError:
        return j(handler, {"error": "not found"}, status=404)
    except EscapeAuthorizationExpiredError as exc:
        return bad(handler, _sanitize_error(exc), 403)
    except ValueError as exc:
        return bad(handler, _sanitize_error(exc), 404)
    if not target.exists() or not target.is_file():
        return j(handler, {"error": "not found"}, status=404)
    ext = target.suffix.lower()
    mime = MIME_MAP.get(ext, "application/octet-stream")
    inline_preview = qs.get("inline", [""])[0] == "1"
    dangerous_types = {"text/html", "application/xhtml+xml", "image/svg+xml"}
    html_inline_ok = inline_preview and mime == "text/html"
    disposition = "attachment" if force_download or (mime in dangerous_types and not html_inline_ok) else "inline"
    sandbox_csp = "sandbox allow-scripts allow-popups allow-popups-to-escape-sandbox"
    # Content-Security-Policy sandboxing is carried through the csp=sandbox_csp handoff below.
    csp = sandbox_csp if (inline_preview and not force_download and disposition == "inline") else None
    if html_inline_ok:
        return _serve_inline_html_preview(handler, target, "no-store", csp=sandbox_csp, anchor_root=anchor_root)
    return _serve_file_bytes(handler, target, mime, disposition, "no-store", csp=csp, anchor_root=anchor_root)


def _sse_with_id(handler, event, data, event_id=None):
    if event_id:
        handler.wfile.write(f"id: {event_id}\n".encode("utf-8"))
    _sse(handler, event, data)


def _session_events_path_session_id(path: str | None) -> str | None:
    path = str(path or "")
    parts = path.strip("/").split("/")
    if len(parts) != 4:
        return None
    if parts[0] != "api" or parts[1] != "sessions" or parts[3] != "events":
        return None
    sid = str(parts[2] or "").strip()
    return sid or None


def _session_events_resume_event_id(handler, parsed) -> str | None:
    headers = getattr(handler, "headers", None)
    raw = None
    if headers is not None:
        try:
            raw = headers.get("Last-Event-ID")
        except Exception:
            raw = None
    raw = str(raw or "").strip()
    if raw:
        return raw
    qs = parse_qs(getattr(parsed, "query", "") or "")
    raw = str(qs.get("after_event_id", [None])[0] or "").strip()
    return raw or None


def _session_snapshot_payload(session, *, active_stream_id: str | None = None) -> dict:
    try:
        payload = session.compact(
            include_runtime=bool(active_stream_id),
            active_stream_ids={active_stream_id} if active_stream_id else None,
        )
    except Exception:
        payload = {"session_id": str(getattr(session, "session_id", "") or "")}
    return {"session": payload}


def _parse_run_journal_event_id(raw: str | None) -> tuple[str | None, int | None]:
    return _shared_parse_run_journal_event_id(raw)


def _chat_stream_resume_cursor(handler, qs: dict, stream_id: str | None = None) -> tuple[int | None, bool, str | None, str | None]:
    """Resolve the client's resume cursor for ``/api/chat/stream``.

    Returns ``(after_seq, resume_requested, raw_cursor, runner_cursor)``:

    - ``after_seq``: the parsed same-run cursor seq, or ``None`` when there is
      no usable same-run (journal-shaped) cursor.
    - ``resume_requested``: True when the client SUPPLIED any cursor — via the
      ``after_event_id`` / ``after_seq`` query params, ``replay=1``, or the
      ``Last-Event-ID`` header — regardless of whether it parsed.
    - ``raw_cursor``: the opaque cursor string exactly as the client supplied it
      (the ``Last-Event-ID`` value, or the explicit ``after_event_id``).
    - ``runner_cursor``: the cursor to hand to the runner observe path, resolved
      with PROVENANCE so the runner adapter (whose cursors are opaque, not
      journal-shaped) gets a cursor it can actually use:

        * a valid ``after_seq`` pairs with whatever ``after_event_id`` was
          supplied — even an opaque runner id like ``event:2`` that the
          journal parser reads as a foreign run — so the paired runner cursor
          resumes at the seq (never ``None`` / full replay, which would
          duplicate events);
        * a header-only opaque runner id (``Last-Event-ID: event:2``) is
          preserved as-is so the runner resumes from it;
        * a malformed or foreign explicit cursor WITHOUT a valid paired
          ``after_seq`` yields ``None`` — it must block the header and replay
          from start (this preserves the r2 malformed-blocks-header rule and
          never forwards an unusable cursor to the runner).

    The presence flag must stay separate from validity: a malformed, foreign-run,
    or ahead-of-stream cursor resolves to ``after_seq=None`` but still means the
    client *asked* to resume. That request must be honored with a
    replay-from-start so no journal events are silently skipped — whereas a
    genuinely cursor-less request is a fresh subscribe (no replay).

    Precedence is decided by query-parameter PRESENCE, not successful parsing:
    when an explicit ``after_seq`` / ``after_event_id`` is supplied (even an
    unparseable one), the ``Last-Event-ID`` header is never consulted, so a
    header can never override an explicit cursor and silently skip events.

    ``Last-Event-ID`` is the cursor every spec-compliant SSE client (browser
    ``EventSource`` auto-reconnect, Android/CLI clients) sends automatically on
    reconnect, carrying the ``id:`` of the last event it received. Every
    journaled event on this stream already emits ``id: stream_id:seq`` via
    ``_sse_with_id()``. Same resolution-chain precedent as
    ``api/kanban_bridge.py`` (``?since=`` → ``Last-Event-ID``).
    """
    after_seq_raw = qs.get("after_seq", [None])[0]
    has_explicit_query = (
        after_seq_raw not in (None, "")
        or bool(qs.get("after_event_id", [None])[0])
        or bool(qs.get("replay", [""])[0])
    )
    if has_explicit_query:
        explicit_raw = str(qs.get("after_event_id", [None])[0] or "").strip() or None
        after_seq = _parse_run_journal_after_seq(qs, stream_id)
        # Runner cursor provenance. ``after_seq`` is authoritative when it
        # parses — it pairs with whatever ``after_event_id`` shape was
        # supplied, including opaque runner ids (``event:2``) that the journal
        # parser reads as a foreign run. Read it directly here (not via
        # ``_parse_run_journal_after_seq``, which checks ``after_event_id``
        # FIRST and would swallow a paired opaque runner id as "foreign").
        paired_seq = _parse_run_journal_after_seq_value(after_seq_raw)
        if paired_seq is not None:
            runner_cursor = str(paired_seq)
        else:
            event_run_id, event_seq = _parse_run_journal_event_id(explicit_raw)
            runner_cursor = (
                explicit_raw
                if explicit_raw and event_seq is not None and (not stream_id or event_run_id == stream_id)
                else None
            )
        return after_seq, True, explicit_raw, runner_cursor
    headers = getattr(handler, "headers", None)
    if headers is None:
        return None, False, None, None
    try:
        raw = headers.get("Last-Event-ID")
    except Exception:
        return None, False, None, None
    raw = str(raw or "").strip()
    if not raw:
        return None, False, None, None
    event_run_id, event_seq = _parse_run_journal_event_id(raw)
    if event_run_id and event_seq is not None:
        if stream_id and event_run_id != stream_id:
            # Foreign-run journal cursor: asked to resume THIS run but the cursor
            # names a different journal run — can't honor it for the journal
            # path (after_seq stays None → replay-from-start). The runner path
            # keys cursors by run_id independently (the cursor is forwarded as
            # an opaque per-run query param), so a ``run:seq`` header still
            # reaches it as-is; this mirrors how a foreign after_event_id on
            # the journal path is rejected while the same client's explicit
            # opaque cursor= would still reach the runner.
            return None, True, raw, raw
        return event_seq, True, raw, raw
    # Malformed as a JOURNAL cursor. A colon-less opaque value is a plausible
    # runner cursor (runner ids need not be journal-shaped), so preserve it for
    # the runner; a value that merely fails int() parsing is unusable anywhere.
    runner_cursor = raw if ":" not in raw else None
    return None, True, raw, runner_cursor


def _parse_run_journal_after_seq_value(raw) -> int | None:
    """Parse a bare ``after_seq`` value, independent of any ``after_event_id``.

    Used by the runner-cursor provenance path, where ``after_seq`` is
    authoritative on its own and must NOT be gated behind the
    ``after_event_id``-first ordering of ``_parse_run_journal_after_seq`` (a
    paired opaque runner id like ``event:2`` would otherwise be read as a
    foreign run and swallow the seq). Mirrors the ``after_seq`` tail of that
    parser: absent/blank → None, non-numeric → 0.
    """
    if raw in (None, ""):
        return None
    try:
        return max(0, int(raw))
    except (TypeError, ValueError):
        return 0


def _parse_run_journal_after_seq(qs: dict, stream_id: str | None = None) -> int | None:
    event_run_id, event_seq = _parse_run_journal_event_id(qs.get("after_event_id", [None])[0])
    if event_run_id:
        if stream_id and event_run_id != stream_id:
            return None
        return event_seq
    raw = qs.get("after_seq", [None])[0]
    if raw in (None, ""):
        return None
    try:
        return max(0, int(raw))
    except (TypeError, ValueError):
        return 0


def _replay_run_journal(
    handler,
    stream_id: str,
    after_seq: int | None,
    *,
    max_seq: int | None = None,
    include_stale: bool = True,
) -> bool:
    summary = find_run_summary(stream_id)
    if not summary:
        return False
    journal = read_run_events(
        str(summary.get("session_id") or ""),
        stream_id,
        after_seq=after_seq,
        max_seq=max_seq,
    )
    for entry in journal.get("events") or []:
        if not journal_replay_visible(entry):
            continue
        _sse_with_id(
            handler,
            entry.get("event") or entry.get("type") or "message",
            entry.get("payload"),
            entry.get("event_id"),
        )
    if include_stale and not summary.get("terminal"):
        stale = stale_interrupted_event(
            str(summary.get("session_id") or ""),
            stream_id,
            after_seq=after_seq,
        )
        if stale:
            _sse_with_id(handler, stale["event"], stale["payload"], stale["event_id"])
    return True


def _run_journal_same_run_seq(event_id: str | None, stream_id: str) -> int | None:
    event_run_id, event_seq = _parse_run_journal_event_id(event_id)
    if event_run_id != stream_id:
        return None
    return event_seq


def _run_journal_covers_offline_gap(
    stream_id: str, after_seq: int | None, cutoff_seq: int | None
) -> bool:
    """Return True when the run journal PROVABLY backfills a dropped-frame gap.

    When StreamChannel evicted frames from its offline buffer
    (``offline_dropped_events > 0``), draining the retained tail to a
    reconnecting client is only safe if the journal replay actually covers
    everything from the client's cursor (*after_seq*, ``None`` = start of run)
    through the snapshot cutoff — journal seqs are assigned contiguously from 1,
    so coverage means every seq in ``(after_seq, cutoff_seq]`` is present. A
    missing journal, a journal that stops short of the cutoff, or a window with
    malformed/dropped lines all mean the evicted frames are unrecoverable here
    and the caller must signal recovery instead of streaming tail-only.

    ``cutoff_seq is None`` (the channel never saw a same-run journaled event id)
    counts as not covered: nothing can be proven against an unknown cutoff.
    """
    if cutoff_seq is None:
        return False
    floor = max(0, int(after_seq)) if after_seq is not None else 0
    if floor >= cutoff_seq:
        # Client cursor is already at/past everything the buffer ever held.
        return True
    try:
        summary = find_run_summary(stream_id)
        if not summary:
            return False
        journal = read_run_events(
            str(summary.get("session_id") or ""),
            stream_id,
            after_seq=floor,
            max_seq=cutoff_seq,
        )
    except Exception:
        logger.debug(
            "Run journal coverage check failed for stream %s", stream_id, exc_info=True
        )
        return False
    cutoff = int(cutoff_seq)
    seqs = set()
    for entry in journal.get("events") or []:
        try:
            seq = int(entry.get("seq") or 0)
        except (TypeError, ValueError):
            continue
        if floor < seq <= cutoff:
            seqs.add(seq)
    # Seqs are unique and bounded to the window, so full coverage means one
    # distinct seq per slot — no need to materialize the whole range.
    return len(seqs) == cutoff - floor


def _sse_replay_run_journal_gap_checked(
    handler, qs: dict, stream_id: str, stream_snapshot: dict,
    *, resume_cursor: tuple[int | None, bool] | None = None,
) -> tuple[bool, int | None]:
    """Journal-replay for a reconnecting client, enforcing offline-gap coverage.

    Returns ``(gap_recovered, replay_cutoff_seq)``. When the channel evicted
    frames from its offline buffer (``offline_dropped_events > 0`` in the
    subscribe snapshot) and the run journal cannot PROVE it backfills the gap
    (see ``_run_journal_covers_offline_gap``), a recovery_control apperror has
    been emitted and the caller must return instead of draining the retained
    tail (``gap_recovered=True``).

    ``resume_cursor`` is the caller-resolved ``(after_seq, resume_requested)``
    pair from ``_chat_stream_resume_cursor``. Presence and validity are kept
    separate: an invalid / foreign-run / unparseable cursor means the client
    *asked* to resume but we couldn't honor it, so it is normalized to
    replay-from-start (``after_seq=None`` inside the replay) rather than
    treated as "no cursor" — which would skip replay and silently drain a
    truncated buffer. A genuinely cursor-less request (``resume_requested``
    False) is a fresh subscribe and returns ``(False, None)`` with no replay.

    Direct callers that only supply ``qs`` keep the historical behavior: the
    cursor is derived from the query params (presence of ``replay`` /
    ``after_seq`` / ``after_event_id`` counts as resume-requested).
    """
    if resume_cursor is None:
        after_seq = _parse_run_journal_after_seq(qs, stream_id)
        resume_requested = (
            bool(qs.get("replay", [""])[0])
            or qs.get("after_seq", [None])[0] not in (None, "")
            or bool(qs.get("after_event_id", [None])[0])
        )
    else:
        after_seq, resume_requested = resume_cursor
    if not resume_requested:
        return False, None
    try:
        offline_dropped = int(stream_snapshot.get("offline_dropped_events") or 0)
    except (TypeError, ValueError):
        offline_dropped = 0
    snapshot_cutoff_seq = _run_journal_same_run_seq(
        str(stream_snapshot.get("last_event_id") or ""),
        stream_id,
    )
    # Normalize an unparseable / foreign cursor (asked to resume, but no usable
    # same-run seq) to replay-from-start so the gap check and dedup operate on
    # a real cursor instead of silently skipping the whole journal.
    if after_seq is None:
        after_seq = 0
    # Normalize a numeric cursor strictly AHEAD of the snapshot's last known
    # frame to replay-from-start too: the client believes it already holds
    # everything, so on a truncated buffer the coverage check's
    # ``floor >= replay_max_seq`` would falsely declare the gap covered, replay
    # nothing, and drain only the retained tail — silently losing every event
    # before it (Codex r2 #2). ``>`` (not ``>=``) — a cursor EQUAL to the
    # cutoff is a valid in-range cursor (see the dedup bound below).
    #
    # An UNKNOWN snapshot cutoff (no parseable ``last_event_id`` — e.g. the
    # channel has not seen an id-bearing frame yet) is treated as fence 0:
    # with no cutoff to bound it, any positive client cursor would otherwise
    # be installed verbatim as the live dedup bound and filter EVERY queued
    # frame — including the terminal ``stream_end`` fence — leaving the
    # reconnect stalled on heartbeats with an empty body (Codex r4). Failing
    # closed to replay-from-start delivers the buffered events (at worst
    # duplicating what the client already holds) instead of silently losing
    # them. Frames dropped while the cutoff is unknown still cannot prove
    # coverage below (``cutoff_seq is None`` → not covered), so the
    # recovery_control fail-closed path is preserved.
    effective_cutoff = snapshot_cutoff_seq if snapshot_cutoff_seq is not None else 0
    if after_seq > effective_cutoff:
        after_seq = 0
    # The subscribe snapshot already queued the retained offline tail, which
    # covers [first buffered frame → snapshot cutoff] by itself. The journal
    # only has to bridge (client cursor → first buffered frame) — and the
    # replay/dedup cutoff must stop there too, or the drain loop's
    # `seq <= replay_cutoff_seq` filter would eat queued frames the journal
    # never emitted. Without a parseable first-frame id (empty buffer, foreign
    # run, unjournaled head frame) fall back to the full (cursor → cutoff]
    # window as before.
    replay_max_seq = snapshot_cutoff_seq
    first_buffered_seq = _run_journal_same_run_seq(
        str(stream_snapshot.get("offline_first_event_id") or ""),
        stream_id,
    )
    if first_buffered_seq is not None:
        replay_max_seq = first_buffered_seq - 1
        if snapshot_cutoff_seq is not None:
            replay_max_seq = min(replay_max_seq, snapshot_cutoff_seq)
    covered = offline_dropped <= 0 or _run_journal_covers_offline_gap(
        stream_id, after_seq, replay_max_seq
    )
    replay_cutoff_seq = None
    replay_failed = False
    if covered:
        try:
            if _replay_run_journal(
                handler,
                stream_id,
                after_seq,
                max_seq=replay_max_seq,
                include_stale=False,
            ):
                replay_cutoff_seq = replay_max_seq
        except _CLIENT_DISCONNECT_ERRORS:
            raise
        except Exception:
            replay_failed = True
            logger.debug("Failed to replay active run journal for stream %s", stream_id, exc_info=True)
    if offline_dropped > 0 and (not covered or replay_failed):
        _sse_offline_gap_recovery(handler, stream_id, offline_dropped)
        return True, None
    # Two distinct dedup bounds feed the drain loop's `seq <=` filter: frames
    # the journal replay just emitted (replay_cutoff_seq, capped at the buffer
    # head so queued frames the journal never sent survive) AND frames the
    # client already holds per its own cursor. A cursor at/inside the retained
    # tail (after_seq >= first buffered frame) would otherwise get the queued
    # copy of frames it already rendered — a double-render, since this filter
    # is the only dedup for replayed streams.
    #
    # Dedup bound semantics: the drain filter skips ``seq <= replay_cutoff_seq``.
    # The event AT the cursor (seq == after_seq) was already delivered to this
    # client, so the cursor must itself enter the bound — equality included —
    # otherwise the buffered copy of that event double-sends. A cursor strictly
    # ahead of the snapshot was already normalized to 0 above; here only
    # in-range cursors (after_seq <= snapshot_cutoff_seq) contribute a bound,
    # and the terminal frame must always survive.
    if after_seq is not None and after_seq > 0:
        if snapshot_cutoff_seq is None or after_seq <= snapshot_cutoff_seq:
            replay_cutoff_seq = (
                after_seq
                if replay_cutoff_seq is None
                else max(replay_cutoff_seq, after_seq)
            )
    return False, replay_cutoff_seq


def _sse_offline_gap_recovery(handler, stream_id: str, offline_dropped: int) -> None:
    """Signal an unrecoverable replay gap instead of streaming tail-only.

    Frames were evicted from the channel's capped offline buffer and the run
    journal cannot prove it backfills (client cursor → snapshot cutoff]:
    draining the retained tail would render a silent transcript hole that ends
    in a normal ``stream_end``. Emit the established ``recovery_control``
    apperror (same client contract as ``run_journal.stale_interrupted_event``)
    so the tab restores the transcript from persisted session state instead.
    """
    # The client only acts on the recovery signal when the payload names its
    # session (eventMatchesCurrent), so fall back to the journal summary when
    # the pre-worker owner registration is already gone.
    try:
        session_id = stream_owner_session_id(stream_id) or ""
        if not session_id:
            session_id = str((find_run_summary(stream_id) or {}).get("session_id") or "")
    except Exception:
        session_id = ""
    _sse(
        handler,
        "apperror",
        {
            "type": "interrupted",
            "recovery_control": True,
            "message": (
                "The live stream's replay buffer overflowed while no tab was "
                "attached and the run journal cannot backfill the dropped frames."
            ),
            "hint": "The transcript was restored to the last saved state.",
            "session_id": session_id,
            "stream_id": stream_id,
            "offline_dropped_events": offline_dropped,
        },
    )


def _runner_stream_cursor_from_query(qs: dict) -> str | None:
    cursor = str(qs.get("cursor", [""])[0] or "").strip()
    if cursor:
        return cursor
    after_seq = _parse_run_journal_after_seq(qs)
    return str(after_seq) if after_seq is not None else None


def _runner_event_name(entry: dict) -> str:
    return str(entry.get("event") or entry.get("type") or "message")


def _runner_event_payload(entry: dict):
    if "payload" in entry:
        return entry.get("payload")
    if "data" in entry:
        return entry.get("data")
    return entry


def _project_runner_event_payload(payload):
    """Strip internal replay fields (api_content, row-id aliases) from a runner
    SSE payload before it is relayed to the browser.

    Runner-backed SSE relays adapter payloads verbatim; a terminal event can
    carry a full ``session`` object (with per-message ``api_content`` sidecars)
    or be session/message-shaped itself. Neither must reach a client, so route
    the transcript-bearing shapes through the same public projection every other
    session emitter uses. Non-session payloads pass through unchanged.
    """
    if not isinstance(payload, dict):
        return payload
    # Terminal events wrap the session under a "session" key.
    if isinstance(payload.get("session"), dict):
        projected = dict(payload)
        projected["session"] = public_session_projection(payload["session"])
        return projected
    # Payload is itself session/message-shaped (has a messages transcript).
    if "messages" in payload or "context_messages" in payload:
        return public_session_projection(payload)
    return payload


def _runner_event_id(run_id: str, entry: dict) -> str | None:
    event_id = entry.get("event_id") or entry.get("id")
    if event_id:
        return str(event_id)
    seq = entry.get("seq")
    if seq not in (None, ""):
        return f"{run_id}:{seq}"
    return None


def _stream_runner_run_events(handler, run_id: str, cursor: str | None = None) -> bool:
    """Stream events from a configured runner without WebUI-owned runtime maps."""
    run_id = str(run_id or "").strip()
    if not run_id:
        return False
    try:
        from api.runtime_adapter import build_runtime_adapter, runtime_adapter_runner_enabled

        if not runtime_adapter_runner_enabled():
            return False
        adapter = build_runtime_adapter(runner_client_factory=_runtime_runner_client_factory)
    except NotImplementedError:
        return False
    if adapter is None:
        return False

    handler.send_response(200)
    handler.send_header("Content-Type", "text/event-stream; charset=utf-8")
    handler.send_header("Cache-Control", "no-cache")
    handler.send_header("X-Accel-Buffering", "no")
    handler.send_header("Connection", "close")
    end_sse_headers(handler)
    cursor_value = cursor
    try:
        while True:
            try:
                event_stream = adapter.observe_run(run_id, cursor=cursor_value)
            except Exception as exc:
                _sse(handler, "error", {"error": _sanitize_error(exc)})
                break
            emitted = False
            terminal = False
            for entry in list(getattr(event_stream, "events", []) or []):
                if not isinstance(entry, dict):
                    continue
                event = _runner_event_name(entry)
                _sse_with_id(handler, event, _project_runner_event_payload(_runner_event_payload(entry)), _runner_event_id(run_id, entry))
                emitted = True
                if event in SSE_RELAY_CLOSE_EVENTS:
                    terminal = True
            next_cursor = getattr(event_stream, "cursor", None)
            if next_cursor not in (None, ""):
                cursor_value = str(next_cursor)
            if terminal:
                break
            if not emitted:
                status = None
                try:
                    status = adapter.get_run(run_id)
                except Exception:
                    status = None
                state = str(getattr(status, "terminal_state", None) or getattr(status, "status", "") or "").lower()
                if state in ("completed", "complete", "failed", "error", "cancelled", "canceled"):
                    _sse(handler, "stream_end", {"run_id": run_id, "status": state})
                    break
                handler.wfile.write(b": heartbeat\n\n")
                handler.wfile.flush()
                time.sleep(_SSE_HEARTBEAT_INTERVAL_SECONDS)
    except _CLIENT_DISCONNECT_ERRORS:
        pass
    return True


def _handle_sse_stream(handler, parsed):
    qs = parse_qs(parsed.query)
    stream_id = qs.get("stream_id", [""])[0]
    if not _stream_id_visible_to_request_profile(handler, stream_id):
        return True
    # Resume cursor: explicit query params (after_event_id/after_seq/replay)
    # win; the Last-Event-ID header that spec-compliant SSE clients auto-send
    # on reconnect is the fallback. Presence is tracked separately from the
    # parsed seq — a client that supplied ANY cursor asked to resume, and an
    # unusable (invalid/foreign/ahead-of-stream) cursor must replay from start
    # rather than silently skip journal events.
    resume_cursor = _chat_stream_resume_cursor(handler, qs, stream_id)
    resume_after_seq, resume_requested, resume_raw_cursor, runner_resume_cursor = resume_cursor
    stream = peek_stream(stream_id)
    if stream is None:
        # Runner-observe path: consume the ALREADY-RESOLVED cursor — do not
        # re-parse query params or re-read the header (Codex r2 #3 / r3). The
        # explicit opaque ``cursor`` query param still wins for runner clients
        # that speak that contract. Otherwise use the resolver's
        # provenance-resolved runner cursor: a valid ``after_seq`` pairs with
        # opaque runner ids (event:2), a header-only opaque runner id resumes
        # as-is, and a malformed/foreign cursor without a valid paired seq
        # yields None (replay from start, never forwarding an unusable cursor).
        runner_cursor = str(qs.get("cursor", [""])[0] or "").strip() or None
        if runner_cursor is None and resume_requested:
            runner_cursor = runner_resume_cursor
        if _stream_runner_run_events(handler, stream_id, runner_cursor):
            return True
        try:
            journal_summary = find_run_summary(stream_id) if stream_id else None
        except Exception:
            journal_summary = None
        if not journal_summary:
            return j(handler, {"error": "stream not found"}, status=404)
        # Normalize a cursor strictly AHEAD of the dead stream's authoritative
        # last_seq to replay-from-start: passing it straight through would make
        # the journal reader emit an empty SSE body for a journal that actually
        # holds events (Codex r2 #2). Equality is in-range — the event at the
        # cursor was already delivered, so the replay correctly resumes after it.
        dead_after_seq = resume_after_seq
        try:
            last_seq = int(journal_summary.get("last_seq") or 0)
        except (TypeError, ValueError):
            last_seq = 0
        if dead_after_seq is not None and dead_after_seq > last_seq:
            dead_after_seq = 0
        handler.send_response(200)
        handler.send_header("Content-Type", "text/event-stream; charset=utf-8")
        handler.send_header("Cache-Control", "no-cache")
        handler.send_header("X-Accel-Buffering", "no")
        handler.send_header("Connection", "close")
        end_sse_headers(handler)
        try:
            _replay_run_journal(handler, stream_id, dead_after_seq)
        except _CLIENT_DISCONNECT_ERRORS:
            pass
        return True
    if hasattr(stream, "subscribe_with_snapshot"):
        subscriber, stream_snapshot = stream.subscribe_with_snapshot()
    else:
        subscriber = stream.subscribe() if hasattr(stream, "subscribe") else stream
        stream_snapshot = {}
    handler.send_response(200)
    handler.send_header("Content-Type", "text/event-stream; charset=utf-8")
    handler.send_header("Cache-Control", "no-cache")
    handler.send_header("X-Accel-Buffering", "no")
    handler.send_header("Connection", "close")
    end_sse_headers(handler)
    _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread
    # Replay shares the drain loop's try/finally so every exit path unsubscribes.
    try:
        gap_recovered, replay_cutoff_seq = _sse_replay_run_journal_gap_checked(
            handler, qs, stream_id, stream_snapshot,
            resume_cursor=(resume_after_seq, resume_requested),
        )
        if gap_recovered:
            return True
        while True:
            try:
                item = subscriber.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                handler.wfile.write(b": heartbeat\n\n")
                handler.wfile.flush()
                continue
            if len(item) >= 3:
                event, data, queued_event_id = item[0], item[1], item[2]
                event_id = queued_event_id
            else:
                event, data = item
                event_id = STREAM_LAST_EVENT_ID.get(stream_id)
            # Stage-364: emit `id:` from STREAM_LAST_EVENT_ID side-channel so
            # the frontend's `_lastRunJournalSeq` cursor advances during live
            # streaming. Without this, mid-stream error→replay would arrive
            # with after_seq=0 and double-render every journaled event.
            event_seq = _run_journal_same_run_seq(event_id, stream_id)
            if replay_cutoff_seq is not None and event_seq is not None and event_seq <= replay_cutoff_seq:
                continue
            if event_id:
                _sse_with_id(handler, event, data, event_id)
            else:
                _sse(handler, event, data)
            if event in SSE_RELAY_CLOSE_EVENTS:
                break
    except _CLIENT_DISCONNECT_ERRORS:
        pass
    finally:
        if subscriber is not stream and hasattr(stream, "unsubscribe"):
            try:
                stream.unsubscribe(subscriber)
            except Exception:
                pass
    return True


def _handle_session_run_journal_stream_for_session(handler, parsed, session_id):
    if not _session_id_visible_to_request_profile(handler, session_id):
        return True
    try:
        session = get_session(session_id, metadata_only=True)
    except KeyError:
        return j(handler, {"error": "Session not found"}, status=404)

    # Parse the resume cursor and baseline the journal BEFORE committing SSE headers
    # (and thus before any run could complete mid-handler). Capturing after
    # end_sse_headers() leaves a window where a run finishing between header commit
    # and baseline is absorbed into the baseline and silently lost. Both operations
    # are side-effect-free (header read + stat-only fingerprint), safe pre-response.
    resume_event_id = _session_events_resume_event_id(handler, parsed)
    _idle_journal_fp = session_journal_fingerprint(session_id)

    handler.send_response(200)
    handler.send_header("Content-Type", "text/event-stream; charset=utf-8")
    handler.send_header("Cache-Control", "no-cache")
    handler.send_header("X-Accel-Buffering", "no")
    # #3103: see _handle_gateway_sse_stream — `Connection: close` causes
    # EventSource reconnect storms in browsers on long-lived SSE.
    end_sse_headers(handler)
    _sse_set_write_deadline(handler)

    active_stream_id = _active_run_stream_for_session(session_id)
    subscriber = None
    subscriber_stream = None
    replay_cutoff_seq = None
    sent_event_ids: set[str] = set()
    sent_event_order = deque()

    def note_sent_event_id(event_id):
        if not event_id:
            return
        sent_event_ids.add(event_id)
        sent_event_order.append(event_id)
        while len(sent_event_order) > _SESSION_SSE_SENT_EVENT_ID_LIMIT:
            sent_event_ids.discard(sent_event_order.popleft())

    def attach_active_stream():
        stream_id = _active_run_stream_for_session(session_id)
        stream = peek_stream(stream_id) if stream_id else None
        if stream is None:
            return None, None, None, stream_id
        if hasattr(stream, "subscribe_with_snapshot"):
            queue_, snapshot = stream.subscribe_with_snapshot()
        else:
            queue_ = stream.subscribe() if hasattr(stream, "subscribe") else stream
            snapshot = {}
        return queue_, stream, snapshot, stream_id

    def emit_replay(events, stream_id, cutoff_seq):
        for entry in events:
            if not journal_replay_visible(entry):
                continue
            event_id = str(entry.get("event_id") or "")
            event_seq = _run_journal_same_run_seq(event_id, stream_id)
            if cutoff_seq is not None and event_seq is not None and event_seq > cutoff_seq:
                continue
            if event_id and event_id in sent_event_ids:
                continue
            _sse_with_id(handler, entry.get("event") or entry.get("type") or "message", entry.get("payload"), event_id)
            if event_id:
                note_sent_event_id(event_id)

    def emit_session_snapshot(active_stream_id):
        try:
            fresh_session = get_session(session_id, metadata_only=True)
        except KeyError:
            fresh_session = session
        _sse(handler, "session_snapshot", _session_snapshot_payload(fresh_session, active_stream_id=active_stream_id))

    try:
        replay_events = []
        replay_ok = False
        if resume_event_id:
            replay = read_session_run_events(session_id, after_event_id=resume_event_id)
            if replay.get("status") != "ok":
                emit_session_snapshot(active_stream_id)
            else:
                replay_ok = True
                replay_events = replay.get("events") or []
        subscriber, subscriber_stream, stream_snapshot, active_stream_id = attach_active_stream()
        if subscriber is None:
            if replay_ok:
                emit_replay(replay_events, active_stream_id, None)
            while True:
                subscriber, subscriber_stream, stream_snapshot, active_stream_id = attach_active_stream()
                if subscriber is not None:
                    break
                # Journal advanced with no live stream to attach → a run completed
                # entirely within the wait (or the first attach). Re-sync via a
                # snapshot boundary (the same honest-recovery contract used for a
                # failed reconciliation), then re-baseline so we only re-sync on
                # genuinely new advances.
                _current_journal_fp = session_journal_fingerprint(session_id)
                if _current_journal_fp != _idle_journal_fp:
                    _idle_journal_fp = _current_journal_fp
                    emit_session_snapshot(active_stream_id)
                handler.wfile.write(b": keepalive\n\n")
                handler.wfile.flush()
                time.sleep(_SSE_HEARTBEAT_INTERVAL_SECONDS)
        if subscriber is None:
            return True
        if replay_ok:
            replay_cutoff_seq = _run_journal_same_run_seq(str(stream_snapshot.get("last_event_id") or ""), active_stream_id)
            reconciled = read_session_run_events(session_id, after_event_id=resume_event_id)
            if reconciled.get("status") == "ok":
                emit_replay(reconciled.get("events") or [], active_stream_id, replay_cutoff_seq)
            else:
                emit_session_snapshot(active_stream_id)
        try:
            while True:
                try:
                    item = subscriber.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
                except queue.Empty:
                    handler.wfile.write(b": keepalive\n\n")
                    handler.wfile.flush()
                    continue
                if len(item) >= 3:
                    event, data, queued_event_id = item[0], item[1], item[2]
                    event_id = queued_event_id
                else:
                    event, data = item
                    event_id = STREAM_LAST_EVENT_ID.get(active_stream_id)
                event_seq = _run_journal_same_run_seq(event_id, active_stream_id)
                _is_terminal = event in SSE_RELAY_CLOSE_EVENTS
                _already_sent = (
                    (replay_cutoff_seq is not None and event_seq is not None and event_seq <= replay_cutoff_seq)
                    or (event_id and event_id in sent_event_ids)
                )
                if _already_sent:
                    # Already delivered via replay/reconciliation (cutoff or dedup).
                    # A terminal event still has to end this loop — otherwise, when
                    # reconciliation replayed the active run's terminal at the cutoff,
                    # the live copy would be skipped here and the handler would stay
                    # blocked on a dead run's queue and miss subsequent session runs.
                    if _is_terminal:
                        break
                    continue
                if event_id:
                    _sse_with_id(handler, event, data, event_id)
                    note_sent_event_id(event_id)
                else:
                    _sse(handler, event, data)
                if _is_terminal:
                    break
        except _CLIENT_DISCONNECT_ERRORS:
            pass
    except _CLIENT_DISCONNECT_ERRORS:
        pass
    finally:
        if subscriber is not None and subscriber is not subscriber_stream and hasattr(subscriber_stream, "unsubscribe"):
            try:
                subscriber_stream.unsubscribe(subscriber)
            except Exception:
                pass
    return True


_handle_session_sse_stream_for_session = _handle_session_run_journal_stream_for_session


def _terminal_session_lookup(body_or_query):
    sid = str(body_or_query.get("session_id", "")).strip()
    if not sid:
        raise ValueError("session_id required")
    try:
        s = get_session(sid)
    except KeyError:
        raise KeyError("Session not found")
    return sid, s


_REMOTE_TERMINAL_BACKEND_UNSUPPORTED_ERROR = "remote_terminal_backend_unsupported"
_REMOTE_TERMINAL_BACKEND_UNSUPPORTED_MESSAGE = (
    "Embedded terminal is only supported for local terminal backends."
)


def _terminal_remote_backend_enabled() -> bool:
    terminal_cfg = get_config().get("terminal", {})
    return _is_remote_terminal_backend(terminal_cfg)


def _handle_terminal_start(handler, body):
    try:
        if not _embedded_terminal_gate_allows(handler):
            return bad(handler, _EMBEDDED_TERMINAL_GATE_DENIED_MESSAGE, 403)
        sid, session = _terminal_session_lookup(body)
        if _terminal_remote_backend_enabled():
            return j(
                handler,
                {
                    "error": _REMOTE_TERMINAL_BACKEND_UNSUPPORTED_ERROR,
                    "message": _REMOTE_TERMINAL_BACKEND_UNSUPPORTED_MESSAGE,
                },
                status=400,
            )
        workspace = resolve_trusted_workspace(getattr(session, "workspace", "") or "", profile=getattr(session, "profile", None))
        from api.terminal import start_terminal
        term = start_terminal(
            sid,
            workspace,
            rows=int(body.get("rows") or 24),
            cols=int(body.get("cols") or 80),
            restart=bool(body.get("restart")),
        )
        return j(
            handler,
            {
                "ok": True,
                "session_id": sid,
                "workspace": term.workspace,
                "running": term.is_alive(),
            },
        )
    except KeyError as e:
        return bad(handler, str(e), 404)
    except ValueError as e:
        return bad(handler, str(e), 400)
    except Exception as e:
        return bad(handler, _sanitize_error(e), 500)


def _handle_terminal_input(handler, body):
    try:
        if not _embedded_terminal_gate_allows(handler):
            return bad(handler, _EMBEDDED_TERMINAL_GATE_DENIED_MESSAGE, 403)
        require(body, "session_id")
        data = str(body.get("data", ""))
        if len(data) > 8192:
            return bad(handler, "input too large", 413)
        from api.terminal import write_terminal
        write_terminal(body["session_id"], data)
        return j(handler, {"ok": True})
    except KeyError as e:
        return bad(handler, str(e), 404)
    except ValueError as e:
        return bad(handler, str(e), 400)
    except Exception as e:
        return bad(handler, _sanitize_error(e), 500)


def _handle_terminal_resize(handler, body):
    try:
        if not _embedded_terminal_gate_allows(handler):
            return bad(handler, _EMBEDDED_TERMINAL_GATE_DENIED_MESSAGE, 403)
        require(body, "session_id")
        from api.terminal import resize_terminal
        resize_terminal(
            body["session_id"],
            rows=int(body.get("rows") or 24),
            cols=int(body.get("cols") or 80),
        )
        return j(handler, {"ok": True})
    except KeyError as e:
        return bad(handler, str(e), 404)
    except ValueError as e:
        return bad(handler, str(e), 400)
    except Exception as e:
        return bad(handler, _sanitize_error(e), 500)


def _handle_terminal_close(handler, body):
    try:
        if not _embedded_terminal_gate_allows(handler):
            return bad(handler, _EMBEDDED_TERMINAL_GATE_DENIED_MESSAGE, 403)
        require(body, "session_id")
        from api.terminal import close_terminal
        closed = close_terminal(body["session_id"])
        return j(handler, {"ok": True, "closed": closed})
    except ValueError as e:
        return bad(handler, str(e), 400)


def _handle_terminal_output(handler, parsed):
    if not _embedded_terminal_gate_allows(handler):
        return bad(handler, _EMBEDDED_TERMINAL_GATE_DENIED_MESSAGE, 403)
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id required")
    from api.terminal import attach_terminal
    # EventSource automatically returns the last received SSE id on transport
    # reconnect. Seed only newer backlog entries in that case so already-rendered
    # terminal bytes (including ANSI cursor controls) are not written twice. A
    # genuinely new viewer has no cursor and receives the full bounded backlog.
    after_seq = None
    last_event_id = str(handler.headers.get("Last-Event-ID", "") or "").strip()
    if last_event_id:
        try:
            after_seq = max(0, int(last_event_id))
        except ValueError:
            pass
    # Look up and subscribe in one atomic step. A separate `get_terminal()` then
    # `term.subscribe()` leaves a window in which the idle reaper can claim and
    # tear the terminal down, leaving this stream attached to a corpse after we
    # already committed a 200. Attaching atomically means we either hold a live
    # viewer (which makes the terminal un-reapable) or learn it is gone in time
    # to answer 404.
    attached = attach_terminal(sid, after_seq=after_seq)
    if attached is None:
        return j(handler, {"error": "terminal not running"}, status=404)
    term, output = attached

    # The subscription is live from here on, so EVERY exit path — including a
    # failure while writing the response headers — must unsubscribe. Writing
    # headers to a client that already dropped raises BrokenPipeError, and if
    # that escaped before the try block the queue would stay in
    # `_subscribers` forever, pinning `unwatched_since` at None and making the
    # terminal permanently unreapable: the exact fd/thread leak this reaper
    # exists to prevent. Hence the try starts immediately after the attach.
    try:
        handler.send_response(200)
        handler.send_header("Content-Type", "text/event-stream; charset=utf-8")
        handler.send_header("Cache-Control", "no-cache")
        handler.send_header("X-Accel-Buffering", "no")
        handler.send_header("Connection", "close")
        end_sse_headers(handler)
        _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread
        while True:
            try:
                event_seq, event, data = output.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                handler.wfile.write(b": terminal heartbeat\n\n")
                handler.wfile.flush()
                if term.closed.is_set() and output.empty():
                    _sse(handler, "terminal_closed", {"exit_code": term.proc.poll()})
                    break
                continue
            _sse_with_id(handler, event, data, event_id=event_seq)
            if event in ("terminal_closed", "terminal_error"):
                break
    except (BrokenPipeError, ConnectionResetError, ConnectionAbortedError):
        pass
    finally:
        term.unsubscribe(output)
    return True


def _gateway_sse_probe_payload(settings, watcher):
    enabled = bool(settings.get('show_cli_sessions'))
    # Use the public is_alive() accessor where available (current GatewayWatcher);
    # fall back to the private _thread check for any older in-memory instance
    # that might still be hanging around mid-upgrade, and for test doubles that
    # don't implement the full public API.
    if watcher is None:
        watcher_alive = False
    elif hasattr(watcher, 'is_alive') and callable(getattr(watcher, 'is_alive')):
        watcher_alive = bool(watcher.is_alive())
    else:
        _t = getattr(watcher, '_thread', None)
        watcher_alive = _t is not None and _t.is_alive()
    payload = {
        'enabled': enabled,
        'fallback_poll_ms': 30000,
        'ok': enabled and watcher_alive,
        'watcher_running': watcher_alive,
        # Cross-client scope markers (hermes-webui/hermes-android#58 follow-up):
        # this probe ONLY describes the optional gateway/agent-sessions stream.
        # Persistent per-session streaming (GET /api/session/stream) is always
        # available and is NOT gated by show_cli_sessions, so a negative gateway
        # probe result must not be read as "session SSE unavailable".
        'scope': 'gateway_sessions',
        'session_stream_available': True,
        'session_stream_path': '/api/session/stream',
    }
    if not enabled:
        payload['error'] = 'agent sessions not enabled'
        return payload, 404
    if not watcher_alive:
        payload['error'] = 'watcher not started'
        return payload, 503
    return payload, 200


def _handle_gateway_sse_stream(handler, parsed):
    """SSE endpoint for real-time gateway session updates.
    Streams change events from the gateway watcher background thread.
    Only active when show_cli_sessions (show_agent_sessions) setting is enabled.

    Probe mode (``?probe=1``) reports the status of THIS optional stream only.
    Its result says nothing about the always-on persistent per-session stream
    (``/api/session/stream``) — the probe payload carries explicit
    ``scope`` / ``session_stream_available`` markers so cross-client consumers
    do not misclassify usable session streaming as unavailable.
    """
    settings = load_settings()

    from api.gateway_watcher import get_watcher
    watcher = get_watcher()

    probe = parse_qs(parsed.query).get('probe', [''])[0].lower() in {'1', 'true', 'yes'}
    if probe:
        payload, status = _gateway_sse_probe_payload(settings, watcher)
        return j(handler, payload, status=status)

    # Check if the feature is enabled
    if not settings.get('show_cli_sessions'):
        return j(handler, {'error': 'agent sessions not enabled'}, status=404)

    # Same watcher_alive semantics as the probe path — centralised via
    # the helper so both branches stay in sync.
    _probe_body, _probe_status = _gateway_sse_probe_payload(settings, watcher)
    if not _probe_body['watcher_running']:
        return j(handler, {'error': 'watcher not started'}, status=503)

    handler.send_response(200)
    handler.send_header('Content-Type', 'text/event-stream; charset=utf-8')
    handler.send_header('Cache-Control', 'no-cache')
    handler.send_header('X-Accel-Buffering', 'no')
    # #3103: do NOT emit `Connection: close` on long-lived SSE streams.
    # The python BaseHTTPServer worker only handles one request per
    # connection anyway, but browsers (Chrome/Firefox) treat the close
    # header as a hard signal that the EventSource lifecycle has ended
    # and trigger an instant reconnect, producing a tight loop of
    # connect/sessions_changed snapshot/disconnect that thrashes the
    # session list every ~1s. Letting the server close the socket
    # naturally after the stream ends is sufficient.
    end_sse_headers(handler)
    _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread

    q = watcher.subscribe()
    try:
        # Send initial snapshot immediately
        from api.models import get_cli_sessions
        initial = get_cli_sessions()
        _sse(handler, 'sessions_changed', {'sessions': initial})

        while True:
            try:
                event_data = q.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                handler.wfile.write(b': keepalive\n\n')
                handler.wfile.flush()
                continue
            if event_data is None:
                break  # watcher is stopping
            _sse(handler, event_data.get('type', 'sessions_changed'), event_data)
    except _CLIENT_DISCONNECT_ERRORS:
        pass
    finally:
        watcher.unsubscribe(q)
    return True


def _handle_session_events_stream(handler):
    """SSE endpoint for lightweight session-list invalidation events."""
    handler.send_response(200)
    handler.send_header('Content-Type', 'text/event-stream; charset=utf-8')
    handler.send_header('Cache-Control', 'no-cache')
    handler.send_header('X-Accel-Buffering', 'no')
    # #3103: see _handle_gateway_sse_stream — `Connection: close` causes
    # EventSource reconnect storms in browsers on long-lived SSE.
    end_sse_headers(handler)
    _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread

    q = subscribe_session_events()
    try:
        while True:
            try:
                event_data = q.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                handler.wfile.write(b': keepalive\n\n')
                handler.wfile.flush()
                continue
            _sse(handler, event_data.get('type', 'sessions_changed'), event_data)
    except _CLIENT_DISCONNECT_ERRORS:
        pass
    finally:
        unsubscribe_session_events(q)
    return True


def _content_disposition_value(disposition: str, filename: str) -> str:
    """Build a latin-1-safe Content-Disposition value with RFC 5987 filename*."""
    import urllib.parse as _up

    safe_name = Path(filename).name.replace("\r", "").replace("\n", "")
    ascii_fallback = "".join(
        ch if 32 <= ord(ch) < 127 and ch not in {'"', '\\'} else "_"
        for ch in safe_name
    ).strip(" .")
    if not ascii_fallback:
        suffix = Path(safe_name).suffix
        ascii_suffix = "".join(
            ch if 32 <= ord(ch) < 127 and ch not in {'"', '\\'} else "_"
            for ch in suffix
        )
        ascii_fallback = f"download{ascii_suffix}" if ascii_suffix else "download"
    quoted_name = _up.quote(safe_name, safe="")
    return (
        f'{disposition}; filename="{ascii_fallback}"; '
        f"filename*=UTF-8''{quoted_name}"
    )


def _parse_range_header(range_header: str, file_size: int) -> tuple[int, int] | None:
    """Parse a single HTTP bytes range into inclusive start/end offsets."""
    if not range_header or not range_header.startswith("bytes=") or file_size < 1:
        return None
    spec = range_header.split("=", 1)[1].strip()
    if "," in spec or "-" not in spec:
        return None
    start_s, end_s = spec.split("-", 1)
    try:
        if start_s == "":
            # suffix range: bytes=-500
            suffix_len = int(end_s)
            if suffix_len <= 0:
                return None
            start = max(0, file_size - suffix_len)
            end = file_size - 1
        else:
            start = int(start_s)
            end = int(end_s) if end_s else file_size - 1
            if start < 0:
                return None
            end = min(end, file_size - 1)
        if start > end or start >= file_size:
            return None
        return start, end
    except ValueError:
        return None


def _open_file_read_fd(target: Path, anchor_root: Path | None = None) -> int:
    if anchor_root is None:
        flags = os.O_RDONLY
        # On Windows, files are opened in text mode by default; O_BINARY
        # prevents CRLF translation that would corrupt binary media files.
        if hasattr(os, "O_BINARY"):
            flags |= os.O_BINARY
        return os.open(str(target), flags)
    return open_anchored_fd(anchor_root, target.resolve(), want_dir=False)


def _close_fd_quietly(fd: int | None) -> None:
    if fd is None:
        return
    try:
        os.close(fd)
    except OSError:
        pass


# Maximum size for which a content-derived ETag is computed.  Files above
# this cap (and all HTML with no-store) are served without ETag to avoid
# hashing every byte of large media / Range requests.
_ETAG_SIZE_CAP = 10 * 1024 * 1024  # 10 MB


def _bytes_etag(data: bytes) -> str:
    """Weak ETag from a content digest of the bytes that will be served.

    The bytes must be an immutable snapshot (e.g. pread or an in-memory copy)
    so the validator cannot diverge from the body under TOCTOU.
    """
    return 'W/"%s"' % hashlib.sha256(data).hexdigest()


def _etag_and_snapshot(fd, *, file_size: int) -> tuple[str | None, bytes | None, int]:
    """Return (weak ETag, snapshot bytes, actual size) for files under the size cap.

    Uses os.lseek + looped os.read to grab an immutable snapshot in a cross-platform
    and short-read-safe manner. The snapshot bytes can be sent directly so the ETag
    and the body can never diverge under TOCTOU (the file may change on disk after read).

    Returns (None, None, file_size) for files above the cap or on unexpected I/O failure.
    Returns (None, None, actual_size) if the file was truncated mid-read (short snapshot).
    The caller must use actual_size (not the original file_size) for Content-Length/Range
    calculations to avoid header/body mismatch.
    """
    if file_size > _ETAG_SIZE_CAP:
        return None, None, file_size

    # Cross-platform: os.lseek + looped os.read instead of POSIX-only os.pread
    # Short reads are possible (interrupted, EOF from truncation), so we loop.
    os.lseek(fd, 0, os.SEEK_SET)
    data = b""
    remaining = file_size
    while remaining > 0:
        chunk = os.read(fd, min(1024 * 1024, remaining))
        if not chunk:  # EOF reached (file was truncated)
            break
        data += chunk
        remaining -= len(chunk)

    actual_size = len(data)
    if actual_size == 0:
        return None, None, 0

    # If the file was truncated after fstat (short snapshot), return the actual
    # size but no ETag — caller will fall back to streaming without ETag.
    # Round-6 fix: keep the captured `data` so the body path serves the immutable
    # snapshot instead of re-reading the fd after headers are committed.
    # Content-Length derives from actual_size == len(data), so header/body match.
    if actual_size != file_size:
        return None, data, actual_size

    return _bytes_etag(data), data, actual_size


def _serve_file_bytes(handler, target: Path, mime: str, disposition: str, cache_control: str, *, csp: str | None = None, anchor_root: Path | None = None, download_name: str | None = None):
    """Serve a file with correct MIME/disposition and optional byte-range support.

    Supports conditional GET via If-None-Match (ETag) — when the ETag matches,
    the request is short-circuited with 304 so revalidating clients (e.g.
    `no-cache` responses) do not re-download unchanged files.

    ``download_name`` overrides the Content-Disposition filename (used when
    serving an immutable media snapshot whose on-disk name is a content digest).
    """
    fd = None
    try:
        fd = _open_file_read_fd(target, anchor_root)
        st = os.fstat(fd)
        file_size = st.st_size
    except PermissionError:
        _close_fd_quietly(fd)
        return bad(handler, "Permission denied", 403)
    except FileNotFoundError:
        _close_fd_quietly(fd)
        return j(handler, {"error": "not found"}, status=404)
    except ValueError as e:
        _close_fd_quietly(fd)
        return bad(handler, _sanitize_error(e), 403)
    except Exception:
        _close_fd_quietly(fd)
        return bad(handler, "Could not stat file", 500)

    try:
        # Pre-commit phase: ETag/snapshot computation and response selection.
        # OSError here is still convertible into a clean 500 because no
        # status line has been written yet. After end_headers() the response
        # is committed, so body-transmission errors must never call bad()
        # again (that would attempt a second send_response on a stream the
        # client may have already closed — see the body phase below).
        try:
            no_store = "no-store" in cache_control
            if no_store or file_size > _ETAG_SIZE_CAP:
                etag = None
                snapshot = None
                # actual_size stays as file_size for over-cap/no-store cases
            else:
                etag, snapshot, actual_size = _etag_and_snapshot(fd, file_size=file_size)
                # If the file was truncated mid-read (short snapshot), use the
                # actual read size for all subsequent calculations so
                # Content-Length/Range match the body we can actually send.
                # Round-6 fix: the captured bytes are kept, so the body path
                # serves the immutable snapshot instead of re-reading the fd.
                if actual_size != file_size:
                    file_size = actual_size  # reconcile to avoid header/body mismatch
        except OSError:
            return bad(handler, "Could not serve file", 500)

        # RFC 7232 §3.2: If-None-Match uses weak comparison (W/ prefixes ignored)
        # and "*" matches any existing resource. On match, GET/HEAD is
        # short-circuited with 304 — processed before Range since a matched
        # conditional request skips the entity entirely.
        if_none_match = handler.headers.get("If-None-Match", "")
        if if_none_match and etag is not None:
            current = etag[2:] if etag.startswith("W/") else etag
            matched = if_none_match.strip() == "*" or any(
                (c.strip()[2:] if c.strip().startswith("W/") else c.strip()) == current
                for c in if_none_match.split(",")
                if c.strip()
            )
            if matched:
                handler.send_response(304)
                handler.send_header("ETag", etag)
                handler.send_header("Cache-Control", cache_control)
                _security_headers(handler)
                handler.end_headers()
                return True

        byte_range = _parse_range_header(handler.headers.get("Range", ""), file_size)
        if handler.headers.get("Range") and byte_range is None:
            handler.send_response(416)
            handler.send_header("Content-Range", f"bytes */{file_size}")
            handler.send_header("Accept-Ranges", "bytes")
            handler.send_header("Content-Length", "0")
            _security_headers(handler)
            handler.end_headers()
            return True

        start, end = byte_range if byte_range else (0, max(0, file_size - 1))
        content_length = end - start + 1 if file_size else 0
        handler.send_response(206 if byte_range else 200)
        handler.send_header("Content-Type", mime)
        handler.send_header("Content-Length", str(content_length))
        handler.send_header("Accept-Ranges", "bytes")
        if etag is not None:
            handler.send_header("ETag", etag)
        if byte_range:
            handler.send_header("Content-Range", f"bytes {start}-{end}/{file_size}")
        handler.send_header("Cache-Control", cache_control)
        handler.send_header("Content-Disposition", _content_disposition_value(disposition, download_name or target.name))
        if csp:
            # Sandboxed inline HTML must remain frameable for workspace previews;
            # X-Frame-Options: DENY would block the iframe before CSP sandbox applies.
            handler.send_header("Content-Security-Policy", csp)
            handler.send_header("X-Content-Type-Options", "nosniff")
            handler.send_header("Referrer-Policy", "same-origin")
            handler.send_header(
                "Permissions-Policy",
                "camera=(), microphone=(self), geolocation=(), clipboard-write=(self)",
            )
        else:
            _security_headers(handler)
        handler.end_headers()

        # Body transmission: the response is committed once end_headers()
        # returns, so attempting bad()/send_response again here would corrupt
        # the stream with a second status line. Client disconnects are normal
        # (tab close, network switch) — log at debug and stop, same contract
        # as _safe_write(). Never emit a 500 after headers are out.
        if content_length:
            try:
                if snapshot is not None:
                    handler.wfile.write(snapshot[start:start + content_length])
                else:
                    with os.fdopen(fd, "rb", closefd=True) as f:
                        fd = None
                        f.seek(start)
                        remaining = content_length
                        while remaining:
                            chunk = f.read(min(1024 * 1024, remaining))
                            if not chunk:
                                break
                            handler.wfile.write(chunk)
                            remaining -= len(chunk)
            except _CLIENT_DISCONNECT_ERRORS as exc:
                logging.getLogger("hermes.webui").debug(
                    "Client disconnected mid-response (%s): %s",
                    type(exc).__name__,
                    getattr(handler, "path", "?"),
                )
            except Exception as exc:
                # Post-commit fail-closed: the status line is already on the
                # wire, so ANY body-transmission error that is not a client
                # disconnect (EIO from a truncated read, a generic OSError,
                # PermissionError, ...) must still be contained here. Letting
                # it escape would reach Handler.do_GET's 500 path and emit a
                # SECOND status line after the committed 200/206, corrupting
                # the HTTP stream. Log at debug and stop — the client already
                # received its headers (and possibly a partial body).
                logging.getLogger("hermes.webui").debug(
                    "Body transmission error after commit (%s): %s",
                    type(exc).__name__,
                    getattr(handler, "path", "?"),
                )
        return True
    finally:
        _close_fd_quietly(fd)



def _normalize_tts_prosody(value, *, unit: str) -> str | None:
    if not value:
        return ""
    value = str(value).strip()
    if not re.fullmatch(r"[+-]?\d{1,3}" + re.escape(unit), value):
        return None
    amount = int(value[: -len(unit)])
    if -100 <= amount <= 100:
        return value
    return None


_TTS_PROXY_MAX_BYTES = 16 * 1024 * 1024
_TTS_LOCALHOST_HOSTS = {"127.0.0.1", "::1", "localhost"}


def _tts_addr_is_blocked(ip_str: str) -> bool:
    """Return True when IP is in a private or otherwise non-routable class.

    The explicit flags below document the concrete SSRF-risk classes, but the
    load-bearing rule is the ``not is_global`` backstop: it blocks every address
    that is not globally routable — including ranges the named flags miss, most
    notably ``100.64.0.0/10`` (RFC 6598 CGNAT, also Tailscale's default address
    space), the ``198.18.0.0/15`` benchmarking range, and any future
    non-global allocation — so a rebinding host cannot reach a victim's tailnet
    or carrier-NAT peer.
    """
    import ipaddress

    try:
        ip = ipaddress.ip_address(ip_str)
    except ValueError:
        return False
    return (
        not ip.is_global
        or ip.is_private
        or ip.is_loopback
        or ip.is_link_local
        or ip.is_reserved
        or ip.is_multicast
        or ip.is_unspecified
    )


def _tts_host_is_blocked_target(hostname: str) -> bool:
    """True if the hostname resolves to (or literally is) a private / loopback /
    link-local / reserved / multicast address — the SSRF-risk targets that an
    OpenAI-compatible TTS base_url must not be allowed to reach. Public hosts
    (a user's own hosted OpenAI-compatible server) are allowed; the explicit
    localhost-over-http dev case is handled separately by the caller."""
    import ipaddress
    import socket

    host = (hostname or "").strip().lower()
    if not host:
        return True

    # Literal IP host?
    try:
        ipaddress.ip_address(host)
        return _tts_addr_is_blocked(host)
    except ValueError:
        pass

    # DNS host: resolve and block if ANY resolved address is a blocked target
    # (defends against a hostname pointing at an internal/link-local address).
    # A DNS-resolution failure is NOT treated as an SSRF block — an unresolvable
    # host simply can't be reached (the outbound request fails naturally), and
    # failing closed here would wrongly reject legitimate public hosts that don't
    # resolve in a sandboxed/offline environment. Only a host that resolves to a
    # blocked address is rejected.
    try:
        infos = socket.getaddrinfo(host, None)
    except Exception:
        return False
    for info in infos:
        sockaddr = info[4]
        if sockaddr and _tts_addr_is_blocked(str(sockaddr[0])):
            return True
    return False


def _tts_resolve_pinned_addresses(hostname: str, port: int | None) -> list[str]:
    """Resolve once, validate the RRset, and preserve candidate dial order."""
    import socket

    host = (hostname or "").strip().lower()
    if not host:
        raise ValueError("invalid OpenAI TTS base_url host")

    try:
        infos = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM)
    except Exception as exc:
        raise ValueError("could not resolve OpenAI TTS base_url host") from exc
    pinned_hosts = []
    for info in infos:
        sockaddr = info[4]
        if not sockaddr:
            continue
        pinned_host = str(sockaddr[0])
        if _tts_addr_is_blocked(pinned_host):
            raise ValueError("resolved OpenAI TTS target is not allowed")
        pinned_hosts.append(pinned_host)
    if not pinned_hosts:
        raise ValueError("could not resolve OpenAI TTS base_url host")
    return pinned_hosts


def _tts_resolve_pinned_address(hostname: str) -> str:
    """Return the first vetted literal address for direct helper callers."""
    return _tts_resolve_pinned_addresses(hostname, None)[0]


def _normalized_openai_tts_base_url(base_url: str) -> str:
    from urllib.parse import urlsplit, urlunsplit

    raw = str(base_url or "").strip()
    parsed = urlsplit(raw)
    hostname = (parsed.hostname or "").strip().lower()
    if parsed.username or parsed.password:
        raise ValueError("invalid OpenAI base_url in config")
    if not parsed.scheme or not parsed.netloc or parsed.query or parsed.fragment:
        raise ValueError("invalid OpenAI base_url in config")
    if parsed.scheme == "https":
        # Public https hosts are allowed (a user's own OpenAI-compatible server),
        # but reject private/loopback/link-local/reserved targets to close the
        # SSRF surface (e.g. https://169.254.169.254, https://10.x internal).
        if _tts_host_is_blocked_target(hostname):
            raise ValueError("invalid OpenAI base_url in config")
    elif parsed.scheme == "http" and hostname in _TTS_LOCALHOST_HOSTS:
        # Explicit localhost-over-http dev/self-hosted case only.
        pass
    else:
        raise ValueError("invalid OpenAI base_url in config")
    path = parsed.path.rstrip("/") or ""
    return urlunsplit((parsed.scheme, parsed.netloc, path, "", ""))


def _buffer_tts_audio_response(resp, *, max_bytes: int | None = None) -> bytes:
    if max_bytes is None:
        max_bytes = _TTS_PROXY_MAX_BYTES
    headers = getattr(resp, "headers", None)
    content_type = ""
    if headers is not None:
        try:
            content_type = str(headers.get("Content-Type") or "")
        except Exception:
            content_type = ""
    if not content_type:
        try:
            info = resp.info()
            content_type = str(info.get("Content-Type") or "")
        except Exception:
            content_type = ""
    # A present Content-Type that isn't audio/* is rejected. A MISSING
    # Content-Type is tolerated: some OpenAI-compatible servers stream audio
    # bytes without setting Content-Type, and the success path defaults the
    # browser-facing type to audio/mpeg (encoded in the tests). The SSRF
    # base-url guard is the primary defense against reaching a non-audio
    # internal endpoint.
    if content_type and not content_type.lower().startswith("audio/"):
        raise ValueError("upstream returned non-audio content")
    audio_data = bytearray()
    while True:
        chunk = resp.read(65536)
        if not chunk:
            break
        audio_data.extend(chunk)
        if len(audio_data) > max_bytes:
            raise ValueError("upstream audio exceeded byte limit")
    return bytes(audio_data)

class _NoRedirectTtsHandler(HTTPRedirectHandler):
    """Refuse to follow redirects on the TTS call.

    A redirect is never a legitimate response to POST /audio/speech and can
    carry the Authorization bearer to a target that bypasses the base_url check.
    """

    def redirect_request(self, req, fp, code, msg, headers, newurl):
        raise ValueError("OpenAI TTS upstream attempted a redirect")


class _PinnedHTTPSConnection(http.client.HTTPSConnection):
    """Connect to a pinned IP while keeping Host and TLS SNI on the hostname."""

    def connect(self):
        sys.audit("http.client.connect", self, self.host, self.port)
        last_error = None
        for pinned_host in _tts_resolve_pinned_addresses(self.host, self.port):
            try:
                self.sock = _socket.create_connection(
                    (pinned_host, self.port), self.timeout, self.source_address
                )
                break
            except OSError as exc:
                last_error = exc
        else:
            if last_error is not None:
                raise last_error
            raise OSError("could not connect to any pinned OpenAI TTS target")
        try:
            self.sock.setsockopt(_socket.IPPROTO_TCP, _socket.TCP_NODELAY, 1)
        except OSError as exc:
            if exc.errno != errno.ENOPROTOOPT:
                raise

        if self._tunnel_host:
            self._tunnel()

        server_hostname = self._tunnel_host or self.host
        self.sock = self._context.wrap_socket(self.sock, server_hostname=server_hostname)


class _PinnedHTTPSHandler(HTTPSHandler):
    def https_open(self, req):
        return self.do_open(_PinnedHTTPSConnection, req, context=self._context)


def _tts_open(req, *, timeout=30, opener_factory=None):
    """Thin network seam for the TTS upstream fetch so tests can intercept it.

    Defaults to a no-redirect opener (built by opener_factory) so an upstream
    redirect can't carry the Authorization bearer to — or SSRF-bounce the
    request into — a different/private target after base-url validation passed.
    Tests monkeypatch this function (or urllib.request.urlopen) to inject a
    stub response."""
    if opener_factory is not None:
        opener = opener_factory()
        return opener.open(req, timeout=timeout)
    from urllib.request import urlopen as _urlopen
    return _urlopen(req, timeout=timeout)


def _handle_tts(handler, parsed):
    """Generate TTS audio via supported server TTS engines. POST JSON body only.

    Design note addressing deep review blocker #4 (synchronous I/O):
    The server uses ThreadingHTTPServer (see server.py:173), so each request
    already runs in its own dedicated thread. A TTS request therefore occupies
    only its own thread during Microsoft network I/O + streaming; other clients
    are unaffected. Combined with early auth, a strict per-client 2 s rate
    limit, 5000-char cap, and voice allowlist, the blocking cost is bounded and
    intentional. All audio chunks are buffered before sending so that a
    Content-Length header can be included. The 5000-char cap bounds audio to
    roughly 1-5 MB, making full buffering safe. Without Content-Length the
    HTTP/1.0 server leaves the response open until a ~31 s timeout fires, and
    the browser cannot play the blob mid-stream.
    If the HTTP layer ever moves to asyncio we can adopt edge_tts's native
    async API at that time.
    """
    text = ""
    voice = "zh-CN-XiaoxiaoNeural"
    rate_str = ""
    pitch_str = ""
    engine = "edge"  # "edge" | "elevenlabs" | "openai" | "browser" (browser is client-side only)

    if handler.command != "POST":
        from api.helpers import bad as _bad
        return _bad(handler, "POST required for /api/tts", 405)

    try:
        data = read_body(handler)
        text = (data.get("text") or "").strip()
        voice = data.get("voice") or voice
        rate_str = _normalize_tts_prosody(data.get("rate"), unit="%")
        pitch_str = _normalize_tts_prosody(data.get("pitch"), unit="Hz")
        request_engine = data.get("engine")
        if "engine" not in data or (
            isinstance(request_engine, str) and not request_engine.strip()
        ):
            persisted_engine = (load_settings() or {}).get("tts_engine")
            if isinstance(persisted_engine, str):
                persisted_engine = persisted_engine.strip().lower()
            else:
                persisted_engine = ""
            engine = (
                persisted_engine
                if persisted_engine in {"edge", "elevenlabs", "openai"}
                else "edge"
            )
        else:
            engine = (request_engine or "edge").strip().lower()
    except Exception:
        from api.helpers import bad as _bad
        return _bad(handler, "invalid request body", 400)

    if rate_str is None:
        from api.helpers import bad as _bad
        return _bad(handler, "invalid rate", 400)
    if pitch_str is None:
        from api.helpers import bad as _bad
        return _bad(handler, "invalid pitch", 400)

    if not text:
        from api.helpers import bad as _bad
        return _bad(handler, "text is required", 400)
    if len(text) > 5000:
        from api.helpers import bad as _bad
        return _bad(handler, "text too long (max 5000 characters)", 400)

    from api.auth import is_auth_enabled, parse_cookie, verify_session
    cv = None
    if is_auth_enabled():
        cv = parse_cookie(handler)
        if not (cv and verify_session(cv)):
            from api.helpers import bad as _bad
            return _bad(handler, "unauthorized", 401)

    # High-quality per-client rate limiting for TTS.
    if not hasattr(_handle_tts, "_tts_limiter"):
        import time as _time, threading as _threading
        class _TtsRateLimiter:
            def __init__(self, window_seconds=2.0, prune_interval=50):
                self.window = window_seconds
                self.prune_interval = prune_interval
                self._hits = {}
                self._lock = _threading.Lock()
                self._checks = 0

            def _get_client_key(self, h):
                trust_proxy = os.getenv("HERMES_WEBUI_TRUST_FORWARDED_FOR", "").strip().lower()
                if trust_proxy in ("1", "true", "yes", "on"):
                    for hdr in ("X-Forwarded-For", "X-Real-IP", "Forwarded"):
                        val = h.headers.get(hdr)
                        if val:
                            ip = val.split(",")[0].strip().split(";")[0].strip()
                            if ip:
                                return ip
                return getattr(h, "client_address", ("unknown",))[0]

            def check(self, handler, session_cookie=None):
                key = self._get_client_key(handler)
                if session_cookie and "." in str(session_cookie):
                    key = str(session_cookie).split(".", 1)[0]
                now = _time.time()
                with self._lock:
                    self._checks += 1
                    if self._checks % self.prune_interval == 0:
                        cutoff = now - (self.window * 10)
                        self._hits = {k: v for k, v in self._hits.items() if v > cutoff}
                    last = self._hits.get(key, 0)
                    if now - last < self.window:
                        return False
                    self._hits[key] = now
                    return True

        _handle_tts._tts_limiter = _TtsRateLimiter(window_seconds=2.0)

    limiter = _handle_tts._tts_limiter
    if not limiter.check(handler, cv):
        logger.warning("TTS rate limit hit for client=%s", limiter._get_client_key(handler))
        from api.helpers import bad as _bad
        return _bad(handler, "rate limit exceeded — please wait", 429)

    # ── ElevenLabs TTS ──────────────────────────────────────────────────
    if engine == "elevenlabs":
        api_key = os.getenv("ELEVENLABS_API_KEY", "").strip()
        if not api_key:
            # Fall back to reading from Hermes .env file
            try:
                from api.onboarding import _load_env_file
                from api.profiles import get_active_hermes_home
                api_key = _load_env_file(get_active_hermes_home() / ".env").get("ELEVENLABS_API_KEY", "")
            except Exception:
                pass
        if not api_key:
            from api.helpers import bad as _bad
            return _bad(handler, "ELEVENLABS_API_KEY not configured", 503)

        # Resolve voice_id from Hermes config.yaml → env fallback
        voice_id = "pNInz6obpgDQGcFmaJgB"  # Adam (same default as hermes-agent config.yaml)
        model_id = "eleven_multilingual_v2"
        try:
            from api.config import get_config
            tts_cfg = (get_config() or {}).get("tts", {})
            if isinstance(tts_cfg, dict):
                el_cfg = tts_cfg.get("elevenlabs", {})
                if isinstance(el_cfg, dict):
                    voice_id = el_cfg.get("voice_id", voice_id)
                    model_id = el_cfg.get("model", model_id) or el_cfg.get("model_id", model_id)
                    # ^ treat empty string as "not set" — fall through to default
        except Exception:
            pass  # fall back to defaults

        # Validate voice_id is a safe path segment (no traversal)
        # fullmatch (not match) so a trailing newline can't slip past the `$`
        # anchor — defense-in-depth on the config-derived voice_id before it
        # goes into the request URL (#3510 review).
        if not re.fullmatch(r'[A-Za-z0-9_-]+', voice_id):
            from api.helpers import bad as _bad
            return _bad(handler, "invalid voice_id in config", 400)

        url = f"https://api.elevenlabs.io/v1/text-to-speech/{voice_id}/stream?output_format=mp3_44100_128"
        req_body = json.dumps({
            "text": text,
            "model_id": model_id,
            "voice_settings": {"stability": 0.5, "similarity_boost": 0.75},
        }).encode("utf-8")

        req = Request(url, data=req_body, headers={
            "xi-api-key": api_key,
            "Content-Type": "application/json",
            "Accept": "audio/mpeg",
        })

        # Buffer the full response before sending first byte.
        # The streaming endpoint is designed for chunked delivery, but urllib's
        # chunked-read path adds per-chunk overhead that dominates short TTS
        # payloads. A hard cap keeps the buffered path bounded even if the
        # upstream misbehaves.
        try:
            with _tts_open(req, timeout=30, opener_factory=lambda: build_opener(ProxyHandler({}), _NoRedirectTtsHandler())) as resp:
                audio_data = _buffer_tts_audio_response(resp)
        except ValueError:
            logger.warning("ElevenLabs TTS rejected an invalid upstream response", exc_info=True)
            from api.helpers import bad as _bad
            return _bad(handler, "ElevenLabs TTS generation failed", 502)
        except Exception:
            logger.exception("ElevenLabs TTS generation failed")
            from api.helpers import bad as _bad
            return _bad(handler, "ElevenLabs TTS generation failed", 500)

        handler.send_response(200)
        handler.send_header("Content-Type", "audio/mpeg")
        handler.send_header("Cache-Control", "no-store")
        handler.send_header("Content-Length", str(len(audio_data)))
        handler.end_headers()
        try:
            handler.wfile.write(audio_data)
        except (BrokenPipeError, ConnectionResetError):
            pass
        return True

    # ── OpenAI-compatible TTS ──────────────────────────────────────────
    if engine == "openai":
        api_key = os.getenv("VOICE_TOOLS_OPENAI_KEY", "").strip()
        if not api_key:
            api_key = os.getenv("OPENAI_API_KEY", "").strip()
        if not api_key:
            try:
                from api.onboarding import _load_env_file
                from api.profiles import get_active_hermes_home
                env_cfg = _load_env_file(get_active_hermes_home() / ".env")
                api_key = env_cfg.get("VOICE_TOOLS_OPENAI_KEY", "") or env_cfg.get("OPENAI_API_KEY", "")
            except Exception:
                pass
        if not api_key:
            from api.helpers import bad as _bad
            return _bad(handler, "OpenAI API key not configured", 503)

        from urllib.parse import urlunsplit as _urlunsplit

        base_url = _urlunsplit(("https", "api.openai.com", "/v1", "", ""))
        model = "gpt-4o-mini-tts"
        oai_voice = "alloy"
        try:
            from api.config import get_config
            tts_cfg = (get_config() or {}).get("tts", {})
            if isinstance(tts_cfg, dict):
                oai_cfg = tts_cfg.get("openai", {})
                if isinstance(oai_cfg, dict):
                    base_url = _normalized_openai_tts_base_url(oai_cfg.get("base_url") or base_url)
                    model = oai_cfg.get("model") or model
                    oai_voice = oai_cfg.get("voice") or oai_voice
                else:
                    base_url = _normalized_openai_tts_base_url(base_url)
            else:
                base_url = _normalized_openai_tts_base_url(base_url)
        except ValueError:
            from api.helpers import bad as _bad
            return _bad(handler, "invalid OpenAI base_url in config", 400)
        except Exception:
            pass

        url = f"{base_url}/audio/speech"
        req_body = json.dumps({
            "model": model,
            "input": text,
            "voice": oai_voice,
        }).encode("utf-8")

        req = Request(url, data=req_body, headers={
            "Authorization": f"Bearer {api_key}",
            "Content-Type": "application/json",
            "Accept": "audio/mpeg",
        })

        # Use a pinned HTTPS opener so the resolved address is the one that gets
        # dialed. Keep the no-redirect handler in the same chain to block
        # bearer leaks and SSRF bounce redirects after hostname validation.
        try:
            with _tts_open(req, timeout=30, opener_factory=lambda: build_opener(ProxyHandler({}), _NoRedirectTtsHandler(), _PinnedHTTPSHandler())) as resp:
                audio_data = _buffer_tts_audio_response(resp)
        except ValueError:
            logger.warning("OpenAI TTS rejected an invalid upstream response", exc_info=True)
            from api.helpers import bad as _bad
            return _bad(handler, "OpenAI TTS generation failed", 502)
        except Exception:
            logger.exception("OpenAI TTS generation failed")
            from api.helpers import bad as _bad
            return _bad(handler, "OpenAI TTS generation failed", 500)

        handler.send_response(200)
        handler.send_header("Content-Type", "audio/mpeg")
        handler.send_header("Cache-Control", "no-store")
        handler.send_header("Content-Length", str(len(audio_data)))
        handler.end_headers()
        try:
            handler.wfile.write(audio_data)
        except (BrokenPipeError, ConnectionResetError):
            pass
        return True

    # ── Edge TTS ────────────────────────────────────────────────────────
    allowed = {
        "zh-CN-XiaoxiaoNeural", "zh-CN-XiaoyiNeural", "zh-CN-YunxiNeural",
        "zh-CN-YunjianNeural", "zh-CN-YunyangNeural",
        "en-US-AriaNeural", "en-US-GuyNeural",
        "fr-CA-AntoineNeural", "fr-CA-JeanNeural",
        "fr-CA-SylvieNeural", "fr-CA-ThierryNeural",
        "fr-FR-DeniseNeural", "fr-FR-EloiseNeural", "fr-FR-HenriNeural",
        "id-ID-GadisNeural",
    }
    if voice not in allowed:
        from api.helpers import bad as _bad
        return _bad(handler, "invalid voice", 400)

    try:
        try:
            import edge_tts
        except ImportError:
            from api.helpers import bad as _bad
            return _bad(handler, "Edge TTS engine not installed on the server. Install it with: pip install edge-tts", 503)

        kwargs = {}
        if rate_str:
            kwargs["rate"] = rate_str
        if pitch_str:
            kwargs["pitch"] = pitch_str

        comm = edge_tts.Communicate(text, voice, **kwargs)

        # Buffer all audio chunks before responding so Content-Length is known.
        # Without it the HTTP/1.0 server holds the connection open until a ~31 s
        # timeout fires and the browser cannot play the resulting blob.
        audio_buf = bytearray()
        for chunk in comm.stream_sync():
            if chunk.get("type") == "audio" and chunk.get("data"):
                audio_buf.extend(chunk["data"])

        if not audio_buf:
            from api.helpers import bad as _bad
            return _bad(handler, "TTS produced no audio", 500)

        handler.send_response(200)
        handler.send_header("Content-Type", "audio/mpeg")
        handler.send_header("Content-Length", str(len(audio_buf)))
        handler.send_header("Cache-Control", "no-store")
        handler.end_headers()
        try:
            handler.wfile.write(audio_buf)
        except (BrokenPipeError, ConnectionResetError):
            pass
        return True

    except BrokenPipeError:
        return True
    except Exception:
        logger.exception("Edge TTS generation failed")
        from api.helpers import bad as _bad
        return _bad(handler, "TTS generation failed", 500)
def _html_preview_with_blank_base(raw: bytes) -> bytes:
    base = '<base target="_blank">'
    text = raw.decode("utf-8", errors="replace")
    if re.search(r"<head(?:\s[^>]*)?>", text, flags=re.IGNORECASE):
        text = re.sub(r"(<head\b[^>]*>)", r"\1" + base, text, count=1, flags=re.IGNORECASE)
    elif re.search(r"<!doctype[^>]*>", text, flags=re.IGNORECASE):
        text = re.sub(
            r"(<!doctype[^>]*>)",
            r"\1<head>" + base + "</head>",
            text,
            count=1,
            flags=re.IGNORECASE,
        )
    else:
        text = "<head>" + base + "</head>" + text
    return text.encode("utf-8")


def _serve_inline_html_preview(handler, target: Path, cache_control: str, *, csp: str, anchor_root: Path | None = None):
    """Serve sandboxed workspace HTML preview with links targeting a new tab."""
    fd = None
    try:
        fd = _open_file_read_fd(target, anchor_root)
        with os.fdopen(fd, "rb", closefd=True) as f:
            fd = None
            body = _html_preview_with_blank_base(f.read())
    except PermissionError:
        return bad(handler, "Permission denied", 403)
    except FileNotFoundError:
        return j(handler, {"error": "not found"}, status=404)
    except ValueError as e:
        return bad(handler, _sanitize_error(e), 403)
    except Exception:
        return bad(handler, "Could not read file", 500)
    finally:
        if fd is not None:
            try:
                os.close(fd)
            except OSError:
                pass

    handler.send_response(200)
    handler.send_header("Content-Type", "text/html; charset=utf-8")
    handler.send_header("Content-Length", str(len(body)))
    handler.send_header("Accept-Ranges", "none")
    handler.send_header("Cache-Control", cache_control)
    handler.send_header("Content-Disposition", _content_disposition_value("inline", target.name))
    handler.send_header("Content-Security-Policy", csp)
    handler.send_header("X-Content-Type-Options", "nosniff")
    handler.send_header("Referrer-Policy", "same-origin")
    handler.send_header(
        "Permissions-Policy",
        "camera=(), microphone=(self), geolocation=(), clipboard-write=(self)",
    )
    handler.end_headers()
    handler.wfile.write(body)
    return True


_MEDIA_TOKEN_RE = re.compile(r"MEDIA:([^\s\)\]]+)")
# #7680 re-gate (9/22): two-pass scan.
#   1. `` `MEDIA:path` `` (backtick-wrapped, inline-code form) → strip
#      the wrapping backticks so the bare-token pass below sees a
#      plain ``MEDIA:path`` and the closing backtick is not consumed
#      as part of the path.
#   2. ``MEDIA:[^\s\)\]]+`` (bare, no backtick in the exclusion
#      class) so a filename that legally contains a backtick
#      (``report`final.png``) is captured in full instead of being
#      truncated at the first backtick.
_BACKTICK_MEDIA_RE = re.compile(r"`MEDIA:([^`\s]+)`")


def _message_content_text(content) -> str:
    if isinstance(content, list):
        parts = []
        for part in content:
            if isinstance(part, dict):
                parts.append(str(part.get("text") or ""))
            else:
                parts.append(str(part or ""))
        return "\n".join(parts)
    return str(content or "")


def _session_media_token_allows_path(sid: str, target: Path, allowed_mimes: set[str]) -> bool:
    """Allow exact safe MEDIA: paths already present in the requested session."""
    sid = str(sid or "").strip()
    if not sid:
        return False
    mime = MIME_MAP.get(target.suffix.lower(), "application/octet-stream")
    if mime not in allowed_mimes:
        return False
    try:
        target_resolved = target.resolve()
    except Exception:
        return False
    try:
        session = get_session(sid)
    except Exception:
        return False

    for message in getattr(session, "messages", []) or []:
        if not isinstance(message, dict):
            continue
        # Only honor MEDIA: tokens that the assistant/tool emitted. User-authored
        # content cannot mint allow-list entries even if it contains a MEDIA:
        # token — keeps the implicit threat model (assistant-emitted artifacts
        # only) explicit.
        role = str(message.get("role") or "").strip().lower()
        if role == "user":
            continue
        # #7565: also inspect typed public assistant commentary carried in
        # ``codex_message_items`` (Agent phase: "commentary"). The
        # concatenated text below is the union of the existing
        # top-level extraction and the new commentary-only helper, so
        # the existing exact-path, owning-session, safe-MIME, URL,
        # hard-denied, and symlink guards are unchanged. The
        # commentary helper is fail-closed (outer role must be
        # assistant; item type/role/phase all constrained; only
        # textual output_text parts are read).
        from api.media_snapshots import codex_commentary_text
        text = "\n".join(
            fragment for fragment in (
                _message_content_text(message.get("content")),
                codex_commentary_text(message),
            ) if fragment
        )
        if "MEDIA:" not in text:
            continue
        # #7680 re-gate: strip backtick wrappers first so the bare
        # class below captures the full path even when the filename
        # itself contains a backtick.
        text = _BACKTICK_MEDIA_RE.sub(lambda m: f"MEDIA:{m.group(1)}", text)
        for ref in _MEDIA_TOKEN_RE.findall(text):
            if "://" in ref:
                continue
            try:
                if Path(ref).expanduser().resolve() == target_resolved:
                    return True
            except Exception:
                continue
    return False


def _session_media_token_allows_image_path(sid: str, target: Path, image_mimes: set[str]) -> bool:
    """Backward-compatible image-only wrapper for existing callers/tests."""
    return _session_media_token_allows_path(sid, target, image_mimes)


def _path_is_within_root(child: Path, root: Path) -> bool:
    """Return True when ``child`` is inside ``root`` without crashing on Windows drives."""
    try:
        return os.path.commonpath([str(child), str(root)]) == str(root)
    except ValueError:
        return False


def _media_deny_reason(target: Path) -> str | None:
    """Return a reason string when ``target`` must be hard-denied, else None.

    The ``/api/media`` #3234 state/profile deny model, extracted so the
    snapshot CAPTURE side (api/media_snapshots.media_capture_allowed) shares
    the EXACT same predicate — anything the serve path refuses is never
    captured in the first place (#6979 Round 2 MUST-FIX 1 deny parity).

    Model: the ACTIVE WORKSPACE is a legitimate-media carve-out — the user is
    entitled to their own workspace files (that is also how the workspace file
    browser reaches them), even when a workspace happens to live under a
    Hermes root. The deny rules target Hermes's OWN internal state, which lives
    OUTSIDE any workspace. So: if the target is inside the active workspace, it
    is never denied here; otherwise we deny known secret/config basenames and
    the internal state subdirectories across every Hermes root the allowlist
    accepts (active-profile HERMES_HOME, base ~/.hermes, the api.profiles
    default home, and STATE_DIR — which also defends sibling profiles).
    """
    import os as _os

    _HOME = Path(_os.path.expanduser("~"))
    _HERMES_HOME = Path(_os.getenv("HERMES_HOME", str(_HOME / ".hermes"))).expanduser()

    _DENY_FILENAMES = {
        "settings.json", "state.db", "state.db-wal", "state.db-shm",
        "auth.json", "auth.lock", "config.yaml", "config.yml", ".env",
        ".signing_key", ".pbkdf2_key", ".sessions.json",
        "google_token.json", "google_client_secret.json",
        "gateway_state.json", "channel_directory.json", "jobs.json",
        "passkeys.json", ".passkey_challenges.json", ".login_attempts.json",
    }
    # Internal state subdirs that are sensitive in their entirety. NOTE:
    # `profiles` is intentionally NOT here — it is a container of profile roots,
    # each of which has its own legitimate workspace/. We instead enumerate each
    # named-profile root below and deny ITS state subdirs, so a sibling profile's
    # secrets are blocked without 403-ing a named-profile workspace. (#3234.)
    _DENY_SUBDIRS = (
        "sessions", "memories", "cron", "logs",
        "checkpoints", "backups",
        # Content-addressed media snapshots (api/media_snapshots.py) are an
        # internal store: digest bytes are only reachable through the validated
        # `snap=` parameter, never as a bare `path=` request. (#media-snapshots)
        "media_snapshots",
    )
    _state_dir = None
    try:
        from api.config import STATE_DIR as _STATE_DIR
        _state_dir = Path(_STATE_DIR).resolve()
    except Exception:
        _state_dir = None
    _base_hermes_home = None
    try:
        from api.profiles import _DEFAULT_HERMES_HOME as _BASE_HH
        _base_hermes_home = Path(_BASE_HH).resolve()
    except Exception:
        _base_hermes_home = None
    _hermes_roots = []
    for _r in (
        _HERMES_HOME.resolve(),
        (_HOME / ".hermes").resolve(),
        _base_hermes_home,
        _state_dir,
    ):
        if _r is not None and _r not in _hermes_roots:
            _hermes_roots.append(_r)
    # Enumerate named-profile roots (<root>/profiles/<name>) and treat each as a
    # Hermes root in its own right, so a sibling/other profile's sensitive subdirs
    # + secret files are denied — WITHOUT denying the whole `profiles` container
    # (which would block a legit named-profile workspace at
    # <root>/profiles/<name>/workspace/). (Codex review #3234.)
    _profile_roots = []
    for _root in list(_hermes_roots):
        _profiles_dir = (_root / "profiles")
        try:
            if _profiles_dir.is_dir():
                for _pchild in _profiles_dir.iterdir():
                    if _pchild.is_dir():
                        _pr = _pchild.resolve()
                        if _pr not in _hermes_roots and _pr not in _profile_roots:
                            _profile_roots.append(_pr)
        except OSError:
            pass
    _hermes_roots.extend(_profile_roots)

    # Case-insensitive path helpers so STATE.DB / Sessions/ casing variants
    # cannot bypass the deny on macOS/Windows filesystems (Codex review #3234).
    def _norm(p):
        return os.path.normcase(str(Path(p).resolve())).casefold()
    def _within_ci(child, root):
        try:
            c, r = _norm(child), _norm(root)
            return os.path.commonpath([c, r]) == r
        except (ValueError, OSError):
            return False
    def _equal_ci(a, b):
        try:
            return _norm(a) == _norm(b)
        except (ValueError, OSError):
            return False

    # State-subdir deny set: each DENY_SUBDIR directly under any Hermes root
    # (which includes STATE_DIR — so STATE_DIR/sessions, STATE_DIR/memories,
    # etc. are covered). These ALWAYS apply — even to a file under the active
    # workspace — so a workspace pointed at (or overlapping) a state dir cannot
    # expose sessions/memories/profiles/etc. We do NOT deny STATE_DIR itself
    # wholesale: the default workspace lives at STATE_DIR/workspace, and that is
    # legitimate user media — direct sensitive files there are still caught by
    # the filename denies below. (Codex review #3234.)
    _deny_dirs = []
    for _root in _hermes_roots:
        for _sub in _DENY_SUBDIRS:
            _deny_dirs.append((_root / _sub).resolve())
        # Per-profile WebUI state lives at <root>/webui_state (api/workspace.py),
        # so its state subdirs (<root>/webui_state/sessions, etc.) must be denied
        # too — they are NOT direct children of <root>. (Codex review #3234.)
        _ws_state = (_root / "webui_state")
        for _sub in _DENY_SUBDIRS:
            _deny_dirs.append((_ws_state / _sub).resolve())
    # The configured media-snapshot store root itself: blobs are internal and
    # only reachable through the validated `snap=` parameter on an authorized
    # path, so a bare `path=` request at or below the store is rejected
    # REGARDLESS of the store's configured name/location
    # (HERMES_WEBUI_MEDIA_SNAPSHOT_DIR may point anywhere, e.g. /tmp/custom-name;
    # the literal "media_snapshots" entry above only covers the default layout).
    # (#6979 Round 2 MUST-FIX 2.)
    try:
        from api.media_snapshots import get_snapshot_dir
        _snap_store = get_snapshot_dir().resolve()
    except Exception:
        _snap_store = None
    if _snap_store is not None and _within_ci(target, _snap_store):
        return "media snapshot store is internal"
    _deny_names_ci = {n.casefold() for n in _DENY_FILENAMES}

    # Active-workspace carve-out: a file inside a genuine PROJECT workspace is
    # the user's own content, so the secret/config FILENAME denies are relaxed
    # for it. The carve-out is DISABLED when the workspace is a broad/internal
    # location ($HOME, a Hermes root itself, an ANCESTOR of a Hermes root, a
    # */profiles dir, a named-profile root, or a state subdir) — honoring those
    # would re-open the disclosure. A workspace that is a proper DESCENDANT of a
    # Hermes root (e.g. STATE_DIR/workspace) is still a legit project workspace
    # and keeps the carve-out. The dir-based denies above are NOT relaxed.
    _active_workspace = None
    try:
        from api.workspace import get_last_workspace
        _aw = Path(get_last_workspace()).resolve()
        if _aw.is_dir():
            _active_workspace = _aw
    except Exception:
        _active_workspace = None

    def _workspace_is_safe_carveout(ws):
        if ws is None:
            return False
        if _equal_ci(ws, _HOME):
            return False
        for _root in _hermes_roots:
            # ws IS a root, or ws is an ANCESTOR of a root → unsafe. (A proper
            # descendant of a root is fine — that's a normal project workspace.)
            if _equal_ci(ws, _root) or _within_ci(_root, ws):
                return False
        if ws.name == "profiles" or ws.parent.name == "profiles":
            return False
        if ws.name in _DENY_SUBDIRS:
            return False
        return True

    _in_active_workspace = (
        _active_workspace is not None
        and _workspace_is_safe_carveout(_active_workspace)
        and _within_ci(target, _active_workspace)
    )

    # Dir-based denies always fire (even inside the active workspace).
    if any(_within_ci(target, d) for d in _deny_dirs):
        return "denied state subdir"
    # Filename-based denies fire for files under a Hermes root, UNLESS the file
    # is inside a genuine project workspace (carve-out).
    if not _in_active_workspace:
        _under_hermes_root = any(_within_ci(target, _root) for _root in _hermes_roots)
        _name_cf = target.name.casefold()
        # Exact secret/state basenames, plus atomic-write temp files for those
        # (api/auth.py and api/passkeys.py write via a `tmp*.<name>.tmp` / `tmp*.tmp`
        # sidecar then rename) — deny those suffixes too so a momentary temp file
        # cannot be fetched. (Codex review #3234.)
        _deny_tmp_suffixes = (".sessions.tmp", ".login_attempts.tmp",
                              ".passkeys.tmp", ".passkey_challenges.tmp")
        if _under_hermes_root and (
            _name_cf in _deny_names_ci
            or _name_cf.endswith(_deny_tmp_suffixes)
        ):
            return "denied state filename"
    return None


def _handle_media(handler, parsed):
    """Serve a local file by absolute path for inline display in the chat.

    Security:
    - Path must resolve to an allowed root (hermes home, /tmp, common dirs)
    - Auth-gated when auth is enabled
    - Safe preview MIME types can render inline when requested; SVG always downloads
    - SVG always served as attachment (XSS risk)
    - No path traversal: resolved path must stay within an allowed root
    - Additional roots can be added via MEDIA_ALLOWED_ROOTS env var
      (os.pathsep-separated list of absolute paths; ":" on POSIX, ";" on Windows)
    """
    import os as _os
    from api.auth import is_auth_enabled, parse_cookie, verify_session
    _HOME = Path(_os.path.expanduser("~"))
    _HERMES_HOME = Path(_os.getenv("HERMES_HOME", str(_HOME / ".hermes"))).expanduser()

    # Auth check
    if is_auth_enabled():
        cv = parse_cookie(handler)
        if not (cv and verify_session(cv)):
            body = b'{"error":"Authentication required"}'
            handler.send_response(401)
            handler.send_header("Content-Type", "application/json")
            handler.send_header("Content-Length", str(len(body)))
            handler.end_headers()
            handler.wfile.write(body)
            return

    qs = parse_qs(parsed.query)
    raw_path = qs.get("path", [""])[0].strip()
    if not raw_path:
        return bad(handler, "path parameter required", 400)

    # Resolve the path and check it is within an allowed root
    try:
        target = Path(raw_path).resolve()
    except Exception:
        return bad(handler, "Invalid path", 400)

    # Allowed roots: hermes home, /tmp, and active workspace.
    # Intentionally NOT the entire home dir — that would expose ~/.ssh,
    # ~/.aws, browser profiles, etc. to any authenticated user.
    allowed_roots = [
        _HERMES_HOME.resolve(),
        Path("/tmp").resolve(),
        (_HOME / ".hermes").resolve(),
    ]
    # Also allow the active workspace directory (where screenshots land)
    try:
        from api.workspace import get_last_workspace
        ws = Path(get_last_workspace()).resolve()
        if ws.is_dir():
            allowed_roots.append(ws)
    except Exception:
        pass

    # Also allow additional roots from MEDIA_ALLOWED_ROOTS env var
    # (os.pathsep-separated list; ":" on POSIX, ";" on Windows).
    extra_roots = _os.environ.get("MEDIA_ALLOWED_ROOTS", "").strip()
    if extra_roots:
        for root in extra_roots.split(_os.pathsep):
            root = root.strip()
            if root:
                try:
                    rp = Path(root).resolve()
                    if rp.is_dir():
                        allowed_roots.append(rp)
                except Exception:
                    pass

    _INLINE_IMAGE_TYPES = {
        "image/png", "image/jpeg", "image/gif", "image/webp",
        "image/x-icon", "image/bmp",
    }
    within_allowed = any(
        _path_is_within_root(target, root)
        for root in allowed_roots
        if root.exists()
    )
    _AUDIO_VIDEO_PDF_TYPES = {
        "audio/mpeg", "audio/wav", "audio/x-wav", "audio/mp4", "audio/aac",
        "audio/ogg", "audio/opus", "audio/flac",
        "video/mp4", "video/quicktime", "video/webm", "video/ogg",
        "application/pdf",
    }
    # Archives are download-only: never added to the inline-preview sets below,
    # so they always get Content-Disposition: attachment.
    _ARCHIVE_TYPES = {"application/zip"}
    _SESSION_MEDIA_TOKEN_TYPES = (
        _INLINE_IMAGE_TYPES | _AUDIO_VIDEO_PDF_TYPES | _ARCHIVE_TYPES | {"text/html"}
    )
    session_media_allowed = _session_media_token_allows_path(
        qs.get("session_id", [""])[0],
        target,
        _SESSION_MEDIA_TOKEN_TYPES,
    )

    # ── #3234: hard-deny Hermes's own state + secret/config files ────────────
    # The allowlist above grants the whole Hermes home (and base ~/.hermes), so
    # an authenticated session rendering attacker-influenced agent output that
    # emits a file:// / MEDIA: link to a state/secret file could fetch it
    # through /api/media. This guard runs BEFORE the allow/serve decision so it
    # covers every entry path (bare file:// URLs, markdown anchors, MEDIA:
    # tokens, and session-token grants).
    #
    # The predicate is SHARED with snapshot capture
    # (api/media_snapshots.media_capture_allowed) so capture and serve can
    # never diverge on what is denied — anything denied here is never
    # snapshotted in the first place. (#6979 Round 2 MUST-FIX 1.)
    deny_reason = _media_deny_reason(target)
    if deny_reason:
        return bad(handler, "Path not in allowed location", 403)
    # ── end #3234 deny ───────────────────────────────────────────────────────

    if not within_allowed and not session_media_allowed:
        return bad(handler, "Path not in allowed location", 403)

    # Determine MIME type from the requested path's extension. Computed BEFORE
    # the existence check because the requested file may have been overwritten
    # or deleted while its message-level snapshot still exists below.
    ext = target.suffix.lower()
    mime = MIME_MAP.get(ext, "application/octet-stream")

    # Only serve safe media/PDF types inline when explicitly requested. HTML is
    # allowed inline only with a CSP sandbox so "open full page" can work without
    # granting same-origin access to the WebUI. SVG is always a download (XSS risk).
    _INLINE_PREVIEW_TYPES = _INLINE_IMAGE_TYPES | _AUDIO_VIDEO_PDF_TYPES
    _DOWNLOAD_TYPES = {"image/svg+xml"}  # SVG: XSS risk, force download
    inline_preview = qs.get("inline", [""])[0] == "1"
    html_inline_ok = inline_preview and mime == "text/html"
    disposition = "inline" if (
        mime not in _DOWNLOAD_TYPES and (
            mime in _INLINE_IMAGE_TYPES or (inline_preview and mime in _INLINE_PREVIEW_TYPES)
            or html_inline_ok
        )
    ) else "attachment"
    # _serve_file_bytes sends Content-Security-Policy when csp is set.
    csp = "sandbox allow-scripts" if html_inline_ok else None

    # ── Message-level snapshot serving (?snap=<sha256>) ─────────────────────
    # Historical chat previews carry a content-addressed snapshot digest of the
    # file as it existed when the message settled (see api/media_snapshots.py).
    # Serving the frozen bytes instead of the live file means an in-place
    # overwrite (same filename) no longer rewrites old previews — the user can
    # still compare old vs new. The allow/deny checks above still gate the
    # request: `snap` only selects WHICH bytes to serve for an already-
    # authorized path; it never grants access to a path that would be denied
    # without it. A missing/evicted snapshot falls back to the live file.
    snap_digest = qs.get("snap", [""])[0].strip().lower()
    snapshot_file = None
    snap_dir = None
    if snap_digest:
        from api.media_snapshots import (
            get_snapshot_dir,
            is_valid_digest,
            snapshot_path_for_digest,
            snapshot_servable_for_path,
        )

        snap_dir = get_snapshot_dir().resolve()
        if is_valid_digest(snap_digest):
            snapshot_file = snapshot_path_for_digest(snap_digest)
            # Server-owned source-path binding (#6979 Round 2 MUST-FIX 1): a
            # digest may only be served back for the EXACT canonical path it
            # was captured from. Replaying a digest through a different
            # (allowed) path must not leak the stored bytes — treat it as an
            # invalid snapshot and fall back to the live file (or 404 when the
            # live file is absent).
            if snapshot_file is not None and not snapshot_servable_for_path(snap_digest, target):
                snapshot_file = None
    if snapshot_file is not None:
        # Content-addressed and immutable: the digest IS the SHA-256 of the
        # exact bytes, so the browser may cache forever and never revalidate.
        # The blob is opened ANCHORED inside the store root (no-follow), and
        # the store root itself is deny-listed from bare path= fetches above
        # (#6979 Round 2 MUST-FIX 2).
        return _serve_file_bytes(
            handler,
            snapshot_file,
            mime,
            disposition,
            "private, max-age=31536000, immutable",
            csp=csp,
            download_name=target.name,
            anchor_root=snap_dir,
        )

    if not target.exists() or not target.is_file():
        return j(handler, {"error": "not found"}, status=404)

    # HTML inline previews change frequently (agent edits + re-renders).
    # Use no-store so the browser always fetches fresh content, avoiding stale
    # previews that require a manual full-page refresh to update.
    # All other media (images, audio, video, PDF) use private, no-cache + ETag
    # revalidation (see _serve_file_bytes): the browser may cache, but must
    # revalidate on every use, so a file replaced in place (same name) is
    # picked up immediately while unchanged files still short-circuit with 304.
    # The `private` directive keeps per-user/per-session media out of shared
    # intermediary caches.
    if mime == "text/html":
        cache_control = "no-store"
    else:
        cache_control = "private, no-cache"
    return _serve_file_bytes(handler, target, mime, disposition, cache_control, csp=csp)


def _file_raw_target(session, sid: str, rel: str) -> tuple[Path, Path] | None:
    """Resolve /api/file/raw paths from the workspace or this session's uploads."""
    workspace_root = Path(session.workspace)
    try:
        target = safe_resolve(workspace_root, rel)
    except ValueError:
        target = None
    if target and target.exists() and target.is_file():
        return workspace_root, target

    # Chat uploads now live in a per-session attachment inbox outside the
    # workspace. Keep the public URL stable while scoping fallback lookup to
    # the requesting session's own attachment directory.
    try:
        from api.upload import _session_attachment_dir

        attachment_root = _session_attachment_dir(sid)
        attachment_target = safe_resolve(attachment_root, rel)
    except Exception:
        return None
    if attachment_target.exists() and attachment_target.is_file():
        return attachment_root, attachment_target
    return None


# ─── /api/folder/download ───────────────────────────────────────────────────
# Configurable caps. Match the HERMES_WEBUI_MAX_UPLOAD_MB style used elsewhere
# (api/config.py) so operators have one consistent env-var convention.
# Bound on per-request wall-clock and bandwidth, not RSS. The zip streams
# straight into handler.wfile, so peak memory is the per-file read buffer
# inside zipfile, not the cap value.
def _folder_zip_max_bytes() -> int:
    try:
        mb = int(os.getenv("HERMES_WEBUI_FOLDER_ZIP_MAX_MB", "1024"))
    except ValueError:
        mb = 1024
    return max(1, mb) * 1024 * 1024


def _folder_zip_max_files() -> int:
    try:
        return max(1, int(os.getenv("HERMES_WEBUI_FOLDER_ZIP_MAX_FILES", "50000")))
    except ValueError:
        return 50000


def _folder_download_collect(target: Path, workspace_root: Path,
                              max_bytes: int, max_files: int):
    """Walk target dir; return (files, total_bytes, hit_limit_reason_or_None).

    files is a list of (filesystem_path, archive_name) tuples. Each filesystem
    path is reopened through the workspace anchor when streamed into the ZIP.
    Symlinks escaping the workspace are skipped.
    """
    import os as _os
    files = []
    total_bytes = 0
    for root, dirs, names in _os.walk(target, followlinks=False):
        root_path = Path(root)
        try:
            if not root_path.resolve().is_relative_to(workspace_root):
                dirs[:] = []
                continue
        except (ValueError, OSError):
            dirs[:] = []
            continue
        for name in names:
            fp = root_path / name
            if fp.is_symlink():
                try:
                    if not fp.resolve().is_relative_to(workspace_root):
                        continue
                except (ValueError, OSError):
                    continue
            try:
                size = fp.stat().st_size
            except OSError:
                continue
            if len(files) >= max_files:
                return files, total_bytes, "max_files"
            if total_bytes + size > max_bytes:
                return files, total_bytes, "max_bytes"
            try:
                arcname = fp.relative_to(target)
            except ValueError:
                continue
            files.append((fp, str(arcname)))
            total_bytes += size
    return files, total_bytes, None


def _handle_folder_download(handler, parsed):
    """GET /api/folder/download?session_id=...&path=...

    Streams a zip of <session.workspace>/<path>. Symlinks escaping the
    workspace are skipped. Empty folders return an empty (valid) zip.
    Respects HERMES_WEBUI_FOLDER_ZIP_MAX_MB and HERMES_WEBUI_FOLDER_ZIP_MAX_FILES.
    Pre-flights the walk so size/count failures return a clean 413 with JSON
    body BEFORE any zip bytes are sent.
    """
    import zipfile
    from urllib.parse import parse_qs

    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)

    rel = qs.get("path", [""])[0]
    try:
        target = safe_resolve(Path(s.workspace), rel)
    except ValueError:
        return bad(handler, "invalid path", 400)
    if not target.exists():
        return j(handler, {"error": "not found"}, status=404)
    if not target.is_dir():
        return bad(handler, "path must be a directory; use /api/file/raw for single files", 400)

    workspace_root = Path(s.workspace).resolve()
    max_bytes = _folder_zip_max_bytes()
    max_files = _folder_zip_max_files()

    files, total_bytes, limit_hit = _folder_download_collect(
        target, workspace_root, max_bytes, max_files
    )
    if limit_hit == "max_files":
        return j(handler, {
            "error": "too many files",
            "limit": max_files,
            "configure": "HERMES_WEBUI_FOLDER_ZIP_MAX_FILES",
        }, status=413)
    if limit_hit == "max_bytes":
        return j(handler, {
            "error": "folder too large",
            "limit_bytes": max_bytes,
            "configure": "HERMES_WEBUI_FOLDER_ZIP_MAX_MB",
        }, status=413)

    zip_name = (target.name or "workspace") + ".zip"
    handler.send_response(200)
    handler.send_header("Content-Type", "application/zip")
    handler.send_header(
        "Content-Disposition",
        _content_disposition_value("attachment", zip_name),
    )
    handler.send_header("Cache-Control", "no-store")
    # Under HTTP/1.1 (Handler.protocol_version, see server.py post-#2836)
    # a response with no Content-Length and no Transfer-Encoding requires
    # Connection: close so the client knows the body ends at FIN. The ZIP
    # is built on-the-fly so we cannot send Content-Length up front; mirror
    # the SSE-endpoint pattern #2836 uses. Without this header the client
    # hangs waiting for the next pipelined response after the central
    # directory bytes finish. Caught by Opus pre-release advisor on
    # stage-batch11.
    handler.send_header("Connection", "close")
    handler.end_headers()

    written = 0
    with zipfile.ZipFile(handler.wfile, mode="w", compression=zipfile.ZIP_DEFLATED, allowZip64=True) as zf:
        for fp, arcname in files:
            fd = None
            try:
                fd = open_anchored_fd(workspace_root, fp.resolve(), want_dir=False)
                info = zipfile.ZipInfo(arcname)
                info.compress_type = zipfile.ZIP_DEFLATED
                with os.fdopen(fd, "rb", closefd=True) as src:
                    fd = None
                    with zf.open(info, "w") as dst:
                        shutil.copyfileobj(src, dst, length=1024 * 1024)
                written += 1
            except (ValueError, OSError, PermissionError) as e:
                logger.warning("folder-download: skipping %s: %s", fp, e)
            finally:
                if fd is not None:
                    try:
                        os.close(fd)
                    except OSError:
                        pass
    logger.info(
        "folder-download: streamed %d/%d files (~%d bytes) from %s",
        written, len(files), total_bytes, target,
    )


def _handle_file_raw(handler, parsed):
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    rel = qs.get("path", [""])[0]
    force_download = qs.get("download", [""])[0] == "1"
    resolved = _file_raw_target(s, sid, rel)
    if resolved is None:
        return j(handler, {"error": "not found"}, status=404)
    anchor_root, target = resolved
    ext = target.suffix.lower()
    mime = MIME_MAP.get(ext, "application/octet-stream")
    # Security: force download for dangerous MIME types to prevent XSS.
    # Exception: ?inline=1 permits text/html to be served inline for the
    # sandboxed workspace HTML preview iframe (sandbox="allow-scripts" with no
    # allow-same-origin, so the iframe cannot access parent cookies/storage).
    inline_preview = qs.get("inline", [""])[0] == "1"
    dangerous_types = {"text/html", "application/xhtml+xml", "image/svg+xml"}
    html_inline_ok = inline_preview and mime == "text/html"
    disposition = "attachment" if force_download or (mime in dangerous_types and not html_inline_ok) else "inline"
    # Defense-in-depth for ?inline=1 HTML: even though the workspace.js iframe
    # sets sandbox="allow-scripts", a user could be tricked into opening the
    # ?inline=1 URL directly in a top-level tab (e.g. via a chat link), which
    # would render the HTML in the WebUI's origin without iframe sandbox. The
    # CSP sandbox directive applies the same isolation server-side: without
    # allow-same-origin, the document is treated as a unique opaque origin and
    # cannot read WebUI cookies, localStorage, or postMessage to the parent.
    sandbox_csp = "sandbox allow-scripts allow-popups allow-popups-to-escape-sandbox"
    csp = sandbox_csp if (inline_preview and not force_download and disposition == "inline") else None
    # _serve_file_bytes sends Content-Security-Policy when csp is set.
    if html_inline_ok:
        return _serve_inline_html_preview(handler, target, "no-store", csp=sandbox_csp, anchor_root=anchor_root)
    return _serve_file_bytes(handler, target, mime, disposition, "no-store", csp=csp, anchor_root=anchor_root)


def _handle_file_read(handler, parsed):
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")
    try:
        s = get_session_for_file_ops(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    rel = qs.get("path", [""])[0]
    if not rel:
        return bad(handler, "path is required")
    try:
        return j(handler, read_file_content(Path(s.workspace), rel))
    except ImportError as e:
        return bad(handler, str(e), 503)
    except (FileNotFoundError, ValueError) as e:
        return bad(handler, _sanitize_error(e), 404)


def _read_anchored_file_bytes(ws_root: Path, target: Path) -> bytes:
    fd = open_anchored_fd(ws_root, target, want_dir=False)
    with os.fdopen(fd, "rb", closefd=True) as fh:
        st = os.fstat(fh.fileno())
        if not _stat.S_ISREG(st.st_mode):
            raise FileNotFoundError(f"Not a file: {target}")
        if st.st_size > MAX_FILE_BYTES:
            raise ValueError(f"File too large ({st.st_size} bytes, max {MAX_FILE_BYTES})")
        return fh.read(MAX_FILE_BYTES + 1)


def _handle_approval_pending(handler, parsed):
    sid = parse_qs(parsed.query).get("session_id", [""])[0]
    with _lock:
        _head, _total, _changed = reconcile_gateway_pending_mirror_locked(sid)
        queue = _pending.get(sid)
        # Support both the new list format and a legacy single-dict value.
        if isinstance(queue, list):
            p = queue[0] if queue else None
            total = len(queue)
        elif queue:
            p = queue
            total = 1
        else:
            p = None
            total = 0
        if p is None:
            gw_queue = _gateway_queues.get(sid) or []
            if gw_queue:
                raw = getattr(gw_queue[0], "data", None) or {}
                if raw:
                    p = raw
                    total = len(gw_queue)
                else:
                    logger.warning("Gateway queue entry for %s has no .data attribute", sid)
    if p:
        return j(handler, {"pending": dict(p), "pending_count": total})
    return j(handler, {"pending": None, "pending_count": 0})


def _handle_approval_sse_stream(handler, parsed):
    """SSE endpoint for real-time approval notifications.

    Long-lived connection that pushes approval events the moment they arrive,
    replacing the 1.5s polling loop.  The frontend uses EventSource and falls
    back to HTTP polling if the connection fails.
    """
    sid = parse_qs(parsed.query).get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")

    # Subscribe AND snapshot atomically under a single _lock acquisition so a
    # submit_pending() that fires between the two cannot be lost. If we
    # snapshot first then subscribe (the naive ordering), an approval that
    # arrives in the gap is appended to _pending (after our snapshot) AND
    # notified to subscribers (before we joined) — leaving the client unaware
    # until the next event arrives.
    q = queue.Queue(maxsize=16)
    initial_pending = None
    initial_count = 0
    with _lock:
        _approval_sse_subscribers.setdefault(sid, []).append(q)
        reconcile_gateway_pending_mirror_locked(sid)
        q_list = _pending.get(sid)
        if isinstance(q_list, list):
            initial_pending = dict(q_list[0]) if q_list else None
            initial_count = len(q_list)
        elif q_list:
            initial_pending = dict(q_list)
            initial_count = 1

    handler.send_response(200)
    handler.send_header('Content-Type', 'text/event-stream; charset=utf-8')
    handler.send_header('Cache-Control', 'no-cache')
    handler.send_header('X-Accel-Buffering', 'no')
    handler.send_header('Connection', 'close')
    end_sse_headers(handler)
    _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread

    from api.streaming import _sse

    # Push initial state immediately so the client doesn't miss anything.
    _sse(handler, 'initial', {"pending": initial_pending, "pending_count": initial_count})

    try:
        while True:
            try:
                payload = q.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                # Keepalive — SSE comment line prevents proxy/CDN timeout.
                handler.wfile.write(b': keepalive\n\n')
                handler.wfile.flush()
                continue
            if payload is None:
                break  # signal to close
            _sse(handler, 'approval', payload)
    except _CLIENT_DISCONNECT_ERRORS:
        pass  # client went away — normal for long-lived connections
    finally:
        _approval_sse_unsubscribe(sid, q)


def _handle_approval_inject(handler, parsed):
    """Inject a fake pending approval -- loopback-only, used by automated tests."""
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    key = qs.get("pattern_key", ["test_pattern"])[0]
    cmd = qs.get("command", ["rm -rf /tmp/test"])[0]
    if sid:
        submit_pending(
            sid,
            {
                "command": cmd,
                "pattern_key": key,
                "pattern_keys": [key],
                "description": "test pattern",
            },
        )
        return j(handler, {"ok": True, "session_id": sid})
    return j(handler, {"error": "session_id required"}, status=400)


def _handle_clarify_pending(handler, parsed):
    sid = parse_qs(parsed.query).get("session_id", [""])[0]
    pending = get_clarify_pending(sid)
    if pending:
        return j(handler, {"pending": pending})
    return j(handler, {"pending": None})


def _handle_clarify_sse_stream(handler, parsed):
    """SSE endpoint for real-time clarify notifications.

    Long-lived connection that pushes clarify events the moment they arrive,
    replacing the 1.5s polling loop.  The frontend uses EventSource and falls
    back to HTTP polling if the connection fails.
    """
    if clarify_sse_subscribe is None:
        return bad(handler, "clarify SSE not available")

    sid = parse_qs(parsed.query).get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")

    # Subscribe AND snapshot atomically.  We import clarify's _lock so that
    # subscribe and the snapshot read happen under the same mutex — same
    # pattern as the approval SSE handler.
    #
    # NOTE: We must NOT call clarify.get_pending() here — it acquires _lock
    # internally, which would deadlock since clarify._lock is a non-reentrant
    # threading.Lock.  Instead, read _gateway_queues / _pending inline under
    # the lock we already hold.
    from api.clarify import (
        _lock as _clarify_lock,
        _clarify_sse_subscribers as _clarify_subs,
        _gateway_queues as _clarify_gateway_queues,
        _pending as _clarify_pending,
    )
    q = queue.Queue(maxsize=16)
    initial_pending = None
    initial_count = 0
    with _clarify_lock:
        _clarify_subs.setdefault(sid, []).append(q)
        gw_q = _clarify_gateway_queues.get(sid) or []
        if gw_q:
            initial_pending = dict(gw_q[0].data)
            initial_count = len(gw_q)
        else:
            _legacy = _clarify_pending.get(sid)
            if _legacy:
                initial_pending = dict(_legacy)
                initial_count = 1

    handler.send_response(200)
    handler.send_header('Content-Type', 'text/event-stream; charset=utf-8')
    handler.send_header('Cache-Control', 'no-cache')
    handler.send_header('X-Accel-Buffering', 'no')
    handler.send_header('Connection', 'close')
    end_sse_headers(handler)
    _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread

    from api.streaming import _sse

    # Push initial state immediately so the client doesn't miss anything.
    _sse(handler, 'initial', {"pending": initial_pending, "pending_count": initial_count})

    try:
        while True:
            try:
                payload = q.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                handler.wfile.write(b': keepalive\n\n')
                handler.wfile.flush()
                continue
            if payload is None:
                break
            _sse(handler, 'clarify', payload)
    except _CLIENT_DISCONNECT_ERRORS:
        pass
    finally:
        clarify_sse_unsubscribe(sid, q)


def _handle_session_sse_stream(handler, parsed):
    """SSE endpoint for the persistent per-session channel (Option X).

    Subscribes to ``api.background_process.SESSION_CHANNELS[sid]`` — a channel
    that lives across agent turns (unlike STREAMS, which is torn down at
    end-of-turn). Used to deliver ``bg_task_complete`` events that fire while
    no agent turn is active.

    Lifecycle: opened by the frontend at session mount, closed at unmount or
    on tab close. Multiple tabs share one SessionChannel (refcounted via
    subscribe/unsubscribe). 30s SSE keepalive comments keep the proxy alive.
    Reaper-driven idle TTL (default 4h) prevents zombie channels.
    """
    sid = parse_qs(parsed.query).get("session_id", [""])[0]
    if not sid:
        return bad(handler, "session_id is required")

    # The (re)subscribing tab reports its last-known message_count via
    # ?known_count=N so the on-subscribe self-heal can detect a server-initiated
    # turn that started AND finished entirely inside this tab's SSE gap (see the
    # "server-initiated turn finished during the gap" self-heal block below).
    # Absent/blank/non-numeric => None ("tab didn't report", never triggers).
    _known_count_raw = parse_qs(parsed.query).get("known_count", [""])[0]
    try:
        subscriber_known_count = int(_known_count_raw) if _known_count_raw != "" else None
    except (TypeError, ValueError):
        subscriber_known_count = None

    from api.background_process import (
        subscribe_to_session_channel,
        active_stream_id_for_session,
        persisted_message_count_for_session,
        should_emit_session_updated,
    )

    # Atomic get-or-create + subscribe under SESSION_CHANNELS_LOCK. Doing these
    # two steps separately (get_or_create_session_channel then ch.subscribe)
    # left a TOCTOU gap where the reaper — which also holds
    # SESSION_CHANNELS_LOCK and collects idle 0-subscriber channels in one
    # critical section — could collect the channel between the two calls,
    # orphaning this subscriber on a channel no longer in SESSION_CHANNELS.
    # bg_task_complete emits would then never reach this queue. See
    # subscribe_to_session_channel for the full rationale (PR #2971 Greptile P1).
    ch, q = subscribe_to_session_channel(sid, maxsize=64)

    # NOTE: ``subscribe_to_session_channel`` above acquires a subscriber slot
    # that MUST be released on every exit path. Header setup
    # (``send_response`` / ``send_header`` / ``end_headers`` /
    # ``_sse_set_write_deadline``) and the initial-frame + on-subscribe
    # recovery writes below all touch the socket and can raise a member of
    # ``_CLIENT_DISCONNECT_ERRORS`` (BrokenPipeError / ConnectionResetError) if
    # the client drops immediately after subscribing. If that happened outside
    # this try/finally the ``ch.unsubscribe(q)`` cleanup would be skipped,
    # permanently leaking a subscriber. Because
    # ``SessionChannel.reaper_should_collect()`` refuses to collect any channel
    # with ``sub_count > 0``, a single ghost subscriber blocks the reaper
    # forever and the channel zombies in SESSION_CHANNELS. So EVERYTHING from
    # the subscribe onward — header setup included — runs inside one
    # try/finally that unconditionally unsubscribes.
    try:
        handler.send_response(200)
        handler.send_header('Content-Type', 'text/event-stream; charset=utf-8')
        handler.send_header('Cache-Control', 'no-cache')
        handler.send_header('X-Accel-Buffering', 'no')
        # #3103: omit the Connection header — rely on the HTTP/1.1 keep-alive
        # default, matching the other long-lived SSE handlers (gateway/session
        # events) that fixed the reconnect-storm. An explicit value here is a
        # third, inconsistent approach (greptile flag).
        end_sse_headers(handler)
        _sse_set_write_deadline(handler)  # Defect A: slow tab can't pin this thread

        from api.streaming import _sse

        # Push an initial frame so the client has confirmation the channel is
        # live (mirrors approval/clarify which send an 'initial' frame). No
        # snapshot data is needed — this channel only carries forward-looking
        # events, not pending state.
        _sse(handler, 'initial', {"session_id": sid})

        # ── Open-tab live-view self-heal (root cause: lost server_turn_started) ──
        # The `server_turn_started` fan-out (routes.start_session_turn) is a
        # fire-and-forget SessionChannel.emit with NO replay buffer: it reaches
        # only the subscribers connected at the exact emit instant. A tab whose
        # per-session EventSource was momentarily absent at that instant — a
        # transient SSE drop, a reverse-proxy idle-timeout, or browser
        # connection-pool starvation (all common behind a corporate proxy) —
        # misses the frame permanently, so a SERVER-initiated wakeup turn never
        # renders live and the user must hard-refresh (the reported defect). The
        # server-side wakeup itself ran and persisted fine; only the live-view
        # was lost. On (re)subscribe, if the session has a live run RIGHT NOW,
        # replay a synthetic `server_turn_started` to THIS new subscriber so the
        # open tab attaches its existing chat-stream renderer (attachLiveStream)
        # and self-heals with no refresh. `recovered: True` lets the frontend
        # use the replay (reconnecting) attach so the renderer picks up the
        # in-progress stream from the run journal rather than expecting token 0.
        # Idempotent: the frontend dedupes by (session_id, stream_id) — if the
        # original frame WAS delivered this is a harmless no-op there.
        try:
            recover_stream_id = active_stream_id_for_session(sid)
            if recover_stream_id:
                pending_started_at = None
                try:
                    recover_session = get_session(sid, metadata_only=True)
                    pending_started_at = getattr(recover_session, "pending_started_at", None)
                except Exception:
                    logger.debug(
                        "session-stream recovery could not read pending_started_at for %s",
                        sid,
                        exc_info=True,
                    )
                _sse(handler, 'server_turn_started', {
                    "session_id": sid,
                    "stream_id": recover_stream_id,
                    "pending_started_at": pending_started_at,
                    "source": "subscribe_recovery",
                    "recovered": True,
                })
            else:
                # ── Server-initiated turn that FINISHED during the SSE gap ──
                # The block above only heals a turn that is live RIGHT NOW. But
                # a server-initiated turn (self-wake / cron / restart hook) can
                # start AND finish entirely inside the gap: the fire-and-forget
                # `server_turn_started` reached no subscriber, and by the time
                # this tab reconnects the run has already cleared from
                # ACTIVE_RUNS — so active_stream_id_for_session returns None and
                # nothing above replays. The turn IS persisted, but this tab's
                # transcript stays stale until a hard refresh (the reported
                # visible-tab defect). Detect it by comparing the persisted
                # message_count against what this (re)subscribing tab last knew
                # (?known_count). If the server is AHEAD, emit a lightweight
                # `session-updated` frame so the tab does an INCREMENTAL,
                # swap-in-place message sync (frontend reuses #5189's
                # keepStaleUntilLoaded loadSession path — NO clear+refetch, so
                # the #5177/#5189 blank-gap jump is not reintroduced). Carries
                # only counts (no transcript) to stay cheap. Skipped
                # entirely when the tab didn't report a count or the persisted
                # count is unknown (legacy sidecar) → never a spurious reload.
                if subscriber_known_count is not None:
                    persisted_count = persisted_message_count_for_session(sid)
                    if should_emit_session_updated(subscriber_known_count, persisted_count):
                        _sse(handler, 'session-updated', {
                            "session_id": sid,
                            "message_count": persisted_count,
                            "known_count": subscriber_known_count,
                            "source": "subscribe_recovery",
                        })
        except _CLIENT_DISCONNECT_ERRORS:
            # Client vanished mid-recovery — re-raise so the outer handler
            # treats it as a normal disconnect and the finally still cleans up.
            raise
        except Exception:
            logger.debug(
                "session-stream on-subscribe recovery failed for %s", sid,
                exc_info=True,
            )

        while True:
            try:
                payload = q.get(timeout=_SSE_HEARTBEAT_INTERVAL_SECONDS)
            except queue.Empty:
                handler.wfile.write(b': keepalive\n\n')
                handler.wfile.flush()
                continue
            if payload is None:
                break
            event_name, data = payload
            _sse(handler, event_name, data)
    except _CLIENT_DISCONNECT_ERRORS:
        pass  # client went away — normal for long-lived connections
    finally:
        ch.unsubscribe(q)


def _handle_clarify_inject(handler, parsed):
    """Inject a fake pending clarify prompt -- loopback-only, used by automated tests."""
    qs = parse_qs(parsed.query)
    sid = qs.get("session_id", [""])[0]
    question = qs.get("question", ["Which option?"])[0]
    choices = qs.get("choices", [])
    if sid:
        submit_clarify_pending(
            sid,
            {
                "question": question,
                "choices_offered": choices,
                "session_id": sid,
                "kind": "clarify",
            },
        )
        return j(handler, {"ok": True, "session_id": sid})
    return j(handler, {"error": "session_id required"}, status=400)


def _handle_live_models(handler, parsed):
    """Return the live model list for a provider.

    Delegates to the agent's provider_model_ids() which handles:
    - OpenRouter: live fetch from /api/v1/models
    - Anthropic: live fetch from /v1/models (API key or OAuth token)
    - Copilot: live fetch from api.githubcopilot.com/models with correct headers
    - openai-codex: Codex OAuth endpoint + local ~/.codex/ cache fallback
    - Nous: live fetch from inference-api.nousresearch.com/v1/models
    - DeepSeek, kimi-coding, opencode-zen/go, custom: generic OpenAI-compat /v1/models
    - ZAI, MiniMax, Google/Gemini: fall back to static list (non-standard endpoints)
    - All others: static _PROVIDER_MODELS fallback

    The agent already maintains all provider-specific auth and endpoint logic
    in one place; the WebUI inherits it rather than duplicating it.

    Query params:
        provider  (optional) — provider ID; defaults to active profile provider
    """
    qs = parse_qs(parsed.query)
    provider = (qs.get("provider", [""])[0] or "").lower().strip()

    try:
        from api.config import get_config as _gc
        cfg = _gc()
        if not provider:
            provider = cfg.get("model", {}).get("provider") or ""
        if not provider:
            return j(handler, {"error": "no_provider", "models": []})

        # Normalize provider alias so 'z.ai' -> 'zai', 'x.ai' -> 'xai', etc.
        # The browser sends whatever active_provider the static endpoint returned;
        # without normalization, provider_model_ids() misses the alias and returns [].
        # Uses the WebUI-owned table (api/config._resolve_provider_alias) which
        # works even when hermes_cli is not on sys.path.
        from api.config import _resolve_provider_alias
        provider = _resolve_provider_alias(provider)

        cache_key = _live_models_cache_key(provider)
        cached = _get_cached_live_models(cache_key)
        if cached is not None:
            return j(handler, cached)

        def _finish(payload: dict):
            _set_cached_live_models(cache_key, payload)
            return j(handler, payload)

        # Delegate to the agent's live-fetch + fallback resolver.
        # provider_model_ids() tries live endpoints first and falls back to
        # the static _PROVIDER_MODELS list — it never raises.
        try:
            import sys as _sys
            import os as _os
            _agent_dir = _os.path.join(_os.path.dirname(_os.path.dirname(_os.path.abspath(__file__))),
                                       "..", "..", ".hermes", "hermes-agent")
            _agent_dir = _os.path.normpath(_agent_dir)
            if _agent_dir not in _sys.path:
                _sys.path.insert(0, _agent_dir)
            from hermes_cli.models import provider_model_ids as _pmi
            ids = _pmi(provider)
        except Exception as _import_err:
            logger.debug("provider_model_ids import failed for %s: %s", provider, _import_err)
            ids = []

        if not ids:
            custom_provider_entry = None

            def _custom_provider_entries_for_request():
                if not (provider == "custom" or provider.startswith("custom:")):
                    return []
                try:
                    from api.config import _custom_provider_slug_from_name
                    _cp_entries = cfg.get("custom_providers", [])
                    if not isinstance(_cp_entries, list):
                        return []
                    _matches = []
                    for _cp in _cp_entries:
                        if not isinstance(_cp, dict):
                            continue
                        _slug = _custom_provider_slug_from_name(_cp.get("name", ""))
                        if provider.startswith("custom:"):
                            if _slug == provider:
                                _matches.append(_cp)
                        elif provider == "custom" and not _slug:
                            _matches.append(_cp)
                    return _matches
                except Exception:
                    return []

            def _custom_provider_model_ids(_cp):
                _ids = []

                def _append(_mid):
                    _mid = str(_mid or "").strip()
                    if _mid and _mid not in _ids:
                        _ids.append(_mid)

                _append(_cp.get("model", ""))
                _models = _cp.get("models")
                if isinstance(_models, dict):
                    for _mid in _models:
                        if isinstance(_mid, str):
                            _append(_mid)
                elif isinstance(_models, list):
                    for _item in _models:
                        if isinstance(_item, str):
                            _append(_item)
                        elif isinstance(_item, dict):
                            _append(_item.get("id") or _item.get("model") or _item.get("name"))
                return _ids

            def _custom_provider_api_key(_cp):
                _raw = _cp.get("api_key")
                if _raw is not None:
                    _key = str(_raw).strip()
                    if _key.startswith("${") and _key.endswith("}") and len(_key) > 3:
                        _key = os.getenv(_key[2:-1], "").strip()
                    if _key:
                        return _key
                _env = str(_cp.get("key_env") or "").strip()
                return os.getenv(_env, "").strip() if _env else ""

            def _custom_provider_models_discover_is_false(_cp):
                """True when ``discover_models`` is an explicit ``false`` opt-out.

                Mirrors ``api.config._provider_discover_allowed`` / Hermes
                Agent ``model_switch_providers._discover_flag``: ``discover_models``
                defaults to True, and the string forms ``"false"``/``"no"``/``"0"``
                (case-insensitive) mean False.  Used to decide whether a
                dict-shaped ``models`` mapping is a hand-pinned allowlist (only
                when discovery is off) rather than per-model metadata.
                """
                _discover = _cp.get("discover_models", True) if isinstance(_cp, dict) else True
                if isinstance(_discover, str):
                    return _discover.strip().lower() in {"false", "no", "0"}
                return not bool(_discover)

            def _custom_provider_allowlist_ids(_cp):
                """Plural ``models`` list only — the explicit allowlist signal.

                The singular ``model`` field is sticky/default metadata, NOT an
                allowlist: it must not gate live-catalog filtering, otherwise a
                provider configured with only ``model: assistant`` (no ``models``
                list) would be collapsed to a single model.  Only an explicit
                ``models`` allowlist expresses "show exactly these models".

                An auto-discovered catalog is NOT an allowlist: when Hermes
                persisted discovery results back into config (``models: {...}``
                plus ``models_discovered: true``), that mapping is a snapshot
                of what the gateway exposed at discovery time.  Gating on it
                would permanently pin the live catalog to the first-discovery
                set, silently dropping any model the user pulls in later
                (LM Studio / Ollama).  ``_provider_models_are_discovered_catalog()``
                is the shared predicate the ``/api/models`` path already uses;
                when it says "discovered", return no allowlist and let the
                live probe win.  An explicit ``discover_models: false`` opt-out
                re-pins the catalog (the predicate accounts for it), so a
                hand-pinned discovered catalog still filters.

                A dict-shaped ``models`` mapping that is NOT marked discovered is
                *per-model metadata* written by the Hermes Agent setup flow
                (``hermes_cli/model_switch.py::_save_custom_provider`` and the
                setup wizard) — e.g. ``{chat-a: {context_length: 128000}}``.  It is
                not a catalog narrow: treating its keys as an allowlist would
                collapse the live picker to the single saved default (keyless
                Ollama) while the CLI live-probe shows the full catalog.  Only an
                explicit ``discover_models: false`` opts into treating the dict
                keys as a pinned allowlist.  List/JSON-array-string/Python-literal
                shapes remain plain allowlists.

                The plural value is decoded through ``_parse_config_string_list()``
                because ``hermes config set`` / JSON-mode editor saves persist
                lists as quoted JSON-array strings (``'["chat-a","chat-b"]'``) or
                Python literals (``"['chat-a']"``).  Handling only ``dict``/``list``
                made a serialized allowlist fall through to ``[]``, which the
                caller reads as "no allowlist configured" and floods the picker
                with the full upstream catalog — the same class of bug as the
                ``skills.disabled`` regression (#7120 / #7134).
                """
                from api.config import _provider_models_are_discovered_catalog

                if _provider_models_are_discovered_catalog(_cp):
                    return []
                _ids = []
                _models = _cp.get("models")
                # Serialized shapes only: ``hermes config set`` / JSON-mode
                # editor saves persist lists as quoted JSON-array strings
                # (``'["chat-a","chat-b"]'``) or Python literals
                # (``"['chat-a']"``).  Decode those through the shared helper so
                # they are honored; native dict/list values are walked below so
                # dict *entries* keep their id|model|name metadata (the decoder
                # str()-ifies list members, which would mangle them).
                if isinstance(_models, str):
                    _models = _parse_config_string_list(_models)
                if isinstance(_models, dict):
                    # A dict-shaped ``models`` is per-model metadata written by
                    # the Hermes Agent setup flow, NOT a catalog narrow.  Only a
                    # ``discover_models: false`` opt-out treats the dict keys as a
                    # pinned allowlist; otherwise return the empty allowlist so the
                    # live probe returns the full catalog.
                    if not _custom_provider_models_discover_is_false(_cp):
                        return []
                    for _mid in _models:
                        if isinstance(_mid, str) and _mid.strip():
                            _ids.append(_mid.strip())
                elif isinstance(_models, (list, tuple)):
                    for _item in _models:
                        if isinstance(_item, str) and _item.strip():
                            _ids.append(_item.strip())
                        elif isinstance(_item, dict):
                            _mid = _item.get("id") or _item.get("model") or _item.get("name")
                            if _mid and str(_mid).strip():
                                _ids.append(str(_mid).strip())
                return _ids

            # For 'custom' and 'custom:*' providers, provider_model_ids()
            # returns [] because they aren't real hermes_cli endpoints.
            # Fall back to the custom_providers entries from config.yaml so
            # the live-model enrichment step can add any models that weren't
            # already in the static list (issue #1619).
            # Collect config-specified model IDs separately so they don't
            # prevent the live fetch below from running (#3718).
            _config_ids = []
            _allowlist_ids = []
            if provider == "custom" or provider.startswith("custom:"):
                for _cp in _custom_provider_entries_for_request():
                    if custom_provider_entry is None:
                        custom_provider_entry = _cp
                    _config_ids.extend(_custom_provider_model_ids(_cp))
                    _allowlist_ids.extend(_custom_provider_allowlist_ids(_cp))
            
            # Always try live fetch for custom providers — config entries are a
            # fallback, not a replacement.  The live endpoint should return ALL
            # models the key has access to, not just what's listed in config.yaml.
            if provider == "custom" or provider.startswith("custom:"):
                _base_url = None
                _api_key = None
                if custom_provider_entry:
                    _base_url = custom_provider_entry.get("base_url")
                    _api_key = _custom_provider_api_key(custom_provider_entry)
                else:
                    _model_cfg = cfg.get("model", {})
                    _base_url = _model_cfg.get("base_url")
                    _api_key = _model_cfg.get("api_key")
                # Fallback: try credential pool for base_url + api_key
                if (not _base_url or not _api_key) and provider.startswith("custom:"):
                    try:
                        from api.config import _has_explicit_pool_credentials

                        if _has_explicit_pool_credentials(provider):
                            from agent.credential_pool import load_pool as _lpool
                            _resolved = _resolve_provider_alias(provider)
                            _lm_pool = _lpool(_resolved)
                            if _lm_pool:
                                _lm_entry = _lm_pool.select()
                                if _lm_entry:
                                    if not _api_key:
                                        _api_key = getattr(_lm_entry, "runtime_api_key", "") or ""
                                    if not _base_url:
                                        _base_url = str(getattr(_lm_entry, "base_url", "") or "").strip()
                    except ImportError:
                        pass
                if _base_url and _api_key:
                    try:
                        import urllib.request
                        import json
                        
                        # Build the models endpoint URL
                        # AxonHub and similar OpenAI-compat endpoints serve /v1/models
                        _ep = _base_url.rstrip("/")
                        # If base_url already ends with /v1, use /models; otherwise add /v1/models
                        if _ep.endswith("/v1"):
                            _models_url = f"{_ep}/models"
                        else:
                            _models_url = f"{_ep}/v1/models"
                        
                        _req = urllib.request.Request(
                            _models_url,
                            headers={"Authorization": f"Bearer {_api_key}"},
                        )
                        
                        with urllib.request.urlopen(_req, timeout=CUSTOM_MODELS_ENDPOINT_TIMEOUT_SECONDS) as _resp:
                            _body = json.loads(_resp.read())
                        
                        # Parse response: {"data": [{"id": "model1", ...}, ...]}
                        if isinstance(_body, dict):
                            _data = _body.get("data", [])
                            if isinstance(_data, list):
                                ids = [m.get("id", "") for m in _data if m.get("id")]
                        elif isinstance(_body, list):
                            ids = [m.get("id", m) if isinstance(m, dict) else m for m in _body]

                        if ids:
                            logger.debug("Live-fetched %d models from custom provider %s", len(ids), _base_url)
                        else:
                            logger.debug("Custom provider returned no models from %s", _base_url)

                    except Exception as _fetch_err:
                        logger.debug("Live fetch from custom provider failed: %s", _fetch_err)

                # If live fetch succeeded, filter the live catalog down to the
                # config-declared models allowlist first, then append any
                # allowlisted models the live endpoint didn't return.  Custom
                # providers (esp. New-API-style gateways) expose their ENTIRE
                # catalog via /v1/models — including image/audio models that
                # are not chat models.  The picker must not surface models the
                # user never declared in config.yaml.  Only a NON-EMPTY explicit
                # ``models`` allowlist gates the filter — a provider configured
                # with just a singular ``model`` (no ``models`` list) keeps the
                # unfiltered live catalog.  When no allowlist is configured,
                # return the live list as-is (preserves the pre-filter
                # discovery behaviour).
                #
                # An empty allowlist (``models: []``, ``models: "[]"``, or a
                # value that decodes to no usable ids) is deliberately treated
                # as "not configured", NOT as "allow nothing".  Gating on
                # declared-ness instead would emit an EMPTY picker and make the
                # provider unselectable — a harder failure than surfacing a few
                # extra models, and unrecoverable from the UI because
                # ``custom_providers`` is hand-edited in config.yaml (the WebUI
                # has no write path for it).  ``[]`` in practice means a
                # leftover/placeholder key, not an intentional deny-all, and no
                # deny-all use case exists: a provider the user wants hidden is
                # removed from ``custom_providers`` outright.
                if ids:
                    if _allowlist_ids:
                        _allowlist_set = set(_allowlist_ids)
                        ids = [m for m in ids if m in _allowlist_set] or []
                        for _aid in _allowlist_ids:
                            if _aid not in ids:
                                ids.append(_aid)
                else:
                    ids = list(_allowlist_ids or _config_ids)

        # ── OpenAI-compat live fetch fallback ──────────────────────────────────
        # When provider_model_ids() is unavailable or returns [] for a provider
        # that exposes a standard /v1/models endpoint, fetch directly.  This
        # eliminates the need to keep _PROVIDER_MODELS in sync for providers
        # that have a discoverable API (#871).
        #
        # WARNING: This uses synchronous urllib.request which blocks the worker
        # thread for up to 8 seconds on timeout. This is acceptable because:
        #  (a) the server uses threading (not async), so other requests continue;
        #  (b) the frontend shows the static list immediately and enriches in
        #      the background via _fetchLiveModels(), so the user never waits.
        if not ids:
            _ep = _OPENAI_COMPAT_ENDPOINTS.get(provider)
            if _ep:
                try:
                    import urllib.request
                    _providers_cfg = cfg.get("providers") or {}
                    _prov = _providers_cfg.get(provider, {}) if isinstance(_providers_cfg, dict) else {}
                    # Only use a provider-scoped key.  A top-level model.api_key
                    # is safe here only when it belongs to the requested provider;
                    # otherwise /api/models/live?provider=<other> could forward
                    # the active provider's credential to the wrong third party.
                    _key = _prov.get("api_key") if isinstance(_prov, dict) else None
                    if not _key:
                        _model_cfg = cfg.get("model", {})
                        if isinstance(_model_cfg, dict):
                            _active_provider = _resolve_provider_alias(
                                (_model_cfg.get("provider") or "").strip().lower()
                            )
                            if _active_provider == provider:
                                _key = _model_cfg.get("api_key")
                    if _key:
                        _req = urllib.request.Request(
                            f"{_ep}/models",
                            headers={"Authorization": f"Bearer {_key}"},
                        )
                        with urllib.request.urlopen(_req, timeout=8) as _resp:
                            _body = json.loads(_resp.read())
                        ids = [m.get("id", "") for m in _body.get("data", []) if m.get("id")]
                        logger.debug("Live-fetched %d models from %s /v1/models", len(ids), provider)
                except Exception as _fetch_err:
                    logger.debug("Live fetch from %s failed: %s", provider, _fetch_err)
                    # Fall through to static list below

        # Static fallback — only reached when live fetch also failed.
        if not ids:
            from api.config import _PROVIDER_MODELS as _pm
            ids = [m["id"] for m in _pm.get(provider, [])]
        if not ids:
            return _finish({"provider": provider, "models": [], "count": 0})

        # Match the same dropdown visibility budget that /api/models uses so
        # background enrichment via _fetchLiveModels() does not re-append an
        # uncapped catalog after the initial picker render. The full catalog
        # still comes from /api/models via extra_models for search/show-all;
        # this endpoint is only a dropdown-enrichment surface. (#1567, #3691)
        if provider == "nous":
            try:
                from api.config import _build_nous_featured_set
                _default_model = (cfg.get("model", {}) or {}).get("model") if isinstance(cfg.get("model"), dict) else None
                _featured, _ = _build_nous_featured_set(ids, selected_model_id=_default_model)
                ids = _featured
            except Exception:
                logger.debug("Failed to apply Nous featured-set cap for /api/models/live")
        else:
            from api.config import _MODEL_PICKER_OVERFLOW_THRESHOLD, _MODEL_PICKER_VISIBLE_TARGET
            if len(ids) > _MODEL_PICKER_OVERFLOW_THRESHOLD:
                ids = ids[:_MODEL_PICKER_VISIBLE_TARGET]

        # Normalise to {id, label} — provider_model_ids() returns plain string IDs.
        # For ollama-cloud use the shared Ollama formatter (handles `:variant` suffix).
        # For all other providers use a simpler hyphen-split capitaliser.
        from api.config import (
            _format_ollama_label as _fmt_ollama,
            _is_openai_family_provider as _is_fast_tier_provider,
            _model_supports_fast_tier_for_provider,
        )

        def _make_label(mid):
            """Best-effort human label from a model ID string."""
            if provider in ("ollama", "ollama-cloud"):
                return _fmt_ollama(mid)
            # Preserve slashes for router IDs like "anthropic/claude-sonnet-4.6"
            display = mid.split("/")[-1] if "/" in mid else mid
            parts = display.split("-")
            result = []
            for p in parts:
                pl = p.lower()
                if pl == "gpt":
                    result.append("GPT")
                elif pl in ("claude", "gemini", "gemma", "llama", "mistral",
                            "qwen", "deepseek", "grok", "kimi", "glm"):
                    result.append(p.capitalize())
                elif p[:1].isdigit():
                    result.append(p)  # version numbers: 5.4, 3.5, 4.6 — unchanged
                else:
                    result.append(p.capitalize())
            label = " ".join(result)
            # Restore well-known uppercase tokens that title-casing breaks
            for orig in ("GPT", "GLM", "API", "AI", "XL", "MoE"):
                label = label.replace(orig.title(), orig)
            return label

        annotate_fast_tier = _is_fast_tier_provider(provider)
        models_out = []
        for mid in ids:
            if not mid:
                continue
            entry = {"id": mid, "label": _make_label(mid)}
            if annotate_fast_tier:
                entry["supports_fast_tier"] = _model_supports_fast_tier_for_provider(mid, provider)
            models_out.append(entry)
        return _finish({"provider": provider, "models": models_out,
                        "count": len(models_out)})

    except Exception as _e:
        logger.debug("_handle_live_models failed for %s: %s", provider, _e)
        return j(handler, {"error": str(_e), "models": []})


def _handle_cron_history(handler, parsed):
    """List cron run output files with metadata (no content).

    Returns lightweight file listing so the frontend can render a run history
    without fetching full output for every run.
    """
    from cron.jobs import OUTPUT_DIR as CRON_OUT
    import re as _re

    qs = parse_qs(parsed.query)
    job_id = qs.get("job_id", [""])[0]
    if not job_id:
        return j(handler, {"error": "job_id required"}, status=400)
    # Defense-in-depth: cron job_ids are 12-char hex from the agent's scheduler.
    # Without validation, a job_id of "../<other>" would let an authenticated
    # caller enumerate .md filenames in adjacent directories under CRON_OUT's
    # parent. Mirror the rollback checkpoint id regex shape.
    # (Opus pre-release advisor finding.)
    if not _re.fullmatch(r"[A-Za-z0-9_-][A-Za-z0-9_.-]{0,63}", job_id) or job_id in (".", ".."):
        return j(handler, {"error": "invalid job_id"}, status=400)
    # Reject malformed offset/limit instead of letting int() raise ValueError
    # and surface as a confusing 500. Clamp to safe ranges.
    try:
        offset = max(0, int(qs.get("offset", ["0"])[0]))
        limit = max(1, min(500, int(qs.get("limit", ["50"])[0])))
    except (ValueError, TypeError):
        return j(handler, {"error": "offset and limit must be integers"}, status=400)
    out_dir = CRON_OUT / job_id
    runs = []
    total = 0
    if out_dir.exists():
        all_files = sorted(out_dir.glob("*.md"), key=lambda f: f.stat().st_mtime, reverse=True)
        total = len(all_files)
        page = all_files[offset:offset + limit]
        for f in page:
            try:
                st = f.stat()
                usage = _cron_output_usage_metadata(
                    f.read_text(encoding="utf-8", errors="replace")
                )
                runs.append({
                    "filename": f.name,
                    "size": st.st_size,
                    "modified": st.st_mtime,
                    "usage": usage,
                })
            except OSError:
                logger.debug("Failed to stat cron output file %s", f)
    return j(handler, {"job_id": job_id, "runs": runs, "total": total, "offset": offset})


def _handle_cron_run_detail(handler, parsed):
    """Return full content of a single cron run output file."""
    from cron.jobs import OUTPUT_DIR as CRON_OUT
    import re as _re

    qs = parse_qs(parsed.query)
    job_id = qs.get("job_id", [""])[0]
    filename = qs.get("filename", [""])[0]
    if not job_id or not filename:
        return j(handler, {"error": "job_id and filename required"}, status=400)
    # Validate job_id shape (defense-in-depth even though the resolve+is_relative_to
    # check below catches traversal — fail-closed at the parameter boundary so
    # malformed job_ids return a 400 from the validator rather than a 400 from
    # the path resolver).
    if not _re.fullmatch(r"[A-Za-z0-9_-][A-Za-z0-9_.-]{0,63}", job_id) or job_id in (".", ".."):
        return j(handler, {"error": "invalid job_id"}, status=400)
    # Prevent path traversal — resolve and verify it stays within the job's output dir
    fpath = (CRON_OUT / job_id / filename).resolve()
    if not fpath.is_relative_to(CRON_OUT.resolve()):
        return j(handler, {"error": "invalid filename"}, status=400)
    if not fpath.exists():
        return j(handler, {"error": "run not found"}, status=404)
    try:
        content = fpath.read_text(encoding="utf-8", errors="replace")
        snippet = _cron_output_snippet(content)
        usage = _cron_output_usage_metadata(content)
        return j(handler, {"job_id": job_id, "filename": filename,
                           "content": content, "snippet": snippet,
                           "usage": usage})
    except Exception as e:
        return j(handler, {"error": str(e)}, status=500)


def _cron_output_usage_metadata(text: str) -> dict:
    """Extract optional token/cost metadata from a cron output markdown file."""
    import re as _re

    head = text.split("## Response", 1)[0].split("# Response", 1)[0]
    usage: dict = {}

    def _intish(value: str):
        cleaned = _re.sub(r"[^0-9]", "", value or "")
        return int(cleaned) if cleaned else None

    def _floatish(value: str):
        match = _re.search(r"[-+]?\d+(?:\.\d+)?", (value or "").replace(",", ""))
        return float(match.group(0)) if match else None

    for raw_line in head.splitlines():
        line = raw_line.strip()
        model_match = _re.match(r"\*\*(?:Model|Model Used):\*\*\s*(.+)$", line, _re.I)
        if model_match:
            usage["model"] = model_match.group(1).strip()
            continue
        provider_match = _re.match(r"\*\*Provider:\*\*\s*(.+)$", line, _re.I)
        if provider_match:
            usage["provider"] = provider_match.group(1).strip()
            continue
        cost_match = _re.match(r"\*\*(?:Estimated cost|Cost):\*\*\s*(.+)$", line, _re.I)
        if cost_match:
            cost = _floatish(cost_match.group(1))
            if cost is not None:
                usage["estimated_cost_usd"] = cost
            continue
        duration_match = _re.match(r"\*\*(?:Duration|Elapsed):\*\*\s*(.+)$", line, _re.I)
        if duration_match:
            seconds = _floatish(duration_match.group(1))
            if seconds is not None:
                usage["duration_seconds"] = seconds
            continue
        tokens_match = _re.match(r"\*\*Tokens:\*\*\s*(.+)$", line, _re.I)
        if tokens_match:
            value = tokens_match.group(1)
            input_match = _re.search(r"([0-9][0-9,]*)\s*(?:input|in)\b", value, _re.I)
            output_match = _re.search(r"([0-9][0-9,]*)\s*(?:output|out)\b", value, _re.I)
            total_match = _re.search(r"([0-9][0-9,]*)\s*(?:total\s*)?tokens?\b", value, _re.I)
            if input_match:
                usage["input_tokens"] = _intish(input_match.group(1))
            if output_match:
                usage["output_tokens"] = _intish(output_match.group(1))
            if total_match and "total_tokens" not in usage:
                usage["total_tokens"] = _intish(total_match.group(1))

    if "total_tokens" not in usage:
        total = sum(int(usage.get(k) or 0) for k in ("input_tokens", "output_tokens"))
        if total:
            usage["total_tokens"] = total
    return usage


def _cron_output_snippet(text: str, limit: int = 600) -> str:
    """Extract the response body from a cron output .md file for preview.

    Contract: cron output files use markdown front-matter followed by a
    ``## Response`` (or ``# Response``) heading that marks the start of the
    agent's reply.  This function locates that heading and returns everything
    after it (up to *limit* chars).  If no heading is found the entire text
    is returned — callers should be aware that front-matter fields (model,
    timestamp, …) may appear in the snippet.
    """
    lines = text.split("\n")
    response_idx = -1
    for i, line in enumerate(lines):
        if line.startswith("## Response") or line.startswith("# Response"):
            response_idx = i
            break
    body = ("\n".join(lines[response_idx + 1:]) if response_idx >= 0 else "\n".join(lines)).strip()
    return body[:limit] or "(empty)"


def _handle_cron_output(handler, parsed):
    from cron.jobs import OUTPUT_DIR as CRON_OUT
    import re as _re

    qs = parse_qs(parsed.query)
    job_id = qs.get("job_id", [""])[0]
    if not job_id:
        return j(handler, {"error": "job_id required"}, status=400)
    # Match the job_id boundary enforced by the newer cron history/detail
    # handlers.  This endpoint also builds CRON_OUT / job_id before globbing
    # markdown outputs, so reject traversal-shaped IDs before path resolution.
    if not _re.fullmatch(r"[A-Za-z0-9_-][A-Za-z0-9_.-]{0,63}", job_id):
        return j(handler, {"error": "invalid job_id"}, status=400)
    # Reject malformed limit instead of letting int() raise ValueError and
    # surface as a confusing 500. Clamp to a safe range; a negative value must
    # never reach the slice below — files is sorted newest-first, so a negative
    # limit on `files[:limit]` slices as `files[:-n]` and drops the n OLDEST
    # entries (or all of them when |n| >= len), returning a truncated/empty list
    # instead of the newest outputs. Mirrors _handle_cron_run_detail.
    try:
        limit = max(1, min(500, int(qs.get("limit", ["5"])[0])))
    except (ValueError, TypeError):
        limit = 5
    out_dir = CRON_OUT / job_id
    outputs = []
    if out_dir.exists():
        files = sorted(out_dir.glob("*.md"), key=lambda f: f.stat().st_mtime, reverse=True)[:limit]
        for f in files:
            try:
                txt = f.read_text(encoding="utf-8", errors="replace")
                outputs.append({"filename": f.name, "content": _cron_output_content_window(txt)})
            except Exception:
                logger.debug("Failed to read cron output file %s", f)
    return j(handler, {"job_id": job_id, "outputs": outputs})


def _handle_cron_status(handler, parsed):
    """Return running status for one or all cron jobs."""
    qs = parse_qs(parsed.query)
    job_id = qs.get("job_id", [""])[0]
    if job_id:
        running, elapsed = _is_cron_running(job_id)
        return j(handler, {"job_id": job_id, "running": running, "elapsed": round(elapsed, 1)})
    # Return status for all running jobs
    with _RUNNING_CRON_LOCK:
        all_running = {jid: round(time.time() - t, 1) for jid, t in _RUNNING_CRON_JOBS.items()}
    return j(handler, {"running": all_running})


def _handle_cron_recent(handler, parsed):
    """Return cron jobs that have completed since a given timestamp."""
    import datetime

    qs = parse_qs(parsed.query)
    # Reject a malformed `since` instead of letting float() raise ValueError and
    # surface as a confusing 500. A bad/absent value means "from the epoch", so
    # the client still gets a well-formed (if unfiltered) response.
    try:
        since = float(qs.get("since", ["0"])[0])
    except (ValueError, TypeError):
        since = 0.0
    try:
        from cron.jobs import list_jobs

        jobs = list_jobs(include_disabled=True)
        completions = []
        for job in jobs:
            job_id = str(job.get("id", "") or "")
            last_run = job.get("last_run_at")
            if not last_run:
                continue
            if isinstance(last_run, str):
                try:
                    ts = datetime.datetime.fromisoformat(
                        last_run.replace("Z", "+00:00")
                    ).timestamp()
                except (ValueError, TypeError):
                    continue
            else:
                ts = float(last_run)
            if ts > since:
                completions.append(
                    {
                        "job_id": job_id,
                        "name": job.get("name", "Unknown"),
                        "status": job.get("last_status", "unknown"),
                        "completed_at": ts,
                        "toast_notifications": job.get("toast_notifications") is not False,
                    }
                )
        latest_session_info = _latest_cron_session_info_for_jobs(
            [job.get("id", "") for job in jobs],
            [c["job_id"] for c in completions],
        )
        for completion in completions:
            info = latest_session_info.get(str(completion.get("job_id", "") or ""), {})
            completion["session_id"] = str(info.get("session_id", "") or "")
            if info.get("message_count") is not None:
                completion["message_count"] = int(info["message_count"])
        return j(handler, {"completions": completions, "since": since})
    except ImportError:
        return j(handler, {"completions": [], "since": since})


_PROJECT_CONTEXT_HERMES_NAMES = (".hermes.md", "HERMES.md")
# Mirror the agent's lowercase filename variants (agents.md / claude.md) so the
# tab does not under-report on case-sensitive filesystems.
_PROJECT_CONTEXT_CWD_NAMES = (
    "AGENTS.md",
    "agents.md",
    "CLAUDE.md",
    "claude.md",
    ".cursorrules",
)
# Modern Cursor rules live in a directory of .mdc files, which the agent also loads.
_PROJECT_CONTEXT_CURSOR_RULES_GLOB = ".cursor/rules/*.mdc"
_PROJECT_CONTEXT_MAX_BYTES = 20_000


def _strip_project_context_frontmatter(content: str) -> str:
    """Strip a leading YAML frontmatter block, mirroring the agent's loader.

    The agent removes a leading ``---\\n ... \\n---`` block before injecting a
    context file. Without this the tab would display frontmatter the agent never
    sends to the model.
    """
    if not content.startswith("---"):
        return content
    lines = content.splitlines(keepends=True)
    if not lines or lines[0].strip() != "---":
        return content
    for idx in range(1, len(lines)):
        if lines[idx].strip() in ("---", "..."):
            return "".join(lines[idx + 1:]).lstrip("\n")
    return content


def _project_context_git_root(start: Path) -> Path | None:
    """Return the nearest git root for project context discovery."""
    current = start.resolve()
    for parent in [current, *current.parents]:
        if (parent / ".git").exists():
            return parent
    return None


def _project_context_candidates(workspace: Path) -> list[Path]:
    """Mirror the agent's first-match project context file priority.

    #4164: in a non-git workspace the agent's ``_find_hermes_md`` walks all
    the way up to filesystem root because its ``stop_at = git_root`` is
    ``None`` — so a workspace at ``/tmp/x/project/subdir`` could surface
    ``/tmp/x/HERMES.md`` (a file *outside* the user's workspace) in the
    Project Context tab.

    The WebUI tab is a read-only mirror of what the agent injects, so the
    safest bound that does not over-promise is: when there is no git root,
    treat the workspace itself as the stop boundary. The cwd is still
    scanned (preserving the in-workspace AGENTS.md / HERMES.md behavior),
    but we no longer walk into the user's home directory or ``/tmp``.

    The agent-side walk in ``agent/prompt_builder._find_hermes_md`` should
    be bounded the same way for full parity; until that ships the WebUI
    will under-report context files that live *above* a non-git workspace,
    which is strictly less surprising than over-reporting them.
    """
    cwd = workspace.resolve()
    candidates: list[Path] = []
    git_root = _project_context_git_root(cwd)
    # When inside a git tree, walk up to the git root as before. When not,
    # bound the walk at the workspace itself so we never surface files
    # above the user's workspace.
    stop_at = git_root if git_root is not None else cwd

    for directory in [cwd, *cwd.parents]:
        for name in _PROJECT_CONTEXT_HERMES_NAMES:
            candidates.append(directory / name)
        if directory == stop_at:
            break

    for name in _PROJECT_CONTEXT_CWD_NAMES:
        candidates.append(cwd / name)

    # Modern Cursor rules: the agent reads .cursor/rules/*.mdc in addition to
    # the legacy .cursorrules file. Sort for deterministic ordering.
    try:
        candidates.extend(sorted(cwd.glob(_PROJECT_CONTEXT_CURSOR_RULES_GLOB)))
    except OSError:
        pass

    return candidates


def _memory_project_context_workspace(parsed) -> Path | None:
    qs = parse_qs(parsed.query or "") if parsed is not None else {}
    sid = qs.get("session_id", [""])[0]
    if sid:
        try:
            # A blank session workspace (freshly-created/draft sessions) must not
            # fall through to Path("").resolve(), which returns the server's own
            # CWD and would surface the install's AGENTS.md/HERMES.md as if it
            # were the user's project context.
            session = get_session(sid)
            ws = (session.workspace or "").strip()
            if not ws:
                return None
            return _resolve_path(ws, profile=getattr(session, "profile", None))
        except Exception:
            return None

    raw_workspace = qs.get("workspace", [""])[0] or os.environ.get("TERMINAL_CWD", "") or get_last_workspace()
    if not raw_workspace:
        return None
    try:
        return resolve_trusted_workspace(raw_workspace)
    except Exception:
        logger.debug("Skipping project context for untrusted workspace %s", raw_workspace, exc_info=True)
        return None


def _read_active_project_context(workspace: Path | None) -> dict:
    payload = {
        "content": "",
        "path": "",
        "mtime": None,
        "workspace": str(workspace) if workspace else "",
        "shadowed": [],
    }
    if not workspace:
        return payload
    try:
        if not workspace.exists() or not workspace.is_dir():
            return payload
    except OSError:
        return payload

    seen: set[str] = set()
    readable: list[dict] = []
    for candidate in _project_context_candidates(workspace):
        try:
            if not candidate.is_file():
                continue
            resolved = candidate.resolve()
            key = os.path.normcase(str(resolved)).casefold()
            if key in seen:
                continue
            seen.add(key)
            content = resolved.read_text(encoding="utf-8", errors="replace")
            # Mirror what the agent actually injects: strip YAML frontmatter and
            # cap each source, so the tab reports the effective context rather
            # than raw file bytes the agent never sends to the model.
            content = _strip_project_context_frontmatter(content)
            if len(content) > _PROJECT_CONTEXT_MAX_BYTES:
                content = content[:_PROJECT_CONTEXT_MAX_BYTES]
            if not content.strip():
                continue
            readable.append(
                {
                    "name": resolved.name,
                    "path": str(resolved),
                    "content": content,
                    "mtime": resolved.stat().st_mtime,
                }
            )
        except Exception:
            logger.debug("Could not read project context candidate %s", candidate, exc_info=True)

    if not readable:
        return payload

    active = readable[0]
    payload.update(
        {
            "content": active["content"],
            "path": active["path"],
            "mtime": active["mtime"],
            "name": active["name"],
            "shadowed": [
                {
                    "name": item["name"],
                    "path": item["path"],
                    "mtime": item["mtime"],
                    "shadowed_by": active["name"],
                    "shadowed_by_path": active["path"],
                }
                for item in readable[1:]
            ],
        }
    )
    return payload


def _handle_memory_read(handler, parsed=None):
    try:
        from api.profiles import get_active_hermes_home

        home = get_active_hermes_home()
        mem_dir = home / "memories"
    except ImportError:
        home = Path.home() / ".hermes"
        mem_dir = home / "memories"

    # Respect memory_enabled and user_profile_enabled config flags (#6406)
    # Use get_config_snapshot() for per-profile isolation — get_config() returns
    # the process-global mutable _cfg_cache which races across profiles.
    # The flags are nested under cfg["memory"] in Hermes Agent's schema.
    cfg = get_config_snapshot()
    mem = cfg.get("memory") if isinstance(cfg, dict) else None
    mem_cfg = mem if isinstance(mem, dict) else {}
    memory_enabled = _webui_truthy(mem_cfg.get("memory_enabled", True))
    user_profile_enabled = _webui_truthy(mem_cfg.get("user_profile_enabled", True))

    mem_file = mem_dir / "MEMORY.md" if memory_enabled else None
    user_file = mem_dir / "USER.md" if user_profile_enabled else None
    soul_file = home / "SOUL.md"
    memory = (
        mem_file.read_text(encoding="utf-8", errors="replace")
        if mem_file and mem_file.exists()
        else ""
    )
    user = (
        user_file.read_text(encoding="utf-8", errors="replace")
        if user_file and user_file.exists()
        else ""
    )
    soul = (
        soul_file.read_text(encoding="utf-8", errors="replace")
        if soul_file.exists()
        else ""
    )
    project_context = _read_active_project_context(_memory_project_context_workspace(parsed))
    return j(
        handler,
        {
            "memory": _redact_text(memory),
            "user": _redact_text(user),
            "soul": _redact_text(soul),
            "project_context": _redact_text(project_context["content"]),
            "memory_path": str(mem_file) if mem_file else "",
            "user_path": str(user_file) if user_file else "",
            "soul_path": str(soul_file),
            "project_context_path": project_context["path"],
            "project_context_name": project_context.get("name", ""),
            "project_context_workspace": project_context["workspace"],
            "memory_mtime": mem_file.stat().st_mtime if mem_file and mem_file.exists() else None,
            "user_mtime": user_file.stat().st_mtime if user_file and user_file.exists() else None,
            "soul_mtime": soul_file.stat().st_mtime if soul_file.exists() else None,
            "project_context_mtime": project_context["mtime"],
            "project_context_shadowed": project_context["shadowed"],
            "external_notes_enabled": _external_notes_sources_enabled(cfg),
        },
    )


# ── POST route helpers ────────────────────────────────────────────────────────


def _handle_sessions_cleanup(handler, body, zero_only=False):
    cleaned = 0
    phase1_removed_ids = set()

    # Phase 1: Clean orphan session files (existing behavior).
    for p in SESSION_DIR.glob("*.json"):
        if p.name.startswith("_"):
            continue
        try:
            s = Session.load(p.stem)
            if zero_only:
                should_delete = s and len(s.messages) == 0
            else:
                should_delete = s and s.title == "Untitled" and len(s.messages) == 0
            if should_delete:
                with LOCK:
                    SESSIONS.pop(p.stem, None)
                p.unlink(missing_ok=True)
                cleaned += 1
                phase1_removed_ids.add(p.stem)
        except Exception:
            logger.debug("Failed to clean up session file %s", p)

    phase1_touched = bool(cleaned)
    phase2_rewrote_index = False

    # Phase 2: Index-only ghost sweep (#5331).
    # Remove index entries that have no backing .json file and no
    # in-memory session.  These are orphaned rows left by a write path that
    # updated _index.json without writing the session sidecar.
    #
    # Title-agnostic (#5331): a session with no backing file has no real
    # data regardless of its title, so even a non-"Untitled" stub in the
    # index is a ghost.  A legitimate session always has a sidecar file.
    #
    # Holds _INDEX_WRITE_LOCK for the full read-modify-write cycle to
    # prevent races with concurrent Session.save() / prune_session_from_index().
    if SESSION_INDEX_FILE.exists():
        try:
            from api.models import _INDEX_WRITE_LOCK, _safe_replace

            with _INDEX_WRITE_LOCK:
                index_file_data = json.loads(
                    SESSION_INDEX_FILE.read_bytes()
                )
                if isinstance(index_file_data, list):
                    live_ids = {
                        p.stem
                        for p in SESSION_DIR.glob("*.json")
                        if not p.name.startswith("_")
                    }
                    with LOCK:
                        in_memory_ids = set(SESSIONS.keys())

                    survivors = []
                    for entry in index_file_data:
                        sid = entry.get("session_id")
                        if not sid or sid in live_ids or sid in in_memory_ids:
                            survivors.append(entry)
                            continue
                        # Phase 1 already removed the backing file for this
                        # sid, so the index entry is stale too.  Drop it
                        # from the index without double-counting.
                        if sid in phase1_removed_ids:
                            continue
                        # Index-only ghost — no backing file, not in memory.
                        cleaned += 1
                        # Ghost not added to survivors — removed from index.

                    if cleaned > 0 and len(survivors) < len(index_file_data):
                        _tmp = SESSION_INDEX_FILE.with_suffix(
                            f".tmp.{os.getpid()}.{threading.current_thread().ident}"
                        )
                        _payload = json.dumps(survivors, ensure_ascii=False, indent=2)
                        try:
                            with open(_tmp, "w", encoding="utf-8") as f:
                                f.write(_payload)
                                f.flush()
                                os.fsync(f.fileno())
                            _safe_replace(_tmp, SESSION_INDEX_FILE)
                            phase2_rewrote_index = True
                        except Exception:
                            try:
                                _tmp.unlink(missing_ok=True)
                            except Exception:
                                pass
                            raise
        except Exception:
            logger.debug(
                "Failed to clean up index-only session entries", exc_info=True
            )

    # Post-cleanup index invalidation.
    # When Phase 1 removed files and Phase 2 didn't already clean the
    # index, delete the index to force a fresh rebuild from disk on the
    # next sidebar poll.  When Phase 2 succeeded the index is already
    # correct, so keep it (avoids a wasteful rebuild).
    if phase1_touched and not phase2_rewrote_index and SESSION_INDEX_FILE.exists():
        SESSION_INDEX_FILE.unlink(missing_ok=True)

    return j(handler, {"ok": True, "cleaned": cleaned})


def _handle_btw(handler, body):
    """POST /api/btw — ephemeral side question using session context.

    Creates a temporary hidden session, streams the answer via SSE, then
    discards the session. The parent session is not modified.
    """
    try:
        require(body, "session_id")
        require(body, "question")
    except ValueError as e:
        return bad(handler, str(e))
    stale_response = _agent_runtime_barrier_response(runner_local_owned=False)
    if stale_response is not None:
        return j(handler, stale_response, status=409)
    if _session_is_subagent_view_only(str(body.get("session_id") or "")):
        return bad(handler, "Subagent sessions are view-only and cannot be used for /btw from WebUI", 400)
    try:
        s = get_session(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    question = str(body["question"]).strip()
    if not question:
        return bad(handler, "question is required")
    # Duplicate-stream guard (same pattern as chat/start)
    current_stream_id = getattr(s, "active_stream_id", None)
    if current_stream_id:
        with STREAMS_LOCK:
            if current_stream_id in STREAMS:
                return j(handler, {"error": "session already has an active stream"}, status=409)
        s.active_stream_id = None
    # Create ephemeral hidden session inheriting context
    from api.models import new_session as _new_session
    model_provider = getattr(s, 'model_provider', None)
    ephemeral = _new_session(
        workspace=s.workspace,
        model=s.model,
        model_provider=model_provider,
        profile=getattr(s, 'profile', None),
    )
    # Copy conversation history for context (agent reads from messages)
    ephemeral.messages = list(s.messages or [])
    ephemeral.title = f"btw: {question[:60]}"
    ephemeral.save()
    stream_id = uuid.uuid4().hex
    ephemeral.active_stream_id = stream_id
    register_session_writeback_owner(ephemeral.session_id, stream_id)
    ephemeral.save()
    stream = create_stream_channel()
    register_stream_owner(stream_id, ephemeral.session_id)
    with STREAMS_LOCK:
        STREAMS[stream_id] = stream
    from api.background import track_btw
    track_btw(body["session_id"], ephemeral.session_id, stream_id, question)
    thr = threading.Thread(
        target=_run_agent_streaming,
        args=(ephemeral.session_id, question, s.model, s.workspace, stream_id, None),
        kwargs={"ephemeral": True, "model_provider": model_provider},
        daemon=True,
    )
    thr.start()
    return j(handler, {"stream_id": stream_id, "session_id": ephemeral.session_id, "parent_session_id": body["session_id"]})


def _handle_background(handler, body):
    """POST /api/background — run prompt in parallel background agent.

    Creates a hidden session, starts streaming in a daemon thread.
    Frontend polls /api/background/status for completed results.
    """
    try:
        require(body, "session_id")
        require(body, "prompt")
    except ValueError as e:
        return bad(handler, str(e))
    stale_response = _agent_runtime_barrier_response(runner_local_owned=False)
    if stale_response is not None:
        return j(handler, stale_response, status=409)
    try:
        s = get_session(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    prompt = str(body["prompt"]).strip()
    if not prompt:
        return bad(handler, "prompt is required")
    from api.models import new_session as _new_session
    model_provider = getattr(s, 'model_provider', None)
    bg = _new_session(
        workspace=s.workspace,
        model=s.model,
        model_provider=model_provider,
        profile=getattr(s, 'profile', None),
    )
    bg.title = f"bg: {prompt[:60]}"
    bg.save()
    stream_id = uuid.uuid4().hex
    bg.active_stream_id = stream_id
    register_session_writeback_owner(bg.session_id, stream_id)
    bg.save()
    stream = create_stream_channel()
    register_stream_owner(stream_id, bg.session_id)
    with STREAMS_LOCK:
        STREAMS[stream_id] = stream
    task_id = uuid.uuid4().hex[:8]
    from api.background import track_background, complete_background
    parent_sid = body["session_id"]
    bg_sid = bg.session_id
    track_background(parent_sid, bg_sid, stream_id, task_id, prompt)

    def _run_bg_and_notify():
        """Run the background agent, then mark the tracked task `done` with the
        last assistant reply so `/api/background/status` can surface it.  Without
        this, `complete_background()` is never called and the result is lost —
        `get_results()` would see a forever-`running` task and return nothing.
        """
        try:
            _run_agent_streaming(
                bg_sid,
                prompt,
                s.model,
                s.workspace,
                stream_id,
                None,
                model_provider=model_provider,
            )
            # Reload the bg session from disk and extract the final assistant reply.
            try:
                from api.models import Session as _Session
                reloaded = _Session.load(bg_sid)
                _answer = ""
                for _m in reversed((reloaded.messages if reloaded else None) or []):
                    if not isinstance(_m, dict) or _m.get("role") != "assistant":
                        continue
                    if _m.get("_error"):
                        continue
                    _content = str(_m.get("content") or "").strip()
                    if _content:
                        _answer = _content
                        break
                complete_background(parent_sid, task_id, _answer or "(no answer produced)")
            except Exception:
                complete_background(parent_sid, task_id, "(background task failed)")
            # Best-effort cleanup of the hidden bg session file so it doesn't
            # clutter the sidebar or SESSION_DIR. The index is pruned on the
            # next rebuild via _index_entry_exists().
            try:
                (SESSION_DIR / f"{bg_sid}.json").unlink(missing_ok=True)
            except Exception:
                pass
        except Exception:
            try:
                complete_background(parent_sid, task_id, "(background task failed)")
            except Exception:
                pass

    thr = threading.Thread(target=_run_bg_and_notify, daemon=True)
    thr.start()
    return j(handler, {"task_id": task_id, "stream_id": stream_id, "session_id": bg.session_id})


def _checkpoint_user_message_for_eager_session_save(s, msg: str, attachments, started_at: float | None, source: str = "webui") -> None:
    """Materialize the current user turn for eager first-turn persistence.

    The streaming thread still receives ``pending_user_message`` so existing
    cancel/recovery/final-merge paths keep their current contract. Eager mode
    only adds a durable display-message checkpoint before the agent launches.
    """
    if not msg:
        return
    existing = list(getattr(s, "messages", None) or [])
    if existing:
        latest = existing[-1]
        if isinstance(latest, dict) and latest.get("role") == "user":
            latest_text = " ".join(str(latest.get("content") or "").split())
            msg_text = " ".join(str(msg or "").split())
            if latest_text == msg_text:
                if str(source or "").strip().lower() == "fork":
                    latest["_fork_child_turn"] = s.session_id
                return
    user_msg = {"role": "user", "content": msg}
    from api.process_event_utils import build_active_turn_token, stamp_message_source

    stamp_message_source(
        user_msg,
        source,
        active_turn_token=build_active_turn_token(getattr(s, "active_stream_id", None), started_at),
    )
    if str(source or "").strip().lower() == "fork":
        user_msg["_fork_child_turn"] = s.session_id
    if isinstance(started_at, (int, float)) and started_at > 0:
        user_msg["timestamp"] = float(started_at)
    if attachments:
        user_msg["attachments"] = list(attachments)
    s.messages.append(user_msg)
    # The new user turn is now committed to messages (#3831): advance the
    # truncation watermark to the new message's timestamp so that
    # merge_session_messages_append_only() still filters out replaced
    # pre-edit rows from state.db whose timestamps fall below the boundary.
    # The merge's sidecar_advanced_past_watermark guard (models.py:5172)
    # allows state.db rows newer than the watermark, so post-edit turns
    # are not dropped. Never 0.0 (the truncate-to-empty sentinel, #2914).
    if getattr(s, "truncation_watermark", None):
        s.truncation_watermark = user_msg.get("timestamp") or time.time()


def _is_default_or_empty_session_title(title) -> bool:
    return str(title or "").strip() in ("", "Untitled", "New Chat")


def _provisional_title_from_prompt(prompt: str, fallback: str = "Untitled") -> str:
    text = str(prompt or "").strip()
    if not text:
        return fallback
    return title_from([{"role": "user", "content": text}], fallback) or fallback


_RETAINED_CONTEXT_USER_UNSET = object()


def _prepare_chat_start_session_for_stream(
    s,
    *,
    msg: str,
    attachments,
    workspace: str,
    model: str,
    model_provider,
    stream_id: str,
    started_at: float | None = None,
    source: str = "webui",
    retained_user=None,
    retained_context_user=_RETAINED_CONTEXT_USER_UNSET,
    defer_save: bool = False,
):
    """Persist chat-start state according to webui.session_save_mode.

    ``deferred`` keeps the existing sidecar/WAL-backed behaviour: save pending
    fields but leave the display transcript empty until the agent merges the
    result. ``eager`` additionally writes the current user turn into messages so
    a process restart immediately after /api/chat/start preserves the prompt as
    a normal session message. Empty sessions are never saved here because this
    helper only runs after a non-empty message is validated.
    """
    effective_source = (
        "fork"
        if str(getattr(s, "session_source", None) or "").strip().lower() == "fork"
        else source
    )
    s.workspace = workspace
    s.model = model
    s.model_provider = model_provider
    s.active_stream_id = stream_id
    register_session_writeback_owner(s.session_id, stream_id)
    s.post_compression_context_tokens_estimate = None
    s.pending_user_message = msg
    s.pending_attachments = attachments
    s.pending_started_at = started_at if started_at is not None else time.time()
    s.pending_user_source = effective_source
    s._webui_pending_user_timestamp_identity = None
    if retained_user is not None:
        from api.process_event_utils import build_active_turn_token

        retained_user["timestamp"] = s.pending_started_at
        active_turn_token = build_active_turn_token(stream_id, s.pending_started_at)
        retained_user["_active_turn_token"] = active_turn_token
        if str(effective_source or "").strip().lower() == "fork":
            retained_user["_fork_child_turn"] = s.session_id
        if retained_context_user is not _RETAINED_CONTEXT_USER_UNSET:
            if retained_context_user is not None and not any(
                row is retained_context_user
                for row in list(getattr(s, "context_messages", None) or [])
            ):
                raise RuntimeError("regeneration retained context row is not installed")
            if isinstance(retained_context_user, dict):
                retained_context_user["timestamp"] = s.pending_started_at
                retained_context_user["_active_turn_token"] = active_turn_token
                if str(effective_source or "").strip().lower() == "fork":
                    retained_context_user["_fork_child_turn"] = s.session_id
        else:
            retained_id = retained_user.get("id") or retained_user.get("message_id")
            retained_old_timestamp = retained_user.get("timestamp")
            retained_old_content = retained_user.get("content")
            for context_row in reversed(list(getattr(s, "context_messages", None) or [])):
                if not isinstance(context_row, dict) or context_row.get("role") != "user":
                    continue
                context_id = context_row.get("id") or context_row.get("message_id")
                id_match = retained_id is not None and context_id == retained_id
                old_shape_match = (
                    retained_old_timestamp is not None
                    and context_row.get("timestamp") == retained_old_timestamp
                    and context_row.get("content") == retained_old_content
                )
                if not (id_match or old_shape_match):
                    continue
                context_row["timestamp"] = s.pending_started_at
                context_row["_active_turn_token"] = active_turn_token
                if str(effective_source or "").strip().lower() == "fork":
                    context_row["_fork_child_turn"] = s.session_id
                break
    current_title = getattr(s, "title", None)
    if retained_user is None and _is_default_or_empty_session_title(current_title):
        provisional_title = _provisional_title_from_prompt(msg, current_title or "Untitled")
        if provisional_title and not _is_default_or_empty_session_title(provisional_title):
            s.title = provisional_title
    if retained_user is None and get_webui_session_save_mode() == "eager":
        _checkpoint_user_message_for_eager_session_save(
            s,
            msg,
            attachments,
            s.pending_started_at,
            source=effective_source,
        )
    if not defer_save:
        s.save()


def _cleanup_chat_start_launch_failure(session, stream_id: str) -> None:
    """Release state registered before a worker thread successfully starts."""
    clear_session_writeback_owner_if_owned(session.session_id, stream_id)
    unregister_stream_owner(stream_id)
    with STREAMS_LOCK:
        STREAMS.pop(stream_id, None)
    STREAM_GOAL_RELATED.pop(stream_id, None)
    # The session-field reset needs the same concurrency discipline as the
    # registry half: hold the per-session lock and re-resolve the canonical
    # session before clearing anything. Mutating the passed-in stale object
    # could wipe a concurrent successor turn's pending fields, and saving it
    # could resurrect a session deleted while the launch was failing. Same
    # pattern as the #1533 race fix (routes.py:3077) and the anchor-scene
    # write guard (routes.py:5140).
    #
    # This runs while the original launch failure is being handled, so it must
    # never raise. Lock acquisition and session resolution can fail on their own
    # (I/O, deserialization), and an escaping error here would mask the launch
    # failure the caller is about to report while leaving the reset half done.
    try:
        with _get_session_agent_lock(session.session_id):
            try:
                canonical = get_session(session.session_id)
            except KeyError:
                return  # session deleted while the thread launch was failing
            if getattr(canonical, "active_stream_id", None) != stream_id:
                return  # a successor turn already owns the session
            canonical.active_stream_id = None
            canonical.pending_user_message = None
            canonical.pending_attachments = []
            canonical.pending_started_at = None
            canonical.pending_user_source = None
            try:
                canonical.save()
            except Exception:
                logger.debug(
                    "Failed to persist chat-start cleanup after worker launch failure for %s",
                    stream_id,
                    exc_info=True,
                )
    except Exception:
        logger.debug(
            "Failed to reset session state after worker launch failure for %s",
            stream_id,
            exc_info=True,
        )


def _is_hidden_empty_session(s) -> bool:
    return (
        getattr(s, "title", "Untitled") == "Untitled"
        and not getattr(s, "messages", None)
        and not getattr(s, "active_stream_id", None)
        and not getattr(s, "pending_user_message", None)
        and not getattr(s, "worktree_path", None)
    )


def _active_stream_blocks_chat_start(session, stream_id: str | None) -> bool:
    """Return whether an active_stream_id still owns this session's next turn.

    ``active_stream_id`` is written before the SSE channel is registered, so a
    very fresh pending turn must also block duplicate chat_start requests. If we
    only check STREAMS here, a second request can race through the registration
    gap and overwrite the sidecar owner.
    """
    if not stream_id:
        return False
    with STREAMS_LOCK:
        if stream_id in STREAMS:
            return True
    try:
        from api import config as _live_config
        with _live_config.ACTIVE_RUNS_LOCK:
            if stream_id in (_live_config.ACTIVE_RUNS or {}):
                return True
    except Exception:
        pass
    if getattr(session, "pending_user_message", None):
        try:
            from api.models import _REPAIR_STALE_PENDING_GRACE_SECONDS
            grace_seconds = float(_REPAIR_STALE_PENDING_GRACE_SECONDS)
        except Exception:
            grace_seconds = 30.0
        try:
            pending_started_at = float(getattr(session, "pending_started_at", None) or 0)
        except Exception:
            pending_started_at = 0.0
        if pending_started_at and time.time() - pending_started_at < grace_seconds:
            return True
    return False


def _start_regeneration_stream_locked(
    s,
    *,
    turn,
    workspace: str,
    model: str,
    model_provider,
    normalized_model: bool,
    diag,
    goal_related: bool,
    source: str,
    moa_config,
    backend_is_gateway: bool,
):
    """Commit a retained-row regeneration before releasing its real worker."""
    from api.session_ops import (
        RegenerationUnavailable,
        apply_regeneration_plan,
        plan_regeneration,
        restore_regeneration_state,
        snapshot_regeneration_state,
    )

    try:
        plan = plan_regeneration(
            s, expected_revision=turn.revision, lock_held=True
        )
        turn = plan.turn
    except RegenerationUnavailable as exc:
        return {
            "error": str(exc),
            "code": exc.code,
            "_status": exc.status,
        }
    # Snapshot only after lock-held authority validation, before mutation.
    snapshot = snapshot_regeneration_state(s)
    if compression_recovery_payload_for_session(s):
        clear_compression_recovery(s)
    stream_id = uuid.uuid4().hex
    gateway_starting = False
    thread_started = False
    save_attempted = False
    accepted = False
    journal_event = {}
    release_worker = threading.Event()
    abort_worker = threading.Event()
    worker_thread = None

    worker_target = (
        _run_gateway_chat_streaming if backend_is_gateway else _run_agent_streaming
    )
    worker_kwargs = {
        "model_provider": model_provider,
        "goal_related": goal_related,
    }
    if backend_is_gateway:
        worker_kwargs["regeneration"] = True
    if moa_config and not backend_is_gateway:
        worker_kwargs["moa_config"] = moa_config

    def _gated_worker():
        release_worker.wait()
        if abort_worker.is_set():
            return
        worker_target(
            s.session_id,
            turn.message_text,
            model,
            workspace,
            stream_id,
            copy.deepcopy(turn.attachments),
            **worker_kwargs,
        )

    def _cleanup_owned_start():
        if goal_related:
            STREAM_GOAL_RELATED.pop(stream_id, None)
        with STREAMS_LOCK:
            STREAMS.pop(stream_id, None)
        unregister_stream_owner(stream_id)
        clear_session_writeback_owner_if_owned(s.session_id, stream_id)
        if gateway_starting:
            try:
                from api.gateway_chat import (
                    _clear_gateway_run_starting,
                    _finish_gateway_run_starting,
                )

                _finish_gateway_run_starting(stream_id)
                _clear_gateway_run_starting(stream_id)
            except Exception:
                logger.debug(
                    "Failed to clear compensated gateway start %s",
                    stream_id,
                    exc_info=True,
                )

    try:
        applied, retained_context_user = apply_regeneration_plan(
            s,
            plan,
            return_context_user=True,
        )
        if not applied:
            restore_regeneration_state(s, snapshot)
            return {
                "error": "Session changed while regeneration was being prepared.",
                "code": "stale_regeneration_revision",
                "_status": 409,
            }
        retained_user = s.messages[-1]
        msg = turn.message_text
        attachments = copy.deepcopy(turn.attachments)
        was_hidden_empty_session = _is_hidden_empty_session(s)
        _prepare_chat_start_session_for_stream(
            s,
            msg=msg,
            attachments=attachments,
            workspace=workspace,
            model=model,
            model_provider=model_provider,
            stream_id=stream_id,
            source=turn.source,
            retained_user=retained_user,
            retained_context_user=retained_context_user,
            defer_save=True,
        )

        diag.stage("turn_journal_submitted") if diag else None
        from api.turn_journal import append_turn_journal_event

        journal_event = append_turn_journal_event(
            s.session_id,
            {
                "event": "submitted",
                "stream_id": stream_id,
                "role": "user",
                "content": msg,
                "attachments": attachments,
                "workspace": workspace,
                "model": model,
                "model_provider": model_provider,
                "created_at": s.pending_started_at,
            },
        )
        diag.stage("stream_registration") if diag else None
        stream = create_stream_channel()
        register_stream_owner(stream_id, s.session_id)
        with STREAMS_LOCK:
            STREAMS[stream_id] = stream
        if goal_related:
            STREAM_GOAL_RELATED[stream_id] = True
        if backend_is_gateway:
            from api.gateway_chat import _mark_gateway_run_starting

            gateway_starting = True
            _mark_gateway_run_starting(stream_id)

        diag.stage("worker_thread_start") if diag else None
        worker_thread = threading.Thread(target=_gated_worker, daemon=True)
        worker_thread.start()
        thread_started = True
        save_attempted = True
        s.save()
        accepted = True
        set_last_workspace(workspace, profile=getattr(s, "profile", None))
        release_worker.set()
    except Exception as exc:
        abort_worker.set()
        release_worker.set()
        if (
            thread_started
            and worker_thread is not None
            and callable(getattr(worker_thread, "join", None))
        ):
            worker_thread.join(timeout=1)
        _cleanup_owned_start()
        if accepted:
            if journal_event:
                try:
                    append_turn_journal_event(
                        s.session_id,
                        {
                            "event": "interrupted",
                            "stream_id": stream_id,
                            "turn_id": journal_event.get("turn_id"),
                            "reason": "post_acceptance_workspace_failure",
                        },
                    )
                except Exception:
                    logger.warning("Failed to close accepted regeneration journal", exc_info=True)
            exc._regeneration_accepted = True
            raise
        restore_regeneration_state(s, snapshot)
        if save_attempted:
            try:
                s.save(touch_updated_at=False)
            except Exception:
                logger.exception(
                    "Failed to persist compensated regeneration for %s",
                    s.session_id,
                )
        if journal_event:
            try:
                append_turn_journal_event(
                    s.session_id,
                    {
                        "event": "interrupted",
                        "stream_id": stream_id,
                        "turn_id": journal_event.get("turn_id"),
                        "reason": "start_compensated",
                    },
                )
            except Exception:
                logger.warning(
                    "Failed to close compensated turn journal event",
                    exc_info=True,
                )
        raise

    release_worker.set()
    if was_hidden_empty_session:
        publish_session_list_changed(
            "session_new",
            profile=getattr(s, "profile", None),
            session_id=getattr(s, "session_id", None),
        )
    response = {
        "stream_id": stream_id,
        "session_id": s.session_id,
        "pending_started_at": s.pending_started_at,
        "turn_id": journal_event.get("turn_id"),
        "title": s.title,
    }
    if normalized_model:
        response["effective_model"] = model
    if model_provider:
        response["effective_model_provider"] = model_provider
    return response


def _active_run_stream_for_session(session_id: str | None) -> str | None:
    """Return a live worker stream for this session even if sidecar stream id is clear.

    cancel_stream() intentionally clears ``session.active_stream_id`` before the
    worker thread fully exits so Stop remains responsive. During that unwind
    window ACTIVE_RUNS is the worker-lifecycle truth; a successor chat/start for
    the same session must wait or it can reuse the cached agent while the old
    interrupt is still landing (#3808).

    Bounded: the post-cancel unwind is short (dominated by the worker finally's
    ``_ckpt_thread.join(timeout=15)``), and ``unregister_active_run`` runs in that
    finally, so a healthy worker leaves ACTIVE_RUNS within seconds. A detached /
    wedged worker that never reaches its finally (e.g. stuck in a provider call,
    or leaked by SIGKILL without restart) must NOT 409 the session forever — so an
    entry older than the unwind ceiling (180s) is treated as stale and ignored
    here. For a phase="cancelling" row the ceiling is anchored on the cancel time
    (``cancelled_at``), never on the original run start: cancel_stream() removes
    STREAMS itself, so absence from STREAMS is not worker-death proof, and a
    long-running turn that was just cancelled must not be reaped (and a successor
    admitted) while the old worker is still alive (#6623). A legitimately
    long-running turn keeps ``active_stream_id`` SET and is handled by
    ``_active_stream_blocks_chat_start`` above; this guard only covers the
    cleared-stream-id unwind window. (Codex brick-gate hardening, #3822.)
    """
    sid = str(session_id or "").strip()
    if not sid:
        return None
    ceiling = 180.0  # generous vs the 15s checkpoint-join unwind; finite to avoid permanent-409
    now = time.time()
    try:
        from api import config as _live_config
        # Snapshot the live-worker set BEFORE taking ACTIVE_RUNS_LOCK (sequential,
        # not nested, so no lock-ordering/deadlock risk). A long turn whose
        # active_stream_id was already cleared during final writeback can still be
        # mid-teardown — its STREAMS entry present — past the age ceiling, so age
        # alone must NOT pop its lifecycle row (health / background-wakeup /
        # active-agent-cache consumers read ACTIVE_RUNS as worker-lifecycle truth).
        with _live_config.STREAMS_LOCK:
            live_stream_ids = set(_live_config.STREAMS.keys())
        stale_stream_ids = []
        with _live_config.ACTIVE_RUNS_LOCK:
            for run_stream_id, raw in list((_live_config.ACTIVE_RUNS or {}).items()):
                stream_id = str((raw or {}).get("stream_id") or run_stream_id or "").strip()
                run_sid = str((raw or {}).get("session_id") or "").strip()
                if run_sid != sid or not stream_id:
                    continue
                try:
                    started_at = float((raw or {}).get("started_at") or 0)
                except (TypeError, ValueError):
                    started_at = 0.0
                # #6623 re-gate: cancel_stream() removes STREAMS itself, so a
                # missing SSE channel is NOT proof that a cancelled worker is
                # dead. For a phase="cancelling" row the unwind ceiling must be
                # anchored on the CANCEL time (cancelled_at, started_at as a
                # legacy fallback), never on the original run start: a turn that
                # ran for minutes and was just cancelled would otherwise be
                # reaped the instant its started_at crosses the ceiling and a
                # successor admitted while the old worker is still alive. A
                # recently cancelled run therefore keeps blocking a successor
                # (this function returns its stream id) until the worker either
                # unwinds (its finally unregisters the row within seconds) or
                # the cancel itself has been outstanding past the ceiling.
                _run_phase = str((raw or {}).get("phase") or "").strip()
                if _run_phase == "cancelling":
                    try:
                        _age_anchor = float(
                            (raw or {}).get("cancelled_at") or started_at or 0
                        )
                    except (TypeError, ValueError):
                        _age_anchor = started_at
                else:
                    _age_anchor = started_at
                # Past the unwind ceiling: never block a successor on it (the
                # anti-permanent-409 guarantee, #3822). Additionally reconcile the
                # zombie out of ACTIVE_RUNS so health/recovery polling stops seeing a
                # half-alive run — but ONLY when the worker is truly gone from
                # STREAMS, so a still-live / still-tearing-down worker keeps its
                # lifecycle row. Pop by the real dict key. (Codex gate, #4492)
                if _age_anchor and (now - _age_anchor) > ceiling:
                    if run_stream_id not in live_stream_ids and stream_id not in live_stream_ids:
                        stale_stream_ids.append(run_stream_id)
                    continue
                return stream_id
            for stale_stream_id in stale_stream_ids:
                (_live_config.ACTIVE_RUNS or {}).pop(stale_stream_id, None)
                # The zombie run is pruned directly here (not via the normal teardown
                # finally / unregister_active_run), so release its stream-owner entry too
                # or STREAM_SESSION_OWNERS leaks for every reconciled zombie. (#5198 gate)
                unregister_stream_owner(stale_stream_id)
    except Exception:
        return None
    return None


def _agent_runtime_barrier_response(
    *,
    runner_local_owned: bool = False,
    external_runtime_owned: bool | None = None,
) -> dict | None:
    """Return the typed stale-runtime response for local in-process turns."""
    if external_runtime_owned is True:
        return None
    if runner_local_owned and webui_gateway_chat_enabled(get_config()):
        return None
    from api.runtime_adapter import runtime_adapter_runner_enabled

    if runner_local_owned and runtime_adapter_runner_enabled():
        return None
    try:
        ensure_agent_runtime_current()
    except AgentRuntimeChangedError as exc:
        return agent_runtime_stale_payload(exc)
    return None


def _start_chat_stream_for_session(
    s,
    *,
    msg: str,
    attachments=None,
    workspace: str,
    model: str,
    model_provider=None,
    normalized_model: bool = False,
    diag=None,
    goal_related: bool = False,
    source: str = "webui",
    moa_config=None,
    external_runtime_owned: bool | None = None,
    regeneration=None,
):
    """Persist pending state, register an SSE channel, and start an agent turn."""
    if external_runtime_owned is None:
        external_runtime_owned = webui_gateway_chat_enabled(get_config())
    backend_is_gateway = bool(external_runtime_owned)
    stale_response = _agent_runtime_barrier_response(
        external_runtime_owned=backend_is_gateway,
    )
    if stale_response is not None:
        stale_response["_status"] = 409
        return stale_response
    attachments = attachments or []
    # Prevent duplicate runs in the same session while a stream is still active.
    # This commonly happens after page refresh/reconnect races and can produce
    # duplicated clarify cards for what appears to be a single user request.
    diag.stage("active_stream_check") if diag else None
    current_stream_id = getattr(s, "active_stream_id", None)
    if current_stream_id:
        if _active_stream_blocks_chat_start(s, current_stream_id):
            diag.stage("response_write") if diag else None
            return {
                "error": "session already has an active stream",
                "active_stream_id": current_stream_id,
                "_status": 409,
            }
        # Stale stream id from a previous run; clear and continue.
        diag.stage("stale_stream_cleanup") if diag else None
        _clear_stale_stream_state(s)

    # #1932: check if this session has a pending goal continuation flag.
    # The streaming hook sets PENDING_GOAL_CONTINUATION when goal_continue fires,
    # so the next chat/start for this session is automatically treated as goal-related.
    if not goal_related and s.session_id in PENDING_GOAL_CONTINUATION:
        goal_related = True
        PENDING_GOAL_CONTINUATION.discard(s.session_id)

    # process_complete wakeup (ours-original, Option B): if this session has a
    # pending process_complete marker (set by api/background_process.py drain),
    # discard it atomically here. Mirrors the goal_continue pattern (#1932).
    # The marker is server-internal telemetry; the actual wakeup is delivered
    # either server-side (Option Z) or via the PR #2279 next-turn drain.
    if s.session_id in PENDING_BG_TASK_COMPLETIONS:
        PENDING_BG_TASK_COMPLETIONS.discard(s.session_id)

    session_lock = _get_session_agent_lock(s.session_id)
    diag.stage("session_lock_wait") if diag else None
    while True:
        with session_lock:
            locked_stream_id = getattr(s, "active_stream_id", None)
            if locked_stream_id:
                if _active_stream_blocks_chat_start(s, locked_stream_id):
                    diag.stage("response_write") if diag else None
                    return {
                        "error": "session already has an active stream",
                        "active_stream_id": locked_stream_id,
                        "_status": 409,
                    }
                needs_stale_cleanup = True
            else:
                blocking_run_stream_id = _active_run_stream_for_session(s.session_id)
                if blocking_run_stream_id:
                    diag.stage("response_write") if diag else None
                    return {
                        "error": "session already has an active stream",
                        "active_stream_id": blocking_run_stream_id,
                        "_status": 409,
                    }
                needs_stale_cleanup = False
                if regeneration is not None:
                    return _start_regeneration_stream_locked(
                        s,
                        turn=regeneration,
                        workspace=workspace,
                        model=model,
                        model_provider=model_provider,
                        normalized_model=normalized_model,
                        diag=diag,
                        goal_related=goal_related,
                        source=source,
                        moa_config=moa_config,
                        backend_is_gateway=backend_is_gateway,
                    )
                stream_id = uuid.uuid4().hex
                diag.stage("save_pending_state") if diag else None
                was_hidden_empty_session = _is_hidden_empty_session(s)
                _prepare_chat_start_session_for_stream(
                    s,
                    msg=msg,
                    attachments=attachments,
                    workspace=workspace,
                    model=model,
                    model_provider=model_provider,
                    stream_id=stream_id,
                    source=source,
                )
                break
        if needs_stale_cleanup:
            diag.stage("stale_stream_cleanup") if diag else None
            cleared = _clear_stale_stream_state(s)
            if not cleared and getattr(s, "active_stream_id", None):
                diag.stage("response_write") if diag else None
                return {
                    "error": "session already has an active stream",
                    "active_stream_id": getattr(s, "active_stream_id", None),
                    "_status": 409,
                }
    if was_hidden_empty_session:
        publish_session_list_changed(
            "session_new",
            profile=getattr(s, "profile", None),
            session_id=getattr(s, "session_id", None),
        )
    diag.stage("turn_journal_submitted") if diag else None
    journal_event = {}
    try:
        from api.turn_journal import append_turn_journal_event
        journal_event = append_turn_journal_event(
            s.session_id,
            {
                "event": "submitted",
                "stream_id": stream_id,
                "role": "user",
                "content": msg,
                "attachments": attachments,
                "workspace": workspace,
                "model": model,
                "model_provider": model_provider,
                "created_at": s.pending_started_at,
            },
        )
    except Exception:
        logger.warning("Failed to append submitted turn journal event", exc_info=True)
    diag.stage("set_last_workspace") if diag else None
    set_last_workspace(workspace, profile=getattr(s, "profile", None))
    diag.stage("stream_registration") if diag else None
    stream = create_stream_channel()
    register_stream_owner(stream_id, s.session_id)
    with STREAMS_LOCK:
        STREAMS[stream_id] = stream
    # #1932: mark stream as goal-related so the streaming hook evaluates the goal.
    if goal_related:
        STREAM_GOAL_RELATED[stream_id] = True
    diag.stage("worker_thread_start") if diag else None
    worker_target = _run_gateway_chat_streaming if backend_is_gateway else _run_agent_streaming
    worker_kwargs = {"model_provider": model_provider, "goal_related": goal_related}
    if moa_config and not backend_is_gateway:
        worker_kwargs["moa_config"] = moa_config
    if backend_is_gateway:
        from api.gateway_chat import _mark_gateway_run_starting
        _mark_gateway_run_starting(stream_id)
    thr = threading.Thread(
        target=worker_target,
        args=(s.session_id, msg, model, workspace, stream_id, attachments),
        kwargs=worker_kwargs,
        daemon=True,
    )
    try:
        thr.start()
    except Exception:
        if backend_is_gateway:
            try:
                from api.gateway_chat import _finish_gateway_run_starting
                _finish_gateway_run_starting(stream_id)
                from api.gateway_chat import _clear_gateway_run_starting
                _clear_gateway_run_starting(stream_id)
            except Exception:
                logger.debug("Failed to record gateway run-start failure for stream %s", stream_id, exc_info=True)
        _cleanup_chat_start_launch_failure(s, stream_id)
        raise
    response = {
        "stream_id": stream_id,
        "session_id": s.session_id,
        "pending_started_at": s.pending_started_at,
        "turn_id": journal_event.get("turn_id"),
        "title": s.title,
    }
    if normalized_model:
        response["effective_model"] = model
    if model_provider:
        response["effective_model_provider"] = model_provider
    return response


def _runtime_runner_client_factory():
    """Return the configured runner-local client.

    `runner-local` remains default-off and bounded: without an explicit runner
    endpoint this factory preserves the existing "runner-local chat backend is
    not configured" 501 path. When
    `HERMES_WEBUI_RUNNER_BASE_URL` is set, the WebUI process only acts as a
    transport client; the runner endpoint owns execution, run ids, replay, and
    controls.
    """
    # Keep this literal here for route-level contract tests and readable 501 provenance:
    # "runner-local chat backend is not configured"
    from api.runner_client import HttpRunnerClient

    return HttpRunnerClient.from_env()


def _chat_start_response_from_run_start(result):
    """Expose only the legacy browser-facing chat-start response fields."""
    payload = dict(getattr(result, "payload", {}) or {})
    response = {}
    for key in (
        "stream_id",
        "session_id",
        "pending_started_at",
        "turn_id",
        "title",
        "effective_model",
        "effective_model_provider",
        "error",
        "code",
        "active_stream_id",
        "_status",
    ):
        if key in payload:
            response[key] = payload[key]
    response.setdefault("stream_id", result.stream_id)
    response.setdefault("session_id", result.session_id)
    return response


def _runtime_adapter_goal_action(goal_args: str) -> str:
    """Return the bounded RuntimeAdapter goal action for WebUI /goal args."""
    action = str(goal_args or "").strip().lower()
    if not action or action == "status":
        return "status"
    if action in ("pause", "resume"):
        return action
    if action in ("clear", "stop", "done"):
        return "clear"
    return "set"


def _start_run(
    s,
    *,
    msg: str,
    attachments,
    workspace: str,
    model,
    model_provider,
    normalized_model,
    source: str,
    route: str,
    diag=None,
    moa_config=None,
    gateway_chat_enabled: bool | None = None,
    regeneration=None,
):
    """Shared start-run helper for /api/chat/start and start_session_turn.

    Centralizes the runtime-adapter selection block (Q-2979-A2 / Copilot
    discussion_r3305864087/r3305864173) so both entrypoints honor
    ``runtime_adapter_enabled()`` / ``runtime_adapter_runner_enabled()`` the
    same way. Prior to this helper ``start_session_turn`` bypassed the
    adapter path entirely, so a process-wakeup turn skipped the adapter that
    a human-typed turn would have hit — a behavioral divergence.

    ``source`` is the StartRunRequest.source (``"webui"`` for browser POSTs,
    ``"process_wakeup"`` for the drain-thread wakeup). ``route`` is the
    metadata.route label that lands on the run record for observability.

    Returns a dict with ``_status`` plus the legacy chat-start response
    fields (``stream_id``, ``session_id``, etc.). Adapter selection that
    returns no adapter is surfaced as ``{"error": str(exc), "_status": 501}``
    so both call sites can map it onto their own HTTP shape.
    """
    from api.runtime_adapter import (
        LegacyJournalRuntimeAdapter,
        StartRunRequest,
        build_runtime_adapter,
        runtime_adapter_enabled,
        runtime_adapter_runner_enabled,
    )

    if runtime_adapter_enabled() or runtime_adapter_runner_enabled():
        if regeneration is not None and runtime_adapter_runner_enabled():
            return {"error": "Regeneration is not supported by the runner backend.", "code": "unsupported_regeneration_backend", "_status": 409}
        def _legacy_start_run(request: StartRunRequest) -> dict:
            return _start_chat_stream_for_session(
                s,
                msg=request.message,
                attachments=request.attachments,
                workspace=request.workspace or workspace,
                model=request.model or model,
                model_provider=request.provider or model_provider,
                normalized_model=normalized_model,
                diag=diag,
                source=request.source or source,
                moa_config=moa_config,
                external_runtime_owned=gateway_chat_enabled,
                regeneration=regeneration,
            )

        def _legacy_adapter_factory():
            return LegacyJournalRuntimeAdapter(start_run_delegate=_legacy_start_run)

        try:
            adapter = build_runtime_adapter(
                legacy_adapter_factory=_legacy_adapter_factory,
                runner_client_factory=_runtime_runner_client_factory,
            )
            if adapter is None:
                raise NotImplementedError("runtime adapter selection returned no adapter")
            result = adapter.start_run(
                StartRunRequest(
                    session_id=s.session_id,
                    message=msg,
                    attachments=attachments,
                    workspace=workspace,
                    profile=getattr(s, "profile", None),
                    provider=model_provider,
                    model=model,
                    source=source,
                    metadata={"route": route},
                )
            )
        except NotImplementedError as exc:
            return {"error": str(exc), "_status": 501}
        return _chat_start_response_from_run_start(result)

    return _start_chat_stream_for_session(
        s,
        msg=msg,
        attachments=attachments,
        workspace=workspace,
        model=model,
        model_provider=model_provider,
        normalized_model=normalized_model,
        diag=diag,
        source=source,
        moa_config=moa_config,
        external_runtime_owned=gateway_chat_enabled,
        regeneration=regeneration,
    )


def _process_wakeup_revalidation_provider(model, provider) -> str:
    """Return the canonical provider id used for wakeup credential revalidation."""
    try:
        _resolved_model, resolved_provider = canonical_model_provider_lane(model, provider)
    except Exception:
        logger.debug(
            "failed to canonicalize process_wakeup revalidation lane for model=%r provider=%r",
            model,
            provider,
            exc_info=True,
        )
        resolved_provider = None
    candidate = resolved_provider if resolved_provider else provider
    return str(candidate or "").strip()


def _process_wakeup_provider_has_recovery_credential(
    session,
    *,
    model,
    provider,
    provider_id: str | None = None,
) -> bool:
    """Check paused credential-pool recovery in the owning session profile."""
    provider_id = str(
        provider_id or _process_wakeup_revalidation_provider(model, provider) or ""
    ).strip()
    if not provider_id:
        return False
    profile_name = str(getattr(session, "profile", "") or "").strip()
    if profile_name and not _is_root_profile(profile_name):
        with profile_scope_for_detached_worker(
            profile_name,
            "process_wakeup credential revalidation",
            logger_override=logger,
        ):
            return provider_has_process_wakeup_recovery_credential(provider_id, refresh=True)
    return provider_has_process_wakeup_recovery_credential(provider_id, refresh=True)


def _refresh_process_wakeup_pause_credential_fingerprint(session) -> bool:
    """Refresh the stored credential fingerprint without clearing the pause."""
    pause = getattr(session, "process_wakeup_pause", None)
    if not isinstance(pause, dict) or not pause.get("paused"):
        return False
    updated = dict(pause)
    updated["credential_state_fingerprint"] = process_wakeup_credential_state_fingerprint(session)
    session.process_wakeup_pause = updated
    return True


def start_session_turn(
    session_id: str,
    message: str,
    *,
    source: str = "process_wakeup",
):
    """Start a server-side agent turn for ``session_id`` with ``message``.

    Option Z primary wakeup entrypoint. This is the minimal, HTTP-handler-free
    core that ``/api/chat/start`` already reaches via ``_handle_chat_start`` →
    ``_start_chat_stream_for_session``. The drain thread
    (``api/background_process._process_one``) calls this directly with a
    synthetic ``[IMPORTANT: …]`` wakeup_prompt so a background process can wake
    the agent server-side with NO browser round-trip — exactly how CLI /
    gateway self-wake from a ``notify_on_complete`` completion.

    Contract:
      - Resolves the session record (profile/workspace/model/model_provider are
        already persisted on it; no user auth needed — same trust level as
        gateway/cron starting a turn).
      - Resolves workspace + model/provider through the SAME helpers
        ``_handle_chat_start`` uses, so a process-wakeup turn is constructed
        identically to a human-typed turn. If the session record has no model
        persisted, ``_resolve_compatible_session_model_state`` falls back to the
        configured default model/provider (documented in the impl report §1).
      - Delegates to ``_start_chat_stream_for_session`` which spawns the agent
        on a daemon worker thread (the drain thread NEVER blocks) and serializes
        on the per-session agent lock + active-stream guard, so a concurrent
        human ``/api/chat/start`` cannot double-start (one wins, the other gets
        the existing 409 "session already has an active stream").

    Returns the same dict ``_start_chat_stream_for_session`` returns, including
    ``_status`` (200 on start, 409 when a turn is already active). On 409 the
    caller must leave the ``PENDING_BG_TASK_COMPLETIONS`` marker in place so the
    PR #2279 next-turn drain delivers the wakeup when the active turn ends.
    """
    msg = str(message or "").strip()
    if _is_silent_control_message(msg):
        return {
            "status": "suppressed",
            "reason": "silent_control_message",
            "_status": 200,
        }
    if not msg:
        return {"error": "message is required", "_status": 400}
    stale_response = _agent_runtime_barrier_response(runner_local_owned=True)
    if stale_response is not None:
        stale_response["_status"] = 409
        return stale_response
    turn_source = str(source or "process_wakeup").strip() or "process_wakeup"
    try:
        s = get_session(session_id)
    except KeyError:
        return {"error": "Session not found", "_status": 404}

    try:
        workspace = _resolve_chat_workspace_with_recovery(s, None)
    except WorkspaceBindingPersistenceError as e:
        return {"error": str(e), "_status": 500}
    except ValueError as e:
        return {"error": str(e), "_status": 400}

    requested_model = s.model
    requested_provider = getattr(s, "model_provider", None)
    # Server-initiated wakeup (Option Z): resolve persisted model via the
    # standard helper in cache-only mode so wakeups never trigger a cold
    # catalog rebuild. Thread the session's PROFILE model defaults through too
    # (mirrors _handle_chat_start) — a brand-new session that spawned a
    # background task before its first human turn has an empty s.model, and
    # without the profile defaults the resolver would fall back to the global
    # DEFAULT_MODEL instead of the profile's configured default (greptile flag).
    _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(s, requested_provider)
    model, model_provider, normalized_model = _resolve_compatible_session_model_state(
        requested_model,
        requested_provider,
        profile_provider=_pp_provider,
        profile_default_model=_pp_default,
        profile_config=_pp_cfg,
        prefer_cached_catalog=True,
    )
    _paused_wakeup_response = None
    with _get_session_agent_lock(s.session_id):
        try:
            s = get_session(session_id)
        except KeyError:
            return {"error": "Session not found", "_status": 404}
        if clear_process_wakeup_pause_if_model_changed(
            s,
            model=model,
            provider=model_provider,
        ):
            try:
                s.save(touch_updated_at=False)
            except Exception:
                logger.debug(
                    "failed to persist process_wakeup pause reset for session %s",
                    session_id,
                    exc_info=True,
                )
        if turn_source == "process_wakeup":
            _credential_state_changed = False
            try:
                _credential_state_changed = process_wakeup_pause_credential_state_changed(s)
            except Exception:
                logger.debug(
                    "failed to compare process_wakeup credential state for session %s",
                    session_id,
                    exc_info=True,
                )
            if process_wakeup_pause_matches(
                s,
                model=model,
                provider=model_provider,
                classification='credential_pool_empty',
            ):
                _credential_recovered = False
                _credential_revalidation_provider = _process_wakeup_revalidation_provider(
                    model,
                    model_provider,
                )
                try:
                    _credential_recovered = _process_wakeup_provider_has_recovery_credential(
                        s,
                        model=model,
                        provider=model_provider,
                        provider_id=_credential_revalidation_provider,
                    )
                except Exception:
                    logger.debug(
                        "failed to revalidate process_wakeup credential availability for session %s",
                        session_id,
                        exc_info=True,
                    )
                if _credential_recovered:
                    _recovery_reason = (
                        'credential_state_changed'
                        if _credential_state_changed
                        else 'credential_recovered'
                    )
                    if clear_process_wakeup_pause(s, reason=_recovery_reason):
                        try:
                            s.save(touch_updated_at=False)
                        except Exception:
                            logger.debug(
                                "failed to persist process_wakeup credential recovery reset for session %s",
                                session_id,
                                exc_info=True,
                            )
                elif _credential_state_changed:
                    if _refresh_process_wakeup_pause_credential_fingerprint(s):
                        try:
                            s.save(touch_updated_at=False)
                        except Exception:
                            logger.debug(
                                "failed to persist process_wakeup credential-state fingerprint refresh for session %s",
                                session_id,
                                exc_info=True,
                            )
            _paused_wakeup = suppress_process_wakeup_for_provider_pause(
                s,
                model=model,
                provider=model_provider,
                classification='credential_pool_empty',
            )
            if _paused_wakeup is not None:
                try:
                    PENDING_BG_TASK_COMPLETIONS.discard(s.session_id)
                except Exception:
                    logger.debug(
                        "failed to discard pending bg-task marker for paused wakeup %s",
                        session_id,
                        exc_info=True,
                    )
                try:
                    s.save(touch_updated_at=False)
                except Exception:
                    logger.debug(
                        "failed to persist process_wakeup suppression for session %s",
                        session_id,
                        exc_info=True,
                    )
                _paused_wakeup_response = {
                    "error": PROCESS_WAKEUP_PAUSE_ERROR,
                    "message": (
                        "Automatic process wakeups are paused for this session because "
                        "the provider credential pool is unavailable."
                    ),
                    "process_wakeup_pause": _paused_wakeup,
                    "_status": 409,
                }
    if _paused_wakeup_response is not None:
        return _paused_wakeup_response
    resp = _start_run(
        s,
        msg=msg,
        attachments=[],
        workspace=workspace,
        model=model,
        model_provider=model_provider,
        normalized_model=normalized_model,
        source=turn_source,
        route="start_session_turn",
    )

    # ── Defect B: live-view of server-initiated turns ──────────────────────
    # Option Z starts this turn server-side, so NO browser EventSource is
    # attached to the new STREAMS[stream_id] (the browser only opens
    # /api/chat/stream when IT POSTs /api/chat/start). An already-open tab
    # would therefore see nothing until a manual refresh re-reads persisted
    # state. Fix: fan a lightweight `server_turn_started` {stream_id} frame
    # onto the persistent per-session live-view channel. messages.js handles
    # it by attaching its EXISTING chat-stream renderer (attachLiveStream) to
    # that stream_id — no second renderer, no chat/start POST.
    #
    # Idempotent with the closed-tab path: get_session_channel() is the
    # NON-creating accessor, so when no tab is open this is a pure no-op and
    # the server-side wakeup (the Option Z headline) is completely unaffected.
    # If the user also has the per-turn chat-stream open, the frontend dedupes
    # by stream_id so there is no double-render.
    try:
        status = int((resp or {}).get("_status", 200) or 200)
        stream_id = (resp or {}).get("stream_id")
        if status < 400 and stream_id:
            from api.background_process import get_session_channel

            ch = get_session_channel(session_id)
            if ch is not None:
                ch.emit(
                    "server_turn_started",
                    {
                        "session_id": str(session_id),
                        "stream_id": str(stream_id),
                        "pending_started_at": (resp or {}).get("pending_started_at"),
                        "source": source,
                    },
                )
    except Exception:
        logger.debug(
            "server_turn_started fan-out failed for session %s", session_id, exc_info=True
        )
    return resp


def _handle_bg_task_complete_ack(handler, body):
    """Acknowledge a bg_task_complete SSE event (diagnostic only).

    Option Z PIVOT: the agent wakeup is now started SERVER-SIDE by the drain
    thread (``api/background_process._process_one`` → ``start_session_turn``)
    with NO browser round-trip — the closed-tab case works (parity with
    CLI/Telegram). The frontend no longer re-POSTs ``wakeup_prompt`` to
    /api/chat/start; the per-session SSE channel is demoted to pure live-view.

    This endpoint is therefore a pure no-op for state — it exists so an open
    tab can confirm receipt of the live-view event and so a future follow-up
    (analytics, telemetry) has a stable hook. ``PENDING_BG_TASK_COMPLETIONS``
    is consumed by ``_start_chat_stream_for_session`` when the server-side
    wakeup turn (or the next human turn / PR #2279 next-turn drain) runs.
    """
    from api.helpers import j

    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))
    sid = str(body.get("session_id") or "").strip()
    try:
        s = get_session(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    # process_id accepted as transitional alias; see Deprecation response header
    # + maintainer decision on removal milestone / future Sunset header. Only
    # flag Deprecation when the alias was ACTUALLY used (i.e. process_id present
    # and not empty), even if task_id is also present.
    _task_id_present = bool(str(body.get("task_id") or "").strip())
    _process_id_present = bool(str(body.get("process_id") or "").strip())
    legacy_process_id_used = _process_id_present
    pid = str(body.get("task_id") or body.get("process_id") or "").strip()
    # Post Option-Z pivot this endpoint owns no state: the server-side drain
    # thread starts the wakeup turn, the browser never re-POSTs /api/chat/start.
    # `noop` is returned so the diagnostic shape stays explicit about that and
    # matches the docstring ("pure no-op for state").
    return j(
        handler,
        {
            "ok": True,
            "session_id": s.session_id,
            "task_id": pid,
            "noop": True,
        },
        extra_headers={"Deprecation": "true"} if legacy_process_id_used else {},
    )


def _handle_session_compression_recovery_start(handler, body):
    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))
    sid = str(body.get("session_id") or "").strip()
    if not sid:
        return bad(handler, "session_id is required")
    if _session_is_subagent_view_only(sid):
        return bad(handler, "Subagent sessions are view-only and cannot start compression recovery from WebUI", 400)
    try:
        source = get_session(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    if not _session_visible_to_active_profile(getattr(source, "profile", None), handler):
        # #7710: same contract as the detail-load endpoint — 409
        # ``session_profile_mismatch`` for a known other profile,
        # 404 only for the None-profile self-heal path.
        _recovery_session_profile = getattr(source, "profile", None)
        if _recovery_session_profile:
            return j(handler, {
                "error": "Session belongs to a different profile",
                "code": "session_profile_mismatch",
                "session_id": sid,
                "profile": _recovery_session_profile,
            }, status=409)
        return bad(handler, "Session not found", 404)
    recovery = compression_recovery_payload_for_session(source)
    if not recovery:
        return bad(handler, "Session does not have a compression recovery action.", 409)
    action = str(recovery.get("recommended_action") or "")
    if action != COMPRESSION_RECOVERY_ACTION_START_FOCUSED:
        return bad(handler, "Unsupported compression recovery action.", 409)

    created = False
    with _COMPRESSION_RECOVERY_START_LOCK:
        source_profile = getattr(source, "profile", None)
        copied_session = find_compression_recovery_session(sid, action, source_profile=source_profile)
        if copied_session is None:
            title = str(getattr(source, "title", None) or "Untitled").strip() or "Untitled"
            if not title.endswith(" (focused continuation)"):
                title = f"{title} (focused continuation)"
            copied_session = Session(
                session_id=uuid.uuid4().hex[:12],
                title=title,
                workspace=getattr(source, "workspace", get_last_workspace()),
                model=getattr(source, "model", None),
                model_provider=getattr(source, "model_provider", None),
                messages=[],
                tool_calls=[],
                pinned=False,
                archived=False,
                project_id=getattr(source, "project_id", None),
                profile=getattr(source, "profile", None),
                session_source="fork",
                personality=getattr(source, "personality", None),
                enabled_toolsets=copy.deepcopy(getattr(source, "enabled_toolsets", None)),
                context_length=getattr(source, "context_length", None),
                threshold_tokens=getattr(source, "threshold_tokens", None),
                gateway_routing=copy.deepcopy(getattr(source, "gateway_routing", None)),
                gateway_routing_history=copy.deepcopy(getattr(source, "gateway_routing_history", None) or []),
                parent_session_id=getattr(source, "session_id", sid),
                worktree_path=getattr(source, "worktree_path", None),
                worktree_branch=getattr(source, "worktree_branch", None),
                worktree_repo_root=getattr(source, "worktree_repo_root", None),
                worktree_created_at=getattr(source, "worktree_created_at", None),
                compression_recovery_source_session_id=sid,
                compression_recovery_action=action,
            )
            # Preserve the workspace/model/profile lane, but intentionally start with an
            # empty model-facing transcript so a focused follow-up does not replay the
            # exhausted state.db/context tail.
            copied_session.context_messages = []
            copied_session.composer_draft = {"text": "", "files": []}
            try:
                copied_session.save()
            except Exception as e:
                logger.exception("failed to persist compression recovery session for %s", sid)
                return bad(handler, f"Failed to start compression recovery: {_sanitize_error(e)}", 500)

            with LOCK:
                SESSIONS[copied_session.session_id] = copied_session
                SESSIONS.move_to_end(copied_session.session_id)
                _evict_sessions_over_cap()
            created = True
    if created:
        publish_session_list_changed(
            "session_compression_recovery",
            profile=getattr(copied_session, "profile", None),
            session_id=getattr(copied_session, "session_id", None),
        )
    session_payload = redact_session_data(copied_session.compact() | {"messages": copied_session.messages})
    return j(
        handler,
        {
            "ok": True,
            "session": session_payload,
            "source_session_id": sid,
            "recommended_recovery_action": action,
            "message": (
                "Started a focused continuation. Describe the next narrow task to continue."
                if created
                else "Opened the existing focused continuation for this exhausted session."
            ),
        },
    )


def _handle_goal_command(handler, body):
    """Handle WebUI /goal command controls and optional kickoff stream."""
    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))
    if _is_silent_control_message(body.get("args") or body.get("text")):
        return j(
            handler,
            {"status": "suppressed", "reason": "silent_control_message"},
            status=200,
        )
    if _session_is_subagent_view_only(str(body.get("session_id") or "")):
        return bad(handler, "Subagent sessions are view-only and cannot run /goal from WebUI", 400)
    try:
        s = get_session(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)

    requested_profile = str(body.get("profile") or "").strip()
    if requested_profile:
        try:
            from api.profiles import _PROFILE_ID_RE

            if requested_profile != "default" and not _PROFILE_ID_RE.fullmatch(requested_profile):
                return bad(handler, "invalid profile", 400)
        except ImportError:
            requested_profile = ""
    if requested_profile and not _profiles_match(getattr(s, "profile", None), requested_profile):
        has_persisted_turns = bool(
            getattr(s, "messages", None)
            or getattr(s, "context_messages", None)
            or getattr(s, "pending_user_message", None)
        )
        if not has_persisted_turns:
            s.profile = requested_profile

    current_stream_id = getattr(s, "active_stream_id", None)
    stream_running = False
    if current_stream_id:
        with STREAMS_LOCK:
            stream_running = current_stream_id in STREAMS
        if not stream_running:
            _clear_stale_stream_state(s)

    try:
        from api.profiles import get_hermes_home_for_profile

        profile_home = get_hermes_home_for_profile(getattr(s, "profile", None))
    except Exception:
        profile_home = None

    from api.goals import goal_command_payload, goal_state_snapshot, restore_goal_state

    goal_args = str(body.get("args", "") or body.get("text", "") or "")
    goal_action = goal_args.strip().lower()
    will_kickoff = bool(
        goal_args.strip()
        and goal_action not in ("status", "pause", "resume", "clear", "stop", "done")
        and not stream_running
    )
    workspace = model = model_provider = normalized_model = None
    explicit_model_pick = bool(body.get("explicit_model_pick"))
    previous_goal_state = None
    if will_kickoff:
        try:
            workspace = str(resolve_trusted_workspace(body.get("workspace") or s.workspace, profile=getattr(s, "profile", None)))
        except ValueError as e:
            return bad(handler, str(e))
        requested_model = body.get("model") or s.model
        requested_provider = (
            body.get("model_provider")
            if "model_provider" in body
            else getattr(s, "model_provider", None)
        )
        # #6703: carry the explicit-pick marker through goal kickoffs. The
        # frontend marks a session-level provider/model choice as explicit (same
        # signal /api/chat/start receives); without it the model resolver treats
        # a persisted cross-provider pick as stale and "repairs" it back to the
        # profile default, silently switching providers mid-session.
        _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(s, requested_provider)
        model, model_provider, normalized_model = _resolve_compatible_session_model_state(
            requested_model,
            requested_provider,
            profile_provider=_pp_provider,
            profile_default_model=_pp_default,
            profile_config=_pp_cfg,
            explicit_model_pick=explicit_model_pick,
        )
        # #5979/#6703 parity with chat-start: record a SIGNATURE of the
        # deliberately-picked model+provider so the streaming resolver can
        # preserve a custom-proxy vendor namespace on a cold catalog. A first
        # /goal launch after a deliberate custom-provider pick must survive a
        # cold streaming catalog exactly like /api/chat/start does; otherwise the
        # provider reverts to the profile default mid-session.
        try:
            if explicit_model_pick:
                from api.models import model_explicit_pick_signature as _mk_sig
                s.model_explicit_pick_signature = _mk_sig(model, model_provider)
        except Exception:
            pass
        previous_goal_state = goal_state_snapshot(s.session_id, profile_home=profile_home)

    from api.runtime_adapter import LegacyJournalRuntimeAdapter, runtime_adapter_enabled

    def _legacy_goal_update(session_id: str, _action: str, text: str) -> dict:
        return goal_command_payload(
            session_id,
            text,
            stream_running=stream_running,
            profile_home=profile_home,
        )

    goal_adapter_action = _runtime_adapter_goal_action(goal_args)
    if runtime_adapter_enabled():
        adapter = LegacyJournalRuntimeAdapter(goal_delegate=_legacy_goal_update)
        control_result = adapter.update_goal(
            s.session_id,
            goal_adapter_action,
            goal_args,
        )
        # Slice 3c keeps the adapter as a structural seam only.  Preserve the
        # public /api/goal response by passing through the legacy payload rather
        # than deriving HTTP behavior from ControlResult.accepted/status.
        payload = dict(control_result.payload)
    else:
        payload = _legacy_goal_update(s.session_id, goal_adapter_action, goal_args)
    if not payload.get("ok", True):
        status = 409 if payload.get("error") == "agent_running" else 400
        return j(handler, payload, status=status)

    kickoff_prompt = str(payload.get("kickoff_prompt") or "").strip()
    if kickoff_prompt:
        if workspace is None:
            try:
                workspace = str(resolve_trusted_workspace(body.get("workspace") or s.workspace, profile=getattr(s, "profile", None)))
            except ValueError as e:
                return bad(handler, str(e))
        if model is None:
            requested_model = body.get("model") or s.model
            requested_provider = (
                body.get("model_provider")
                if "model_provider" in body
                else getattr(s, "model_provider", None)
            )
            _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(s, requested_provider)
            model, model_provider, normalized_model = _resolve_compatible_session_model_state(
                requested_model,
                requested_provider,
                profile_provider=_pp_provider,
                profile_default_model=_pp_default,
                profile_config=_pp_cfg,
                explicit_model_pick=explicit_model_pick,
            )
            # #6703 parity: same explicit-pick signature stamping on the
            # kickoff-prompt fallback resolution path as /api/chat/start.
            try:
                if explicit_model_pick:
                    from api.models import model_explicit_pick_signature as _mk_sig
                    s.model_explicit_pick_signature = _mk_sig(model, model_provider)
            except Exception:
                pass
        stream_response = _start_chat_stream_for_session(
            s,
            msg=kickoff_prompt,
            attachments=[],
            workspace=workspace,
            model=model,
            model_provider=model_provider,
            normalized_model=normalized_model,
            goal_related=True,
            external_runtime_owned=webui_gateway_chat_enabled(get_config()),
        )
        status = int(stream_response.pop("_status", 200) or 200)
        payload.update(stream_response)
        if status >= 400:
            restore_goal_state(s.session_id, previous_goal_state, profile_home=profile_home)
            payload["ok"] = False
            return j(handler, payload, status=status)

    return j(handler, payload)


def _is_silent_control_message(message) -> bool:
    """Return True only for the scheduler's exact suppression sentinel.

    ``[SILENT]`` is control-plane output, never conversation content. If a wake
    relay POSTs it and 8701 restarts while the turn is pending, recovery
    materializes it as a visible ``_recovered`` user message. Suppress it before
    session lookup or pending-state mutation. Matching stays exact and
    case-sensitive so ordinary user text is unaffected.
    """
    return str(message or "").strip() == "[SILENT]"


def _handle_chat_start(handler, body, diag=None):
    try:
        diag.stage("validate_session_id") if diag else None
        try:
            require(body, "session_id")
        except ValueError as e:
            return bad(handler, str(e))
        if _is_silent_control_message(body.get("message")):
            return j(
                handler,
                {"status": "suppressed", "reason": "silent_control_message"},
                status=200,
            )
        if body.get("regenerate") is True:
            from api.runtime_adapter import runtime_adapter_runner_enabled

            if runtime_adapter_runner_enabled():
                return j(handler, {
                    "error": "Regeneration is not supported by the runner backend.",
                    "code": "unsupported_regeneration_backend",
                }, status=409)
        # Reject a stale local Agent runtime before materialising, claiming, or
        # mutating any session state. Gateway-backed turns run in the gateway's
        # process and do not depend on this WebUI process's imported checkout.
        stale_response = _agent_runtime_barrier_response(runner_local_owned=True)
        if stale_response is not None:
            return j(handler, stale_response, status=409)
        diag.stage("get_session") if diag else None
        try:
            s = _get_or_materialize_session(
                body["session_id"],
                refresh_cli_messages=body.get("regenerate") is not True,
            )
        except KeyError:
            # No WebUI sidecar. If this is a foreign-origin session (CLI,
            # TUI, Desktop) with recoverable state.db messages, claim it by
            # materialising a WebUI-owned Session and persisting it as a
            # sidecar. This closes the GET-vs-POST asymmetry where a
            # TUI/Desktop session loads read-only via GET /api/session but
            # 404s on the first POST /api/chat/start, making the typed
            # message disappear into the empty state.
            synth, reason = _claim_or_synthesize_cli_session(body["session_id"])
            if synth is None:
                # 'was_webui' (deleted WebUI session, client should self-heal
                # via the existing 404 path), 'no_foreign_state' (sid has
                # no recoverable state anywhere), or 'invalid_sid' (path
                # safety violation). All collapse to 404 — the client only
                # knows the right thing to do for "this session is gone".
                return bad(handler, "Session not found", 404)
            if reason == "not_claimable":
                # Foreign store says this session is read-only / owned by
                # a non-WebUI process (messaging, claude_code,
                # external_agent, cron, gateway/unknown, or explicit
                # read_only flag). The session is real and viewable, but
                # the WebUI must not take write ownership of it — that
                # would be an ownership-boundary violation (#4911 review).
                # 403 (not 404) because 404 triggers the frontend's
                # empty-state self-heal handler which strips the URL and
                # clears localStorage; for a legitimately-listed read-only
                # session the user should keep their URL and see a refusal,
                # not have their session vanish.
                return bad(
                    handler,
                    "session is read-only in its foreign store; cannot be claimed writeable in WebUI",
                    403,
                )
            try:
                synth.save()
            except Exception as _save_err:
                # Persisting the sidecar failed: surface a generic 500 to
                # the client (paths sanitised, see _sanitize_error) and log
                # the full exception server-side. Returning the raw str(exc)
                # would leak /root/.hermes/webui/sessions/<sid>.json or any
                # other absolute filesystem path the OSError happened to
                # carry — #4911 review feedback.
                logger.exception(
                    "failed to persist materialised sidecar for foreign session %s",
                    body["session_id"],
                )
                return bad(
                    handler,
                    f"failed to claim session: {_sanitize_error(_save_err)}",
                    500,
                )
            s = synth
            try:
                with LOCK:
                    SESSIONS[s.session_id] = s
                    SESSIONS.move_to_end(s.session_id)
            except Exception:
                # If the in-memory LRU refuses the new session, fall through
                # with the just-persisted sidecar; _start_run will load it
                # from disk if needed.
                pass
        except PermissionError:
            return bad(handler, "Read-only imported sessions cannot be continued from WebUI", 403)
        diag.stage("validate_profile") if diag else None
        requested_profile = str(body.get("profile") or "").strip()
        active_profile = _get_active_profile_name()
        if requested_profile:
            try:
                from api.profiles import _PROFILE_ID_RE

                if requested_profile != "default" and not _PROFILE_ID_RE.fullmatch(requested_profile):
                    return bad(handler, "invalid profile", 400)
            except ImportError:
                requested_profile = ""
        session_profile = getattr(s, "profile", None)
        has_persisted_turns = bool(
            getattr(s, "messages", None)
            or getattr(s, "context_messages", None)
            or getattr(s, "pending_user_message", None)
        )
        if not _session_visible_to_active_profile(session_profile, handler):
            if (
                requested_profile
                and _profiles_match(requested_profile, active_profile)
                and not has_persisted_turns
            ):
                # Empty placeholders can still be retagged when the
                # requested profile matches the active request profile.
                s.profile = requested_profile
            elif session_profile:
                # #7710: known other profile → 409 ``session_profile_mismatch``
                # so the client can offer to switch to it (#5419).
                # 404 is preserved only for the None-profile
                # (unknown/legacy) self-heal case.
                return j(handler, {
                    "error": "Session belongs to a different profile",
                    "code": "session_profile_mismatch",
                    "session_id": body.get("session_id", ""),
                    "profile": session_profile,
                }, status=409)
            else:
                return bad(handler, "Session not found", 404)
        # Resolve durable rotations before any workspace/model/pending mutation.
        # GET navigation adopts the tip; POST never silently replays a user turn.
        from api.compression_continuation import durable_compression_continuation
        sealed, continuation = durable_compression_continuation(s)
        if sealed:
            return j(handler, {
                "error": "This session was compressed. Open its continuation before sending.",
                "code": "session_rotated",
                "continuation_session_id": continuation,
            }, status=409)
        regeneration = None
        if body.get("regenerate") is True:
            if any(key in body for key in ("message", "attachments", "keep_count", "prompt", "prompt_index")):
                return j(handler, {"error": "regeneration accepts only regeneration_revision", "code": "invalid_regeneration_request"}, status=400)
            if not isinstance(body.get("regeneration_revision"), str):
                return j(handler, {"error": "regeneration_revision is required", "code": "stale_regeneration_revision"}, status=409)
            try:
                from api.session_ops import plan_regeneration, RegenerationUnavailable
                regeneration = plan_regeneration(
                    s, expected_revision=body["regeneration_revision"]
                )
            except RegenerationUnavailable as exc:
                return j(handler, {"error": str(exc), "code": exc.code}, status=exc.status)
            msg = regeneration.turn.message_text
            attachments = copy.deepcopy(regeneration.turn.attachments)[:20]
        else:
            msg = None
            attachments = None
        diag.stage("normalize_message") if diag else None
        msg = str(msg if msg is not None else body.get("message", "")).strip()
        if not msg:
            return bad(handler, "message is required")
        diag.stage("normalize_attachments") if diag else None
        if attachments is None:
            attachments = _normalize_chat_attachments(body.get("attachments") or [])[:20]
        recovery = compression_recovery_payload_for_session(s)
        if recovery and not attachments and is_generic_continuation_intent(msg):
            return j(
                handler,
                {
                    "error": "This session exhausted context compression. Start a focused continuation, then describe the next narrow task.",
                    "type": "compression_recovery_required",
                    "recommended_recovery_action": recovery.get("recommended_action"),
                    "compression_recovery": recovery,
                    "session_id": getattr(s, "session_id", body["session_id"]),
                },
                status=409,
            )
        diag.stage("resolve_workspace") if diag else None
        try:
            if regeneration is not None:
                workspace = _resolve_chat_workspace_for_regeneration(s, body.get("workspace"))
            else:
                workspace = _resolve_chat_workspace_with_recovery(s, body.get("workspace"))
        except WorkspaceBindingPersistenceError as e:
            return bad(handler, str(e), 500)
        except ValueError as e:
            return bad(handler, str(e))
        requested_model = body.get("model") or s.model
        requested_provider = (
            body.get("model_provider")
            if "model_provider" in body
            else getattr(s, "model_provider", None)
        )
        _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(s, requested_provider)
        explicit_model_pick = bool(body.get("explicit_model_pick"))
        moa_config = None
        config_snapshot = get_config_snapshot()
        gateway_chat_enabled = webui_gateway_chat_enabled(config_snapshot)
        if body.get("moa_config"):
            if gateway_chat_enabled:
                return bad(handler, "MoA override is unavailable on gateway-backed sessions", 409)
            from api.commands import resolve_moa_config

            try:
                moa_config = resolve_moa_config()
            except RuntimeError as e:
                return bad(handler, str(e), 503)
        diag.stage("resolve_model_provider") if diag else None
        model, model_provider, normalized_model = _resolve_compatible_session_model_state(
            requested_model,
            requested_provider,
            profile_provider=_pp_provider,
            profile_default_model=_pp_default,
            profile_config=_pp_cfg,
            explicit_model_pick=explicit_model_pick,
        )
        # #5979: record a SIGNATURE of the deliberately-picked model+provider so
        # the streaming resolver can preserve a custom-proxy vendor namespace on a
        # cold catalog — but ONLY while the routing context still matches. On a
        # fresh explicit pick, stamp the signature of the resolved model+provider;
        # otherwise leave any prior signature in place (it self-invalidates when
        # the model/provider changes, since the streaming side recomputes and
        # compares). This survives same-model follow-up sends (the onchange marker
        # is one-shot) yet can't outlive a real switch.
        try:
            if explicit_model_pick and regeneration is None:
                from api.models import model_explicit_pick_signature as _mk_sig
                s.model_explicit_pick_signature = _mk_sig(model, model_provider)
        except Exception:
            pass
        catalog_profile_provider = _pp_provider
        if catalog_profile_provider is None and isinstance(_pp_cfg, dict):
            profile_model_config = _pp_cfg.get("model") or {}
            if isinstance(profile_model_config, dict):
                catalog_profile_provider = profile_model_config.get("provider")
        model_provider = _repair_foreign_session_model_provider(
            s,
            requested_model=requested_model,
            requested_provider=requested_provider,
            resolved_model=model,
            resolved_provider=model_provider,
            explicit_model_pick=explicit_model_pick,
            profile_provider=catalog_profile_provider,
        )
        if model_provider == "moa" and gateway_chat_enabled:
            from api.config import get_effective_default_model

            model_config = config_snapshot.get("model") if isinstance(config_snapshot, dict) else None
            configured_default, configured_default_provider, configured_default_is_moa = (
                _moa_fast_path_model_state(get_effective_default_model(config_snapshot))
            )
            configured_provider = _clean_session_model_provider(
                model_config.get("provider") if isinstance(model_config, dict) else None
            )
            if configured_provider is None and configured_default_is_moa:
                configured_provider = configured_default_provider
            if (
                configured_provider != "moa"
                or model != configured_default
                or explicit_model_pick
            ):
                return bad(handler, "MoA override is unavailable on gateway-backed sessions", 409)
        elif model_provider == "moa" and moa_config is None:
            from api.commands import resolve_moa_config

            try:
                moa_config = resolve_moa_config(model)
            except RuntimeError as e:
                return bad(handler, str(e), 503)
        # NOTE: runtime-adapter selection is delegated to _start_run (shared
        # with start_session_turn so both entry points behave identically
        # under runtime_adapter_enabled() / runtime_adapter_runner_enabled()
        # — Q-2979-A2 / Copilot discussion_r3305864087/r3305864173).
        start_run_kwargs = {
            "msg": msg,
            "attachments": attachments,
            "workspace": workspace,
            "model": model,
            "model_provider": model_provider,
            "normalized_model": normalized_model,
            "source": "webui",
            "route": "/api/chat/start",
            "diag": diag,
            "gateway_chat_enabled": gateway_chat_enabled,
            "regeneration": regeneration,
        }
        if not gateway_chat_enabled and moa_config is not None:
            start_run_kwargs["moa_config"] = moa_config
        recovery_cleared_for_start = None
        def _restore_cleared_recovery():
            if recovery_cleared_for_start is None:
                return None
            s.compression_recovery = recovery_cleared_for_start
            s.recommended_recovery_action = recovery_cleared_for_start.get("recommended_action")
            try:
                s.save()
            except Exception as restore_err:
                logger.exception("failed to restore compression recovery after chat start rejection for %s", getattr(s, "session_id", None))
                return restore_err
            return None

        if recovery and regeneration is None:
            recovery_cleared_for_start = copy.deepcopy(recovery)
            clear_compression_recovery(s)
        try:
            response = _start_run(
                s,
                **start_run_kwargs,
            )
        except Exception as exc:
            if not getattr(exc, "_regeneration_accepted", False):
                _restore_cleared_recovery()
            raise
        # Map adapter-selection NotImplementedError (501) onto the legacy
        # bad-request response shape that this route exposed historically
        # before the helper extraction.
        if response.get("_status") == 501 and "error" in response:
            restore_err = _restore_cleared_recovery()
            if restore_err is not None:
                return bad(handler, f"failed to restore compression recovery: {_sanitize_error(restore_err)}", 500)
            return j(handler, {"error": response["error"]}, status=501)
        status = int(response.pop("_status", 200) or 200)
        if status >= 400 and recovery_cleared_for_start is not None:
            restore_err = _restore_cleared_recovery()
            if restore_err is not None:
                return bad(handler, f"failed to restore compression recovery: {_sanitize_error(restore_err)}", 500)
        diag.stage("response_write") if diag else None
        return j(handler, response, status=status)
    finally:
        if diag:
            diag.finish()



def _resolve_chat_workspace_with_recovery(s, requested_workspace) -> str:
    """Recover stale implicit session workspaces without hiding explicit errors."""
    _session_profile = getattr(s, "profile", None)
    explicit = requested_workspace not in (None, "")
    if explicit:
        try:
            return str(resolve_trusted_workspace(requested_workspace, profile=_session_profile))
        except TypeError:
            return str(resolve_trusted_workspace(requested_workspace))
    stored_workspace = getattr(s, "workspace", None)
    try:
        workspace, recovered = resolve_implicit_workspace_with_recovery(
            stored_workspace,
            get_last_workspace,
            profile=_session_profile,
        )
    except TypeError:
        workspace, recovered = resolve_implicit_workspace_with_recovery(
            stored_workspace,
            get_last_workspace,
        )
    if not recovered:
        return str(workspace)
    persisted = persist_recovered_workspace_binding(
        s,
        workspace,
        expected_workspace=stored_workspace,
    )
    return str(persisted.workspace)


def _resolve_chat_workspace_for_regeneration(s, requested_workspace) -> str:
    """Resolve regeneration's workspace without persisting before start acceptance."""
    _session_profile = getattr(s, "profile", None) or None
    if requested_workspace not in (None, ""):
        try:
            return str(resolve_trusted_workspace(requested_workspace, profile=_session_profile))
        except TypeError:
            return str(resolve_trusted_workspace(requested_workspace))
    try:
        workspace, _recovered = resolve_implicit_workspace_with_recovery(
            getattr(s, "workspace", None),
            get_last_workspace,
            profile=_session_profile,
        )
    except TypeError:
        workspace, _recovered = resolve_implicit_workspace_with_recovery(
            getattr(s, "workspace", None),
            get_last_workspace,
        )
    return str(workspace)


def _normalize_chat_attachments(raw_attachments):
    """Normalize attachment payloads from the browser.

    Older clients send a list of filenames. Newer clients send upload result
    objects containing name/path/mime/size so image attachments can be supplied
    to Hermes as native multimodal inputs for the current turn.
    """
    normalized = []
    if not isinstance(raw_attachments, list):
        return normalized
    for item in raw_attachments:
        if isinstance(item, dict):
            name = str(item.get("name") or item.get("filename") or "").strip()
            path = str(item.get("path") or "").strip()
            mime = str(item.get("mime") or "").strip()
            att = {"name": name or path, "path": path, "mime": mime}
            size = item.get("size")
            if isinstance(size, int):
                att["size"] = size
            is_image = item.get("is_image")
            if isinstance(is_image, bool):
                att["is_image"] = is_image
            normalized.append(att)
        else:
            value = str(item).strip()
            if value:
                normalized.append({"name": value, "path": "", "mime": ""})
    return normalized


def _handle_chat_sync(handler, body):
    """Fallback synchronous chat endpoint (POST /api/chat). Not used by frontend."""
    stale_response = _agent_runtime_barrier_response(runner_local_owned=False)
    if stale_response is not None:
        return j(handler, stale_response, status=409)
    if _session_is_subagent_view_only(str(body.get("session_id") or "")):
        return bad(handler, "Subagent sessions are view-only and cannot be written from WebUI", 400)
    s = get_session(body["session_id"])
    msg = str(body.get("message", "")).strip()
    if not msg:
        return j(handler, {"error": "empty message"}, status=400)
    try:
        try:
            workspace = str(resolve_trusted_workspace(body.get("workspace") or s.workspace, profile=getattr(s, "profile", None)))
        except TypeError:
            workspace = str(resolve_trusted_workspace(body.get("workspace") or s.workspace))
    except ValueError as e:
        return bad(handler, str(e))
    with _get_session_agent_lock(s.session_id):
        s.workspace = workspace
        _sync_requested_provider = (
            body.get("model_provider") if "model_provider" in body else getattr(s, "model_provider", None)
        )
        _pp_provider, _pp_default, _pp_cfg = _read_profile_model_config(s, _sync_requested_provider)
        model, model_provider = _resolve_compatible_session_model_state(
            body.get("model") or s.model,
            _sync_requested_provider,
            profile_provider=_pp_provider,
            profile_default_model=_pp_default,
            profile_config=_pp_cfg,
        )[:2]
        s.model = model
        s.model_provider = model_provider
    from api.streaming import _ENV_LOCK

    with _ENV_LOCK:
        old_cwd = os.environ.get("TERMINAL_CWD")
        os.environ["TERMINAL_CWD"] = str(workspace)
        old_exec_ask = os.environ.get("HERMES_EXEC_ASK")
        old_session_key = os.environ.get("HERMES_SESSION_KEY")
        os.environ["HERMES_EXEC_ASK"] = "1"
        os.environ["HERMES_SESSION_KEY"] = s.session_id
    try:
        AIAgent = require_ai_agent_class()

        with CHAT_LOCK:
            from api.config import resolve_model_provider

            _model, _provider, _base_url = resolve_model_provider(
                model_with_provider_context(s.model, getattr(s, "model_provider", None))
            )
            # Resolve API key via Hermes runtime provider (matches gateway behaviour)
            _api_key = None
            _rt = None
            try:
                from api.oauth import resolve_runtime_provider_with_anthropic_env_lock
                from hermes_cli.runtime_provider import resolve_runtime_provider

                _rt = resolve_runtime_provider_with_anthropic_env_lock(
                    resolve_runtime_provider,
                    requested=_provider,
                )
                _api_key = _rt.get("api_key")
                # Also use runtime provider/base_url if the webui config didn't resolve them
                if not _provider:
                    _provider = _rt.get("provider")
                if not _base_url:
                    _base_url = _rt.get("base_url")
            except Exception as _e:
                print(
                    f"[webui] WARNING: resolve_runtime_provider failed: {_e}",
                    flush=True,
                )
            # Apply the named custom provider's OWN record atomically, as one
            # COMPLETE bundle. The fill-only form this replaced kept a truthy
            # runtime value, so a slug present in BOTH custom_providers[] and
            # providers: sent the list row's URL with the keyed row's API key;
            # the connection-only view that followed still truncated the record's
            # api_mode / credential_pool / ACP transport before the constructor.
            try:
                _bundle = _resolve_agent_connection_bundle(
                    _provider, _api_key, _base_url, _rt
                )
            except api_config.CustomProviderRouteError as _route_err:
                # The named route resolved no usable connection. Constructing
                # AIAgent with the incomplete pair would send this turn through
                # ``_routed_client_kwargs()`` to whatever provider init resolves
                # next, so answer with the actionable cause instead. 400, not
                # 500: it is a user-fixable provider misconfiguration, exactly
                # like the ambiguous-slug collision.
                logger.warning(
                    "Chat blocked by unroutable custom provider: %s", _route_err.message
                )
                return j(handler, {
                    "error": _route_err.message,
                    "type": "custom_provider_unroutable",
                    "reason": _route_err.reason,
                    "hint": _route_err.hint,
                }, status=400)
            _provider = _bundle["provider"]
            _api_key = _bundle["api_key"]
            _base_url = _bundle["base_url"]
            agent = AIAgent(
                model=_model,
                provider=_provider,
                base_url=_base_url,
                api_key=_api_key,
                # Identify browser-originated sessions as WebUI so Hermes Agent
                # does not inject CLI-specific terminal/output guidance.
                platform="webui",
                quiet_mode=True,
                enabled_toolsets=_resolve_cli_toolsets(),
                session_id=s.session_id,
                **_agent_bundle_kwargs(AIAgent, _bundle),
            )
            from api.streaming import (
                _WEBUI_PROGRESS_PROMPT,
                _active_turn_boundary,
                _assign_stable_message_ids,
                _dedupe_replayed_context_messages,
                _find_active_turn_checkpoint_index,
                _merge_display_messages_after_agent_result,
                _resolve_active_turn_authority,
                _restore_display_reasoning_metadata,
                _restore_reasoning_metadata_before_boundary,
                _settle_current_turn_boundary,
                _sanitize_messages_for_agent,
                _compact_session_image_parts_for_persistence,
                _context_messages_for_new_turn,
                _workspace_context_prefix,
            )
            workspace_ctx = _workspace_context_prefix(str(s.workspace))
            workspace_system_msg = (
                f"Active workspace at session start: {s.workspace}\n"
                "Every user message is prefixed with [Workspace::v1: /absolute/path] indicating the "
                "workspace the user has selected in the web UI at the time they sent that message. "
                "This tag is the single authoritative source of the active workspace and updates "
                "with every message. It overrides any prior workspace mentioned in this system "
                "prompt, memory, or conversation history. Always use the value from the most recent "
                "[Workspace::v1: ...] tag as your default working directory for ALL file operations: "
                "write_file, read_file, search_files, terminal workdir, and patch. "
                "Never fall back to a hardcoded path when this tag is present.\n\n"
                f"{_WEBUI_PROGRESS_PROMPT}\n\n"
                "WebUI external-notes/durable-memory policy: Do not copy or dump this browser transcript "
                "into external notes or durable memory by default. Write or update durable "
                "notes only for explicit captures, durable preferences, decisions, blockers/open "
                "issues, runbook-worthy workflows, or other clearly reusable signals; otherwise "
                "leave external notes and durable memory unchanged. When you do write or update a durable note, briefly tell "
                "the user what note or section changed so the write is reviewable."
            )

            _previous_messages = list(s.messages or [])
            _previous_context_messages = list(_context_messages_for_new_turn(s, msg))

            result = agent.run_conversation(
                user_message=workspace_ctx + msg,
                system_message=workspace_system_msg,
                conversation_history=_sanitize_messages_for_agent(
                    _previous_context_messages,
                    cfg=get_config(),
                    effective_model=_model,
                    effective_provider=_provider,
                    effective_base_url=_base_url,
                ),
                task_id=s.session_id,
                persist_user_message=msg,
            )
    finally:
        with _ENV_LOCK:
            if old_cwd is None:
                os.environ.pop("TERMINAL_CWD", None)
            else:
                os.environ["TERMINAL_CWD"] = old_cwd
            if old_exec_ask is None:
                os.environ.pop("HERMES_EXEC_ASK", None)
            else:
                os.environ["HERMES_EXEC_ASK"] = old_exec_ask
            if old_session_key is None:
                os.environ.pop("HERMES_SESSION_KEY", None)
            else:
                os.environ["HERMES_SESSION_KEY"] = old_session_key
    with _get_session_agent_lock(s.session_id):
        _result_messages = result.get("messages") or _previous_context_messages
        # Active-turn boundary is fixed BEFORE any restoration (same as streaming),
        # using whatever exact turn authority the result/Agent pair exported.
        _active_turn_identity = _resolve_active_turn_authority(
            {"token": None, "text": msg, "current_turn_user_idx": None, "turn_id": ""},
            result=result,
            agent=agent,
        )
        if (
            isinstance(_active_turn_identity, dict)
            and _active_turn_identity.get("agent_turn_boundary_resolved") is True
            and not _active_turn_identity.get("token")
        ):
            _active_image_index = _find_active_turn_checkpoint_index(
                _result_messages,
                _previous_context_messages,
                _active_turn_identity,
                msg,
            )
            _active_image_content = (
                _result_messages[_active_image_index].get("content")
                if _active_image_index is not None
                else None
            )
            if isinstance(_active_image_content, list) and any(
                isinstance(part, dict)
                and part.get("type") in {"image", "image_url", "input_image"}
                for part in _active_image_content
            ):
                from api.process_event_utils import build_active_turn_token

                _active_turn_identity["token"] = build_active_turn_token(
                    f"sync:{s.session_id}:{_active_turn_identity['turn_id']}",
                    time.time(),
                )
        _turn_boundary = _active_turn_boundary(
            _result_messages, _previous_context_messages, _active_turn_identity, msg,
        )
        _next_context_messages = _restore_reasoning_metadata_before_boundary(
            _previous_context_messages,
            _result_messages,
            _turn_boundary,
        )
        # Mint ids on the shared result rows BEFORE dedupe deep-copies any
        # stale-user boundary row, so both arrays share the id (#5564).
        _assign_stable_message_ids(
            _result_messages, _previous_messages, _previous_context_messages
        )
        _next_context_messages = _dedupe_replayed_context_messages(
            _previous_context_messages,
            _next_context_messages,
            msg,
        )
        if _active_turn_identity.get("token"):
            _next_context_messages = _settle_current_turn_boundary(
                _previous_context_messages,
                _next_context_messages,
                _active_turn_identity,
                msg,
                getattr(s, "pending_user_source", None) or "webui",
            )
        s.context_messages = _next_context_messages
        s.messages = _merge_display_messages_after_agent_result(
            _previous_messages,
            _previous_context_messages,
            _restore_display_reasoning_metadata(
                _previous_messages, _result_messages, current_turn_boundary=_turn_boundary,
            ),
            msg,
            source=getattr(s, "pending_user_source", None) or "webui",
            verification_nudge_provenance={
                "active_turn_identity": _active_turn_identity,
            },
        )
        _compact_session_image_parts_for_persistence(s)
        # Only auto-generate title when still default; preserves user renames
        if s.title == "Untitled":
            s.title = title_from(s.messages, s.title)
        s.save()
    # Sync to state.db for /insights (opt-in setting)
    try:
        if load_settings().get("sync_to_insights"):
            from api.state_sync import sync_session_usage

            sync_session_usage(
                session_id=s.session_id,
                input_tokens=s.input_tokens or 0,
                output_tokens=s.output_tokens or 0,
                estimated_cost=s.estimated_cost,
                model=s.model,
                title=s.title,
                message_count=len(s.messages),
                cache_read_tokens=s.cache_read_tokens or 0,
                cache_write_tokens=s.cache_write_tokens or 0,
                # #2762 / #2827 parity with api/streaming.py:5078: pass the
                # session's profile explicitly so a future refactor that
                # backgrounds this handler doesn't silently leak writes to
                # the wrong profile's state.db. HTTP thread today, but
                # defense-in-depth. Opus pre-release advisor MUST-FIX.
                profile=getattr(s, 'profile', None),
            )
    except Exception:
        logger.debug("Failed to update session cost tracking")
    return j(
        handler,
        {
            "answer": result.get("final_response") or "",
            "status": "done" if result.get("completed", True) else "partial",
            "session": public_session_projection(s.compact() | {"messages": s.messages}),
            "result": {k: v for k, v in result.items() if k != "messages"},
        },
    )


def _selected_profile_snapshot_updates(
    profile: str | None,
    *,
    provider,
    model,
) -> dict[str, str | None]:
    selected_profile = str(profile or "").strip()
    if not selected_profile or (provider is not None and model is not None):
        return {}

    try:
        from api.profiles import profile_env_for_background_worker
        from cron.jobs import _compute_provider_model_snapshots
    except Exception:
        logger.warning(
            "Selected-profile cron snapshot repair unavailable; saving ambient snapshots",
            exc_info=True,
        )
        return {}

    try:
        with _CRON_CREATE_SNAPSHOT_LOCK:
            with profile_env_for_background_worker(
                selected_profile,
                "cron create snapshot",
                logger_override=logger,
            ):
                provider_snapshot, model_snapshot = _compute_provider_model_snapshots(
                    provider=provider,
                    model=model,
                    base_url=None,
                    no_agent=False,
                )
    except Exception:
        logger.warning(
            "Selected-profile cron snapshot repair failed for %s; saving ambient snapshots",
            selected_profile,
            exc_info=True,
        )
        return {}

    updates = {}
    if provider is None:
        updates["provider_snapshot"] = provider_snapshot
    if model is None:
        updates["model_snapshot"] = model_snapshot
    return updates


def _handle_cron_create(handler, body):
    try:
        require(body, "prompt", "schedule")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        from cron.jobs import create_job, update_job

        profile = _normalize_cron_profile_value(body.get("profile"))
        toast_notifications = body.get("toast_notifications") is not False
        requested_model = body.get("model") or None
        requested_provider = body.get("provider") or None
        job = create_job(
            prompt=body["prompt"],
            schedule=body["schedule"],
            name=body.get("name") or None,
            deliver=body.get("deliver") or "local",
            skills=body.get("skills") or [],
            model=requested_model,
            provider=requested_provider,
        )
        post_create_updates = {}
        if profile is not None:
            post_create_updates["profile"] = profile
            post_create_updates.update(
                _selected_profile_snapshot_updates(
                    profile,
                    provider=requested_provider,
                    model=requested_model,
                )
            )
        if not toast_notifications:
            post_create_updates["toast_notifications"] = False
        if post_create_updates:
            job = update_job(job["id"], post_create_updates) or job
        return j(handler, {"ok": True, "job": _cron_job_for_api(job)})
    except Exception as e:
        return j(handler, {"error": str(e)}, status=400)


def _handle_cron_delivery_options(handler):
    """Return available delivery platforms for cron jobs."""
    # The Agent moved this authority from ``cron.scheduler`` to
    # ``cron.scheduler_delivery``. Try the current location first and fall back
    # to the legacy one so the picker keeps working across Agent versions.
    # Without the fallback chain a bare ImportError silently degraded this
    # endpoint to local/origin only, dropping every messaging platform from the
    # cron delivery picker (telegram, discord, slack, feishu, ...).
    _KNOWN_DELIVERY_PLATFORMS = frozenset()
    import importlib
    for _module_name in ("cron.scheduler_delivery", "cron.scheduler"):
        try:
            _mod = importlib.import_module(_module_name)
        except Exception:
            continue
        _known = getattr(_mod, "_KNOWN_DELIVERY_PLATFORMS", None)
        if _known:
            _KNOWN_DELIVERY_PLATFORMS = frozenset(_known)
            break
    platforms = [
        {"value": "local", "label": "Local (save output only)"},
        {"value": "origin", "label": "Origin (reply to creator)"}
    ]
    for name in sorted(_KNOWN_DELIVERY_PLATFORMS):
        platforms.append({"value": name, "label": name.capitalize()})
    return j(handler, {"platforms": platforms})


def _handle_cron_update(handler, body):
    try:
        require(body, "job_id")
    except ValueError as e:
        return bad(handler, str(e))
    from cron.jobs import update_job

    try:
        updates = {}
        for k, v in body.items():
            if k == "job_id":
                continue
            if k == "profile":
                updates[k] = _normalize_cron_profile_value(v)
            elif k in ("model", "provider"):
                updates[k] = v if v else None
            elif v is not None:
                updates[k] = v
    except ValueError as e:
        return bad(handler, str(e))
    # #7352: ``update_job`` re-parses the updated schedule through
    # ``cron.jobs.parse_schedule``, which raises ``ValueError`` for
    # display-form input like ``"once at 2026-08-28 16:05"`` (or any other
    # user-typed garbage). That exception previously escaped to a 500
    # because the earlier normalization try block didn't cover it. Return
    # 400 with the parser message so the WebUI can surface a real
    # validation error instead of an opaque Internal Server Error.
    try:
        job = update_job(body["job_id"], updates)
    except ValueError as e:
        return bad(handler, str(e), 400)
    if not job:
        return bad(handler, "Job not found", 404)
    return j(handler, {"ok": True, "job": _cron_job_for_api(job)})


def _handle_cron_delete(handler, body):
    try:
        require(body, "job_id")
    except ValueError as e:
        return bad(handler, str(e))
    from cron.jobs import remove_job

    ok = remove_job(body["job_id"])
    if not ok:
        return bad(handler, "Job not found", 404)
    return j(handler, {"ok": True, "job_id": body["job_id"]})


def _handle_cron_run(handler, body):
    job_id = body.get("job_id", "")
    if not job_id:
        return bad(handler, "job_id required")
    from cron.jobs import get_job

    job = get_job(job_id)
    if not job:
        return bad(handler, "Job not found", 404)
    # Prevent double-run: reject if the job is already tracked as running
    already_running, elapsed = _is_cron_running(job_id)
    if already_running:
        return j(handler, {"ok": False, "job_id": job_id, "status": "already_running",
                            "elapsed": round(elapsed, 1)})
    _mark_cron_running(job_id)
    # Capture the TLS-active profile home now — the thread runs after the
    # request finishes, so TLS is gone by then.
    #
    # Resolve directly without a try/except: get_active_hermes_home() does
    # in-memory dict reads + a single Path.is_dir() stat, so the only way
    # it could raise from inside a request handler is if api.profiles
    # itself partially failed to import (in which case we'd already be
    # 500-ing the whole request). A silent fallback to None here would
    # re-introduce the exact bug #1573 fixes — the worker thread would
    # run unpinned against the process-global HERMES_HOME — so we'd
    # rather let any unexpected exception 500 the request than corrupt
    # cross-profile state.
    from api.profiles import get_active_hermes_home

    _profile_home = get_active_hermes_home()
    _execution_profile_home = _profile_home_for_cron_job(job)
    _event_profile = _event_profile_for_cron_job(job)
    threading.Thread(target=_run_cron_tracked, args=(job, _profile_home, _execution_profile_home, _event_profile), daemon=True).start()
    return j(handler, {"ok": True, "job_id": job_id, "status": "running"})


def _handle_cron_pause(handler, body):
    job_id = body.get("job_id", "")
    if not job_id:
        return bad(handler, "job_id required")
    from cron.jobs import pause_job

    result = pause_job(job_id, reason=body.get("reason"))
    if result:
        return j(handler, {"ok": True, "job": result})
    return bad(handler, "Job not found", 404)


def _handle_cron_resume(handler, body):
    job_id = body.get("job_id", "")
    if not job_id:
        return bad(handler, "job_id required")
    from cron.jobs import resume_job

    result = resume_job(job_id)
    if result:
        return j(handler, {"ok": True, "job": result})
    return bad(handler, "Job not found", 404)


def _git_session(handler, session_id: str):
    if not session_id:
        bad(handler, "session_id required")
        return None
    try:
        return get_session(session_id)
    except KeyError:
        bad(handler, "Session not found", 404)
        return None


def _git_session_workspace(handler, session_id: str):
    session = _git_session(handler, session_id)
    if session is None:
        return None
    return Path(session.workspace)


def _git_session_and_workspace(handler, session_id: str):
    session = _git_session(handler, session_id)
    if session is None:
        return None, None
    return session, Path(session.workspace)


def _git_locked_by_active_stream(session) -> bool:
    stream_id = getattr(session, "active_stream_id", None)
    if not stream_id:
        return False
    try:
        from api.config import STREAMS, STREAMS_LOCK

        with STREAMS_LOCK:
            return stream_id in STREAMS
    except Exception:
        return False


def _git_reject_destructive_if_unsafe(handler, session) -> bool:
    from api.workspace_git import (
        GitWorkspaceError,
        WORKSPACE_GIT_DESTRUCTIVE_ENV,
        workspace_git_destructive_enabled,
    )

    if not workspace_git_destructive_enabled():
        _git_bad(
            handler,
            GitWorkspaceError(
                f"Destructive workspace Git operations are disabled. Set {WORKSPACE_GIT_DESTRUCTIVE_ENV}=1 to enable them.",
                "destructive_git_disabled",
            ),
            status=403,
        )
        return True
    if _git_locked_by_active_stream(session):
        _git_bad(
            handler,
            GitWorkspaceError(
                "A session run is active. Wait for it to finish before running this Git operation.",
                "active_stream",
            ),
            status=409,
        )
        return True
    return False


def _handle_git_status(handler, parsed):
    qs = parse_qs(parsed.query)
    workspace = _git_session_workspace(handler, qs.get("session_id", [""])[0])
    if workspace is None:
        return True
    try:
        from api.workspace_git import GitWorkspaceError, git_status

        return j(handler, {"git": git_status(workspace)})
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_branches(handler, parsed):
    qs = parse_qs(parsed.query)
    workspace = _git_session_workspace(handler, qs.get("session_id", [""])[0])
    if workspace is None:
        return True
    try:
        from api.workspace_git import GitWorkspaceError, git_branches

        return j(handler, {"branches": git_branches(workspace)})
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_diff(handler, parsed):
    qs = parse_qs(parsed.query)
    workspace = _git_session_workspace(handler, qs.get("session_id", [""])[0])
    if workspace is None:
        return True
    path = qs.get("path", [""])[0]
    kind = qs.get("kind", ["unstaged"])[0]
    if not path:
        return bad(handler, "path required")
    try:
        from api.workspace_git import GitWorkspaceError, git_diff

        return j(handler, {"diff": git_diff(workspace, path, kind)})
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _git_bad(handler, err, status: int = 400):
    return j(
        handler,
        {
            "error": _sanitize_error(err),
            "code": getattr(err, "code", "git_failed") or "git_failed",
        },
        status=status,
    )


def _git_paths_from_body(body) -> list[str]:
    raw_paths = body.get("paths")
    if raw_paths is None and body.get("path"):
        raw_paths = [body.get("path")]
    if isinstance(raw_paths, str):
        raw_paths = [raw_paths]
    if not isinstance(raw_paths, list):
        raise ValueError("paths must be a list")
    return [str(path) for path in raw_paths]


def _handle_git_stage(handler, body):
    try:
        require(body, "session_id")
        paths = _git_paths_from_body(body)
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_stage

        return j(handler, {"ok": True, "git": git_stage(workspace, paths)})
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_unstage(handler, body):
    try:
        require(body, "session_id")
        paths = _git_paths_from_body(body)
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_unstage

        return j(handler, {"ok": True, "git": git_unstage(workspace, paths)})
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_discard(handler, body):
    try:
        require(body, "session_id")
        paths = _git_paths_from_body(body)
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_discard

        return j(
            handler,
            {
                "ok": True,
                "git": git_discard(
                    workspace,
                    paths,
                    delete_untracked=bool(body.get("delete_untracked")),
                ),
            },
        )
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _llm_git_commit_message(system_prompt: str, user_prompt: str, session=None) -> str:
    from api import profiles as profiles_api

    active_profile = profiles_api.get_active_profile_name() or "default"
    with profiles_api.profile_env_for_background_worker(
        active_profile,
        "git commit message",
        logger_override=logger,
    ):
        from api.config import (
            get_effective_default_model,
            model_with_provider_context,
            resolve_model_provider,
        )

        session_model = str(getattr(session, "model", "") or "").strip()
        session_provider = str(getattr(session, "model_provider", "") or "").strip() or None
        model_for_resolution = (
            model_with_provider_context(session_model, session_provider)
            if session_model
            else get_effective_default_model()
        )
        _main_model, _main_provider, _main_base_url = resolve_model_provider(model_for_resolution)
        _main_api_key = None
        _rt = None
        try:
            from api.oauth import resolve_runtime_provider_with_anthropic_env_lock
            from hermes_cli.runtime_provider import resolve_runtime_provider

            _rt = resolve_runtime_provider_with_anthropic_env_lock(
                resolve_runtime_provider,
                requested=_main_provider,
            )
            _main_api_key = _rt.get("api_key")
            if not _main_provider:
                _main_provider = _rt.get("provider")
            if not _main_base_url:
                _main_base_url = _rt.get("base_url")
        except Exception as _e:
            logger.debug("git commit message runtime provider resolution failed: %s", _e)
        # Atomic custom-provider authority (see the /api/chat note): the record
        # that supplies the endpoint must also supply the credential — and the
        # wire protocol, credential pool and ACP transport that go with it.
        _bundle = _resolve_agent_connection_bundle(
            _main_provider, _main_api_key, _main_base_url, _rt
        )
        _main_provider = _bundle["provider"]
        _main_api_key = _bundle["api_key"]
        _main_base_url = _bundle["base_url"]

        messages = [
            {"role": "system", "content": system_prompt},
            {"role": "user", "content": user_prompt},
        ]
        main_runtime = _auxiliary_main_runtime(_bundle, _main_model)
        ensure_agent_runtime_current()
        try:
            from agent.auxiliary_client import get_text_auxiliary_client

            aux_client, aux_model = get_text_auxiliary_client(
                "compression",
                main_runtime=main_runtime,
            )
            if aux_client is not None and aux_model:
                response = aux_client.chat.completions.create(
                    model=aux_model,
                    messages=messages,
                )
                return str(response.choices[0].message.content or "").strip()
        except Exception as _e:
            logger.debug("git commit message auxiliary model failed; falling back to main model: %s", _e)

        AIAgent = require_ai_agent_class()

        agent = AIAgent(
            model=_main_model,
            provider=_main_provider,
            base_url=_main_base_url,
            api_key=_main_api_key,
            platform="webui",
            quiet_mode=True,
            enabled_toolsets=[],
            session_id=f"git-commit-message-{uuid.uuid4().hex[:8]}",
            **_agent_bundle_kwargs(AIAgent, _bundle),
        )
        result = agent.run_conversation(
            user_message=user_prompt,
            system_message=system_prompt,
            conversation_history=[],
            task_id=f"git-commit-message-{uuid.uuid4().hex[:8]}",
        )
        return str(result.get("final_response") or "").strip()


def _handle_git_commit_message(handler, body):
    from api.workspace_git import (
        GitWorkspaceError,
        clean_generated_commit_message,
        staged_commit_message_prompt,
    )

    try:
        require(body, "session_id")
        session = get_session(body["session_id"])
        workspace = Path(session.workspace)

        prompt = staged_commit_message_prompt(workspace)
        message = clean_generated_commit_message(
            _llm_git_commit_message(prompt["system_prompt"], prompt["user_prompt"], session=session)
        )
        if not message:
            raise GitWorkspaceError("No commit message was generated")
        return j(handler, {"ok": True, "message": message, "truncated": bool(prompt.get("truncated"))})
    except KeyError:
        return bad(handler, "Session not found", 404)
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)
    except AgentRuntimeChangedError as e:
        return j(handler, agent_runtime_stale_payload(e), status=409)
    except Exception as e:
        logger.exception("git commit message generation failed")
        return bad(handler, _sanitize_error(e), 500)


def _handle_git_commit_message_selected(handler, body):
    from api.workspace_git import (
        GitWorkspaceError,
        clean_generated_commit_message,
        selected_commit_message_prompt,
    )

    try:
        require(body, "session_id")
        paths = _git_paths_from_body(body)
        session = get_session(body["session_id"])
        workspace = Path(session.workspace)

        prompt = selected_commit_message_prompt(workspace, paths)
        message = clean_generated_commit_message(
            _llm_git_commit_message(prompt["system_prompt"], prompt["user_prompt"], session=session)
        )
        if not message:
            raise GitWorkspaceError("No commit message was generated")
        return j(handler, {"ok": True, "message": message, "truncated": bool(prompt.get("truncated"))})
    except KeyError:
        return bad(handler, "Session not found", 404)
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)
    except AgentRuntimeChangedError as e:
        return j(handler, agent_runtime_stale_payload(e), status=409)
    except Exception as e:
        logger.exception("selected git commit message generation failed")
        return bad(handler, _sanitize_error(e), 500)


def _handle_git_commit(handler, body):
    try:
        require(body, "session_id", "message")
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_commit

        return j(handler, git_commit(workspace, body.get("message", "")))
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_commit_selected(handler, body):
    try:
        require(body, "session_id", "message")
        paths = _git_paths_from_body(body)
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_commit_selected

        return j(handler, git_commit_selected(workspace, body.get("message", ""), paths))
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_remote_action(handler, body, action: str):
    try:
        require(body, "session_id")
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if action in {"pull", "push"} and _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_fetch, git_pull, git_push

        actions = {
            "fetch": git_fetch,
            "pull": git_pull,
            "push": git_push,
        }
        return j(handler, actions[action](workspace))
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_checkout(handler, body):
    try:
        require(body, "session_id", "ref", "mode")
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_checkout

        result = git_checkout(
            workspace,
            str(body.get("ref", "")),
            str(body.get("mode", "local")),
            new_branch=body.get("new_branch"),
            track=bool(body.get("track")),
            dirty_mode=str(body.get("dirty_mode", "block")),
        )
        return j(
            handler,
            {
                "ok": True,
                "git": result.get("status"),
                "branches": result.get("branches"),
                "current_branch": result.get("current_branch"),
                "message": result.get("message", ""),
            },
        )
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_git_stash_checkout(handler, body):
    try:
        require(body, "session_id", "ref", "mode")
        session, workspace = _git_session_and_workspace(handler, body["session_id"])
        if workspace is None:
            return True
        if _git_reject_destructive_if_unsafe(handler, session):
            return True
        from api.workspace_git import GitWorkspaceError, git_stash_and_checkout

        result = git_stash_and_checkout(
            workspace,
            str(body.get("ref", "")),
            str(body.get("mode", "local")),
            new_branch=body.get("new_branch"),
            track=bool(body.get("track")),
        )
        return j(
            handler,
            {
                "ok": True,
                "git": result.get("status"),
                "branches": result.get("branches"),
                "current_branch": result.get("current_branch"),
                "message": result.get("message", ""),
                "stash_name": result.get("stash_name", ""),
                "stashed": bool(result.get("stashed")),
                "restored_stash": result.get("restored_stash"),
                "restore_failed": bool(result.get("restore_failed")),
                "restore_error": result.get("restore_error", ""),
                "restore_stash": result.get("restore_stash"),
            },
        )
    except ValueError as e:
        return bad(handler, str(e))
    except GitWorkspaceError as e:
        return _git_bad(handler, e)


def _handle_file_delete(handler, body):
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        target = safe_resolve(ws_root, body["path"])
        # Reject a symlinked entry BEFORE the follow-based exists() check: a
        # dangling symlink resolves to a missing target, so an exists()-first
        # order would misclassify it as 404 "File not found" and leave it
        # permanently undeletable. is_symlink() is a no-follow lstat on the
        # lexically-requested path, so it catches both live and dangling links.
        if (ws_root / body["path"]).is_symlink():
            return bad(handler, "Cannot delete a symlinked entry")
        if not target.exists():
            return bad(handler, "File not found", 404)
        if target.is_dir():
            if not body.get("recursive"):
                return bad(handler, "Set recursive=true to delete directories")
            rmtree_anchored(ws_root, target)
        else:
            unlink_anchored(ws_root, target)
        return j(handler, {"ok": True, "path": body["path"]})
    except (ValueError, FileNotFoundError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_save(handler, body):
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        target = safe_resolve(ws_root, body["path"])
        if (ws_root / body["path"]).is_symlink():
            return bad(handler, "Cannot save to a symlinked entry")
        if not target.exists():
            return bad(handler, "File not found", 404)
        if target.is_dir():
            return bad(handler, "Cannot save: path is a directory")
        if Path(str(body["path"])).suffix.lower() in {".docx", ".xlsx", ".pptx"}:
            return bad(handler, "Use /api/file/office-save for Office documents")
        data = str(body.get("content", "")).encode("utf-8")
        fd = open_anchored_write_fd(ws_root, target)
        with os.fdopen(fd, "wb", closefd=True) as fh:
            fh.write(data)
        return j(
            handler, {"ok": True, "path": body["path"], "size": len(data)}
        )
    except (ValueError, FileNotFoundError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_office_file_save(handler, body):
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        target = safe_resolve(ws_root, body["path"])
        if (ws_root / body["path"]).is_symlink():
            return bad(handler, "Cannot save to a symlinked entry")
        if not target.exists():
            return bad(handler, "File not found", 404)
        if target.is_dir():
            return bad(handler, "Cannot save: path is a directory")
        if Path(str(body["path"])).suffix.lower() not in {".docx", ".xlsx", ".pptx"}:
            return bad(handler, "Office save is only available for .docx, .xlsx, and .pptx files")
        from api.office_documents import save_office_document

        current_bytes = _read_anchored_file_bytes(ws_root, target)
        preview, updated_bytes = save_office_document(body["path"], current_bytes, body.get("content", ""))
        fd = open_anchored_write_fd(ws_root, target)
        with os.fdopen(fd, "wb", closefd=True) as fh:
            fh.write(updated_bytes)
        preview.update({"ok": True, "path": body["path"], "size": len(updated_bytes)})
        return j(handler, preview)
    except ImportError as e:
        return bad(handler, str(e), 503)
    except (ValueError, FileNotFoundError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_create(handler, body):
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        target = safe_resolve(ws_root, body["path"])
        if target.exists():
            return bad(handler, "File already exists")
        data = str(body.get("content", "")).encode("utf-8")
        fd = open_anchored_create_fd(ws_root, target)
        with os.fdopen(fd, "wb", closefd=True) as fh:
            fh.write(data)
        return j(
            handler, {"ok": True, "path": target.relative_to(ws_root.resolve()).as_posix()}
        )
    except FileExistsError:
        return bad(handler, "File already exists")
    except (ValueError, FileNotFoundError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_rename(handler, body):
    try:
        require(body, "session_id", "path", "new_name")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        ws_root_resolved = ws_root.resolve()
        source = safe_resolve(ws_root, body["path"])
        # Reject a symlinked entry BEFORE the follow-based exists() check (see
        # _handle_file_delete): a dangling symlink would otherwise 404 and stay
        # unrenameable. is_symlink() is a no-follow lstat on the requested path.
        if (ws_root / body["path"]).is_symlink():
            return bad(handler, "Cannot rename a symlinked entry")
        if not source.exists():
            return bad(handler, "File not found", 404)
        new_name = body["new_name"].strip()
        if not new_name or "/" in new_name or "\\" in new_name or ".." in new_name:
            return bad(handler, "Invalid file name")
        dest = source.parent / new_name
        if dest.exists():
            return bad(handler, f'A file named "{new_name}" already exists')
        rename_anchored(ws_root, source, dest)
        new_rel = dest.relative_to(ws_root_resolved).as_posix()
        return j(handler, {"ok": True, "old_path": body["path"], "new_path": new_rel})
    except FileExistsError:
        return bad(handler, f'A file named "{body.get("new_name", "")}" already exists')
    except (ValueError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_move(handler, body):
    try:
        require(body, "session_id", "path", "dest_dir")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        # safe_resolve() returns paths under the RESOLVED root, so compute
        # returned relative paths against the resolved root too — otherwise a
        # symlinked workspace root (e.g. macOS /tmp -> /private/tmp) makes
        # dest.relative_to(ws_root) raise after a successful on-disk move,
        # returning a confusing 400 for a move that actually happened.
        ws_root_resolved = ws_root.resolve()
        source = safe_resolve(ws_root, body["path"])
        # Reject a symlinked SOURCE entry BEFORE the follow-based exists() check.
        # safe_resolve() follows the final symlink, so source.name/source.parent
        # would point at the link's TARGET, not the dragged entry — moving
        # link.txt would silently move dir/real.txt and leave link.txt dangling.
        # Detect the symlink on the lexically-requested final component (lstat,
        # no-follow) and refuse; running this before exists() also means a
        # dangling symlink is rejected (400) rather than misclassified as 404
        # (matches the delete/rename ordering).
        if (ws_root / body["path"]).is_symlink():
            return bad(handler, "Cannot move a symlinked entry")
        if not source.exists():
            return bad(handler, "File not found", 404)
        dest_dir_raw = (body.get("dest_dir") or ".").strip()
        if not dest_dir_raw:
            dest_dir_raw = "."
        if ".." in dest_dir_raw.split("/"):
            return bad(handler, "Invalid destination")
        dest_parent = safe_resolve(ws_root, dest_dir_raw)
        if not dest_parent.is_dir():
            return bad(handler, "Destination folder not found", 404)
        if source.is_dir():
            try:
                dest_parent.resolve().relative_to(source.resolve())
                return bad(handler, "Cannot move a folder into itself or its subfolder")
            except ValueError:
                pass
        dest = dest_parent / source.name
        if dest.resolve() == source.resolve():
            new_rel = source.relative_to(ws_root_resolved).as_posix()
            return j(
                handler,
                {"ok": True, "old_path": body["path"], "new_path": new_rel},
            )
        # Perform the move race-safely. The path-based checks above can be raced
        # (TOCTOU): between validating dest_parent and renaming, dest_dir could be
        # swapped to a symlink pointing outside the workspace, and a path-based
        # rename would follow it. Open BOTH parent directories as workspace-anchored
        # fds (openat + O_NOFOLLOW — every component verified non-symlink), do the
        # collision check by fd, then rename via src_dir_fd/dst_dir_fd so the kernel
        # operates on the verified directories, not re-resolved pathnames.
        leaf = source.name
        if os.open in getattr(os, "supports_dir_fd", set()):
            src_parent_fd = open_anchored_fd(ws_root, source.parent, want_dir=True)
            try:
                dst_parent_fd = open_anchored_fd(ws_root, dest_parent, want_dir=True)
                try:
                    try:
                        os.stat(leaf, dir_fd=dst_parent_fd, follow_symlinks=False)
                        return bad(
                            handler,
                            f'A file named "{leaf}" already exists in that folder',
                        )
                    except FileNotFoundError:
                        pass
                    os.rename(
                        leaf, leaf,
                        src_dir_fd=src_parent_fd, dst_dir_fd=dst_parent_fd,
                    )
                finally:
                    os.close(dst_parent_fd)
            finally:
                os.close(src_parent_fd)
        else:
            # Windows / no openat: no new race protection available, but creating
            # symlinks needs admin there. Fall back to the path-based rename.
            if dest.exists():
                return bad(
                    handler,
                    f'A file named "{source.name}" already exists in that folder',
                )
            source.rename(dest)
        new_rel = dest.relative_to(ws_root_resolved).as_posix()
        return j(
            handler,
            {"ok": True, "old_path": body["path"], "new_path": new_rel},
        )
    except (ValueError, FileNotFoundError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_create_dir(handler, body):
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        ws_root = Path(s.workspace)
        target = safe_resolve(ws_root, body["path"])
        if target.exists():
            return bad(handler, "Path already exists")
        make_anchored_dir(ws_root, target)
        return j(
            handler, {"ok": True, "path": target.relative_to(ws_root.resolve()).as_posix()}
        )
    except (ValueError, FileNotFoundError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_reveal(handler, body):
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        target = safe_resolve(Path(s.workspace), body["path"])
        if not target.exists():
            # Include the resolved server-side path in the error message so
            # the frontend toast can show *which* file the system expected.
            # Useful when a stale session row still references a deleted file
            # (#1764 — Cygnus's screenshot showed a "Failed to reveal: not
            # found" toast that dropped the path entirely, leaving no clue
            # what was missing).
            return bad(handler, f"File not found: {target}", 404)

        target_str = str(target)

        # Optional Docker host/container path translation (mirrors _handle_file_open_vscode).
        from api.config import get_config as _get_cfg  # noqa: PLC0415
        vscode_cfg = _get_cfg().get("vscode", {})
        if not isinstance(vscode_cfg, dict):
            vscode_cfg = {}
        container_prefix = vscode_cfg.get("container_path_prefix", "")
        host_prefix = vscode_cfg.get("host_path_prefix", "")
        if container_prefix and host_prefix:
            _norm = container_prefix.rstrip('/') + '/'
            if target_str.startswith(_norm) or target_str == container_prefix.rstrip('/'):
                target_str = host_prefix + target_str[len(container_prefix):]

        system = platform.system()
        if system == "Darwin":
            subprocess.Popen(["open", "-R", target_str])
        elif system == "Windows":
            subprocess.Popen(["explorer.exe", "/select," + target_str])
        else:
            # Linux / other — open parent directory
            subprocess.Popen(["xdg-open", str(Path(target_str).parent)])

        return j(handler, {"ok": True, "path": body["path"]})
    except (ValueError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_path(handler, body):
    """Resolve a relative workspace-rooted path into an absolute on-disk path.

    The right-click "Copy file path" action (#1764) wants to put the
    absolute path on the user's clipboard so they can paste it into a
    terminal, editor, or anywhere else without having to round-trip through
    the OS file browser. The frontend can't compute the absolute path on
    its own — `safe_resolve` joins against the session's workspace root
    which only the server knows. The handler here is a thin lookup; no
    filesystem mutation, no OS-specific dispatch. We do NOT require the
    target to exist (unlike `_handle_file_reveal`) — copying the path of a
    just-deleted file is still useful, and refusing would force callers
    to special-case 404s for an action that cannot fail destructively.
    """
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        target = safe_resolve(Path(s.workspace), body["path"])
        return j(handler, {"ok": True, "path": str(target)})
    except (ValueError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_file_open_vscode(handler, body):
    """Open a workspace file or folder in VS Code (#2735).

    Reads optional ``vscode`` config block from config.yaml:

        vscode:
          command: code          # executable on PATH; defaults to "code"
          host_path_prefix: /home/user/projects       # Docker host path
          container_path_prefix: /app/workspace       # matching container path

    If ``host_path_prefix`` and ``container_path_prefix`` are both set,
    paths that begin with ``container_path_prefix`` are translated to the
    host prefix before being handed to VS Code.  This lets users running
    Hermes WebUI inside Docker still open files in their local editor.
    """
    try:
        require(body, "session_id", "path")
    except ValueError as e:
        return bad(handler, str(e))
    try:
        s = get_session_for_file_ops(body["session_id"])
    except KeyError:
        return bad(handler, "Session not found", 404)
    try:
        target = safe_resolve(Path(s.workspace), body["path"])
        if not target.exists():
            return bad(handler, f"File not found: {target}", 404)

        target_str = str(target)

        # Optional Docker host/container path translation
        from api.config import get_config as _get_cfg  # noqa: PLC0415
        vscode_cfg = _get_cfg().get("vscode", {})
        if not isinstance(vscode_cfg, dict):
            vscode_cfg = {}
        container_prefix = vscode_cfg.get("container_path_prefix", "")
        host_prefix = vscode_cfg.get("host_path_prefix", "")
        if container_prefix and host_prefix:
            _norm = container_prefix.rstrip('/') + '/'
            if target_str.startswith(_norm) or target_str == container_prefix.rstrip('/'):
                target_str = host_prefix + target_str[len(container_prefix):]

        cmd = vscode_cfg.get("command", "code")
        # Resolve the command to an absolute path so subprocess.Popen finds it
        # even when the server process inherits a minimal PATH (e.g. when
        # launched via start.sh on macOS where /usr/local/bin may be absent).
        resolved_cmd = shutil.which(cmd)
        if resolved_cmd is None:
            # Try common VS Code installation paths as fallback.
            # macOS: /usr/local/bin/code (symlink) or app bundle CLI
            # Linux: /usr/bin/code or snap
            # Windows: user-install under %LOCALAPPDATA%, system-install under %PROGRAMFILES%
            _local_app_data = os.environ.get("LOCALAPPDATA", "")
            _prog_files = os.environ.get("PROGRAMFILES", "C:\\Program Files")
            _prog_files_x86 = os.environ.get("PROGRAMFILES(X86)", "C:\\Program Files (x86)")
            _vscode_fallbacks = [
                # macOS
                "/usr/local/bin/code",
                "/Applications/Visual Studio Code.app/Contents/Resources/app/bin/code",
                # Linux
                "/usr/bin/code",
                "/snap/bin/code",
                # Windows (user install)
                os.path.join(_local_app_data, "Programs", "Microsoft VS Code", "bin", "code.cmd"),
                # Windows (system install)
                os.path.join(_prog_files, "Microsoft VS Code", "bin", "code.cmd"),
                os.path.join(_prog_files_x86, "Microsoft VS Code", "bin", "code.cmd"),
            ]
            for fb in _vscode_fallbacks:
                if fb and Path(fb).exists():
                    resolved_cmd = fb
                    break
        if resolved_cmd is None:
            return bad(
                handler,
                f"VS Code command not found: {cmd!r}. "
                "Install VS Code and ensure the 'code' CLI is on PATH, "
                "or set vscode.command in config.yaml to the full path.",
            )
        subprocess.Popen([resolved_cmd, target_str])

        return j(handler, {"ok": True, "path": body["path"]})
    except (ValueError, PermissionError, OSError) as e:
        return bad(handler, _sanitize_error(e))


def _handle_workspace_add(handler, body):
    # Strip surrounding paired quotes BEFORE any further processing — macOS
    # Finder's "Copy as Pathname" wraps paths in single quotes, and users
    # routinely paste those quoted strings into the Add Space input.
    # Doing this at the route entry means every downstream check (blocked
    # system path, validate_workspace_to_add, duplicate detection) sees the
    # cleaned form.
    path_str = _strip_surrounding_quotes(body.get("path", "").strip())
    name = body.get("name", "").strip()
    auto_create = body.get("create", False)
    if not path_str:
        return bad(handler, "path is required")
    # Validate the path is NOT a blocked system root BEFORE any filesystem mutation.
    # This prevents creating orphan directories on rejected paths (#782 review).
    # _is_blocked_system_path honours user-tmp carve-outs (e.g. /var/folders on
    # macOS) so pytest's tmp_path_factory paths and other legit user-tmp dirs
    # still register cleanly.
    try:
        from api.workspace import _remote_terminal_workspace_candidate, _resolve_path
        from api.profiles import get_active_profile_name
        active_profile = get_active_profile_name()
        remote_candidate = _remote_terminal_workspace_candidate(path_str, profile=active_profile)
        candidate = _resolve_path(path_str, profile=active_profile)
    except (ValueError, OSError, RuntimeError) as e:
        # Invalid path (e.g. embedded null byte) — fail closed with a clean 400
        # instead of letting .resolve() raise an uncaught 500.
        return bad(handler, f"Invalid path: {_sanitize_error(e)}")
    if remote_candidate is None:
        if _is_blocked_system_path(candidate):
            # Home-directory carve-out, mirroring the validators
            # (resolve_trusted_workspace / validate_workspace_to_add): a workspace
            # at or under the active user's home must stay allowed even when that
            # home lives under an otherwise-blocked root (e.g. systemd-homed
            # /var/home/<user>/...). Without this the route rejects valid
            # /var/home workspaces before validate_workspace_to_add()'s carve-out
            # can run.
            _home = _home_path()
            if not (_home != Path("/") and (candidate == _home or _is_within(candidate, _home))):
                return bad(handler, f"Path points to a system directory: {candidate}")
        # Now safe to create the directory if requested
        if auto_create:
            try:
                candidate.mkdir(parents=True, exist_ok=True)
            except (OSError, PermissionError) as e:
                return bad(handler, f"Could not create directory: {_sanitize_error(e)}")
    # Full validation (exists, is_dir) — should pass now that dir exists
    try:
        p = validate_workspace_to_add(path_str, profile=active_profile)
    except ValueError as e:
        return bad(handler, str(e))
    try:
        wss = load_workspaces(profile=active_profile)
    except TypeError:
        wss = load_workspaces()
    if any(w["path"] == str(p) for w in wss):
        return bad(handler, "Workspace already in list")
    wss.append({"path": str(p), "name": name or p.name})
    try:
        save_workspaces(wss, profile=active_profile)
    except TypeError:
        save_workspaces(wss)
    return j(handler, {"ok": True, "workspaces": wss})


def _handle_workspace_remove(handler, body):
    path_str = body.get("path", "").strip()
    if not path_str:
        return bad(handler, "path is required")
    from api.profiles import get_active_profile_name
    active_profile = get_active_profile_name()
    try:
        wss = load_workspaces(profile=active_profile)
    except TypeError:
        wss = load_workspaces()
    wss = [w for w in wss if w["path"] != path_str]
    try:
        save_workspaces(wss, profile=active_profile)
    except TypeError:
        save_workspaces(wss)
    return j(handler, {"ok": True, "workspaces": wss})


def _handle_workspace_rename(handler, body):
    path_str = body.get("path", "").strip()
    name = body.get("name", "").strip()
    if not path_str or not name:
        return bad(handler, "path and name are required")
    from api.profiles import get_active_profile_name
    active_profile = get_active_profile_name()
    try:
        wss = load_workspaces(profile=active_profile)
    except TypeError:
        wss = load_workspaces()
    for w in wss:
        if w["path"] == path_str:
            w["name"] = name
            break
    else:
        return bad(handler, "Workspace not found", 404)
    try:
        save_workspaces(wss, profile=active_profile)
    except TypeError:
        save_workspaces(wss)
    return j(handler, {"ok": True, "workspaces": wss})


def _handle_workspace_reorder(handler, body):
    """Reorder workspaces by providing an ordered list of paths.

    Accepts {"paths": ["path1", "path2", ...]}. The workspaces list is
    rewritten so that entries appear in the given order. Any workspace
    not included in the request is appended at the end (preserves data).
    """
    paths = body.get("paths", [])
    if not paths or not isinstance(paths, list):
        return bad(handler, "paths is required and must be a list")
    from api.profiles import get_active_profile_name
    active_profile = get_active_profile_name()
    try:
        wss = load_workspaces(profile=active_profile)
    except TypeError:
        wss = load_workspaces()
    by_path = {w["path"]: w for w in wss}
    # Build reordered list: given order first, then any omitted entries
    reordered = []
    seen = set()
    for p in paths:
        p = p.strip()
        if p in by_path and p not in seen:
            reordered.append(by_path[p])
            seen.add(p)
    # Append any workspaces not mentioned (safety net)
    for w in wss:
        if w["path"] not in seen:
            reordered.append(w)
    try:
        save_workspaces(reordered, profile=active_profile)
    except TypeError:
        # Legacy signature (test doubles with single-arg lambdas, older forks).
        save_workspaces(reordered)
    return j(handler, {"ok": True, "workspaces": reordered})


def _resolve_approval_legacy(sid: str, approval_id: str, choice: str, run_id: str = "") -> bool:
    """Resolve an approval through the existing callback path.

    Slice 3b keeps the RuntimeAdapter as a protocol translator: it delegates to
    this legacy helper rather than owning approval queues or callback state.
    """
    # Pop the targeted entry from the pending queue by approval_id. Old clients
    # that omit approval_id still resolve the oldest entry for compatibility.
    pending = None
    found_target = False
    gateway_keys = []
    local_gateway_approval_id = ""
    with _lock:
        reconcile_gateway_pending_mirror_locked(sid)
        queue = _pending.get(sid)
        if isinstance(queue, list):
            if approval_id:
                # Prefer a local exact-id match over a mirrored one, so a
                # preceding remote mirror cannot consume the user's local choice.
                preferred_index = None
                fallback_index = None
                for i, entry in enumerate(queue):
                    if entry.get("approval_id") != approval_id:
                        continue
                    if run_id and str(entry.get("run_id") or "").strip() != run_id:
                        continue
                    if not entry.get(_GATEWAY_MIRROR_FLAG) or not str(entry.get("run_id") or "").strip():
                        preferred_index = i
                        break
                    if fallback_index is None:
                        fallback_index = i
                match_index = preferred_index if preferred_index is not None else fallback_index
                if match_index is not None:
                    pending = queue.pop(match_index)
                    found_target = True
                else:
                    # A stale explicit id must not accidentally approve the
                    # oldest queued command; duplicate/stale responses are
                    # bounded as not-active by the adapter route.
                    pending = None
            else:
                pending = queue.pop(0) if queue else None
                found_target = pending is not None
            if not queue:
                _pending.pop(sid, None)
        elif queue:
            # Legacy single-dict value.
            if (
                not approval_id
                or (
                    queue.get("approval_id") == approval_id
                    and (not run_id or str(queue.get("run_id") or "").strip() == run_id)
                )
            ):
                pending = _pending.pop(sid, None)
                found_target = pending is not None
        # When no _pending entry found AND no explicit approval_id was
        # given, peek into _gateway_queues for pattern_keys so legacy
        # no-id clients still work. When approval_id IS given but not
        # found, the caller sent a stale/duplicate id — do NOT fall
        # through to the gateway queue, or a stale click on approval A
        # would resolve the unrelated live approval B.
        if not pending and not approval_id:
            gw_queue = _gateway_queues.get(sid)
            if gw_queue and len(gw_queue) > 0:
                gw_entry = gw_queue[0]
                # _gateway_queues stores _ApprovalEntry objects; their
                # .data dict carries command, pattern_key, pattern_keys.
                gw_data = getattr(gw_entry, 'data', None) or {}
                gateway_keys = gw_data.get("pattern_keys") or [gw_data.get("pattern_key", "")]
                # Peek is not strict — a concurrent resolver may pop a
                # different gateway entry before we reach
                # resolve_gateway_approval below, but approve_session is
                # idempotent over the session key set so the outcome is
                # the same regardless of which entry wins the race.
                found_target = True
        elif approval_id:
            gw_queue = _gateway_queues.get(sid)
            if gw_queue and len(gw_queue) > 0:
                gw_entry = gw_queue[0]
                gw_data = getattr(gw_entry, "data", None) or {}
                gw_approval_id = str(gw_data.get("approval_id") or "").strip()
                gw_run_id = str(gw_data.get("run_id") or "").strip()
                if gw_approval_id == approval_id and (not run_id or gw_run_id == run_id):
                    local_gateway_approval_id = approval_id
                elif not run_id and found_target and pending:
                    # The no-run mirror may belong to a NON-head producer
                    # (multiple parked entries, #7093). The queue head's own
                    # token won't match a non-head mirror, so scan every live
                    # producer for a token/approval_id match instead of only
                    # comparing against `_gateway_queues[0]`.
                    pending_token = str(pending.get(_GATEWAY_MIRROR_TOKEN) or "").strip()
                    matched_data = None
                    for _cand in gw_queue:
                        _cand_data = getattr(_cand, "data", None) or {}
                        _cand_token = str(_cand_data.get("_webui_mirror_token") or "").strip()
                        if pending_token and _cand_token == pending_token:
                            matched_data = _cand_data
                            break
                        if (str(_cand_data.get("approval_id") or "").strip() == approval_id
                                and not str(_cand_data.get("run_id") or "").strip()):
                            matched_data = _cand_data
                            break
                    if matched_data is not None:
                        matched_data["approval_id"] = approval_id
                        local_gateway_approval_id = approval_id
        # Notify SSE subscribers of the new head (or empty state) so the UI
        # surfaces any trailing approvals that were queued behind this one
        # without waiting for the next submit_pending. Without this, a parallel
        # tool-call scenario (#527) would leave the second approval invisible
        # in the SSE path until the next event ever fired (the agent thread
        # would be parked indefinitely from the user's perspective).
        if not local_gateway_approval_id:
            if isinstance(_pending.get(sid), list) and _pending[sid]:
                _approval_sse_notify_locked(sid, _pending[sid][0], len(_pending[sid]))
            else:
                _approval_sse_notify_locked(sid, None, 0)

    # Collect keys from both _pending and _gateway_queues
    keys_from_pending = pending.get("pattern_keys") or [pending.get("pattern_key", "")] if pending else []
    all_keys = [k for k in keys_from_pending if k] + [k for k in gateway_keys if k]
    if choice == "session":
        for k in all_keys:
            approve_session(sid, k)
    elif choice == "always":
        for k in all_keys:
            approve_session(sid, k)
            approve_permanent(k)
        save_permanent_allowlist(_permanent_approved)
    # choice == "once": no persistence — approval lasts this single call only.
    # resolve_gateway_approval() below unblocks the parked agent thread for
    # every choice, so "once" still lets the current tool run; we just must not
    # call approve_session() here, or the next matching guarded call would find
    # the pattern already session-approved and skip its approval card (#6017).
    # Unblock the agent thread waiting in the gateway approval queue.
    # This is the primary signal when streaming is active — the agent
    # thread is parked in entry.event.wait() and needs to be woken up.
    gateway_resolved = 0
    local_gateway_resolved = 0
    if approval_id and found_target and not run_id and local_gateway_approval_id:
        local_gateway_resolved, _head, _total = resolve_gateway_pending_local(
            sid, local_gateway_approval_id, choice
        )
    elif approval_id and found_target and run_id:
        gateway_resolved, _head, _total = resolve_gateway_pending_run(
            sid, approval_id, run_id, choice
        )
    elif not approval_id:
        gateway_resolved = resolve_gateway_approval(sid, choice, resolve_all=False) or 0
    # Keep the historical no-id response path truthy for old clients/tests while
    # making stale explicit ids bounded as not-active for Slice 3b.
    resolved = bool(pending) or bool(gateway_resolved) or bool(local_gateway_resolved) or not bool(approval_id)
    if resolved:
        publish_session_list_changed("attention_resolved")
    return resolved


_GATEWAY_APPROVAL_RELAY_UNAVAILABLE = (
    "Gateway approval could not be relayed because the active run is unavailable. "
    "Reopen the session or retry after it reconnects."
)
_GATEWAY_APPROVAL_RELAY_IN_PROGRESS = (
    "Another approval response for this Gateway run is already in progress. "
    "Wait for it to finish, then retry if the card is still visible."
)


def _gateway_approval_failure(
    sid: str,
    choice: str,
    *,
    code: str,
    error: str,
    status: int,
    enable_yolo: bool,
    relayed: bool = False,
) -> tuple[dict, int]:
    """Build a failed relay response with authoritative session-YOLO state."""
    payload = {
        "ok": False,
        "choice": choice,
        "relayed": relayed,
        "code": code,
        "error": error,
    }
    if enable_yolo:
        payload["yolo_enabled"] = bool(is_session_yolo_enabled(sid))
    return payload, status


def _relay_gateway_run_approval(
    sid: str,
    mirror: dict,
    choice: str,
    *,
    enable_yolo: bool,
) -> tuple[dict, int]:
    """Relay one exact run-backed mirror under the shared `(session, run)` owner.

    The mirror remains actionable unless the remote Runs API confirms success.
    Both the approval-card endpoint and the ordinary session-YOLO endpoint use
    this chokepoint so one tab cannot retire another tab's parked remote run.
    """
    from api.config import gateway_supports_approval_identity_v1
    from api.gateway_chat import gateway_run_endpoint
    from api.runner_client import HttpRunnerClient, RunnerClientError

    run_id = str(mirror.get("run_id") or "").strip()
    approval_id = str(mirror.get("approval_id") or "").strip()
    mirror_token = str(mirror.get(_GATEWAY_MIRROR_TOKEN) or "").strip()
    if not run_id or not approval_id:
        return _gateway_approval_failure(
            sid,
            choice,
            code="gateway_run_unavailable",
            error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
            status=409,
            enable_yolo=enable_yolo,
        )
    if not claim_gateway_approval_relay_owner(sid, run_id, approval_id):
        return _gateway_approval_failure(
            sid,
            choice,
            code="gateway_approval_in_progress",
            error=_GATEWAY_APPROVAL_RELAY_IN_PROGRESS,
            status=409,
            enable_yolo=enable_yolo,
        )

    try:
        current_mirror = gateway_pending_mirror(
            sid,
            approval_id=approval_id,
            run_id=run_id,
            mirror_token=mirror_token,
        )
        if not current_mirror:
            return _gateway_approval_failure(
                sid,
                choice,
                code="gateway_run_unavailable",
                error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                status=409,
                enable_yolo=enable_yolo,
            )

        base_url, api_key = gateway_run_endpoint(run_id)
        identity_v1 = bool(current_mirror.get(_GATEWAY_AGENT_IDENTITY_V1)) and (
            gateway_supports_approval_identity_v1(base_url, api_key)
        )
        if not identity_v1:
            run_head = gateway_pending_mirror(sid, run_id=run_id)
            if not run_head or str(run_head.get("approval_id") or "").strip() != approval_id:
                return _gateway_approval_failure(
                    sid,
                    choice,
                    code="gateway_run_unavailable",
                    error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                    status=409,
                    enable_yolo=enable_yolo,
                )

        yolo_transition = begin_session_yolo_transition(sid) if enable_yolo else None
        relay_error = None
        relay_succeeded = False
        try:
            HttpRunnerClient(base_url=base_url, api_key=api_key).respond_approval(
                run_id,
                approval_id if identity_v1 else "",
                choice,
            )
            relay_succeeded = True
        except (RunnerClientError, ValueError) as exc:
            relay_error = str(exc)
        finally:
            finish_session_yolo_transition(sid, yolo_transition, succeeded=relay_succeeded)

        if relay_error is not None:
            return _gateway_approval_failure(
                sid,
                choice,
                code="gateway_approval_relay_failed",
                error=relay_error,
                status=502,
                enable_yolo=enable_yolo,
                relayed=True,
            )

        # The outbound relay resumes the remote run. Retire the local projection
        # only after that succeeds, then settle any matching in-process mirror.
        _resolve_approval_legacy(sid, approval_id, choice, run_id=run_id)
        retire_gateway_pending_mirror(
            sid,
            approval_id=approval_id,
            run_id=run_id,
            mirror_token=mirror_token,
        )
        return {
            "ok": True,
            "choice": choice,
            "relayed": True,
            **(
                {"yolo_enabled": bool(is_session_yolo_enabled(sid))}
                if enable_yolo
                else {}
            ),
        }, 200
    finally:
        release_gateway_approval_relay_owner(sid, run_id, approval_id)


def _pending_approval_owner_state(
    sid: str,
    approval_id: str,
    run_id: str = "",
    mirror_token: str = "",
) -> tuple[bool, bool]:
    """Return `(exact_owner_exists, any_pending_exists)` under queue authority."""
    approval_id = str(approval_id or "").strip()
    run_id = str(run_id or "").strip()
    mirror_token = str(mirror_token or "").strip()
    with _lock:
        reconcile_gateway_pending_mirror_locked(sid)
        queue = _pending.get(sid)
        entries = queue if isinstance(queue, list) else [queue] if queue else []
        exact = False
        for entry in entries:
            if not isinstance(entry, dict) or str(entry.get("approval_id") or "") != approval_id:
                continue
            entry_run_id = str(entry.get("run_id") or "").strip()
            entry_mirror_token = str(entry.get(_GATEWAY_MIRROR_TOKEN) or "").strip()
            if run_id and entry_run_id != run_id:
                continue
            if mirror_token and entry_mirror_token != mirror_token:
                continue
            if (run_id or mirror_token) and not entry.get(_GATEWAY_MIRROR_FLAG):
                continue
            exact = True
            break
        return exact, bool(entries or _gateway_queues.get(sid))


def _enable_session_yolo_and_release_pending(
    sid: str,
    *,
    choice: str,
    approval_id: str = "",
    run_id: str = "",
    mirror_token: str = "",
    include_choice: bool = False,
) -> tuple[dict, int]:
    """Relay every parked remote approval, drain local waiters, then commit YOLO."""
    approval_id = str(approval_id or "").strip()
    run_id = str(run_id or "").strip()
    mirror_token = str(mirror_token or "").strip()
    has_exact_remote_owner = bool(run_id or mirror_token)
    if has_exact_remote_owner and (not approval_id or not run_id or not mirror_token):
        return _gateway_approval_failure(
            sid,
            choice,
            code="gateway_run_unavailable",
            error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
            status=409,
            enable_yolo=True,
        )

    yolo_transition = None
    try:
        with gateway_yolo_handoff(sid):
            yolo_transition = begin_session_yolo_transition(sid)
            stale_cleared = False
            if approval_id:
                exact_owner, any_pending = _pending_approval_owner_state(
                    sid,
                    approval_id,
                    run_id,
                    mirror_token,
                )
                if not exact_owner and (has_exact_remote_owner or any_pending):
                    return _gateway_approval_failure(
                        sid,
                        choice,
                        code="gateway_run_unavailable",
                        error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                        status=409,
                        enable_yolo=True,
                    )
                stale_cleared = not exact_owner

            run_mirrors = gateway_pending_mirrors(sid)
            relayed = 0
            for mirror in run_mirrors:
                relay_payload, relay_status = _relay_gateway_run_approval(
                    sid,
                    mirror,
                    choice,
                    enable_yolo=False,
                )
                if relay_status != 200 or not relay_payload.get("ok"):
                    finish_session_yolo_transition(
                        sid,
                        yolo_transition,
                        succeeded=False,
                    )
                    yolo_transition = None
                    return {
                        **relay_payload,
                        "yolo_enabled": bool(is_session_yolo_enabled(sid)),
                    }, relay_status
                relayed += 1

            resolve_gateway_pending_local_all(
                sid,
                choice,
            )
            finish_session_yolo_transition(sid, yolo_transition, succeeded=True)
            yolo_transition = None
            return {
                "ok": True,
                "yolo_enabled": bool(is_session_yolo_enabled(sid)),
                **({"choice": choice} if include_choice or relayed else {}),
                **({"relayed": True} if relayed else {}),
                **({"stale_cleared": True} if stale_cleared else {}),
            }, 200
    finally:
        if yolo_transition is not None:
            finish_session_yolo_transition(sid, yolo_transition, succeeded=False)


def _gateway_pending_approval_without_run_id(sid: str, approval_id: str) -> bool:
    with _lock:
        reconcile_gateway_pending_mirror_locked(sid)
        queue = _pending.get(sid)
        if isinstance(queue, list):
            entries = queue
        elif queue:
            entries = [queue]
        else:
            entries = []
        if approval_id:
            for entry in entries:
                if isinstance(entry, dict) and entry.get("approval_id") == approval_id:
                    return bool(entry.get(_GATEWAY_MIRROR_FLAG)) and not str(entry.get("run_id") or "").strip()
            return False
        if not entries or not isinstance(entries[0], dict):
            return False
        return bool(entries[0].get(_GATEWAY_MIRROR_FLAG)) and not str(entries[0].get("run_id") or "").strip()


def _session_has_pending_approval(sid: str) -> bool:
    """True when the session still has any live pending approval to act on.

    Used to tell a benign STALE-CARD click (the card's approval already
    resolved or its stream ended, so nothing is pending) apart from a stale
    explicit-id click made WHILE a different approval is still live (which must
    stay unresolved so it can't accidentally approve the wrong command — #527).
    Reconciles the gateway mirror first so a purged orphan is not counted.
    """
    with _lock:
        reconcile_gateway_pending_mirror_locked(sid)
        queue = _pending.get(sid)
        if isinstance(queue, list):
            if queue:
                return True
        elif queue:
            return True
        gw_queue = _gateway_queues.get(sid)
        return bool(gw_queue)


def _handle_approval_respond(handler, body):
    sid = body.get("session_id", "")
    if not sid:
        return bad(handler, "session_id is required")
    choice = body.get("choice", "deny")
    if choice not in ("once", "session", "always", "deny"):
        return bad(handler, f"Invalid choice: {choice}")
    approval_id = body.get("approval_id", "")
    enable_yolo = body.get("yolo") is True
    requested_run_id = str(body.get("run_id") or "").strip()
    requested_mirror_token = str(body.get("mirror_token") or "").strip()

    if enable_yolo:
        payload, status = _enable_session_yolo_and_release_pending(
            sid,
            choice=choice,
            approval_id=approval_id,
            run_id=requested_run_id,
            mirror_token=requested_mirror_token,
            include_choice=True,
        )
        return j(handler, payload, status=status)

    if requested_run_id or requested_mirror_token:
        if not approval_id or not requested_run_id or not requested_mirror_token:
            relay_payload, relay_status = _gateway_approval_failure(
                sid,
                choice,
                code="gateway_run_unavailable",
                error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                status=409,
                enable_yolo=False,
            )
            return j(handler, relay_payload, status=relay_status)
        exact_mirror = gateway_pending_mirror(
            sid,
            approval_id=approval_id,
            run_id=requested_run_id,
            mirror_token=requested_mirror_token,
        )
        if exact_mirror is None:
            relay_payload, relay_status = _gateway_approval_failure(
                sid,
                choice,
                code="gateway_run_unavailable",
                error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                status=409,
                enable_yolo=False,
            )
            return j(handler, relay_payload, status=relay_status)
        relay_payload, relay_status = _relay_gateway_run_approval(
            sid,
            exact_mirror,
            choice,
            enable_yolo=False,
        )
        return j(handler, relay_payload, status=relay_status)

    # Gateway relay: forward choice to the runs API when session has an active run,
    # or recover the run_id from the mirrored gateway approval entry if the
    # stream pointer has already been cleared.
    try:
        from api.gateway_chat import (
            _STREAM_RUN_IDS,
            webui_gateway_chat_enabled,
        )
        from api.config import get_config as _get_config
        s = get_session(sid)
        _candidate_run_id = None
        if s is not None:
            active_sid = getattr(s, "active_stream_id", None)
            if active_sid:
                _candidate_run_id = _STREAM_RUN_IDS.get(active_sid)
        local_match = False
        run_backed_gateway_matches = 0
        same_run_stale_without_token = False
        with _lock:
            queue = _pending.get(sid)
            entries = queue if isinstance(queue, list) else [queue] if queue else []
            if approval_id:
                local_match = any(
                    isinstance(entry, dict)
                    and entry.get("approval_id") == approval_id
                    and (
                        not entry.get(_GATEWAY_MIRROR_FLAG)
                        or not str(entry.get("run_id") or "").strip()
                    )
                    for entry in entries
                )
                run_backed_gateway_matches = sum(
                    1
                    for entry in entries
                    if isinstance(entry, dict)
                    and entry.get("approval_id") == approval_id
                    and entry.get(_GATEWAY_MIRROR_FLAG)
                    and str(entry.get("run_id") or "").strip()
                )
                gateway_queue = _gateway_queues.get(sid) or []
                live_head_data = getattr(gateway_queue[0], "data", None) or {} if gateway_queue else {}
                live_head_run_id = str(live_head_data.get("run_id") or "").strip()
                live_head_token = (
                    _gateway_mirror_entry_token(gateway_queue[0])
                    if gateway_queue and live_head_data
                    else None
                )
                live_head_approval_id = str(live_head_data.get("approval_id") or "").strip()
                if not live_head_approval_id and live_head_token and live_head_run_id:
                    live_head_approval_id = f"gwrun:{live_head_run_id}:{live_head_token}"
                stale_same_run_id = _candidate_run_id or live_head_run_id
                if (
                    stale_same_run_id
                    and live_head_run_id == stale_same_run_id
                    and live_head_approval_id
                    and live_head_approval_id != approval_id
                ):
                    same_run_stale_without_token = any(
                        isinstance(entry, dict)
                        and entry.get("approval_id") == approval_id
                        and entry.get(_GATEWAY_MIRROR_FLAG)
                        and str(entry.get("run_id") or "").strip() == stale_same_run_id
                        and not str(entry.get(_GATEWAY_MIRROR_TOKEN) or "").strip()
                        for entry in entries
                    )
            else:
                local_match = any(
                    isinstance(entry, dict)
                    and (
                        not entry.get(_GATEWAY_MIRROR_FLAG)
                        or not str(entry.get("run_id") or "").strip()
                    )
                    for entry in entries
                )
        if local_match:
            _candidate_run_id = None
        if approval_id and not local_match and same_run_stale_without_token:
            relay_payload, relay_status = _gateway_approval_failure(
                sid,
                choice,
                code="gateway_run_unavailable",
                error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                status=409,
                enable_yolo=enable_yolo,
            )
            return j(handler, relay_payload, status=relay_status)
        matched_mirror = (
            gateway_pending_mirror(sid, approval_id=approval_id, run_id=_candidate_run_id)
            if approval_id and not local_match
            else None
        )
        _run_id = matched_mirror["run_id"] if matched_mirror else None
        if not matched_mirror and approval_id:
            if local_match:
                _candidate_run_id = None
            elif run_backed_gateway_matches > 1 and not _candidate_run_id:
                relay_payload, relay_status = _gateway_approval_failure(
                    sid,
                    choice,
                    code="gateway_run_unavailable",
                    error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                    status=409,
                    enable_yolo=enable_yolo,
                )
                return j(handler, relay_payload, status=relay_status)
        if _run_id:
            if enable_yolo:
                # The visible card path must serialize the same session-wide
                # handoff as the ordinary /api/session/yolo route and the Runs
                # stream. Revalidate the exact mirror after acquiring it so a
                # later approval cannot be parked while this relay commits YOLO.
                with gateway_yolo_handoff(sid):
                    current_mirror = gateway_pending_mirror(
                        sid,
                        approval_id=approval_id,
                        run_id=_run_id,
                    )
                    if current_mirror is None:
                        relay_payload, relay_status = _gateway_approval_failure(
                            sid,
                            choice,
                            code="gateway_run_unavailable",
                            error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                            status=409,
                            enable_yolo=True,
                        )
                    else:
                        relay_payload, relay_status = _relay_gateway_run_approval(
                            sid,
                            current_mirror,
                            choice,
                            enable_yolo=True,
                        )
            else:
                relay_payload, relay_status = _relay_gateway_run_approval(
                    sid,
                    matched_mirror or {},
                    choice,
                    enable_yolo=False,
                )
            return j(handler, relay_payload, status=relay_status)
        if _candidate_run_id:
            relay_payload, relay_status = _gateway_approval_failure(
                sid,
                choice,
                code="gateway_run_unavailable",
                error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                status=409,
                enable_yolo=enable_yolo,
            )
            return j(handler, relay_payload, status=relay_status)
        # A no-run mirror is local visibility state only. Resolve it only while
        # the exact parked producer still exists; otherwise keep the card live
        # and fail closed instead of claiming success.
        if webui_gateway_chat_enabled(_get_config()):
            handled_no_run_mirror, resolved_count, _, _ = resolve_gateway_pending_local_no_run_mirror(
                sid, approval_id, choice
            )
            if handled_no_run_mirror and resolved_count == 1:
                if enable_yolo:
                    set_session_yolo_enabled(sid, True)
                return j(handler, {
                    "ok": True,
                    "choice": choice,
                    "local_retired": True,
                    **(
                        {"yolo_enabled": bool(is_session_yolo_enabled(sid))}
                        if enable_yolo
                        else {}
                    ),
                })
            if handled_no_run_mirror:
                relay_payload, relay_status = _gateway_approval_failure(
                    sid,
                    choice,
                    code="gateway_run_unavailable",
                    error=_GATEWAY_APPROVAL_RELAY_UNAVAILABLE,
                    status=409,
                    enable_yolo=enable_yolo,
                )
                return j(handler, relay_payload, status=relay_status)
    except Exception:
        pass  # fall through to local approval path

    from api.runtime_adapter import LegacyJournalRuntimeAdapter, runtime_adapter_enabled

    if runtime_adapter_enabled():
        adapter = LegacyJournalRuntimeAdapter(approval_delegate=_resolve_approval_legacy)
        ok = adapter.respond_approval(sid, approval_id, choice).accepted
    else:
        ok = _resolve_approval_legacy(sid, approval_id, choice)
    if not ok and not _session_has_pending_approval(sid):
        # The local resolution path returns False when an explicit approval_id
        # was sent but no matching pending entry exists. There are two distinct
        # causes, and only one is an error:
        #   (a) a STALE CARD — the approval the card was rendered from already
        #       resolved or its stream ended (cancel / fork / provider error /
        #       completion while pending), so the agent's gateway entry was
        #       dropped and reconcile purged the mirror. Nothing is pending for
        #       this session anymore. Before #4771 the frontend was
        #       fire-and-forget and this silently cleared the card; #4771 began
        #       surfacing the bare {ok:false} as "Approval response not
        #       accepted." with a STUCK card (reported by Jamie on .666 / b3nw;
        #       the local-backend variant of #4948).
        #   (b) a STALE EXPLICIT ID while a DIFFERENT approval IS live — that
        #       MUST stay ok:false so a stale click on resolved approval A can
        #       never resolve the unrelated live approval B (#527 guard).
        # Distinguish them: when the session has NO pending approval at all,
        # the click is benign — report it resolved so the UI clears the orphan
        # card instead of dead-ending. When something IS still pending, keep
        # the protective ok:false. `stale_cleared` lets the frontend log/branch
        # without showing an error toast.
        if enable_yolo:
            set_session_yolo_enabled(sid, True)
        return j(handler, {
            "ok": True,
            "choice": choice,
            "stale_cleared": True,
            **(
                {"yolo_enabled": bool(is_session_yolo_enabled(sid))}
                if enable_yolo
                else {}
            ),
        })
    if ok and enable_yolo:
        set_session_yolo_enabled(sid, True)
    return j(handler, {
        "ok": ok,
        "choice": choice,
        **(
            {"yolo_enabled": bool(is_session_yolo_enabled(sid))}
            if ok and enable_yolo
            else {}
        ),
    })


def _resolve_clarify_legacy(sid: str, clarify_id: str, response: str) -> bool:
    """Resolve clarify through the existing callback path without new state."""
    # When a stable clarify_id is provided, match the specific entry so stale
    # or late responses from the frontend are reliably rejected (issue #2639).
    if clarify_id:
        from api.clarify import resolve_clarify_by_id
        return resolve_clarify_by_id(sid, clarify_id, response)
    # Legacy path: resolve the oldest pending entry.  Return the REAL result
    # instead of the old unconditional True so the frontend can detect when
    # there is no pending prompt to resolve.
    resolved = resolve_clarify(sid, response, resolve_all=False)
    return bool(resolved)


def _handle_clarify_respond(handler, body):
    sid = body.get("session_id", "")
    if not sid:
        return bad(handler, "session_id is required")
    response = body.get("response")
    if response is None:
        response = body.get("answer")
    if response is None:
        response = body.get("choice")
    response = str(response or "").strip()
    if not response:
        return bad(handler, "response is required")
    clarify_id = body.get("clarify_id", "")

    from api.runtime_adapter import LegacyJournalRuntimeAdapter, runtime_adapter_enabled

    if runtime_adapter_enabled():
        adapter = LegacyJournalRuntimeAdapter(clarify_delegate=_resolve_clarify_legacy)
        ok = adapter.respond_clarify(sid, clarify_id, response).accepted
    else:
        ok = _resolve_clarify_legacy(sid, clarify_id, response)

    if not ok:
        # Both the runtime adapter and legacy paths set ok=False for
        # stale/expired/wrong-session responses.  The 409 status applies
        # uniformly regardless of which path resolved the clarify request.
        return j(handler, {
            "ok": False,
            "error": "Clarification prompt expired or not found. The agent may have already proceeded.",
            "stale": True,
        }, status=409)

    return j(handler, {"ok": True, "response": response})


class _ManualCompressionMemoryHandler:
    def __init__(self):
        self.wfile = io.BytesIO()
        self.status = None
        self.sent_headers = {}

    def send_response(self, status):
        self.status = status

    def send_header(self, key, value):
        self.sent_headers[key] = value

    def end_headers(self):
        pass

    def payload(self):
        raw = self.wfile.getvalue().decode("utf-8")
        return json.loads(raw) if raw else {}


def _manual_compression_cleanup_locked(now=None):
    now = time.time() if now is None else now
    for sid, job in list(_MANUAL_COMPRESSION_JOBS.items()):
        if job.get("status") == "running":
            continue
        updated_at = float(job.get("updated_at") or job.get("started_at") or now)
        if now - updated_at > _MANUAL_COMPRESSION_JOB_TTL_SECONDS:
            _MANUAL_COMPRESSION_JOBS.pop(sid, None)


def _manual_compression_status_payload(job):
    status = job.get("status") or "running"
    payload = {
        "ok": status not in {"error", "cancelled"},
        "status": status,
        "session_id": job.get("session_id"),
        "focus_topic": job.get("focus_topic"),
        "started_at": job.get("started_at"),
        "updated_at": job.get("updated_at"),
    }
    if status == "done":
        result = job.get("result")
        if isinstance(result, dict):
            payload.update(result)
        payload["status"] = "done"
        payload["ok"] = True
    elif status == "error":
        payload["ok"] = False
        payload["error"] = job.get("error") or "Compression failed"
        payload["error_status"] = int(job.get("error_status") or 400)
        if job.get("error_type"):
            payload["type"] = job["error_type"]
        if job.get("retryable") is not None:
            payload["retryable"] = bool(job["retryable"])
        if job.get("restart_scheduled") is not None:
            payload["restart_scheduled"] = bool(job["restart_scheduled"])
        if job.get("agent_update_state") is not None:
            payload["agent_update_state"] = job["agent_update_state"]
    elif status == "cancelled":
        payload["ok"] = False
        payload["error"] = job.get("error") or "Compression cancelled"
        payload["error_status"] = int(job.get("error_status") or 409)
    return payload


def _run_manual_compression_job(sid, body):
    memory_handler = _ManualCompressionMemoryHandler()
    try:
        try:
            session = get_session(sid)
        except KeyError:
            session = None
        if session is not None:
            from api import profiles as profiles_api

            with profiles_api.profile_env_for_background_worker(session, "manual compression", logger_override=logger):
                _handle_session_compress(memory_handler, body)
        else:
            _handle_session_compress(memory_handler, body)
        status = int(memory_handler.status or 500)
        payload = memory_handler.payload()
        with _MANUAL_COMPRESSION_JOBS_LOCK:
            job = _MANUAL_COMPRESSION_JOBS.get(sid)
            if not job:
                return
            now = time.time()
            if status >= 400 or not isinstance(payload, dict) or payload.get("error"):
                job.update(
                    {
                        "status": "error",
                        "error": str((payload or {}).get("error") or "Compression failed"),
                        "error_status": status,
                        "error_type": (payload or {}).get("type"),
                        "retryable": (payload or {}).get("retryable"),
                        "restart_scheduled": (payload or {}).get("restart_scheduled"),
                        "agent_update_state": (payload or {}).get("agent_update_state"),
                        "updated_at": now,
                    }
                )
            else:
                job.update(
                    {
                        "status": "done",
                        "result": payload,
                        "updated_at": now,
                    }
                )
    except AgentRuntimeChangedError as exc:
        logger.warning("Manual compression worker found stale Agent runtime for session %s", sid)
        stale_payload = agent_runtime_stale_payload(exc)
        with _MANUAL_COMPRESSION_JOBS_LOCK:
            job = _MANUAL_COMPRESSION_JOBS.get(sid)
            if job:
                job.update(
                    {
                        "status": "error",
                        "error": stale_payload["error"],
                        "error_status": 409,
                        "error_type": stale_payload["type"],
                        "retryable": stale_payload["retryable"],
                        "restart_scheduled": stale_payload.get("restart_scheduled"),
                        "agent_update_state": stale_payload.get("agent_update_state"),
                        "updated_at": time.time(),
                    }
                )
    except Exception as exc:
        logger.warning("Manual compression worker failed for session %s: %s", sid, exc)
        with _MANUAL_COMPRESSION_JOBS_LOCK:
            job = _MANUAL_COMPRESSION_JOBS.get(sid)
            if job:
                job.update(
                    {
                        "status": "error",
                        "error": f"Compression failed: {_sanitize_error(exc)}",
                        "error_status": 500,
                        "updated_at": time.time(),
                    }
                )


def _handle_session_compress_start(handler, body):
    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))

    sid = str(body.get("session_id") or "").strip()
    if not sid:
        return bad(handler, "session_id is required")
    try:
        s = get_session(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)
    if getattr(s, "active_stream_id", None):
        return bad(handler, "Session is still streaming; wait for the current turn to finish.", 409)

    focus_topic = str(body.get("focus_topic") or body.get("topic") or "").strip()[:500] or None
    job_body = {"session_id": sid}
    if focus_topic:
        job_body["focus_topic"] = focus_topic

    # Repeated start requests observe an existing running job and do not admit
    # new Agent work, so preserve that idempotent read even if the checkout has
    # since changed. Do not hold the job lock while running Git subprocesses.
    now = time.time()
    with _MANUAL_COMPRESSION_JOBS_LOCK:
        _manual_compression_cleanup_locked(now)
        existing = _MANUAL_COMPRESSION_JOBS.get(sid)
        if existing:
            existing_payload = _manual_compression_status_payload(existing)
            if existing_payload.get("status") == "running":
                return j(handler, existing_payload)

    # Reject a stale local Agent runtime before creating the asynchronous job.
    try:
        ensure_agent_runtime_current()
    except AgentRuntimeChangedError as exc:
        return j(
            handler,
            agent_runtime_stale_payload(exc),
            status=409,
        )

    # Another start request may have admitted a job while the runtime check ran.
    # Re-check under the lock so only one worker is created.
    now = time.time()
    with _MANUAL_COMPRESSION_JOBS_LOCK:
        _manual_compression_cleanup_locked(now)
        existing = _MANUAL_COMPRESSION_JOBS.get(sid)
        if existing:
            existing_payload = _manual_compression_status_payload(existing)
            if existing_payload.get("status") == "running":
                return j(handler, existing_payload)
            # Stage-344 Opus SHOULD-FIX (#2128): always start fresh on re-invoke.
            # The prior implementation short-circuited and returned a stale `done`
            # payload for the full 10-minute TTL window when /compress/start was
            # re-invoked, so a user closing the tab mid-compress and re-running
            # /compress on a fresh open would get the previous result back rather
            # than a new compression. Drop the entry and fall through to the
            # fresh-worker path below.
            _MANUAL_COMPRESSION_JOBS.pop(sid, None)
        job = {
            "session_id": sid,
            "focus_topic": focus_topic,
            "status": "running",
            "started_at": now,
            "updated_at": now,
        }
        _MANUAL_COMPRESSION_JOBS[sid] = job

    worker = threading.Thread(
        target=_run_manual_compression_job,
        args=(sid, job_body),
        name=f"manual-compress-{sid[:8]}",
        daemon=True,
    )
    worker.start()

    with _MANUAL_COMPRESSION_JOBS_LOCK:
        return j(handler, _manual_compression_status_payload(_MANUAL_COMPRESSION_JOBS.get(sid, job)))


def _handle_session_compress_status(handler, sid):
    sid = str(sid or "").strip()
    if not sid:
        return bad(handler, "session_id is required")
    with _MANUAL_COMPRESSION_JOBS_LOCK:
        _manual_compression_cleanup_locked()
        job = _MANUAL_COMPRESSION_JOBS.get(sid)
        if not job:
            return j(handler, {"ok": True, "status": "idle", "session_id": sid})
        payload = _manual_compression_status_payload(job)
        # Stage-344 Opus SHOULD-FIX (#2128): do not pop the job on first
        # read of a `done` payload. The session may be open in multiple
        # tabs, and the first tab's poll would otherwise leave the second
        # tab with `idle` and a "Compression job is no longer available"
        # toast. Let the 10-minute TTL handle eviction so all open tabs
        # see the same terminal payload.
        return j(handler, payload)


def _handle_session_compress(handler, body):
    def _anchor_message_key(m):
        if not isinstance(m, dict):
            return None
        role = str(m.get("role") or "")
        if not role or role == "tool":
            return None
        content = m.get("content", "")
        if isinstance(content, list):
            text = "\n".join(
                str(p.get("text") or p.get("content") or "")
                for p in content
                if isinstance(p, dict) and p.get("type") == "text"
            )
        else:
            text = str(content or "")
        norm = " ".join(text.split()).strip()[:160]
        ts = m.get("_ts") or m.get("timestamp")
        attachments = m.get("attachments")
        attach_count = len(attachments) if isinstance(attachments, list) else 0
        if not norm and not attach_count and not ts:
            return None
        return {"role": role, "ts": ts, "text": norm, "attachments": attach_count}

    def _compression_summary_from_messages(messages):
        text = None
        for m in reversed(messages or []):
            if not isinstance(m, dict):
                continue
            role = str(m.get("role") or "").lower()
            if role != "assistant":
                continue
            if not isinstance(m.get("content"), str):
                continue
            content = str(m.get("content") or "").strip()
            if not content:
                continue
            norm = re.sub(r"\s+", " ", content).strip()
            if (
                "context compaction" in norm.lower()
                or "context compression" in norm.lower()
            ):
                return norm
        return None

    def _compact_summary_text(raw_text):
        if not isinstance(raw_text, str):
            return None
        txt = raw_text.strip()
        if not txt:
            return None
        return re.sub(r"\s+", " ", txt)

    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))

    sid = str(body.get("session_id") or "").strip()
    if not sid:
        return bad(handler, "session_id is required")
    if _session_is_subagent_view_only(sid):
        return bad(handler, "Subagent sessions are view-only and cannot be compressed from WebUI", 400)

    # Cap focus_topic to 500 chars — matches the defensive input-size pattern
    # used elsewhere (session title :80, first-exchange snippets :500) and
    # prevents a user from forwarding an unbounded string into the compressor
    # prompt path. No privilege boundary here (user prompting themself), just
    # cheap bound-checking.
    focus_topic = str(body.get("focus_topic") or body.get("topic") or "").strip()[:500] or None

    try:
        s = get_session(sid)
    except KeyError:
        return bad(handler, "Session not found", 404)

    if getattr(s, "active_stream_id", None):
        return bad(handler, "Session is still streaming; wait for the current turn to finish.", 409)

    try:
        from api.streaming import _sanitize_messages_for_api

        messages = _sanitize_messages_for_api(s.messages)
        if len(messages) < 4:
            return bad(handler, "Not enough conversation to compress (need at least 4 messages).")

        def _fallback_estimate_messages_tokens_rough(msgs):
            """Fallback heuristic token estimate when runtime metadata helpers are absent.

            Uses whitespace token-like word counting only. This intentionally
            over/under-estimates BPE token counts (roughly around x3/x4 scale),
            and is only for resilient fallback behavior.
            """
            total = 0
            for m in msgs or []:
                if not isinstance(m, dict):
                    continue
                content = m.get("content", "")
                if isinstance(content, list):
                    content_text = "\n".join(
                        str(p.get("text") or p.get("content") or "")
                        for p in content
                        if isinstance(p, dict)
                    )
                else:
                    content_text = str(content or "")
                total += len(content_text.split())
            return max(1, total)

        def _fallback_summarize_manual_compression(original_messages, compressed_messages, before_tokens, after_tokens, focus_topic=None):
            """Lightweight fallback summary to keep /session/compress usable in tests/runtime."""
            after_tokens = after_tokens if after_tokens is not None else _fallback_estimate_messages_tokens_rough(compressed_messages)
            headline = f"Compressed: {len(original_messages)} \u2192 {len(compressed_messages)} messages"
            summary = {
                "headline": headline,
                "token_line": f"Rough transcript estimate: ~{before_tokens} \u2192 ~{after_tokens} tokens",
                "note": f"Focus: {focus_topic}" if focus_topic else None,
            }
            summary["reference_message"] = (
                f"[CONTEXT COMPACTION \u2014 REFERENCE ONLY] {headline}\n"
                f"{summary['token_line']}\n"
                + (summary["note"] + "\n" if summary.get("note") else "")
                + "Compression completed."
            )
            return summary

        def _estimate_messages_tokens_rough(msgs):
            try:
                from agent.model_metadata import estimate_messages_tokens_rough

                return estimate_messages_tokens_rough(msgs)
            except Exception:
                return _fallback_estimate_messages_tokens_rough(msgs)

        def _summarize_manual_compression(
            original_messages,
            compressed_messages,
            before_tokens,
            after_tokens,
            focus_topic=None,
        ):
            try:
                from agent.manual_compression_feedback import summarize_manual_compression

                return summarize_manual_compression(
                    original_messages,
                    compressed_messages,
                    before_tokens,
                    after_tokens,
                )
            except Exception:
                return _fallback_summarize_manual_compression(
                    original_messages,
                    compressed_messages,
                    before_tokens,
                    after_tokens,
                    focus_topic,
                )

        ensure_agent_runtime_current()
        import api.config as _cfg
        from api.oauth import resolve_runtime_provider_with_anthropic_env_lock
        import hermes_cli.runtime_provider as _runtime_provider
        AIAgent = require_ai_agent_class()

        resolved_model, resolved_provider, resolved_base_url = _cfg.resolve_model_provider(
            _cfg.model_with_provider_context(s.model, getattr(s, "model_provider", None))
        )

        resolved_api_key = None
        _rt = None
        try:
            _rt = resolve_runtime_provider_with_anthropic_env_lock(
                _runtime_provider.resolve_runtime_provider,
                requested=resolved_provider,
            )
            resolved_api_key = _rt.get("api_key")
            if not resolved_provider:
                resolved_provider = _rt.get("provider")
            if not resolved_base_url:
                resolved_base_url = _rt.get("base_url")
        except Exception as _e:
            logger.warning("resolve_runtime_provider failed for compression: %s", _e)

        # Atomic custom-provider authority: the deterministic list-row URL must
        # not be paired with a keyed/runtime API key (see the /api/chat note),
        # and the record's own api_mode / credential_pool / ACP transport must
        # reach the constructor with it rather than being truncated away.
        _bundle = _resolve_agent_connection_bundle(
            resolved_provider, resolved_api_key, resolved_base_url, _rt
        )
        resolved_provider = _bundle["provider"]
        resolved_api_key = _bundle["api_key"]
        resolved_base_url = _bundle["base_url"]

        if not resolved_api_key:
            return bad(handler, "No provider configured -- cannot compress.")

        # Compute compression *outside* the lock — the LLM round-trip can take
        # many seconds and we must not block cancel_stream or other writers.
        # Lock contract: hold for the in-memory mutation only, never across
        # network I/O.
        original_messages = list(messages)
        original_stream_state = (
            getattr(s, "active_stream_id", None),
            getattr(s, "pending_user_message", None),
            copy.deepcopy(getattr(s, "pending_attachments", None)),
            getattr(s, "pending_started_at", None),
        )
        approx_tokens = _estimate_messages_tokens_rough(original_messages)

        agent = AIAgent(
            model=resolved_model,
            provider=resolved_provider,
            base_url=resolved_base_url,
            api_key=resolved_api_key,
            # Identify browser-originated sessions as WebUI so Hermes Agent
            # does not inject CLI-specific terminal/output guidance.
            platform="webui",
            quiet_mode=True,
            enabled_toolsets=_resolve_cli_toolsets(),
            session_id=sid,
            **_agent_bundle_kwargs(AIAgent, _bundle),
        )
        compressed = agent.context_compressor.compress(
            original_messages,
            current_tokens=approx_tokens,
            focus_topic=focus_topic,
        )
        new_tokens = _estimate_messages_tokens_rough(compressed)
        summary = _summarize_manual_compression(
            original_messages,
            compressed,
            approx_tokens,
            new_tokens,
            focus_topic=focus_topic,
        )

        with _cfg._get_session_agent_lock(sid):
            # Re-read messages to detect concurrent edits during the LLM call.
            # If the history changed, the compression result is stale — abort.
            current_stream_state = (
                getattr(s, "active_stream_id", None),
                getattr(s, "pending_user_message", None),
                copy.deepcopy(getattr(s, "pending_attachments", None)),
                getattr(s, "pending_started_at", None),
            )
            if current_stream_state != original_stream_state:
                return bad(handler, "Session stream state changed during compression; please retry.", 409)
            if _sanitize_messages_for_api(s.messages) != original_messages:
                return bad(handler, "Session was modified during compression; please retry.", 409)

            from api.session_ops import _truncation_watermark_for
            from api.streaming import _stamp_missing_message_timestamps

            compressed_copy = copy.deepcopy(compressed)
            _stamp_missing_message_timestamps(compressed_copy)
            s.context_messages = compressed_copy
            s.active_stream_id = None
            s.pending_user_message = None
            s.pending_attachments = []
            s.pending_started_at = None
            s.pending_user_source = None
            visible_after = visible_messages_for_anchor(s.messages, auto_compression=False)
            s.compression_anchor_visible_idx = max(0, len(visible_after) - 1) if visible_after else None
            s.compression_anchor_message_key = _anchor_message_key(visible_after[-1]) if visible_after else None
            summary_text = None
            if isinstance(summary, dict):
                summary_text = summary.get("reference_message") or summary.get("token_line") or summary.get("headline")
            s.compression_anchor_summary = _compact_summary_text(
                summary_text or _compression_summary_from_messages(compressed) or ""
            )
            # Persist an intentional-shrink boundary so append-only state.db
            # reconciliation does not replay pre-compression rows (#4836).
            compress_watermark = _truncation_watermark_for(compressed_copy)
            s.truncation_watermark = compress_watermark
            s.truncation_boundary = compress_watermark
            s.compression_anchor_mode = "manual"
            s.last_prompt_tokens = new_tokens
            s.save()
            # Drop stale backups that would undo an intentional manual compress.
            try:
                s.path.with_suffix(".json.bak").unlink(missing_ok=True)
            except OSError:
                pass

        session_payload = redact_session_data(
            s.compact() | {
                "messages": s.messages,
                "tool_calls": s.tool_calls,
                "active_stream_id": s.active_stream_id,
                "pending_user_message": s.pending_user_message,
                "pending_attachments": s.pending_attachments,
                "pending_started_at": s.pending_started_at,
                "compression_anchor_visible_idx": getattr(s, "compression_anchor_visible_idx", None),
                "compression_anchor_message_key": getattr(s, "compression_anchor_message_key", None),
            }
        )
        return j(
            handler,
            {
                "ok": True,
                "session": session_payload,
                "summary": summary,
                "focus_topic": focus_topic,
            },
        )
    except AgentRuntimeChangedError as e:
        return j(handler, agent_runtime_stale_payload(e), status=409)
    except Exception as e:
        logger.warning("Manual session compression failed: %s", e)
        return bad(handler, f"Compression failed: {_sanitize_error(e)}")


def _handle_conversation_rounds(handler, body):
    """Return conversation-round count for a gateway session.

    Request body::

        { "session_id": "...", "since": <unix_ts_or_iso> }

    Response::

        { "ok": true, "rounds": 12, "threshold": 10, "should_show": true }
    """
    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))

    sid = str(body.get("session_id") or "").strip()
    if not sid:
        return bad(handler, "session_id is required")

    since = body.get("since")
    if since is not None:
        try:
            since = float(since)
        except (TypeError, ValueError):
            return bad(handler, "since must be a unix timestamp (number)")

    from api.models import count_conversation_rounds, CONVERSATION_ROUND_THRESHOLD

    rounds = count_conversation_rounds(sid, since=since)
    return j(handler, {
        "ok": True,
        "rounds": rounds,
        "threshold": CONVERSATION_ROUND_THRESHOLD,
        "should_show": rounds >= CONVERSATION_ROUND_THRESHOLD,
    })


def _build_handoff_summary_tool_message(
    sid: str,
    summary: str,
    channel: str | None,
    rounds: int | None = None,
    fallback: bool = False,
) -> dict:
    """Build a compact tool-role transcript marker for persistence."""
    now = time.time()
    return {
        "role": "tool",
        # Keep this intentionally empty so API-history sanitization drops it from
        # model context (it is display-only data).
        "tool_call_id": "",
        "name": "handoff_summary",
        "timestamp": now,
        "_ts": now,
        "content": json.dumps({
            "_handoff_summary_card": True,
            "session_id": sid,
            "summary": str(summary or "").strip(),
            "channel": (str(channel or "").strip() or None),
            "rounds": rounds,
            "fallback": bool(fallback),
            "generated_at": now,
        }, ensure_ascii=False),
    }


def _extract_handoff_summary_payload(message: dict) -> dict | None:
    """Return a normalized handoff-summary payload if *message* is a tool marker."""
    if not isinstance(message, dict):
        return None
    if message.get("role") != "tool" or message.get("name") != "handoff_summary":
        return None

    content = message.get("content")
    if isinstance(content, dict):
        payload = content
    else:
        try:
            payload = json.loads(content or "")
        except Exception:
            return None

    if not isinstance(payload, dict) or not payload.get("_handoff_summary_card"):
        return None
    if payload.get("session_id") is None:
        return None
    return {
        "session_id": str(payload.get("session_id")),
        "summary": str(payload.get("summary", "")),
        "channel": payload.get("channel"),
        "rounds": payload.get("rounds"),
        "fallback": bool(payload.get("fallback")),
        "_handoff_summary_card": True,
    }


def _is_matching_handoff_summary_message(existing: dict, target: dict) -> bool:
    """Return True when two message payloads represent the same handoff summary."""
    existing_payload = _extract_handoff_summary_payload(existing)
    target_payload = _extract_handoff_summary_payload(target)
    if not existing_payload or not target_payload:
        return False
    return (
        existing_payload.get("session_id") == target_payload.get("session_id") and
        existing_payload.get("summary") == target_payload.get("summary") and
        existing_payload.get("channel") == target_payload.get("channel") and
        existing_payload.get("rounds") == target_payload.get("rounds") and
        existing_payload.get("fallback") == target_payload.get("fallback") and
        existing_payload.get("_handoff_summary_card") == target_payload.get("_handoff_summary_card")
    )


def _is_matching_handoff_summary_content(content: object, target_payload: dict | None) -> bool:
    """Return True if DB content JSON matches an expected handoff summary payload."""
    if target_payload is None:
        return False
    try:
        payload = json.loads(content or "")
    except Exception:
        return False
    if not isinstance(payload, dict):
        return False
    if payload.get("session_id") is None:
        return False
    return (
        payload.get("_handoff_summary_card") is True and
        str(payload.get("session_id")) == str(target_payload.get("session_id")) and
        str(payload.get("summary", "")) == str(target_payload.get("summary", "")) and
        payload.get("channel") == target_payload.get("channel") and
        payload.get("rounds") == target_payload.get("rounds") and
        bool(payload.get("fallback")) == bool(target_payload.get("fallback"))
    )


def _persist_handoff_summary_locally(sid: str, message: dict) -> bool:
    """Persist a handoff summary marker into a local WebUI session file."""
    try:
        from api.models import get_session

        s = get_session(sid)
    except KeyError:
        return False

    try:
        if s.messages and _is_matching_handoff_summary_message(s.messages[-1], message):
            return True
        s.messages.append(message)
        s.save()
        return True
    except Exception as e:
        logger.warning("Failed to persist handoff summary marker in local session %s: %s", sid, e)
        return False


def _persist_handoff_summary_to_state_db(sid: str, message: dict) -> bool:
    """Persist a handoff summary marker into CLI sessions state.db.

    This keeps summary cards available after hard-refresh for imported gateway
    sessions that are not in local session JSON yet.
    """
    import os

    try:
        import sqlite3
    except ImportError:
        return False

    try:
        from api.profiles import get_active_hermes_home

        hermes_home = Path(get_active_hermes_home()).expanduser().resolve()
    except Exception:
        hermes_home = Path(os.getenv("HERMES_HOME", str(Path.home() / ".hermes"))).expanduser().resolve()

    db_path = hermes_home / "state.db"
    if not db_path.exists():
        return False

    ts = message.get("timestamp", time.time())
    content = message.get("content", "")
    if not isinstance(content, str):
        content = json.dumps(content, ensure_ascii=False)

    marker_payload = _extract_handoff_summary_payload(message)
    try:
        with closing(sqlite3.connect(str(db_path))) as conn:
            try:
                if marker_payload is not None:
                    cur = conn.execute(
                        "SELECT content FROM messages WHERE session_id = ? AND role = 'tool' "
                        "ORDER BY rowid DESC LIMIT 1",
                        (sid,),
                    )
                    row = cur.fetchone()
                    if row is not None and _is_matching_handoff_summary_content(row[0], marker_payload):
                        return True
            except Exception:
                # If tail-read fails, continue with a best-effort write.
                logger.debug("Unable to read tail handoff marker from state.db for %s", sid)

            conn.execute(
                "INSERT INTO messages (session_id, role, content, timestamp) "
                "VALUES (?, 'tool', ?, ?)",
                (sid, content, ts),
            )
            # Keep session row message_count/last-activity aligned with displayed
            # transcript length. session rows are optional in some test DBs, so
            # this update is best-effort.
            conn.execute(
                "UPDATE sessions SET message_count = COALESCE(message_count, 0) + 1 "
                "WHERE id = ?",
                (sid,),
            )
            conn.commit()
        return True
    except Exception as e:
        logger.warning("Failed to persist handoff summary marker in state.db for %s: %s", sid, e)
        return False


def _persist_handoff_summary(sid: str, summary: str, channel: str | None, rounds: int | None, fallback: bool = False) -> dict:
    """Persist a handoff summary marker across local/session backends."""
    marker = _build_handoff_summary_tool_message(sid, summary, channel, rounds, fallback)
    is_messaging_session = _is_messaging_session_id(sid)
    if is_messaging_session:
        _persist_handoff_summary_to_state_db(sid, marker)
        _persist_handoff_summary_locally(sid, marker)
        return marker
    persisted_local = _persist_handoff_summary_locally(sid, marker)
    if persisted_local:
        return marker
    return marker if _persist_handoff_summary_to_state_db(sid, marker) else marker


def _handle_handoff_summary(handler, body):
    """Generate an on-demand handoff summary for a gateway session.

    Request body::

        { "session_id": "...", "since": <unix_ts_or_iso> }

    Uses the session's configured model to produce a concise summary of
    recent conversation activity.  Returns the summary text so the caller
    can display it in a tool-card.
    """
    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))

    sid = str(body.get("session_id") or "").strip()
    if not sid:
        return bad(handler, "session_id is required")

    since = body.get("since")
    if since is not None:
        try:
            since = float(since)
        except (TypeError, ValueError):
            return bad(handler, "since must be a unix timestamp (number)")

    from api.models import get_cli_session_messages, count_conversation_rounds, CONVERSATION_ROUND_THRESHOLD

    if _session_is_subagent_view_only(sid):
        return bad(handler, "Subagent sessions are view-only and cannot be summarized from WebUI", 400)
    rounds = count_conversation_rounds(sid, since=since)
    if rounds < CONVERSATION_ROUND_THRESHOLD:
        return bad(handler, "Not enough conversation rounds to generate a summary.", 400)

    # Filter messages by ``since``.
    all_msgs = get_cli_session_messages(sid)
    if since is not None:
        import datetime as _dt
        filtered = []
        for m in all_msgs:
            ts_raw = m.get("timestamp")
            if ts_raw is None:
                continue
            try:
                if isinstance(ts_raw, (int, float)):
                    ts_val = float(ts_raw)
                else:
                    ts_val = _dt.datetime.fromisoformat(
                        str(ts_raw).replace("Z", "+00:00")
                    ).timestamp()
                if ts_val > since:
                    filtered.append(m)
            except Exception:
                pass
        msgs = filtered
    else:
        msgs = all_msgs

    # Cap to last 50 messages.
    msgs = msgs[-50:]

    if len(msgs) < 2:
        return bad(handler, "Not enough messages to summarize.", 400)

    def _extract_handoff_text(raw_content):
        if isinstance(raw_content, list):
            return " ".join(
                str(p.get("text") or p.get("content") or "")
                for p in raw_content
                if isinstance(p, dict)
            ).strip()
        return str(raw_content or "").strip()

    def _contains_chinese(text):
        return any("\u4e00" <= ch <= "\u9fff" for ch in str(text))

    transcript_is_chinese = any(
        _contains_chinese(_extract_handoff_text(m.get("content")))
        for m in msgs
    )
    # Build a lightweight conversation transcript for the LLM.
    lines = []
    for m in msgs:
        role = m.get("role", "")
        content = _extract_handoff_text(m.get("content"))
        content = str(content or "").strip()[:1000]
        if role in ("user", "assistant") and content:
            lines.append(content)
    transcript = "\n".join(lines)

    def _fallback_handoff_summary(items):
        """Return a deterministic summary when LLM summary generation is unavailable."""
        user_points = []
        assistant_points = []

        def _summarize_snippet(raw_text, max_len=78):
            text = " ".join(str(raw_text or "").split()).strip()
            if not text:
                return ""
            if len(text) <= max_len:
                return text
            return text[: max_len - 1].rstrip() + "…"

        for m in items:
            role = m.get("role", "")
            content = _summarize_snippet(_extract_handoff_text(m.get("content")), 82)
            if role in ("user", "assistant") and content:
                if role == "user":
                    user_points.append(content)
                else:
                    assistant_points.append(content)
        if not user_points and not assistant_points:
            return (
                "近期可读文本不足，无法生成更完整的交接摘要，请补充一条消息后重试。"
                if transcript_is_chinese
                else "Not enough readable text to create a useful handoff summary; please send one more message and retry."
            )

        if transcript_is_chinese:
            bullets = []
            if user_points:
                bullets.append(f"- 你刚讨论了：{user_points[-1]}。")
            if assistant_points:
                bullets.append(f"- 助手已回复：{assistant_points[-1]}。")
            if len(user_points) + len(assistant_points) >= 2:
                bullets.append("- 当前对话存在尚未确认的后续动作。")
            else:
                bullets.append("- 当前信息偏少，建议补充关键点后再切换。")
            return "\n".join(bullets)

        bullets = []
        if user_points:
            bullets.append(f"- You asked: {user_points[-1]}.")
        if assistant_points:
            bullets.append(f"- The assistant responded: {assistant_points[-1]}.")
        if len(user_points) + len(assistant_points) >= 2:
            bullets.append("- There is pending context to continue next.")
        else:
            bullets.append("- The conversation is still short; add one more turn before summarizing.")
        return "\n".join(bullets)

    def _summary_output_incomplete(text):
        """Best-effort guard for truncated summaries when LLM signals are unavailable."""
        if not isinstance(text, str):
            text = str(text or "")
        text = text.strip()
        if not text:
            return True
        if text.endswith("...") or text.endswith("…"):
            return True
        lines = [line.strip() for line in text.splitlines() if line.strip()]
        if not lines:
            return True
        last_line = lines[-1]
        if re.search(r"[。！？；!?.；]$", last_line):
            return False
        if len(last_line) >= 56 and not re.search(r"\b(and|or|so|then|because|if|when|but|so|as)\b$", last_line, re.IGNORECASE):
            return True
        return bool(re.search(r"\b(and|or|but|so|because|if|when)$", last_line, re.IGNORECASE))

    def _agent_summary_incomplete(summary_result):
        if not isinstance(summary_result, dict):
            return True
        reason = (summary_result.get("finish_reason") or "").strip().lower()
        if reason == "length":
            return True
        stop_reason = (summary_result.get("stop_reason") or "").strip().lower()
        if stop_reason in {"max_tokens", "length"}:
            return True
        return _summary_output_incomplete(summary_result.get("text", ""))

    def _resolve_handoff_channel_label():
        channel_label = None
        try:
            from api.models import get_session as _get_session, get_cli_sessions

            session_meta = _get_session(sid)
            channel_label = (
                session_meta.source_label
                or session_meta.raw_source
                or session_meta.source_tag
                or session_meta.session_source
            )
            if not channel_label:
                for candidate in get_cli_sessions():
                    if candidate.get("session_id") == sid:
                        channel_label = (
                            candidate.get("source_label")
                            or candidate.get("raw_source")
                            or candidate.get("source_tag")
                            or candidate.get("source")
                        )
                        break
        except Exception:
            pass
        return channel_label

    def _agent_text_completion(agent, system_prompt, user_text, max_tokens=700):
        """Use the current Hermes Agent transport without mutating conversation history."""
        api_messages = [
            {"role": "system", "content": system_prompt},
            {"role": "user", "content": user_text},
        ]
        result = {
            "text": "",
            "finish_reason": None,
            "stop_reason": None,
            "incomplete": True,
        }
        disabled_reasoning = {"enabled": False}
        previous_reasoning = getattr(agent, "reasoning_config", None)
        try:
            agent.reasoning_config = disabled_reasoning
            if getattr(agent, "api_mode", "") == "codex_responses":
                codex_kwargs = agent._build_api_kwargs(api_messages)
                codex_kwargs.pop("tools", None)
                codex_provider = str(getattr(agent, "provider", "") or "").strip().lower()
                codex_base_url = str(getattr(agent, "base_url", "") or "").strip().lower()
                is_chatgpt_codex = (
                    codex_provider == "openai-codex"
                    or (
                        urlsplit(codex_base_url).hostname == "chatgpt.com"
                        and "/backend-api/codex" in codex_base_url
                    )
                )
                if not is_chatgpt_codex:
                    codex_kwargs["max_output_tokens"] = max_tokens
                resp = agent._run_codex_stream(codex_kwargs)
                normalized = agent._get_transport("codex_responses").normalize_response(resp)
                result["text"] = str((normalized.content or "") if normalized else "").strip()
                result["incomplete"] = _summary_output_incomplete(result["text"])
                return result

            if getattr(agent, "api_mode", "") == "anthropic_messages":
                from agent.anthropic_adapter import build_anthropic_kwargs

                ant_kwargs = build_anthropic_kwargs(
                    model=agent.model,
                    messages=api_messages,
                    tools=None,
                    max_tokens=max_tokens,
                    reasoning_config=disabled_reasoning,
                    is_oauth=getattr(agent, "_is_anthropic_oauth", False),
                    preserve_dots=agent._anthropic_preserve_dots(),
                    base_url=getattr(agent, "_anthropic_base_url", None),
                )
                resp = agent._anthropic_messages_create(ant_kwargs)
                normalized = agent._get_transport().normalize_response(
                    resp,
                    strip_tool_prefix=getattr(agent, "_is_anthropic_oauth", False),
                )
                result["text"] = str((normalized.content or "") if normalized else "").strip()
                result["incomplete"] = _summary_output_incomplete(result["text"])
                return result

            api_kwargs = agent._build_api_kwargs(api_messages)
            api_kwargs.pop("tools", None)
            api_kwargs["temperature"] = 0.2
            api_kwargs["timeout"] = 30.0
            if "max_completion_tokens" in api_kwargs:
                api_kwargs["max_completion_tokens"] = max_tokens
            else:
                api_kwargs["max_tokens"] = max_tokens
            resp = agent._ensure_primary_openai_client(reason="handoff_summary").chat.completions.create(
                **api_kwargs,
            )
            choice = (getattr(resp, "choices", None) or [None])[0]
            msg = getattr(choice, "message", None) if choice is not None else None
            result["text"] = str(getattr(msg, "content", "") or "").strip()
            result["finish_reason"] = getattr(choice, "finish_reason", None)
            result["stop_reason"] = getattr(choice, "stop_reason", None)
            result["incomplete"] = _agent_summary_incomplete(result)
            return result
        finally:
            agent.reasoning_config = previous_reasoning

        # Call LLM for summary.
    try:
        ensure_agent_runtime_current()
        import api.config as _cfg
        from api.oauth import resolve_runtime_provider_with_anthropic_env_lock
        import hermes_cli.runtime_provider as _runtime_provider
        AIAgent = require_ai_agent_class()

        # Try to resolve model from an existing session, fall back to default.
        resolved_model = None
        resolved_provider = None
        resolved_base_url = None
        session_model_provider = None
        try:
            from api.models import get_session
            s_obj = get_session(sid)
            resolved_model = getattr(s_obj, "model", None)
            # Carry the session's OWN selected provider into resolution. Without
            # it, a bare resolve_model_provider(model) routes the summary through
            # whatever main provider is active — so a session pinned to custom:A
            # gets its handoff summary rerouted to the active custom:B when both
            # providers list the same model id (overlapping-id misroute, sibling
            # of the resolve_model_provider fix). model_with_provider_context
            # encodes it as @custom:A:model so the resolver honors the session's
            # endpoint; base_url is backfilled from that provider's own custom
            # entry by the custom-provider authority block below.
            session_model_provider = getattr(s_obj, "model_provider", None)
        except Exception:
            pass

        model_for_resolution = _cfg.model_with_provider_context(
            resolved_model, session_model_provider
        )
        resolved_model, resolved_provider, resolved_base_url = _cfg.resolve_model_provider(model_for_resolution)

        resolved_api_key = None
        _rt = None
        try:
            _rt = resolve_runtime_provider_with_anthropic_env_lock(
                _runtime_provider.resolve_runtime_provider,
                requested=resolved_provider,
            )
            resolved_api_key = _rt.get("api_key")
            if not resolved_provider:
                resolved_provider = _rt.get("provider")
            if not resolved_base_url:
                resolved_base_url = _rt.get("base_url")
        except Exception as _e:
            logger.warning("resolve_runtime_provider failed for handoff summary: %s", _e)

        # Atomic custom-provider authority (see the /api/chat note): the session's
        # own custom_providers[] row owns BOTH the endpoint and the credential,
        # plus the api_mode / credential_pool / ACP transport that travel with
        # them as one constructor bundle.
        _bundle = _resolve_agent_connection_bundle(
            resolved_provider, resolved_api_key, resolved_base_url, _rt
        )
        resolved_provider = _bundle["provider"]
        resolved_api_key = _bundle["api_key"]
        resolved_base_url = _bundle["base_url"]

        if not resolved_api_key:
            summary_text = _fallback_handoff_summary(msgs)
            try:
                _persist_handoff_summary(
                    sid,
                    summary_text,
                    _resolve_handoff_channel_label(),
                    rounds,
                    fallback=True,
                )
            except Exception:
                pass
            return j(handler, {
                "ok": True,
                "summary": summary_text,
                "message_count": len(msgs),
                "rounds": rounds,
                "fallback": True,
            })

        agent = AIAgent(
            model=resolved_model,
            provider=resolved_provider,
            base_url=resolved_base_url,
            api_key=resolved_api_key,
            platform="webui",
            quiet_mode=True,
            enabled_toolsets=[],
            session_id=sid,
            **_agent_bundle_kwargs(AIAgent, _bundle),
        )

        summary_system_prompt = (
            "You are summarizing an external-channel conversation so a Web UI reader "
            "can quickly catch up after switching contexts.\n\n"
            "Only use the latest messages, and never copy raw transcript lines.\n"
            "Do not output role labels (no “你:” / “assistant:” / “user:” / “assistant”).\n"
            "Use direct 2–5 bullet points in the conversation language.\n"
            "English: speak using “you”.\n"
            "中文: 使用“你”。\n\n"
            "Focus on:\n"
            "- Unfinished tasks or action items\n"
            "- Pending questions that need replies\n"
            "- Key decisions made\n"
            "- Open disagreements or TBD items\n\n"
            "If the conversation is purely casual with no actionable items, "
            "say so in one sentence."
        )
        summary_user_text = f"Conversation transcript:\n{transcript}"

        try:
            first_pass = _agent_text_completion(
                agent,
                summary_system_prompt,
                summary_user_text,
                max_tokens=700,
            )
            summary_text = first_pass.get("text") if isinstance(first_pass, dict) else ""
            if _agent_summary_incomplete(first_pass):
                second_pass = _agent_text_completion(
                    agent,
                    summary_system_prompt,
                    summary_user_text,
                    max_tokens=1400,
                )
                summary_text = second_pass.get("text") if isinstance(second_pass, dict) else ""
                if _agent_summary_incomplete(second_pass):
                    summary_text = _fallback_handoff_summary(msgs)
                    fallback = True
                else:
                    fallback = False
            else:
                fallback = False
        finally:
            try:
                agent.release_clients()
            except Exception:
                pass
        if not summary_text:
            summary_text = _fallback_handoff_summary(msgs)
            fallback = True
        elif _summary_output_incomplete(summary_text):
            if not fallback:
                fallback = True

        channel_label = _resolve_handoff_channel_label()
        _persist_handoff_summary(
            sid,
            summary_text,
            channel_label,
            rounds,
            fallback=fallback,
        )

        return j(handler, {
            "ok": True,
            "summary": summary_text,
            "message_count": len(msgs),
            "rounds": rounds,
            "fallback": fallback,
        })
    except AgentRuntimeChangedError as e:
        return j(handler, agent_runtime_stale_payload(e), status=409)
    except api_config.AmbiguousCustomProviderError as e:
        # A custom-provider slug collision is a user-fixable misconfiguration,
        # not a transient summary failure. Return 400 with the actionable rename
        # message so the UI shows it, instead of degrading to a 200 local
        # fallback that the client treats as success and that hides the fix.
        logger.warning("Handoff summary blocked by ambiguous custom provider: %s", e.message)
        return j(handler, {
            "error": e.message,
            "type": "custom_provider_ambiguous",
        }, status=400)
    except Exception as e:
        logger.warning("Handoff summary generation failed: %s", e)
        summary_text = _fallback_handoff_summary(msgs)
        try:
            _persist_handoff_summary(
                sid,
                summary_text,
                _resolve_handoff_channel_label(),
                rounds,
                fallback=True,
            )
        except Exception:
            pass
        return j(handler, {
            "ok": True,
            "summary": summary_text,
            "message_count": len(msgs),
            "rounds": rounds,
            "fallback": True,
            "warning": f"Summary generation used local fallback: {_sanitize_error(e)}",
        })


def _handle_skill_save(handler, body):
    try:
        require(body, "name", "content")
    except ValueError as e:
        return bad(handler, str(e))
    skill_name = body["name"].strip().lower().replace(" ", "-")
    if not skill_name or "/" in skill_name or ".." in skill_name:
        return bad(handler, "Invalid skill name")
    category = body.get("category", "").strip()
    if category and ("/" in category or ".." in category):
        return bad(handler, "Invalid category")
    skills_dir = _active_skills_dir()

    if category:
        skill_dir = skills_dir / category / skill_name
    else:
        skill_dir = skills_dir / skill_name
    # Validate resolved path stays within the active profile skills dir.
    try:
        skill_dir.resolve().relative_to(skills_dir.resolve())
    except ValueError:
        return bad(handler, "Invalid skill path")
    skill_dir.mkdir(parents=True, exist_ok=True)
    skill_file = skill_dir / "SKILL.md"
    if skill_file.is_symlink():
        return bad(handler, "Cannot save to a symlinked skill file")
    skill_file.write_text(body["content"], encoding="utf-8")
    _SKILLS_STATS_CACHE.clear()
    return j(handler, {"ok": True, "name": skill_name, "path": str(skill_file)})


def _handle_skill_delete(handler, body):
    try:
        require(body, "name")
    except ValueError as e:
        return bad(handler, str(e))
    import shutil

    skill_name = str(body["name"]).strip().lower().replace(" ", "-")
    if not skill_name or "/" in skill_name or ".." in skill_name:
        return bad(handler, "Invalid skill name")
    skills_dir = _active_skills_dir()
    matches = [p for p in skills_dir.rglob("SKILL.md") if p.parent.name == skill_name]
    if not matches:
        return bad(handler, "Skill not found", 404)
    skill_dir = matches[0].parent
    shutil.rmtree(str(skill_dir))
    _SKILLS_STATS_CACHE.clear()
    return j(handler, {"ok": True, "name": body["name"]})


def _normalize_names_list(names) -> list[str]:
    """Normalize a config value (None/str/list) into a deduplicated str list."""
    if names is None:
        return []
    if isinstance(names, str):
        names = _parse_config_string_list(names)
    elif not isinstance(names, list):
        names = list(names) if names else []
    return list(dict.fromkeys(str(d).strip() for d in names if str(d).strip()))


def _toggle_name_in_list(names, name: str, enabled: bool) -> list[str]:
    """Add or remove *name* from *names*, returning a new list."""
    names = _normalize_names_list(names)
    if enabled:
        return [d for d in names if d != name]
    if name not in names:
        names.append(name)
    return names


def _handle_skill_toggle(handler, body):
    """Toggle a skill's enabled/disabled state in the active profile's config.yaml.

    Writes through to ``skills.platform_disabled.webui`` when that key exists
    so the toggle takes effect for WebUI sessions (the agent's
    ``get_disabled_skill_names`` checks platform-specific lists first when
    ``HERMES_SESSION_PLATFORM`` is set).
    """
    try:
        require(body, "name", "enabled")
    except ValueError as e:
        return bad(handler, str(e))

    name = body["name"].strip()
    enabled = bool(body["enabled"])

    # Validate the skill exists in the filesystem
    skills_dir = _active_skills_dir()
    search_dirs = _active_skill_search_dirs(skills_dir)
    skill_dir, skill_md = _find_skill_in_dirs(name, search_dirs)
    if not skill_md:
        return bad(handler, f"Skill '{name}' not found", 404)

    config_path = _active_profile_config_path()
    with _cfg_lock:
        cfg = _load_yaml_config_file(config_path)

        # Ensure skills section exists as a dict
        if "skills" not in cfg or not isinstance(cfg["skills"], dict):
            cfg["skills"] = {}
        skills_cfg = cfg["skills"]

        # Always update the global disabled list
        skills_cfg["disabled"] = _toggle_name_in_list(
            skills_cfg.get("disabled"), name, enabled
        )

        # Write-through to platform_disabled.webui if it exists so that the
        # toggle takes effect for WebUI sessions (the agent checks the
        # platform-specific list first when HERMES_SESSION_PLATFORM=webui).
        platform_disabled = skills_cfg.get("platform_disabled")
        if isinstance(platform_disabled, dict) and "webui" in platform_disabled:
            platform_disabled["webui"] = _toggle_name_in_list(
                platform_disabled["webui"], name, enabled
            )

        cfg["skills"] = skills_cfg
        _save_yaml_config_file(config_path, cfg)

    reload_config()  # outside with block — reload_config() acquires the lock itself
    _SKILLS_STATS_CACHE.clear()
    return j(handler, {"ok": True, "name": name, "enabled": enabled})


def _handle_memory_write(handler, body):
    try:
        require(body, "section", "content")
    except ValueError as e:
        return bad(handler, str(e))
    section = body["section"]

    # Respect memory_enabled and user_profile_enabled config flags (#6406)
    # Use get_config_snapshot() for per-profile isolation — get_config() returns
    # the process-global mutable _cfg_cache which races across profiles.
    # The flags are nested under cfg["memory"] in Hermes Agent's schema.
    cfg = get_config_snapshot()
    mem = cfg.get("memory") if isinstance(cfg, dict) else None
    mem_cfg = mem if isinstance(mem, dict) else {}
    if section == "memory":
        if not _webui_truthy(mem_cfg.get("memory_enabled", True)):
            return bad(handler, "Memory is disabled by configuration (memory_enabled: false)", 403)
    elif section == "user":
        if not _webui_truthy(mem_cfg.get("user_profile_enabled", True)):
            return bad(handler, "User profile is disabled by configuration (user_profile_enabled: false)", 403)

    try:
        from api.profiles import get_active_hermes_home

        home = get_active_hermes_home()
        mem_dir = home / "memories"
    except ImportError:
        home = Path.home() / ".hermes"
        mem_dir = home / "memories"
    mem_dir.mkdir(parents=True, exist_ok=True)
    if section == "memory":
        target = mem_dir / "MEMORY.md"
    elif section == "user":
        target = mem_dir / "USER.md"
    elif section == "soul":
        target = home / "SOUL.md"
    else:
        return bad(handler, 'section must be "memory", "user", or "soul"')
    # Refuse to write through a symlinked target file: a symlink planted at the
    # memory path (e.g. via a restored/imported workspace) would otherwise let a
    # memory write clobber an arbitrary file outside the memories directory. This
    # mirrors the symlink-rejection hardening already shipped for skills/plugins
    # (#4217/#4234/#4240).
    if target.is_symlink():
        return bad(handler, "Cannot write to a symlinked memory file")
    try:
        target.write_text(body["content"], encoding="utf-8")
    except OSError as exc:
        if not isinstance(exc, PermissionError) and getattr(exc, "errno", None) != errno.EROFS:
            raise
        mode_hint = ""
        try:
            mode_hint = f" (mode {target.stat().st_mode & 0o777:o})"
        except OSError:
            pass
        return bad(
            handler,
            (
                f"{target.name} is not writable{mode_hint}: {target}. "
                "Run chmod 644 on the file or fix ownership on the shared volume."
            ),
            403,
        )
    return j(handler, {"ok": True, "section": section, "path": str(target)})


def _normalize_message_for_import_refresh(message: object) -> object:
    """Normalize message payloads for import refresh prefix checks.

    The strict dict comparison previously failed when existing messages held
    integer timestamps while refreshed messages held floating-point timestamps.
    Strip timing keys before comparison so we can safely treat semantic
    prefixes as equivalent.
    """
    if not isinstance(message, dict):
        return message
    normalized = dict(message)
    normalized.pop("timestamp", None)
    normalized.pop("_ts", None)
    # These are WebUI/Agent replay bookkeeping aliases at the message's top
    # level.  Strip only those exact keys; nested business payloads are opaque
    # and must remain part of the semantic import comparison.
    for key in ("api_content", "_state_db_row_id", "_db_row_id", "state_db_row_id"):
        normalized.pop(key, None)
    return normalized


def _message_has_cli_tool_metadata(message: object) -> bool:
    if not isinstance(message, dict):
        return False
    if message.get("role") == "assistant" and message.get("tool_calls"):
        return True
    if message.get("role") == "tool" and (message.get("tool_call_id") or message.get("tool_name") or message.get("name")):
        return True
    return False


def _strip_cli_tool_metadata_for_refresh(message: object) -> object:
    if not isinstance(message, dict):
        return _normalize_message_for_import_refresh(message)
    normalized = _normalize_message_for_import_refresh(message)
    if not isinstance(normalized, dict):
        return normalized
    for key in ("tool_calls", "tool_call_id", "tool_name", "name"):
        normalized.pop(key, None)
    return normalized


def _is_cli_tool_metadata_enrichment(existing_messages: list, fresh_messages: list) -> bool:
    """Return True when fresh messages only add CLI tool metadata.

    Older imports from get_cli_session_messages() persisted assistant/tool rows
    without tool_calls, tool_call_id, or tool_name. After #1772 the refreshed
    transcript can have the same length but richer metadata, so re-imports must
    rebuild the stored sidecar even without a new row.
    """
    if not isinstance(existing_messages, list) or not isinstance(fresh_messages, list):
        return False
    if len(existing_messages) != len(fresh_messages):
        return False
    if any(_message_has_cli_tool_metadata(m) for m in existing_messages):
        return False
    if not any(_message_has_cli_tool_metadata(m) for m in fresh_messages):
        return False
    for idx, existing_message in enumerate(existing_messages):
        if _strip_cli_tool_metadata_for_refresh(existing_message) != _strip_cli_tool_metadata_for_refresh(fresh_messages[idx]):
            return False
    return True


def _is_messages_refresh_prefix_match(existing_messages: list, fresh_messages: list) -> bool:
    """Return True when existing_messages is a prefix of fresh_messages by value.

    This is a semantic comparison intended for import refresh, not deep
    structural equality. It intentionally ignores timing fields that may differ
    in type/precision between storage layers.
    """
    if not isinstance(existing_messages, list) or not isinstance(fresh_messages, list):
        return False
    if len(existing_messages) > len(fresh_messages):
        return False
    for idx, existing_message in enumerate(existing_messages):
        fresh_message = fresh_messages[idx]
        if _normalize_message_for_import_refresh(existing_message) != _normalize_message_for_import_refresh(fresh_message):
            return False
    return True


def _handle_session_import_cli(handler, body):
    """Import a single CLI session into the WebUI store."""
    try:
        require(body, "session_id")
    except ValueError as e:
        return bad(handler, str(e))

    sid = str(body["session_id"])
    requested_profile = _normalize_import_profile_value((body or {}).get("profile"))
    if requested_profile == "":
        return bad(handler, "invalid profile", 400)
    allow_all_profiles = _request_wants_all_profiles_import(body)
    if allow_all_profiles and _is_isolated_profile_mode():
        return bad(handler, "all_profiles import is not allowed in isolated profile mode", 403)
    if allow_all_profiles and not requested_profile:
        return bad(handler, "profile is required for all_profiles import", 400)

    # Check if already imported — refresh messages from CLI store if new ones arrived
    existing = Session.load(sid)
    if existing:
        # Cross-profile boundary: an unqualified (non-all-profiles) request must not
        # read or refresh a session that belongs to another profile, even though the
        # WebUI session store (SESSION_DIR) is a single global directory. This mirrors
        # the /api/session detail and /api/session/export profile-scoping gates.
        # An explicit all_profiles import is still allowed, but only when the request's
        # profile matches the stored session's profile.
        existing_profile = getattr(existing, "profile", None)
        if allow_all_profiles:
            if requested_profile and not _profiles_match(existing_profile, requested_profile):
                return bad(handler, "Session not found in CLI store", 404)
        elif not _session_visible_to_active_profile(existing_profile, handler):
            # #7710: same contract as the detail-load endpoint —
            # 409 ``session_profile_mismatch`` for a known other
            # profile, 404 only for the None-profile self-heal path.
            if existing_profile:
                return j(handler, {
                    "error": "Session belongs to a different profile",
                    "code": "session_profile_mismatch",
                    "session_id": sid,
                    "profile": existing_profile,
                }, status=409)
            return bad(handler, "Session not found in CLI store", 404)
        refresh_profile = requested_profile or existing_profile
        cli_meta = _resolve_cli_import_metadata(
            sid,
            requested_profile=refresh_profile,
            allow_all_profiles=allow_all_profiles,
        )
        fresh_msgs = get_cli_session_messages(
            sid,
            profile=(cli_meta or {}).get("profile") or refresh_profile,
        )
        changed = False
        if fresh_msgs and len(fresh_msgs) > len(existing.messages):
            # Prefix-equality guard: only extend if existing messages are a prefix of
            # the fresh CLI messages. Prevents silently dropping WebUI-added messages
            # on hybrid sessions (user sent messages via WebUI while CLI continued).
            if _is_messages_refresh_prefix_match(existing.messages, fresh_msgs):
                existing.messages = fresh_msgs
                changed = True
        elif fresh_msgs and _is_cli_tool_metadata_enrichment(existing.messages, fresh_msgs):
            # Same row count, richer payload: rebuild sidecars imported before
            # CLI tool metadata was preserved (#1772).
            existing.messages = fresh_msgs
            changed = True
        if cli_meta:
            # A subagent child must never be flipped to CLI-classified /
            # writable on an existing-session refresh either (#5307).
            _existing_is_sa = (
                (existing.source_tag or existing.raw_source or "").strip().lower() == "subagent"
                or (cli_meta.get("source_tag") or cli_meta.get("raw_source") or "").strip().lower() == "subagent"
                or _is_subagent_child_session_id(sid)
            )
            updates = {
                "is_cli_session": (False if _existing_is_sa else True),
                "source_tag": existing.source_tag or cli_meta.get("source_tag"),
                "raw_source": existing.raw_source or cli_meta.get("raw_source") or cli_meta.get("source_tag"),
                "session_source": existing.session_source or cli_meta.get("session_source"),
                "source_label": existing.source_label or cli_meta.get("source_label"),
                "parent_session_id": existing.parent_session_id or cli_meta.get("parent_session_id"),
            }
            # A subagent child is view-only: also coerce read_only=True on the
            # persisted sidecar so a stale writable (pre-fix) sidecar can't be
            # used to start a WebUI turn (#5307).
            if _existing_is_sa:
                updates["read_only"] = True
            for attr, value in updates.items():
                if getattr(existing, attr, None) != value:
                    setattr(existing, attr, value)
                    changed = True
        else:
            _existing_is_sa = (
                (existing.source_tag or existing.raw_source or "").strip().lower() == "subagent"
                or _is_subagent_child_session_id(sid)
            )
        if changed:
            existing.save(touch_updated_at=False)
            publish_session_list_changed(
                "session_import_cli",
                profile=getattr(existing, "profile", None),
            )
        return j(
            handler,
            {
                "session": public_session_projection(
                    existing.compact()
                    | {
                        "messages": existing.messages,
                        "is_cli_session": (False if _existing_is_sa else True),
                        # Greptile #4911 follow-up: read read_only from
                        # the persisted Session, NOT from cli_meta.  This
                        # refresh path is for an already-WebUI-owned
                        # session; the WebUI's persisted view is the
                        # source of truth for the response, not the
                        # foreign store's current value.  (Mirrors the
                        # GET /api/session fix.)
                        "read_only": bool(getattr(existing, "read_only", False)),
                    }
                ),
                "imported": False,
            },
        )

    # Fetch messages from CLI store
    cli_meta = _resolve_cli_import_metadata(
        sid,
        requested_profile=requested_profile,
        allow_all_profiles=allow_all_profiles,
    )
    profile = cli_meta.get("profile") if cli_meta else (requested_profile if allow_all_profiles else None)
    msgs = get_cli_session_messages(sid, profile=profile)
    if not msgs:
        return bad(handler, "Session not found in CLI store", 404)

    # Get profile, model, timestamps, and title from CLI session metadata
    created_at = cli_meta.get("created_at") if cli_meta else None
    updated_at = cli_meta.get("updated_at") if cli_meta else None
    cli_title = cli_meta.get("title") if cli_meta else None
    cli_source_tag = cli_meta.get("source_tag") if cli_meta else None
    model = cli_meta.get("model", "unknown") if cli_meta else "unknown"
    cli_raw_source = cli_meta.get("raw_source") if cli_meta else None
    cli_session_source = cli_meta.get("session_source") if cli_meta else None
    cli_source_label = cli_meta.get("source_label") if cli_meta else None
    cli_user_id = cli_meta.get("user_id") if cli_meta else None
    cli_chat_id = cli_meta.get("chat_id") if cli_meta else None
    cli_chat_type = cli_meta.get("chat_type") if cli_meta else None
    cli_thread_id = cli_meta.get("thread_id") if cli_meta else None
    cli_session_key = cli_meta.get("session_key") if cli_meta else None
    cli_platform = cli_meta.get("platform") if cli_meta else None
    cli_parent_session_id = cli_meta.get("parent_session_id") if cli_meta else None
    cli_read_only = bool((cli_meta or {}).get("read_only"))
    # Delegated subagent children (#5307) are recovered VIEW-ONLY: they must
    # never be materialized as a writable WebUI sidecar via this endpoint, or a
    # subsequent chat-start/composer write would take ownership of a session
    # that belongs to the delegate runner. Treat them like an explicitly
    # read-only source (return the read-only stub payload, do not import), and
    # keep them out of the _isExternalSession frontend gates (is_cli_session=False).
    _sa_child = _is_subagent_child_session_id(sid)
    # Also treat a resolved-metadata subagent source as view-only: with
    # all_profiles=true, cli_meta is resolved from the requested (possibly
    # non-active) profile, so the active-profile state.db check (_sa_child)
    # can miss it (#5307 cross-profile edge).
    _cli_sa = (cli_source_tag or cli_raw_source or "").strip().lower() == "subagent"
    _sa_child = _sa_child or _cli_sa
    _read_only_view = cli_read_only or _sa_child

    # Use the CLI session title if available (e.g., cron job name), otherwise derive from messages
    title = cli_title or title_from(msgs, "CLI Session")

    # Auto-assign cron sessions to the dedicated "Cron Jobs" project (#1079),
    # gated on whether this profile has opted into project organization (#5379)
    cron_project_id = None
    if is_cron_session(sid, cli_source_tag):
        cron_project_id = ensure_cron_project(create=_profile_has_user_projects())

    if _read_only_view:
        session_payload = {
            "session_id": sid,
            "title": title,
            "workspace": str(get_last_workspace(profile=profile)),
            "model": model,
            "message_count": len(msgs),
            "created_at": created_at,
            "updated_at": updated_at,
            "last_message_at": updated_at or created_at,
            "pinned": False,
            "archived": False,
            "project_id": None,
            "profile": profile,
            # Subagent children (#5307) are recovered view-only and must NOT be
            # CLI-classified (keeps them out of the frontend _isExternalSession
            # gates); other explicitly-read-only sources keep is_cli_session=True.
            "is_cli_session": (False if _sa_child else True),
            "source_tag": cli_source_tag,
            "raw_source": cli_raw_source or cli_source_tag,
            "session_source": cli_session_source,
            "source_label": cli_source_label,
            "parent_session_id": cli_parent_session_id,
            "read_only": True,
            "messages": msgs,
            "tool_calls": [],
        }
        return j(
            handler,
            {
                "session": public_session_projection(session_payload),
                "imported": False,
            },
        )

    s = import_cli_session(
        sid,
        title,
        msgs,
        model,
        profile=profile,
        created_at=created_at,
        updated_at=updated_at,
        parent_session_id=cli_parent_session_id,
    )
    if cron_project_id:
        s.project_id = cron_project_id
    s.is_cli_session = True
    s.source_tag = cli_source_tag
    s.raw_source = cli_raw_source or cli_source_tag
    s.session_source = cli_session_source
    s.source_label = cli_source_label
    s.user_id = cli_user_id
    s.chat_id = cli_chat_id
    s.chat_type = cli_chat_type
    s.thread_id = cli_thread_id
    s.session_key = cli_session_key
    s.platform = cli_platform
    s._cli_origin = sid
    s.save(touch_updated_at=False)
    publish_session_list_changed(
        "session_import_cli",
        profile=getattr(s, "profile", None),
    )
    _queue_generated_title_for_imported_session(
        s,
        {
            "title": cli_title,
            "source_tag": cli_source_tag,
            "raw_source": cli_raw_source,
            "session_source": cli_session_source,
            "source_label": cli_source_label,
            "read_only": cli_read_only,
        },
    )
    return j(
        handler,
        {
            "session": public_session_projection(
                s.compact()
                | {
                    "messages": msgs,
                    "is_cli_session": True,
                }
            ),
            "imported": True,
        },
    )


def _handle_session_import(handler, body):
    """Import a session from a JSON export. Creates a new session with a new ID."""
    if not body or not isinstance(body, dict):
        return bad(handler, "Request body must be a JSON object")
    messages = strip_public_internal_fields(
        body.get("messages"),
        message_records=True,
    )
    if not isinstance(messages, list):
        return bad(handler, 'JSON must contain a "messages" array')
    raw_tool_calls = body.get("tool_calls", [])
    if not isinstance(raw_tool_calls, list):
        return bad(handler, 'JSON "tool_calls" must be an array')
    title = body.get("title", "Imported session")
    try:
        workspace = str(resolve_trusted_workspace(body.get("workspace", str(DEFAULT_WORKSPACE))))
    except (TypeError, ValueError) as e:
        return bad(handler, str(e))
    model = body.get("model", DEFAULT_MODEL)
    s = Session(
        title=title,
        workspace=workspace,
        model=model,
        messages=messages,
        tool_calls=strip_public_internal_fields(raw_tool_calls),
        profile=get_active_profile_name(),
    )
    s.pinned = body.get("pinned", False)
    with LOCK:
        SESSIONS[s.session_id] = s
        SESSIONS.move_to_end(s.session_id)
        _evict_sessions_over_cap()  # #4765: safe LRU eviction (never active/unsaved)
    s.save()
    publish_session_list_changed("session_import")
    return j(
        handler,
        {
            "ok": True,
            "session": public_session_projection(s.compact() | {"messages": s.messages}),
        },
    )


def _mask_secrets(obj):
    """Mask sensitive values in env vars and headers."""
    if not isinstance(obj, dict):
        return obj
    sensitive = ("auth", "token", "key", "secret", "password", "credential")
    masked = {}
    for k, v in obj.items():
        if isinstance(v, str) and any(s in k.lower() for s in sensitive):
            masked[k] = "••••••"
        elif isinstance(v, dict):
            masked[k] = _mask_secrets(v)
        else:
            masked[k] = v
    return masked


def _parse_mcp_enabled(value) -> bool:
    """Parse Hermes MCP ``enabled`` values without raising on bad config."""
    if value is None:
        return True
    if isinstance(value, bool):
        return value
    if isinstance(value, (int, float)):
        return value != 0
    if isinstance(value, str):
        normalized = value.strip().lower()
        if normalized in {"true", "1", "yes", "on"}:
            return True
        if normalized in {"false", "0", "no", "off"}:
            return False
    return True


def _mcp_runtime_status_by_name(servers=None, view=None) -> dict[str, dict]:
    """Return already-known MCP runtime status without starting servers.

    ``tools.mcp_tool.get_mcp_status()`` only reads the existing MCP registry and
    configuration; it does not probe or spawn MCP subprocesses. If Hermes Agent
    is unavailable, fall back to an empty map so the API remains safe.

    Call it inside ``mcp_runtime_scope()`` and pass its ``view``: the agent
    filters a routed profile's connections itself, but its launch-profile view
    is process-wide, so rows are narrowed to the connections serving ``view``
    (``api.mcp_runtime.filter_runtime_status_to_view``). ``servers`` (the
    profile's ``mcp_servers`` WebUI displays) is passed as ``configured`` when
    supported so both read the same config.
    """
    try:
        from api.agent_compat import agent_attr
        from api.mcp_runtime import accepts_keywords, filter_runtime_status_to_view
        get_mcp_status = agent_attr("tools.mcp_tool", "get_mcp_status", "tools.mcp_tool_discovery")
        if isinstance(servers, dict) and accepts_keywords(get_mcp_status, "configured"):
            # Invalid entries are summarized as invalid_config by WebUI; keep them
            # out of the agent call so one bad entry cannot blank every status.
            statuses = get_mcp_status(configured={
                str(name): scfg for name, scfg in servers.items() if isinstance(scfg, dict)
            })
        else:
            statuses = get_mcp_status()
        if view is not None and isinstance(statuses, list):
            statuses = filter_runtime_status_to_view(statuses, view)
    except Exception:
        return {}
    if not isinstance(statuses, list):
        return {}
    return {
        str(entry.get("name")): entry
        for entry in statuses
        if isinstance(entry, dict) and entry.get("name")
    }


def _server_summary(name, cfg, runtime_status=None):
    """Return a safe summary of an MCP server config."""
    runtime_status = runtime_status if isinstance(runtime_status, dict) else {}
    out = {"name": name}
    if not isinstance(cfg, dict):
        out.update({
            "transport": "invalid",
            "timeout": 120,
            "connect_timeout": 60,
            "enabled": False,
            "active": False,
            "status": "invalid_config",
            "tool_count": None,
        })
        return out

    enabled = _parse_mcp_enabled(cfg.get("enabled", True))
    connected = bool(runtime_status.get("connected")) if enabled else False
    if "url" in cfg:
        out["transport"] = "http"
        # Mask auth headers
        if "headers" in cfg:
            out["headers"] = _mask_secrets(cfg["headers"])
        out["url"] = cfg["url"]
    elif "command" in cfg:
        out["transport"] = "stdio"
        out["command"] = cfg.get("command", "")
        out["args"] = cfg.get("args", [])
        if "env" in cfg:
            out["env"] = _mask_secrets(cfg["env"])
    else:
        out["transport"] = "invalid"
        enabled = False
        connected = False

    out["timeout"] = cfg.get("timeout", 120)
    out["connect_timeout"] = cfg.get("connect_timeout", 60)
    out["enabled"] = enabled
    out["active"] = connected
    if out["transport"] == "invalid":
        out["status"] = "invalid_config"
    elif not enabled:
        out["status"] = "disabled"
    elif connected:
        out["status"] = "active"
    else:
        out["status"] = "configured"
    out["tool_count"] = runtime_status.get("tools") if runtime_status else None
    return out


def _mcp_safe_display_text(value, *, limit: int) -> str:
    """Return redacted, bounded MCP text safe for WebUI inventory rows."""
    if not isinstance(value, str):
        value = "" if value is None else str(value)
    value = _redact_text(value).strip()
    value = re.sub(r"Authorization:\s*Bearer\s+\S+", "[REDACTED CREDENTIAL]", value, flags=re.I)
    if len(value) > limit:
        value = value[: max(0, limit - 1)].rstrip() + "…"
    return value


def _mcp_schema_type(schema) -> str:
    """Return a compact, non-sensitive display type for a JSON schema node."""
    if not isinstance(schema, dict):
        return "unknown"
    typ = schema.get("type")
    if isinstance(typ, list):
        typ = "/".join(str(t) for t in typ if t)
    if isinstance(typ, str) and typ:
        return typ
    for composite in ("anyOf", "oneOf", "allOf"):
        if isinstance(schema.get(composite), list) and schema[composite]:
            return composite
    if "enum" in schema:
        return "enum"
    return "unknown"


def _mcp_schema_summary(schema, *, limit: int = 12) -> list[dict]:
    """Summarize an MCP input schema without exposing raw defaults/examples.

    The WebUI only needs searchable/displayable argument hints. Returning raw
    JSON Schema can overexpose server-provided defaults, examples, enums, or
    vendor extensions, so this strips each parameter down to name/type/required
    and a redacted description.
    """
    if not isinstance(schema, dict):
        return []
    properties = schema.get("properties")
    if not isinstance(properties, dict):
        return []
    required = schema.get("required")
    required_names = set(required) if isinstance(required, list) else set()
    out = []
    for name, prop in properties.items():
        if len(out) >= limit:
            break
        if not isinstance(name, str):
            continue
        prop = prop if isinstance(prop, dict) else {}
        desc = prop.get("description", "")
        if not isinstance(desc, str):
            desc = ""
        desc = _mcp_safe_display_text(desc, limit=180)
        out.append({
            "name": name,
            "type": _mcp_schema_type(prop),
            "required": name in required_names,
            "description": desc,
        })
    return out


def _mcp_tool_schema_from_payload(tool):
    if not isinstance(tool, dict):
        return {}
    for key in ("parameters", "inputSchema", "input_schema", "schema"):
        value = tool.get(key)
        if isinstance(value, dict):
            if key == "schema" and isinstance(value.get("parameters"), dict):
                return value["parameters"]
            return value
    return {}


def _mcp_tool_summary(name, tool, server_summary):
    """Return a safe global inventory row for one MCP tool."""
    server_summary = server_summary if isinstance(server_summary, dict) else {}
    if isinstance(tool, str):
        tool = {"name": tool}
    elif not isinstance(tool, dict):
        tool = {}
    tool_name = str(tool.get("name") or name or "")
    description = tool.get("description") or ""
    if not isinstance(description, str):
        description = str(description)
    description = _mcp_safe_display_text(description, limit=360)
    return {
        "name": tool_name,
        "server": str(server_summary.get("name") or ""),
        "description": description,
        "active": bool(server_summary.get("active")),
        "enabled": bool(server_summary.get("enabled")),
        "status": server_summary.get("status") or "unknown",
        "schema_summary": _mcp_schema_summary(_mcp_tool_schema_from_payload(tool)),
    }


def _mcp_tools_from_runtime_status(runtime_by_name, server_summaries):
    """Read detailed MCP tool payloads from runtime status when available."""
    tools = []
    if not isinstance(runtime_by_name, dict):
        return tools
    for server_name, runtime in runtime_by_name.items():
        if not isinstance(runtime, dict):
            continue
        raw_tools = runtime.get("tools")
        if not isinstance(raw_tools, list):
            raw_tools = runtime.get("tool_schemas")
        if not isinstance(raw_tools, list):
            continue
        server_summary = server_summaries.get(str(server_name), {"name": str(server_name)})
        for index, tool in enumerate(raw_tools):
            fallback_name = f"{server_name}:{index}"
            summary = _mcp_tool_summary(fallback_name, tool, server_summary)
            if summary["name"]:
                tools.append(summary)
    return tools


def _mcp_tools_from_registry(server_summaries, view=None):
    """Read already-registered MCP tool schemas without probing MCP servers.

    With a profile ``view`` (see ``api.mcp_runtime``), only tools registered in
    that profile's own registry slot are listed. The slot is the isolation check;
    the raw ``mcp_servers`` config is not an allowlist: the agent merges portable
    plugin servers into the running config at runtime, and their tools are
    registered in the same slot without a ``config.yaml`` entry.
    """
    try:
        from tools.registry import registry
        from api.mcp_runtime import registry_tool_owned_by_view
    except Exception:
        return []
    tools = []
    try:
        names = registry.get_all_tool_names()
    except Exception:
        return []
    for tool_name in names:
        try:
            toolset = registry.get_toolset_for_tool(tool_name)
        except Exception:
            continue
        if not isinstance(toolset, str) or not toolset.startswith("mcp-"):
            continue
        server_name = toolset[len("mcp-"):]
        if view is not None and not view.legacy and not registry_tool_owned_by_view(
            registry, tool_name, view
        ):
            continue
        schema = registry.get_schema(tool_name) or {}
        server_summary = server_summaries.get(server_name, {
            "name": server_name,
            "enabled": True,
            "active": False,
            "status": "configured",
        })
        tools.append(_mcp_tool_summary(tool_name, schema, server_summary))
    return tools


def _mcp_profile_runtime_inventory(servers, purpose, *, include_tools=True):
    """Build server summaries and tools from ONE runtime view of the request profile.

    Status, tool count and inventory are all read inside the same
    ``mcp_runtime_scope()`` so they describe the same profile's connections.
    When that profile scope cannot be confirmed, runtime data is withheld rather
    than showing another profile's connection. Passive: never starts or probes
    MCP servers.
    """
    from api.mcp_runtime import mcp_runtime_scope

    runtime = {}
    tools = []
    source = "none"
    with mcp_runtime_scope(purpose) as view:
        if view.trusted:
            runtime = _mcp_runtime_status_by_name(servers, view)
        server_summaries = {
            str(name): _server_summary(str(name), scfg, runtime.get(str(name)))
            for name, scfg in servers.items()
        }
        if include_tools and view.trusted:
            tools = _mcp_tools_from_runtime_status(runtime, server_summaries)
            source = "mcp_runtime_status"
            if not tools:
                tools = _mcp_tools_from_registry(server_summaries, view)
                source = "tool_registry" if tools else "none"
    return server_summaries, tools, source, view.scope_label


def _handle_mcp_tools_list(handler):
    """List known MCP tools from already-available runtime inventory only."""
    cfg = get_config_for_profile_home(get_active_hermes_home())
    servers = cfg.get("mcp_servers", {})
    if not isinstance(servers, dict):
        servers = {}
    server_summaries, tools, source, runtime_scope = _mcp_profile_runtime_inventory(
        servers, "/api/mcp/tools"
    )
    tools.sort(key=lambda row: (row.get("server", ""), row.get("name", "")))
    unavailable_servers = [
        summary["name"] for summary in server_summaries.values()
        if summary.get("enabled") and not summary.get("active")
    ]
    return j(handler, {
        "tools": tools,
        "total": len(tools),
        "source": source,
        "inventory_scope": "already_known_runtime_only",
        "runtime_scope": runtime_scope,
        "unavailable_servers": unavailable_servers,
    })


def _webui_truthy(value) -> bool:
    return str(value or "").strip().lower() in {"1", "true", "yes", "on"}


def _external_notes_sources_enabled(config_data: dict | None = None) -> bool:
    """Return whether the third-party notes drawer is explicitly enabled.

    The Memory panel is a primary surface, so this power-user drawer stays
    default-off unless a deployment opts in through config or environment.
    """
    env_value = os.getenv("HERMES_WEBUI_EXTERNAL_NOTES_SOURCES", "")
    if env_value:
        return _webui_truthy(env_value)
    cfg = config_data if isinstance(config_data, dict) else get_config()
    if not isinstance(cfg, dict):
        return False
    return _webui_truthy(
        cfg.get("webui_external_notes_sources")
        or cfg.get("external_notes_sources")
        or cfg.get("notes_sources_drawer")
    )


_NOTES_SOURCE_SERVER_HINTS = {
    "joplin", "obsidian", "notion", "llm-wiki", "llmwiki", "wiki",
    "notes", "note", "knowledge", "kb", "readwise", "logseq",
}
_NOTES_SOURCE_TOOL_HINTS = {
    "note", "notes", "notebook", "page", "pages", "wiki", "knowledge",
    "search_notes", "get_note", "list_notes", "read_note",
}
_NOTES_SOURCE_CONFIGURED_TOOL_HINTS = {
    "joplin": [
        {"name": "search_notes", "description": "Search Joplin notes by keyword."},
        {"name": "list_notes", "description": "List notes from a Joplin notebook."},
        {"name": "get_note", "description": "Read a specific Joplin note by ID."},
    ],
    "obsidian": [
        {"name": "search_notes", "description": "Search Obsidian notes by keyword."},
        {"name": "read_note", "description": "Read a specific Obsidian note or file."},
    ],
    "notion": [
        {"name": "search_pages", "description": "Search Notion pages or databases."},
        {"name": "get_page", "description": "Read a specific Notion page."},
    ],
    "llm-wiki": [
        {"name": "query_knowledge_base", "description": "Query the LLM Wiki knowledge base."},
        {"name": "read_page", "description": "Read a specific wiki page."},
    ],
    "llmwiki": [
        {"name": "query_knowledge_base", "description": "Query the LLM Wiki knowledge base."},
        {"name": "read_page", "description": "Read a specific wiki page."},
    ],
}


def _note_source_label(name: str) -> str:
    labels = {
        "joplin": "Joplin",
        "obsidian": "Obsidian",
        "notion": "Notion",
        "llm-wiki": "LLM Wiki",
        "llmwiki": "LLM Wiki",
        "readwise": "Readwise",
        "logseq": "Logseq",
    }
    lowered = str(name or "").strip().lower()
    return labels.get(lowered, str(name or "").replace("_", " ").replace("-", " ").title())


def _looks_like_notes_source(server_name: str, tool_rows: list[dict]) -> bool:
    server_l = str(server_name or "").lower()
    if any(hint in server_l for hint in _NOTES_SOURCE_SERVER_HINTS):
        return True
    for tool in tool_rows:
        haystack = " ".join([
            str(tool.get("name") or ""),
            str(tool.get("description") or ""),
        ]).lower()
        if any(hint in haystack for hint in _NOTES_SOURCE_TOOL_HINTS):
            return True
    return False


def _configured_note_tool_hints(server_name: str) -> list[dict]:
    """Return safe expected note-tool hints for configured known sources."""
    server_l = str(server_name or "").strip().lower()
    hints = _NOTES_SOURCE_CONFIGURED_TOOL_HINTS.get(server_l)
    if hints is None:
        if any(hint in server_l for hint in ("wiki", "knowledge", "kb")):
            hints = [
                {"name": "search", "description": "Search this configured knowledge source."},
                {"name": "read", "description": "Read an item from this configured knowledge source."},
            ]
        elif any(hint in server_l for hint in ("note", "notes")):
            hints = [
                {"name": "search_notes", "description": "Search this configured notes source."},
                {"name": "read_note", "description": "Read a note from this configured notes source."},
            ]
        else:
            hints = []
    return [
        {
            "name": _mcp_safe_display_text(row.get("name") or "", limit=96),
            "description": _mcp_safe_display_text(row.get("description") or "", limit=180),
            "inferred": True,
        }
        for row in hints
        if isinstance(row, dict)
    ]


def _notes_sources_from_mcp_inventory(server_summaries: dict, tools: list[dict]) -> list[dict]:
    """Build a safe notes/knowledge-source inventory from MCP servers/tools.

    Some WebUI deployments can read ``mcp_servers`` from config before their
    local runtime/tool registry has hydrated MCP tool metadata.  Still show
    configured note/knowledge servers (for example Joplin) in that case so the
    drawer reflects connection/configuration state instead of appearing empty.
    """
    by_server: dict[str, list[dict]] = {}
    for tool in tools or []:
        if not isinstance(tool, dict):
            continue
        server = str(tool.get("server") or "").strip()
        if not server:
            continue
        by_server.setdefault(server, []).append(tool)

    if isinstance(server_summaries, dict):
        for server, summary in server_summaries.items():
            server_name = str(server or "").strip()
            if not server_name or server_name in by_server:
                continue
            if _looks_like_notes_source(server_name, []):
                by_server.setdefault(server_name, [])

    sources = []
    for server, tool_rows in by_server.items():
        if not _looks_like_notes_source(server, tool_rows):
            continue
        summary = server_summaries.get(server, {"name": server}) if isinstance(server_summaries, dict) else {"name": server}
        safe_tools = []
        tool_source = "runtime"
        for tool in tool_rows[:8]:
            desc = _mcp_safe_display_text(tool.get("description") or "", limit=180)
            desc = re.sub(r"(?i)\b(api[_-]?key|token|password|secret)\s*[:=]\s*\S+", "[REDACTED]", desc)
            safe_tools.append({
                "name": _mcp_safe_display_text(tool.get("name") or "", limit=96),
                "description": desc,
            })
        if not safe_tools:
            safe_tools = _configured_note_tool_hints(server)
            if safe_tools:
                tool_source = "configured_hint"
        sources.append({
            "name": server,
            "label": _note_source_label(server),
            "enabled": bool(summary.get("enabled", True)),
            "active": bool(summary.get("active")),
            "status": summary.get("status") or "unknown",
            "tool_count": len(safe_tools),
            "tool_source": tool_source,
            "tools": safe_tools,
        })
    sources.sort(key=lambda row: (not row.get("active"), row.get("label", "")))
    return sources


def _handle_notes_sources_list(handler):
    """List note/knowledge MCP sources for the WebUI Notes drawer."""
    cfg = get_config()
    if not _external_notes_sources_enabled(cfg):
        return j(handler, {
            "enabled": False,
            "sources": [],
            "source": "disabled",
            "inventory_scope": "disabled_by_default",
            "attach_supported": False,
            "automatic_recall_unchanged": True,
            "recent_ai_notes": [],
        })
    servers = cfg.get("mcp_servers", {})
    if not isinstance(servers, dict):
        servers = {}
    server_summaries, tools, source, runtime_scope = _mcp_profile_runtime_inventory(
        servers, "/api/notes/sources"
    )
    return j(handler, {
        "enabled": True,
        "sources": _notes_sources_from_mcp_inventory(server_summaries, tools),
        "source": source,
        "inventory_scope": "already_known_runtime_only",
        "runtime_scope": runtime_scope,
        "attach_supported": False,
        "automatic_recall_unchanged": True,
        "recent_ai_notes": _joplin_recent_ai_notes(limit=6),
    })


def _notes_configured_server(source: str) -> dict:
    cfg = get_config()
    servers = cfg.get("mcp_servers", {}) if isinstance(cfg, dict) else {}
    if not isinstance(servers, dict):
        return {}
    source_l = str(source or "").strip().lower()
    for name, server_cfg in servers.items():
        if str(name or "").strip().lower() == source_l and isinstance(server_cfg, dict):
            return server_cfg
    return {}


def _joplin_connection_from_config() -> tuple[str, str]:
    cfg = _notes_configured_server("joplin")
    env = cfg.get("env", {}) if isinstance(cfg, dict) else {}
    if not isinstance(env, dict):
        env = {}
    url = str(env.get("JOPLIN_URL") or os.environ.get("JOPLIN_URL") or "http://127.0.0.1:41184").rstrip("/")
    token = str(env.get("JOPLIN_TOKEN") or os.environ.get("JOPLIN_TOKEN") or "")
    return url, token


def _joplin_api_get(path: str, params: dict | None = None) -> dict:
    """Call the local Joplin Web Clipper API without logging credentials."""
    from urllib.parse import urlencode
    from urllib.request import Request, urlopen
    from urllib.error import HTTPError, URLError

    base_url, token = _joplin_connection_from_config()
    if not token:
        raise ValueError("Joplin token is not configured")
    safe_path = "/" + str(path or "").lstrip("/")
    query = dict(params or {})
    # Joplin Web Clipper builds can reject header-only auth on /search even when
    # they accept it elsewhere. Keep the Authorization header for defense in
    # depth and add the query token only for /search compatibility.
    if safe_path == "/search":
        query["token"] = token
    url = f"{base_url}{safe_path}?{urlencode(query)}"
    request = Request(url, headers={"Authorization": f"token {token}"})
    try:
        with urlopen(request, timeout=8) as response:
            raw = response.read(2_000_000).decode("utf-8", errors="replace")
    except HTTPError as exc:
        raise ValueError(f"Joplin API returned HTTP {exc.code}") from None
    except (URLError, TimeoutError) as exc:
        # A bare socket-connect TimeoutError from urlopen(timeout=8) is NOT
        # always URLError-wrapped, so catch it explicitly here. Otherwise it
        # propagates past _handle_notes_search's `except ValueError` to the
        # request dispatch, where the consolidated client-disconnect handler
        # (#3210) would swallow it as a fake disconnect — silent empty response,
        # no log. Convert it to a normal "not reachable" ValueError instead.
        raise ValueError("Joplin API is not reachable") from None
    try:
        data = json.loads(raw)
    except Exception:
        raise ValueError("Joplin API returned invalid JSON") from None
    return data if isinstance(data, dict) else {}


def _note_snippet(body: str, query: str = "", *, limit: int = 220) -> str:
    text = re.sub(r"\s+", " ", str(body or "")).strip()
    if not text:
        return ""
    q = str(query or "").strip().lower()
    if q:
        idx = text.lower().find(q)
        if idx > 40:
            text = "…" + text[max(0, idx - 60):]
    if len(text) > limit:
        return text[:limit].rstrip() + "…"
    return text


def _joplin_search_notes(query: str, *, limit: int = 20) -> list[dict]:
    query = str(query or "").strip()
    if not query:
        return []
    limit = max(1, min(int(limit or 20), 50))
    data = _joplin_api_get("/search", {
        "query": query,
        "type": "note",
        "fields": "id,title,body,parent_id,updated_time",
        "limit": limit,
    })
    rows = data.get("items") if isinstance(data, dict) else []
    results = []
    for row in rows if isinstance(rows, list) else []:
        if not isinstance(row, dict):
            continue
        note_id = _mcp_safe_display_text(row.get("id") or "", limit=64)
        if not note_id:
            continue
        title = _mcp_safe_display_text(row.get("title") or "Untitled", limit=180)
        body = str(row.get("body") or "")
        results.append({
            "id": note_id,
            "title": title,
            "snippet": _mcp_safe_display_text(_note_snippet(body, query), limit=260),
            "parent_id": _mcp_safe_display_text(row.get("parent_id") or "", limit=64),
            "updated_time": row.get("updated_time"),
            "source": "joplin",
        })
    return results


def _joplin_get_note(note_id: str) -> dict:
    note_id = str(note_id or "").strip()
    if not re.fullmatch(r"[A-Za-z0-9]{16,64}", note_id):
        raise ValueError("Invalid Joplin note id")
    data = _joplin_api_get(f"/notes/{note_id}", {
        "fields": "id,title,body,parent_id,updated_time,created_time",
    })
    if not data.get("id"):
        raise ValueError("Joplin note not found")
    body = str(data.get("body") or "")
    if len(body) > 50_000:
        body = body[:50_000].rstrip() + "\n\n[Preview truncated at 50,000 characters]"
    return {
        "id": _mcp_safe_display_text(data.get("id") or "", limit=64),
        "title": _mcp_safe_display_text(data.get("title") or "Untitled", limit=180),
        "body": _redact_text(body),
        "parent_id": _mcp_safe_display_text(data.get("parent_id") or "", limit=64),
        "updated_time": data.get("updated_time"),
        "created_time": data.get("created_time"),
        "source": "joplin",
    }


_JOPLIN_AI_RECALL_NOTE_PRIORITY = [
    ("CURRENT_CONTEXT_ID", "Current Context"),
    ("OPEN_ISSUES_ID", "Open Issues"),
    ("AGENT_MEMORY_ID", "Agent Memory"),
    ("CONVENTIONS_ID", "Conventions / Preferences"),
    ("INFRA_ID", "Infrastructure"),
    ("SERVICES_ID", "Services"),
]


def _script_path_from_config_value(path_value) -> Path | None:
    """Return the likely recall script path from a string or argv-style hook."""
    if not path_value:
        return None
    try:
        if isinstance(path_value, (list, tuple)):
            candidates = [str(part).strip() for part in path_value if str(part).strip()]
        else:
            raw = str(path_value).strip()
            raw_path = Path(raw).expanduser()
            if raw and raw_path.exists():
                return raw_path
            candidates = shlex.split(raw)
        # Hooks commonly use either [python, /path/to/script.py] or the string
        # form "python /path/to/script.py". Prefer the first script-like argument
        # over the interpreter so AI-recent notes reflect the configured recall
        # source rather than "python3".
        for candidate in candidates:
            if candidate.endswith((".py", ".sh", ".bash")):
                return Path(candidate).expanduser()
        if candidates:
            return Path(candidates[-1]).expanduser()
        return None
    except Exception:
        return None


def _joplin_prefill_script_path() -> Path | None:
    cfg = get_config()
    if not isinstance(cfg, dict):
        return None
    # The browser notes drawer should mirror the WebUI-specific recall hook when
    # configured. Fall back to the legacy generic session prefill script only for
    # deployments that have not opted into WebUI dynamic recall.
    return _script_path_from_config_value(
        os.getenv("HERMES_WEBUI_PREFILL_MESSAGES_SCRIPT", "")
        or cfg.get("webui_prefill_messages_script")
        or cfg.get("prefill_messages_script")
    )


def _joplin_recall_note_refs(script_path: Path | None = None) -> list[dict]:
    """Find stable Joplin note IDs referenced by the configured recall script.

    This keeps the WebUI generic: it does not hard-code a user's note IDs, but
    can still surface the notes that the configured AI prefill/recall script is
    known to read for automatic context.
    """
    script_path = script_path or _joplin_prefill_script_path()
    if not script_path or not script_path.exists() or not script_path.is_file():
        return []
    try:
        text = script_path.read_text(encoding="utf-8", errors="replace")
    except Exception:
        return []
    constants = {
        match.group(1): match.group(2)
        for match in re.finditer(r'(?m)^\s*([A-Z0-9_]+_ID)\s*=\s*["\']([A-Fa-f0-9]{16,64})["\']', text)
    }
    refs = []
    seen = set()
    for const_name, label in _JOPLIN_AI_RECALL_NOTE_PRIORITY:
        note_id = constants.get(const_name)
        if not note_id or note_id in seen:
            continue
        seen.add(note_id)
        refs.append({
            "id": note_id,
            "label": label,
            "constant": const_name,
            "used_by": "ai_prefill",
            "used_reason": "automatic_recall",
        })
    return refs


def _joplin_recent_ai_notes(*, limit: int = 6) -> list[dict]:
    """Return safe Joplin notes that the configured AI recall path recently uses."""
    try:
        limit = max(1, min(int(limit or 6), 20))
    except Exception:
        limit = 6
    notes = []
    for ref in _joplin_recall_note_refs()[:limit]:
        try:
            data = _joplin_api_get(f"/notes/{ref['id']}", {
                "fields": "id,title,parent_id,updated_time,user_updated_time,created_time",
            })
        except Exception:
            continue
        note_id = _mcp_safe_display_text(data.get("id") or ref.get("id") or "", limit=64)
        if not note_id:
            continue
        notes.append({
            "id": note_id,
            "title": _mcp_safe_display_text(data.get("title") or ref.get("label") or "Untitled", limit=180),
            "label": _mcp_safe_display_text(ref.get("label") or "", limit=120),
            "parent_id": _mcp_safe_display_text(data.get("parent_id") or "", limit=64),
            "updated_time": data.get("user_updated_time") or data.get("updated_time"),
            "created_time": data.get("created_time"),
            "source": "joplin",
            "used_by": ref.get("used_by") or "ai_prefill",
            "used_reason": ref.get("used_reason") or "automatic_recall",
        })
    return notes


def _handle_notes_search(handler, parsed):
    if not _external_notes_sources_enabled():
        return j(handler, {"source": "disabled", "results": [], "error": "External notes sources are disabled."}, status=404)
    query = parse_qs(parsed.query or "")
    source = str(query.get("source", ["joplin"])[0] or "joplin").strip().lower()
    q = str(query.get("q", [""])[0] or "").strip()
    try:
        limit = int(query.get("limit", ["20"])[0] or 20)
    except Exception:
        limit = 20
    if source != "joplin":
        return j(handler, {"source": source, "results": [], "error": "Search is currently implemented for Joplin sources only."}, status=400)
    try:
        return j(handler, {"source": "joplin", "query": q, "results": _joplin_search_notes(q, limit=limit)})
    except ValueError as exc:
        return j(handler, {"source": "joplin", "query": q, "results": [], "error": str(exc)}, status=502)


def _handle_notes_item(handler, parsed):
    if not _external_notes_sources_enabled():
        return j(handler, {"source": "disabled", "error": "External notes sources are disabled."}, status=404)
    query = parse_qs(parsed.query or "")
    source = str(query.get("source", ["joplin"])[0] or "joplin").strip().lower()
    note_id = str(query.get("id", [""])[0] or "").strip()
    if source != "joplin":
        return j(handler, {"source": source, "error": "Preview is currently implemented for Joplin sources only."}, status=400)
    try:
        return j(handler, {"source": "joplin", "note": _joplin_get_note(note_id)})
    except ValueError as exc:
        return j(handler, {"source": "joplin", "error": str(exc)}, status=502)


def _handle_mcp_servers_list(handler):
    """List configured MCP servers with safe, read-only runtime visibility."""
    cfg = get_config_for_profile_home(get_active_hermes_home())
    servers = cfg.get("mcp_servers", {})
    if not isinstance(servers, dict):
        servers = {}
    server_summaries, _tools, _source, runtime_scope = _mcp_profile_runtime_inventory(
        servers, "/api/mcp/servers", include_tools=False
    )
    return j(handler, {
        "servers": list(server_summaries.values()),
        "toggle_supported": True,
        "reload_required": True,
        "runtime_scope": runtime_scope,
    })


def _handle_mcp_server_delete(handler, name):
    """Delete an MCP server by name."""
    from urllib.parse import unquote
    name = unquote(name)
    if not name:
        return bad(handler, "name is required")
    cfg = get_config()
    servers = cfg.get("mcp_servers", {})
    if not isinstance(servers, dict):
        servers = {}
    if name not in servers:
        return bad(handler, f"MCP server '{name}' not found", 404)
    del servers[name]
    cfg["mcp_servers"] = servers
    _save_yaml_config_file(_get_config_path(), cfg)
    reload_config()
    return j(handler, {"ok": True, "deleted": name})


def _handle_mcp_server_toggle(handler, name, body):
    """Toggle enabled state for an MCP server (PATCH /api/mcp/servers/{name})."""
    from urllib.parse import unquote
    name = unquote(name)
    if not name:
        return bad(handler, "name is required")
    if "enabled" not in body:
        return bad(handler, "enabled field is required")
    enabled = bool(body["enabled"])
    cfg = get_config()
    servers = cfg.get("mcp_servers", {})
    if not isinstance(servers, dict):
        servers = {}
    if name not in servers:
        return bad(handler, f"MCP server '{name}' not found", 404)
    if not isinstance(servers[name], dict):
        return bad(handler, f"MCP server '{name}' has invalid config", 400)
    servers[name]["enabled"] = enabled
    cfg["mcp_servers"] = servers
    _save_yaml_config_file(_get_config_path(), cfg)
    reload_config()
    return j(handler, {"ok": True, "name": name, "enabled": enabled})


_MASKED_PLACEHOLDER = "••••••"


def _strip_masked_values(submitted, existing):
    """Remove masked placeholder values from submitted dict, keeping originals."""
    if not isinstance(submitted, dict) or not isinstance(existing, dict):
        return submitted
    cleaned = {}
    for k, v in submitted.items():
        if isinstance(v, str) and v == _MASKED_PLACEHOLDER:
            if k in existing and isinstance(existing[k], str):
                cleaned[k] = existing[k]  # preserve original real value
                continue
        elif isinstance(v, dict) and k in existing and isinstance(existing[k], dict):
            cleaned[k] = _strip_masked_values(v, existing[k])
        else:
            cleaned[k] = v
    return cleaned


def _handle_mcp_server_update(handler, name, body):
    """Add or update an MCP server."""
    from urllib.parse import unquote
    name = unquote(name)
    if not name:
        return bad(handler, "name is required")
    # Validate: must have url (http) or command (stdio)
    server_cfg = {}
    cfg = get_config()
    servers = cfg.get("mcp_servers", {})
    if not isinstance(servers, dict):
        servers = {}
    existing_cfg = servers.get(name, {})
    if body.get("url"):
        server_cfg["url"] = body["url"].strip()
        if body.get("headers"):
            server_cfg["headers"] = _strip_masked_values(body["headers"], existing_cfg.get("headers", {}))
    elif body.get("command"):
        server_cfg["command"] = body["command"].strip()
        if body.get("args"):
            server_cfg["args"] = body["args"] if isinstance(body["args"], list) else [body["args"]]
        if body.get("env"):
            server_cfg["env"] = _strip_masked_values(body["env"], existing_cfg.get("env", {}))
    else:
        return bad(handler, "url or command is required")
    if body.get("timeout") is not None:
        try:
            server_cfg["timeout"] = int(body["timeout"])
        except (ValueError, TypeError):
            pass
    servers[name] = server_cfg
    cfg["mcp_servers"] = servers
    _save_yaml_config_file(_get_config_path(), cfg)
    reload_config()
    return j(handler, {"ok": True, "server": _server_summary(name, server_cfg)})
