"""Auto-generate short session titles from the user's opening message.

Two stages, both off the critical path: an **instant** deterministic title (written before the model
is called, cannot fail), then an **upgrade** from one small-model call (cheap tier, thinking off,
JSON-constrained). Storage enforces provenance ``derived < llm < user``: stage 2 only replaces stage 1
and neither replaces a name the user typed."""

import json
import logging
import os
import re
import threading
import time
import weakref
from contextlib import suppress
from typing import Any, Callable, Optional

from agent.auxiliary_client import call_llm
from agent.context_compressor import LEGACY_SUMMARY_PREFIX
from agent.delegation_context import is_dispatcher_owned_worker_context
from agent.message_content import flatten_message_text

logger = logging.getLogger(__name__)

# In-flight stage-2 upgrade threads. They bill their aux usage to the session from a daemon thread,
# so a process that reads the ledger right before exit (``-z --usage-file``) must be able to join
# them (bounded) instead of racing the write (#112848).
_UPGRADE_THREADS: "weakref.WeakSet[threading.Thread]" = weakref.WeakSet()


def wait_for_title_upgrades(timeout: float = 10.0) -> None:
    """Bounded join of the auto-title threads still running; never raises."""
    deadline = time.monotonic() + timeout
    for thread in list(_UPGRADE_THREADS):
        thread.join(max(0.0, deadline - time.monotonic()))

# (task_name, exception) -> None; surfaces auxiliary failures so silent drops don't pile up as NULL titles.
FailureCallback = Callable[[str, BaseException], None]
# (title, source) -> None; source is the persisted provenance (``derived`` / ``llm``). Consumers paying a
# rate-limited remote rename per title (Discord thread, Telegram topic) should act on ``llm`` only.
TitleCallback = Callable[[str, str], None]
# () -> bool, called right before the LLM request; False skips (e.g. the user switched models and
# the request would reload one the runtime already evicted).
# Validation callback: () -> bool. See #19027.
RuntimeValidator = Callable[[], bool]

# Text budget handed to the model (Claude Code / OpenClaw converged on 1000).
MAX_TITLE_INPUT_CHARS = 1000
_PASTE_PREVIEW_LABEL = "\n\nPasted content:\n"
_ATTACHMENT_REF_RE = re.compile(r"@(?:file|folder):\S+")
# Footers the @-reference expander appends below the typed text (agent/context_references.py).
_CONTEXT_FOOTER_RE = re.compile(r"\n+--- (?:Context Warnings|Attached Context) ---\n.*", re.DOTALL)
# Cap on the instant derived title; a raw fragment reads worse the longer it runs.
MAX_DERIVED_TITLE_CHARS = 48
# Answer-shaped guard: a tiny model sometimes answers instead of titling; longer is rejected, not truncated.
# Upper bound on accepted title word count. Titling is a 3-7 word task; a small tiny-model sometimes ignores
# the task and answers the user's message instead — that answer must never become the session title (see the
# answer-shaped output guard in generate_title; port of can1357/oh-my-pi#7306). 12 leaves headroom for
# legitimate wordy titles while excluding full-sentence answers.
_MAX_TITLE_WORDS = 12
# Output budget for the title call: room for a fenced/prefixed JSON reply and for a reasoning model that
# thinks despite the thinking-disabled request, without letting a runaway reply burn minutes.
TITLE_MAX_TOKENS = 512

# The example titles shown to the model in the prompt, and the echo-guard
# set: when the opening message carries little topical signal, a small model
# sometimes takes the cheapest schema-valid answer and parrots one of these
# back verbatim — most visibly "Fix login button on mobile" naming sessions
# that have nothing to do with a login button. The prompt's example lines are
# rendered from these constants so the guard set and the prompt cannot drift
# apart. Port of QwenLM/qwen-code#9709.
_PROMPT_GOOD_EXAMPLES = (
    "Fix login button on mobile",
    "Postgres connection pool exhaustion",
    "Friendly greeting",
)
_PROMPT_VAGUE_EXAMPLE = "Code changes"

# "Friendly greeting" is deliberately NOT in the reject set: the prompt
# instructs the model to produce it for bare greetings, so it is a legitimate
# output, not an echo failure. The too-vague example is rejected too — a model
# repeating the counter-example says nothing about the session, and the
# derived title the guard falls back to is strictly more informative.
_EXAMPLE_ECHO_REJECT = frozenset(
    t.lower() for t in _PROMPT_GOOD_EXAMPLES if t != "Friendly greeting"
) | {_PROMPT_VAGUE_EXAMPLE.lower()}

# "Friendly greeting" is what the prompt asks for when the opener has no topic yet, so
# it is the one model title that must NOT settle the session: it is persisted at
# ``derived`` authority (a placeholder, like the instant title) and the next substantive
# turn upgrades it. Kept as a prompt example on purpose — a predictable placeholder is
# detectable, an improvised one ("Casual check-in chat") would lock the title as ``llm``.
_PROVISIONAL_GREETING_TITLE = "friendly greeting"


_TITLE_PROMPT_TEMPLATE = (
    "You name chat sessions. Given the user's opening message, write a title "
    "that lets them find this conversation again in a list.\n\n"
    "Rules:\n"
    "- 3 to 7 words, sentence case (capitalize only the first word and proper nouns).\n"
    "- Name what the user wants DONE, not that they asked a question.\n"
    "- Keep technical terms, filenames, numbers, and error codes exact.\n"
    "- Drop filler words: the, this, my, a, an.\n"
    "- No trailing punctuation, no quotes, no tool names, no 'Title:' prefix.\n"
    "- Never answer the message. Name it.\n"
    "- Always produce something, even for a bare greeting.\n"
    "__LANGUAGE_RULE__\n"
    + "".join(f'Good: {{"title": "{t}"}}\n' for t in _PROMPT_GOOD_EXAMPLES)
    + f'Too vague: {{"title": "{_PROMPT_VAGUE_EXAMPLE}"}}\n'
    'Too long: {"title": "Investigate and fix the issue where the login button '
    'does not respond on mobile devices"}\n\n'
    'Reply with JSON only: {"title": "..."}'
)

_LANGUAGE_RULE_MATCH_USER = "- Write the title in the same language as the user's message."
_LANGUAGE_RULE_PINNED = "- Write the title in {language}."

# Constrains the response to a single title field ("model answered instead of titling" failure class).
_TITLE_RESPONSE_FORMAT = {
    "type": "json_schema",
    "json_schema": {"name": "session_title", "strict": True, "schema": {
        "type": "object", "properties": {"title": {"type": "string"}}, "required": ["title"], "additionalProperties": False}},
}

# Control-tag wrappers around machine-authored content inside a nominal "user" message (Codex CLI's
# RECOGNIZED_CONTROL_WRAPPERS): stripped, titling continues on what remains.
_CONTROL_WRAPPERS = tuple(
    (f"<{tag}>", f"</{tag}>")
    for tag in ("command-message", "command-name", "command-args", "local-command-caveat", "local-command-stderr",
                "local-command-stdout", "task-notification", "system-reminder", "ide_opened_file", "ide_selection")
)

# Hermes' own machine-authored openers: a compaction handoff or resumed session must not be titled after them.
_MACHINE_PREFIXES = (
    "[CONTEXT COMPACTION", LEGACY_SUMMARY_PREFIX, "[Runtime note:", "[System note:", "[SYSTEM]",
    # tui_gateway.server._MODEL_SWITCH_MARKER_PREFIX (keep in sync); persisted as role="user" because
    # strict providers reject a non-first system message.
    # Model-switch marker from tui_gateway.server._append_model_switch_marker. It is persisted with
    # role="user" (strict OpenAI-compatible providers reject a system message that is not first — #48338),
    # so without this entry it looks like a real opening turn: switching models before the first real
    # message titled the session "[System: The active model for this chat has…" instead of the user's actual
    # question.
    "[System: The active model for this chat has changed to ",
)


def _title_config() -> dict:
    """``auxiliary.title_generation`` (lazy read-only import: no hermes_cli cycle, no migration writes)."""
    from hermes_cli.config import load_config_readonly
    return ((load_config_readonly() or {}).get("auxiliary") or {}).get("title_generation") or {}


def _title_language() -> str:
    """Configured title language, or "" to match the user."""
    try:
        return str(_title_config().get("language", "")).strip()
    except Exception:
        return ""


def _auto_title_enabled() -> bool:
    try:
        from utils import is_truthy_value
        return is_truthy_value(_title_config().get("enabled"), default=True)
    except Exception:
        logger.debug("Failed to read title_generation.enabled", exc_info=True)
        return True


def _model_title_upgrade_enabled() -> bool:
    """Distinct from ``enabled``: keep the instant derived title, skip the background model call (#85194)."""
    try:
        from utils import is_truthy_value
        return is_truthy_value(_title_config().get("model_upgrade_enabled"), default=True)
    except Exception:
        logger.debug("Failed to read title_generation.model_upgrade_enabled", exc_info=True)
        return True


def title_upgrade_must_wait_for_turn(main_runtime: Optional[dict]) -> bool:
    """True when the model title call would hit the SAME self-hosted endpoint as the turn's own request.

    A self-hosted main route (``_is_self_hosted_provider``: custom, lmstudio, local and their aliases)
    whose ``auxiliary.title_generation`` is not pinned elsewhere (a pin naming the same custom route —
    ``custom:<name>``, bare ``<name>`` or its display name — is not "elsewhere", #120558)
    shares one local server between the streaming main request and the
    concurrent ``response_format: json_schema`` title request. Single-slot servers then serve the
    title grammar/completion into the main turn: the user's reply arrives as ``{"title": ...}``, is
    persisted as a genuine assistant row and replayed, and the model adopts the format (#117296).
    Running the title call after the turn settles keeps the two requests off the wire at once.
    Hosted providers multiplex requests independently and keep the turn-start timing.
    """
    provider = str((main_runtime or {}).get("provider") or "").strip().lower()
    if not _is_self_hosted_provider(provider):
        return False
    try:
        cfg = _title_config()
        pinned_provider = str(cfg.get("provider") or "").strip().lower()
        main_base_url = str((main_runtime or {}).get("base_url") or "").strip().rstrip("/")
        if pinned_provider not in ("", "auto") and not _title_pin_may_share_endpoint(
                pinned_provider, provider, main_base_url):
            return False
    except Exception:
        return True
    pinned_base_url = str(cfg.get("base_url") or "").strip().rstrip("/")
    return not pinned_base_url or pinned_base_url == main_base_url


def _is_self_hosted_provider(provider: str) -> bool:
    """``custom``, a named ``custom:<name>`` route, LM Studio or ``local`` (vllm/llama.cpp): one local server per route.

    Normalised here so the main route and the title pin resolve aliases (``ollama``, ``lm-studio``…) the same way.
    """
    from hermes_cli.providers import normalize_provider
    provider = normalize_provider(provider)
    return provider in ("custom", "lmstudio", "local") or provider.startswith("custom:")


def _title_pin_may_share_endpoint(pinned_provider: str, main_provider: str, main_base_url: str) -> bool:
    """A title pin that can land on the turn's own self-hosted server (only its ``base_url`` can prove otherwise).

    Hosted pins (``openrouter``…) multiplex and never share the slot. A pin to ``custom``/``lmstudio``/``local``/any
    ``custom:<name>`` is assumed to share until the caller compares ``base_url``, and a bare ``<name>`` /
    display-name pin is the same endpoint when it aliases the main ``custom:<name>`` route
    (``hermes_cli.providers.custom_provider_aliases`` — the resolver's own identity set) or resolves to a
    configured custom entry serving ``main_base_url`` (a keyed ``providers:`` entry's display name does not
    alias its ``custom:<key>`` id).
    """
    from hermes_cli.config import get_compatible_custom_providers, load_config_readonly
    from hermes_cli.providers import custom_provider_aliases, resolve_custom_provider
    if _is_self_hosted_provider(pinned_provider):
        return True
    if custom_provider_aliases(pinned_provider) & custom_provider_aliases(main_provider):
        return True
    pdef = resolve_custom_provider(pinned_provider, get_compatible_custom_providers(load_config_readonly()))
    return bool(pdef and main_base_url and pdef.base_url.strip().rstrip("/") == main_base_url)


def start_title_upgrade(upgrade: Optional[threading.Thread]) -> None:
    """Start a (deferred) title upgrade thread; joinable via ``wait_for_title_upgrades`` only once started."""
    if upgrade is None or upgrade.ident is not None:
        return
    _UPGRADE_THREADS.add(upgrade)
    upgrade.start()


def strip_control_wrappers(text: str) -> str:
    """Remove leading control wrappers (nested too) so a slash-command turn reduces to the prose the user typed."""
    current = (text or "").strip()
    for _ in range(len(_CONTROL_WRAPPERS) * 2):  # bounded: each pass must remove a wrapper or we stop
        stripped = _strip_one_wrapper(current)
        if stripped == current:
            break
        current = stripped
    return current


def _strip_one_wrapper(text: str) -> str:
    lowered = text.lower()
    for open_tag, close_tag in _CONTROL_WRAPPERS:
        if not lowered.startswith(open_tag):
            continue
        end = lowered.find(close_tag)
        if end == -1:  # unterminated wrapper: drop the opening tag and keep the body
            return text[len(open_tag):].strip()
        # Prefer trailing prose; otherwise the wrapper body is all we have.
        return (text[end + len(close_tag):].strip() or text[len(open_tag):end].strip()).strip()
    return text


# Matches an ``@file:``/``@folder:`` context reference the way the canonical
# ``agent.context_references`` reference pattern does: an unquoted ``\S+``
# value, or a backtick/double/single-quoted value (a space-bearing path is
# written backtick-quoted). Kept local so the titler doesn't import the
# context-reference machinery (circularity / startup cost) just to detect the
# attachment-only shape (#92068).
_QUOTED = r"(?:`[^`\n]+`|\"[^\"\n]+\"|'[^'\n]+')"
# Mirrors the canonical agent.context_references value shape: a quoted
# (space-bearing) path may carry a ``:start[-end]`` line-range suffix, and a
# bare path swallows its own range via ``\S+``.
_CONTEXT_REFERENCE_TOKEN_RE = re.compile(
    rf"(?<![\w/])@(?:file|folder):(?:{_QUOTED}(?::\d+(?:-\d+)?)?|\S+)"
)


def _attachment_only_opener(message: str) -> bool:
    """True when nothing but context references (and their expansion footer)
    remain — a manual-attach opener with no typed request and no paste
    preview has no topic of its own to title (#92068)."""
    residual = _CONTEXT_REFERENCE_TOKEN_RE.sub("", _CONTEXT_FOOTER_RE.sub("", message))
    return not residual.strip()


def _summarize_user_message(user_message: str) -> str:
    """Text worth titling: describe a ``/skill`` invocation (it embeds the whole skill body), then strip wrappers."""
    if not user_message:
        return ""
    described = None
    try:
        from agent.skill_commands import describe_skill_invocation
        described = describe_skill_invocation(user_message)
    except Exception:
        logger.debug("Skill-scaffolding summary failed; titling raw", exc_info=True)
    return strip_control_wrappers(user_message if described is None else described)


def build_title_input(user_message: str, title_preview: str | None = None) -> str:
    """Combine the opening text with a bounded Desktop-generated paste preview.

    ``title_preview`` is deliberately an explicit, auxiliary-only value: ordinary
    attachments never populate it, and it is never returned to the main turn.
    Keep enough of a separately typed request to preserve a useful instruction,
    then spend the remaining title budget on the beginning of the pasted topic.
    """
    message = _summarize_user_message(user_message)
    preview = title_preview.strip() if isinstance(title_preview, str) else ""
    if not preview:
        return message[:MAX_TITLE_INPUT_CHARS]
    # The titler sees the message AFTER @-reference expansion, so the generated ref drags a
    # warnings/attached-context footer along; the preview already carries the topic, so drop it.
    message = _CONTEXT_FOOTER_RE.sub("", message).strip()
    # A paste-only opener is just the generated `@file:` ref: the preview IS the topic, so it
    # leads (derive_title takes the first line, and a file path is not a title).
    if not _ATTACHMENT_REF_RE.sub("", message).strip():
        return preview[:MAX_TITLE_INPUT_CHARS]
    message_budget = min(len(message), MAX_TITLE_INPUT_CHARS // 2)
    preview_budget = MAX_TITLE_INPUT_CHARS - message_budget - len(_PASTE_PREVIEW_LABEL)
    if preview_budget <= 0:
        return message[:MAX_TITLE_INPUT_CHARS]
    return message[:message_budget] + _PASTE_PREVIEW_LABEL + preview[:preview_budget]


def is_titleable_user_message(user_message: str) -> bool:
    """False for machine-authored openers and turns that reduce to nothing once scaffolding is stripped."""
    return (isinstance(user_message, str) and bool(user_message.strip()) and not user_message.lstrip().startswith(_MACHINE_PREFIXES)
            and bool(_summarize_user_message(user_message).strip())
            # An attachment-only opener (manual attach, no paste preview) is a
            # file drop, not a request: deriving its "title" from the message
            # names the session after the truncated file path (#92068).
            and not _attachment_only_opener(user_message))


def derive_title(user_message: str, title_preview: str | None = None) -> Optional[str]:
    """Instant title: first meaningful line trimmed to a word boundary. No model, never fails."""
    # Attachment-only opener, no paste preview: a file drop has no topic —
    # refuse rather than name the session after the truncated path (#92068).
    if not title_preview and _attachment_only_opener(user_message):
        return None
    line = " ".join(_first_line(build_title_input(user_message, title_preview)).split())
    if len(line) > MAX_DERIVED_TITLE_CHARS:
        cut = line[:MAX_DERIVED_TITLE_CHARS]
        space = cut.rfind(" ")
        line = (cut[:space] if space > MAX_DERIVED_TITLE_CHARS // 2 else cut).rstrip(" ,.;:—-") + "…"
    return line or None


def _strip_title_prefix(text: str) -> str:
    return text[6:].strip() if text.lower().startswith("title:") else text


def _first_line(text: str) -> str:
    return next((ln.strip() for ln in text.splitlines() if ln.strip()), "")


def _extract_json_title(raw: str) -> Optional[str]:
    """Title from a ``{"title": ...}`` payload — strict parse, then a loose ``"title": "..."`` scan; None when absent."""
    try:
        parsed = json.loads(raw)
        if isinstance(parsed, dict) and isinstance(parsed.get("title"), str):
            return parsed["title"].strip()
    except (ValueError, TypeError):
        pass
    match = re.search(r'"title\"\s*:\s*"((?:[^"\\]|\\.)*)"', raw)
    if match:
        with suppress(ValueError):
            return json.loads(f'"{match.group(1)}"').strip()
        return match.group(1).strip()
    return None


def _is_truncated_structured_output(raw: str) -> bool:
    """Structured output the token cap cut before its closing quote/brace/fence (``{"title``, a bare fence opener).

    Checked only after the JSON paths failed, and on structure alone (a JSON-shaped opener, a fence
    opener that is never closed) so quoted, *emphasized* or ``[WIP]``-prefixed prose titles are
    untouched (#83903)."""
    return raw.startswith(('{"', '["', "[{")) or (raw.startswith("```") and raw.count("```") % 2 == 1)


def _extract_title_text(content: str) -> str:
    """Strict JSON, then a loose JSON scan, then first-line prose (a provider ignoring ``response_format`` still titles).

    A truncated structured payload is dropped rather than handed to the prose fallback: the fragment
    would otherwise be persisted as the session title."""
    if not content:
        return ""
    raw = content.strip()
    fenced = re.match(r"^```(?:json)?\s*(.*?)\s*```$", raw, re.DOTALL)
    if fenced:
        raw = fenced.group(1).strip()
    title = _extract_json_title(raw)
    if title is not None:
        return title
    if _is_truncated_structured_output(raw):
        return ""
    # Prose fallback: scrub <think> blocks so reasoning can't leak into a title.
    try:
        from agent.agent_runtime_helpers import strip_think_blocks
        raw = strip_think_blocks(None, raw).strip()
    except Exception:
        logger.debug("strip_think_blocks unavailable for title output", exc_info=True)
    return _strip_title_prefix(_first_line(raw)).strip("\"'").strip()


def _title_from_reasoning(message: Any) -> str:
    """The ``{"title": ...}`` payload when a reasoning model put it in ``reasoning_content`` / ``reasoning``
    and left ``content`` empty (glm-5 / minimax under ``json_schema``, #82291). Structured extraction only:
    chain-of-thought prose is never a title, so there is no prose fallback here."""
    for field in ("reasoning_content", "reasoning"):
        text = getattr(message, field, None)
        if isinstance(text, str) and text.strip():
            title = _extract_json_title(text.strip())
            if title:
                return title
    return ""


def _clean_title(text: str) -> Optional[str]:
    """Normalize a model-produced title, or None when nothing usable remains."""
    title = _strip_title_prefix(" ".join((text or "").split()).strip("\"'").strip()).rstrip(".!,;:")
    if len(title) > 80:
        title = title[:77].rstrip() + "..."
    return title or None


def _safe_callback(callback: Optional[Callable], args: tuple, log_fmt: str, label: str) -> None:
    """Invoke an optional consumer callback, never raising."""
    try:
        if callback is not None:
            callback(*args)
    except Exception:
        logger.debug(log_fmt, label, exc_info=True)


def _report_failure(failure_callback: Optional[FailureCallback], exc: BaseException, label: str) -> None:
    _safe_callback(failure_callback, ("title generation", exc), "%s failure_callback raised", label)


def _notify_title(title_callback: Optional[TitleCallback], title: str, source: str, label: str) -> None:
    _safe_callback(title_callback, (title, source), "%s callback failed", label)


def _is_provisional_greeting_title(title: str) -> bool:
    """The prompt's greeting placeholder (also "Friendly greeting in chat" and quoted/bracketed variants)."""
    normalized = re.sub(r"^[\W_]+|[\W_]+$", "", title.strip(), flags=re.UNICODE).lower()
    return normalized in (_PROVISIONAL_GREETING_TITLE, _PROVISIONAL_GREETING_TITLE + " in chat")


def _is_prompt_example_echo(title: str) -> bool:
    """Return True when *title* is one of the prompt's own example titles.

    Comparison is case-insensitive after stripping any leading/trailing run of
    non-letter/non-digit characters, so bracket/quote wrappers cannot bypass
    the guard — while ``_clean_title`` keeps brackets for real titles like
    "(WIP) Fix build". Unicode-aware so full-width wrappers are covered too.
    """
    normalized = re.sub(r"^[\W_]+|[\W_]+$", "", title.strip(), flags=re.UNICODE).lower()
    return normalized in _EXAMPLE_ECHO_REJECT


def generate_title(
    user_message: str,
    timeout: Optional[float] = None,
    failure_callback: Optional[FailureCallback] = None,
    main_runtime: dict = None,
    runtime_validator: Optional[RuntimeValidator] = None,
    title_preview: str | None = None,
) -> Optional[str]:
    """Title from the opening message alone (waiting for the assistant made this slow and bought
    nothing). ``runtime_validator`` runs right before the request; False skips silently.

    If it returns False (e.g. the user's model was switched since the background thread captured its runtime
    snapshot), the call is skipped silently — no request is sent, so a stale title request can't reload a
    model the runtime already unloaded (#19027).
    """
    if not _auto_title_enabled():
        logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false")
        return None
    try:
        if runtime_validator is not None and not runtime_validator():
            logger.debug("Title generation skipped: runtime validator returned False")
            return None
    except Exception:  # fail open: a broken validator must not disable titling
        logger.debug("Title runtime validator raised; proceeding", exc_info=True)
    user_snippet = build_title_input(user_message, title_preview)
    if not user_snippet.strip() or (
        # An attachment-only opener (manual attach, no paste preview) reaches
        # here via auto_title_session, which lacks the instant-title guard:
        # refuse it rather than titling the session after the file path (#92068).
        not title_preview and _attachment_only_opener(user_snippet)
    ):
        return None
    language = _title_language()
    # str.replace, not str.format: the prompt embeds literal JSON braces.
    prompt = _TITLE_PROMPT_TEMPLATE.replace(
        "__LANGUAGE_RULE__", _LANGUAGE_RULE_PINNED.format(language=language) if language else _LANGUAGE_RULE_MATCH_USER,
    )
    try:
        # Use the provider's default temperature instead of forcing 0.3.
        # Some models (e.g. GPT-5.6) only accept their server-side default
        # and reject explicit temperature values, causing the daemon title
        # thread to fail with "Unsupported value: 'temperature'".
        # See: #72351, #51083, #51157
        response = call_llm(
            task="title_generation",
            messages=[{"role": "system", "content": prompt}, {"role": "user", "content": user_snippet}],
            # A title is a handful of tokens, but 64 was cut mid-JSON by fenced/prefixed replies and by
            # reasoning models whose thinking survives the disable below (#83903, #82291). A model that
            # honours the JSON contract stops after ~15 tokens regardless, so the ceiling only costs on
            # replies that would have been garbage anyway. temperature=None: omitted from the wire so
            # default-only reasoning models accept the first request (#72351).
            max_tokens=TITLE_MAX_TOKENS, temperature=None, timeout=timeout, main_runtime=main_runtime,
            extra_body={"response_format": _TITLE_RESPONSE_FORMAT},
            # The module contract above promises thinking-disabled operation,
            # but nothing enforced it: with the aux default reasoning_effort
            # "" (provider default), Gemini enables internal thinking and
            # bills thought tokens against max_tokens=64 — the JSON payload
            # never lands, and the prose fallback stores the opening fence
            # ("```json") as the session title (#91927).
            reasoning_config={"enabled": False},
        )
        message = response.choices[0].message
        title = _clean_title(_extract_title_text(message.content or "") or _title_from_reasoning(message))
        # Answer-shaped output guard: titling is a 3-7 word task, so a title with many words is a model that
        # ignored the task and answered the user's message instead ("I don't have context on X — that's not
        # something I recognize..."). Truncating would store half an assistant blob as the session title,
        # which is still an assistant blob — reject instead so the caller retries on the next exchange
        # (maybe_auto_title retries a placeholder title through the third exchange). Port of can1357/oh-my-pi#7306.
        if title is not None and len(title.split()) > _MAX_TITLE_WORDS:
            # Answer-shaped output: reject (not truncate) so the caller retries next exchange.
            logger.debug("Rejecting answer-shaped title output (%d words > %d)", len(title.split()), _MAX_TITLE_WORDS)
            return None
        # Example-echo guard: a title that parrots one of the prompt's own
        # examples back verbatim says nothing about the session — reject it so
        # the instant derived title (a slice of the user's actual words)
        # survives instead. Exact match after wrapper-stripping, deliberately
        # not fuzzy, so a genuinely topical title that merely resembles an
        # example still passes. Wrappers are stripped for the comparison only
        # ("(Fix login button on mobile)" is the same canned echo as the bare
        # example). Port of QwenLM/qwen-code#9709.
        if title is not None and _is_prompt_example_echo(title):
            logger.debug("Rejecting prompt-example echo title: %r", title)
            return None
        return title
    except Exception as e:
        # WARNING so it shows in agent.log without debug mode; stack at debug.
        logger.warning("Title generation failed: %s", e)
        logger.debug("Title generation traceback", exc_info=True)
        _report_failure(failure_callback, e, "Title generation")
        return None


def _has_upgraded_title(session_db, session_id: str) -> bool:
    """True when the session already carries an ``llm``/``user`` title (or the check fails)."""
    try:
        source_fn = getattr(session_db, "get_session_title_source", None)
        if source_fn is not None:
            return source_fn(session_id) not in (None, "derived")
        return bool(session_db.get_session_title(session_id))
    except Exception:
        return True


def _persist_session_title(session_db, session_id, title, *, source, dedupe=True):
    """Persist at *source* authority via ``set_auto_title`` (precedence check + write in one
    transaction, so a manual ``/title`` is never overwritten); None when a higher authority held the row.
    ``ValueError`` = unique-title index collision → append ``#N`` via ``get_next_title_in_lineage``;
    ``dedupe=False`` re-raises instead (the derived title is on the critical path, collides constantly
    on "hi", and the model replaces it a second later anyway).

    ``ValueError`` means the name is taken by an unrelated session (the unique-title index); rather than
    leave the session untitled (#50537), append a ``#N`` suffix via ``get_next_title_in_lineage``.
    """
    auto_fn = getattr(session_db, "set_auto_title", None)

    def _set(candidate):
        if auto_fn is not None:
            if auto_fn(session_id, candidate, source=source):
                return candidate
            logger.debug("Skipping %s title: a higher-authority title already holds session %s", source, session_id)
            return None
        legacy_fn = getattr(session_db, "set_auto_title_if_empty", None)  # older store without provenance
        if legacy_fn is not None:
            return candidate if legacy_fn(session_id, candidate) else None
        if session_db.set_session_title(session_id, candidate) is False:
            raise RuntimeError(f"session {session_id} not found when storing title")
        return candidate

    try:
        return _set(title)
    except ValueError:
        next_title_fn = getattr(session_db, "get_next_title_in_lineage", None)
        deduped = next_title_fn(title) if dedupe and next_title_fn is not None else None
        if not deduped or deduped == title:
            raise
        return _set(deduped)


def apply_subagent_title(session_db, session_id: str, goal: str) -> Optional[str]:
    """Title a delegate run ``Subagent: <goal's first line>`` at ``derived`` authority. No model call:
    runs fan out in bulk, and the prefix alone is what tells them apart from conversations wherever
    ``sessions.show_subagents`` lists them (#97202). Collisions get ``#N``. Never raises."""
    try:
        derived = derive_title(goal) if is_titleable_user_message(goal) else None
        return _persist_session_title(session_db, session_id, f"Subagent: {derived}", source="derived") if derived else None
    except Exception:
        logger.debug("Subagent title failed for %s", session_id, exc_info=True)
        return None


def apply_instant_title(
    session_db, session_id: str, user_message: str, title_callback: Optional[TitleCallback] = None,
    title_preview: str | None = None,
) -> Optional[str]:
    """Write the derived title inline. Returns it, or None (no usable text, or a ``derived``+ title exists). Never raises.

    ``title_preview`` must reach this stage too: the model upgrade's own ``derive_title`` fallback writes
    ``derived`` provenance, which never replaces the ``derived`` title written here."""
    if not session_db or not session_id:
        return None
    try:
        title = derive_title(user_message, title_preview) if is_titleable_user_message(user_message) else None
        persisted = _persist_session_title(session_db, session_id, title, source="derived", dedupe=False) if title else None
        if persisted:
            _notify_title(title_callback, persisted, "derived", "Instant-title")
        return persisted
    except Exception:
        logger.debug("Instant title failed", exc_info=True)
        return None


def auto_title_session(
    session_db,
    session_id: str,
    user_message: str,
    failure_callback: Optional[FailureCallback] = None,
    main_runtime: dict = None,
    title_callback: Optional[TitleCallback] = None,
    runtime_validator: Optional[RuntimeValidator] = None,
    title_preview: str | None = None,
) -> None:
    """Generate and store the model title (daemon-thread target); skips sessions already carrying an
    ``llm``/``user`` title (a ``derived`` one is expected — upgrading it is the point). Never lets an
    exception escape (the threading excepthook would spray a traceback into the terminal); the canonical
    trigger is the post-``hermes update`` window where lazy imports read NEW source against OLD modules."""
    try:
        if not session_db or not session_id or _has_upgraded_title(session_db, session_id):
            return
        # This thread starts AFTER the turn's ambient context was reset; republish it so the call carries
        # the same Portal ``conversation=`` tag (root-of-lineage) and bills usage to this session.
        from agent.aux_accounting import set_accounting_context
        from agent.portal_tags import set_conversation_context
        conversation_id = session_id
        with suppress(Exception):
            conversation_id = session_db.get_conversation_root(session_id) or session_id
        set_conversation_context(conversation_id)
        # Same for the accounting context, so the title call's token usage is recorded against this session
        # (task='title_generation', #23270).
        set_accounting_context(session_db, session_id)
        title, source = generate_title(
            user_message, failure_callback=failure_callback, main_runtime=main_runtime,
            runtime_validator=runtime_validator, title_preview=title_preview,
        ), "llm"
        if title and _is_provisional_greeting_title(title):
            source = "derived"
        if not title:  # the inline attempt declined collisions; off the critical path the lineage scan is affordable
            title, source = derive_title(user_message, title_preview), "derived"
        if not title:
            return
        try:
            persisted = _persist_session_title(session_db, session_id, title, source=source)
        except Exception as e:
            logger.debug("Failed to set auto-generated title: %s", e)
            return
        if persisted is not None:
            logger.debug("Auto-generated session title: %s", persisted)
            _notify_title(title_callback, persisted, source, "Auto-title")
    except Exception as e:
        # WARNING so operators see it in agent.log; names the likely cause.
        logger.warning("Auto-title failed (harmless; if this started after an update, restart the running Hermes process): %s", e)
        logger.debug("Auto-title traceback", exc_info=True)
        _report_failure(failure_callback, e, "Auto-title")


def _is_real_user_turn(message: Any) -> bool:
    """A question a person actually asked (Hermes persists machinery under ``role="user"``)."""
    if not isinstance(message, dict) or message.get("role") != "user":
        return False
    content = message.get("content")
    return is_titleable_user_message(content if isinstance(content, str) else flatten_message_text(content))


def _session_is_untitled(session_db, session_id: str) -> bool:
    """No title of any provenance; False when it can't tell (no model call per turn for an unreadable title)."""
    getter = getattr(session_db, "get_session_title", None)
    try:
        return callable(getter) and not str(getter(session_id) or "").strip()
    except Exception:
        logger.debug("Untitled check failed for %s", session_id, exc_info=True)
        return False


def _kanban_task_title() -> Optional[str]:
    """Kanban worker: the card's title, or ``Kanban task <id>`` when the board can't be read; None elsewhere
    (including delegate_task children of the worker, which inherit the env var but are not the card)."""
    task_id = (os.environ.get("HERMES_KANBAN_TASK") or "").strip()
    if not task_id or not is_dispatcher_owned_worker_context():
        return None
    try:
        from hermes_cli import kanban_db, kanban_db_connect
        from hermes_state import SessionDB
        with kanban_db_connect.connect_closing() as conn:
            task = kanban_db.get_task(conn, task_id)
        title = " ".join((task.title or "").split()) if task is not None else ""
        # Cards have no length cap; the title store rejects past MAX_TITLE_LENGTH (and the ``#N``
        # retry suffix needs room), which would leave the worker nameless.
        cap = SessionDB.MAX_TITLE_LENGTH - 4
        if len(title) > cap:
            title = title[: cap - 1].rstrip() + "…"
    except Exception:
        logger.debug("Kanban task %s unreadable; naming the session after its id", task_id, exc_info=True)
        title = ""
    return title or f"Kanban task {task_id}"


def maybe_auto_title(
    session_db,
    session_id: str,
    user_message: str,
    conversation_history: Optional[list] = None,
    failure_callback: Optional[FailureCallback] = None,
    main_runtime: dict = None,
    title_callback: Optional[TitleCallback] = None,
    runtime_validator: Optional[RuntimeValidator] = None,
    title_preview: str | None = None,
) -> Optional[threading.Thread]:
    """Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model.

    Returns the upgrade thread: already started, or — when ``title_upgrade_must_wait_for_turn`` — left
    UNSTARTED for the caller to hand to ``start_title_upgrade`` once the turn's model request settled."""
    if not session_db or not session_id or not user_message:
        return None
    # History may be pre- or post-message. Past the opening turn, skip once the session holds an
    # ``llm``/``user`` name: count alone left a machinery-opened session nameless, and a ``derived``
    # name is still a placeholder (instant slice, or the model's greeting title for a bare "hi") that
    # the first substantive turn should replace. Untitled sessions always get another shot; a
    # placeholder gets turns 2-3, so a failing title model costs at most three calls, not one per turn.
    user_msg_count = sum(1 for m in (conversation_history or []) if _is_real_user_turn(m))
    if user_msg_count > 1 and (
        _has_upgraded_title(session_db, session_id)
        or (user_msg_count > 3 and not _session_is_untitled(session_db, session_id))
    ):
        return None
    kanban_title = _kanban_task_title()
    if kanban_title:
        # The card already carries a human-written name; an auxiliary model call per spawned worker
        # only competes with the worker for capacity (#111166). Final (``llm``) authority: nothing
        # upgrades it later, and a manual ``/title`` still wins inside ``set_auto_title``.
        with suppress(Exception):
            persisted = _persist_session_title(session_db, session_id, kanban_title, source="llm")
            if persisted:
                _notify_title(title_callback, persisted, "llm", "Kanban task title")
        return None
    if not is_titleable_user_message(user_message):
        return None
    if not _auto_title_enabled():  # config read after the cheap guards so the file isn't touched every turn
        logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false")
        return None
    apply_instant_title(session_db, session_id, user_message, title_callback, title_preview=title_preview)
    if not _model_title_upgrade_enabled():
        logger.debug("Instant title persisted; model upgrade disabled by auxiliary.title_generation.model_upgrade_enabled=false")
        return None
    # The thread must resolve auxiliary.title_generation (config, provider key, language) for the
    # profile whose turn this is: a bare Thread starts with an empty context and lands on the launch
    # profile under multiplex, titling X's session with the default profile's model and billing its key.
    from agent.memory_provider import spawn_context_thread
    upgrade_kwargs = dict(failure_callback=failure_callback, main_runtime=main_runtime, title_callback=title_callback,
                          runtime_validator=runtime_validator)
    if isinstance(title_preview, str) and title_preview.strip():
        upgrade_kwargs["title_preview"] = title_preview
    upgrade = spawn_context_thread(
        auto_title_session, name="auto-title",
        args=(session_db, session_id, user_message),
        kwargs=upgrade_kwargs,
    )
    if title_upgrade_must_wait_for_turn(main_runtime):
        logger.debug("Auto-title upgrade deferred past the turn: shares the self-hosted endpoint with the main request")
        return upgrade
    start_title_upgrade(upgrade)
    return upgrade
