"""Config, env var and provider custom-endpoint dashboard routes.

Extracted from ``hermes_cli.web_server``; helpers/state that tests monkeypatch on
``web_server`` stay there and are late-bound (cycle-safe).
"""

import contextlib
import logging
import re
import asyncio
import time
import urllib.parse
from fastapi import APIRouter
from hermes_cli.web_routers._common import (
    REDACTED_CREDENTIAL_WRITE_DETAIL, http_failure, is_redacted_credential_preview,
    redacted_credential_preview, scoped_to_thread,
)
from hermes_cli.web_deps import LateState, late
from hermes_cli.web_server_config import (
    _apply_main_model_assignment, _denormalize_config_from_web, _normalize_config_for_web, _schema_with_dynamic_provider_options,
    _validated_main_model_selection,
)
from hermes_cli.web_server_profiles import (
    _approval_mode_of, _broadcast_gateway_session_info, _is_other_profile, _parse_model_entries,
)
from fastapi import HTTPException, Request
from hermes_cli.config import DEFAULT_CONFIG, OPTIONAL_ENV_VARS, read_raw_config, require_readable_config_before_write, custom_endpoint_key_env, coerce_provider_id, find_provider_entry, get_compatible_custom_providers, _ENV_REF_RE, _deep_merge
from hermes_cli.config_providers import _canonical_api_mode, _custom_provider_entry_to_provider_config
from hermes_cli.web_models import ConfigUpdate, EnvVarUpdate, EnvVarDelete, EnvVarReveal, CustomEndpointUpdate
from typing import Any, Dict, List, Optional, Tuple

_log = logging.getLogger("hermes_cli.web_server")
config_router = APIRouter()
router = APIRouter()

# Late-bound so a test's monkeypatch on the owning module wins at call time.
_channel_managed_env_keys = late("_channel_managed_env_keys", "hermes_cli.web_server_messaging")
_config_profile_scope = late("_config_profile_scope", "hermes_cli.web_server_profiles")
_profile_scope = late("_profile_scope", "hermes_cli.web_server_profiles")
_require_token = late("_require_token")
load_config = late("load_config", "hermes_cli.config")
load_env = late("load_env", "hermes_cli.config")
remove_env_value = late("remove_env_value", "hermes_cli.config")
save_config = late("save_config", "hermes_cli.config")
save_env_value = late("save_env_value", "hermes_cli.config")
_CONFIG_MUTATION_LOCK = LateState("_CONFIG_MUTATION_LOCK")

# Simple rate limiter for the reveal endpoint
_reveal_timestamps: List[float] = []
_REVEAL_MAX_PER_WINDOW = 5
_REVEAL_WINDOW_SECONDS = 30

# Display order for tabs — unlisted categories sort alphabetically after these.
_CATEGORY_ORDER = [
    "general", "agent", "terminal", "display", "delegation",
    "memory", "compression", "security", "browser", "voice",
    "tts", "stt", "logging", "discord", "auxiliary",
]


@contextlib.contextmanager
def _env_write_errors(log_msg: str):
    """``ValueError`` -> 400 with its message (save/remove_env_value reject
    invalid names and denylisted keys — LD_PRELOAD, PATH, PYTHONPATH, …, the
    credential lifecycle rejects keys a managed install or administrator pins,
    and the SPA needs the reason, not an opaque 500); ``HTTPException`` (the
    profile scope's 404 for an unknown ``?profile=``) passes through; anything
    else is logged and becomes 500 "Internal server error"."""
    try:
        yield
    except HTTPException:
        raise
    except ValueError as exc:
        raise HTTPException(status_code=400, detail=str(exc)) from exc
    except Exception:
        _log.exception(log_msg)
        raise HTTPException(status_code=500, detail="Internal server error")


@config_router.get("/api/config")
async def get_config(profile: Optional[str] = None, include_defaults: bool = True):
    # _profile_scope blocks on the process-wide _SKILLS_PROFILE_LOCK and
    # load_config() reads from disk; a slow lock-holder on the event loop froze
    # the whole gateway for >1s. asyncio.to_thread copies the contextvar
    # context, so the profile override stays scoped to the worker thread.
    # Opt in to saved values so clients can distinguish user choices from defaults.
    config = await scoped_to_thread(
        profile, lambda: _normalize_config_for_web(load_config() if include_defaults else read_raw_config())
    )
    # Strip internal keys that the frontend shouldn't see or send back
    return {k: v for k, v in config.items() if not k.startswith("_")}


@config_router.get("/api/config/defaults")
async def get_defaults():
    return DEFAULT_CONFIG


@config_router.get("/api/config/schema")
async def get_schema(profile: Optional[str] = None):
    # Discovery-driven provider options (voice command providers + memory
    # provider plugins) are merged per-request so providers added after server
    # start still show up, scoped to the requested profile's config.
    with _config_profile_scope(profile):
        fields = _schema_with_dynamic_provider_options()
    return {"fields": fields, "category_order": _CATEGORY_ORDER}


@config_router.get("/api/egress/status")
async def get_egress_status(profile: Optional[str] = None):
    """Dashboard/Desktop-readable egress proxy status and remediation text."""
    from hermes_cli.proxy_cli import format_status_text
    with _config_profile_scope(profile):  # reads the profile's ``proxy:`` config block
        return {"text": format_status_text()}


@router.put("/api/config")
async def update_config(
    body: ConfigUpdate, profile: Optional[str] = None, preserve_language: bool = False
):
    def _run():
        approvals_mode_changed = False
        with _profile_scope(body.profile or profile):
            # The dashboard form is schema-driven; root keys absent from the
            # schema (``custom_providers``, ``agent.personalities``, ...) are not
            # in the PUT body, so deep-merge incoming over disk rather than
            # full-replace — the frontend can only overwrite what it sends.
            with _CONFIG_MUTATION_LOCK:
                # Strict read: the merge below builds a new dict, so a swallowed read error here
                # would save the PUT body alone over the whole file.
                existing = require_readable_config_before_write()
                incoming = _denormalize_config_from_web(body.config)
                merged = _deep_merge(existing, incoming)
                # Compare normalized approvals.mode across the in-memory
                # documents, not config blocks and not cache re-reads: the page
                # PUTs the defaulted GET record while disk holds sparse YAML (a
                # block compare is always-unequal), and a post-save reload can
                # serve the pre-save cache on an (mtime_ns, size) collision.
                # Only approvals.mode feeds session.info, so it is the trigger.
                approvals_mode_changed = _approval_mode_of(merged) != _approval_mode_of(existing)
                # Explicit English must survive default stripping: an absent
                # language lets the desktop follow the OS on its next launch.
                # Ordinary settings saves include merged defaults, not a choice.
                save_config(
                    merged, preserve_keys={("display", "language")} if preserve_language else None
                )
        # REST saves bypass the config.set RPC (which re-emits itself), so
        # refresh live sessions' cached approval/YOLO indicators after a mode
        # change. Own-profile saves only: a profile-scoped save targets a
        # different HERMES_HOME than this process's gateway sessions.
        if approvals_mode_changed and not _is_other_profile(body.profile or profile):
            _broadcast_gateway_session_info()
        return {"ok": True}

    with http_failure("PUT /api/config failed", 500, detail="Internal server error"):
        return await asyncio.to_thread(_run)


def _provider_card(d, description: str, url, *, is_password: bool, advanced: bool) -> dict:
    return {
        "provider": d.slug, "provider_label": d.label, "description": description, "url": url,
        "is_password": is_password, "advanced": advanced, "category": "provider",
    }


_AUTH_TYPE_ENV_VARS = {
    "aws_sdk": (
        ("AWS_REGION", lambda d, var: f"{d.label} ({var})"),
        ("AWS_PROFILE", lambda d, var: f"{d.label} ({var})"),
    ),
    "vertex": (("VERTEX_CREDENTIALS_PATH", lambda d, var: f"{d.label} — service account JSON path (or use ADC)"),),
}


def _catalog_provider_env_metadata() -> dict:
    """Map provider env vars -> desktop card metadata, derived from the catalog.

    Returns ``{env_var: {provider, provider_label, description, url, is_password,
    advanced}}`` for every API-key provider in the unified ``provider_catalog()``
    (the ``hermes model`` universe). When multiple providers intentionally share
    one env var, ``provider_profiles`` preserves every provider identity while
    the legacy singular fields keep describing the first provider. Hand
    ``OPTIONAL_ENV_VARS`` prose is layered on top in the endpoint; this only
    supplies membership + grouping + fallbacks.
    """
    try:
        from hermes_cli.provider_catalog import provider_catalog
    except Exception:
        return {}

    # Env vars declared with a NON-provider category (e.g. the shared
    # GITHUB_TOKEN, a Skills-Hub "tool" credential) must not be promoted into a
    # provider card even when a provider (Copilot) lists them as auth aliases.
    _non_provider_keys = {
        k for k, v in OPTIONAL_ENV_VARS.items()
        if (v or {}).get("category") and (v or {}).get("category") != "provider"
    }

    meta: dict = {}

    def _profile(entry: dict) -> dict:
        """Return the provider-specific part of a shared credential row."""
        return {
            "provider": entry["provider"],
            "provider_label": entry["provider_label"],
            "description": entry["description"],
            "url": entry["url"],
            "primary": bool(entry.get("provider_primary")),
        }

    def _add_provider_env(env_var: str, entry: dict) -> None:
        """Add one provider without discarding peers that share ``env_var``."""
        existing = meta.get(env_var)
        if existing is None:
            meta[env_var] = entry
            return
        if existing.get("provider") == entry.get("provider"):
            return
        profiles = existing.setdefault("provider_profiles", [_profile(existing)])
        if not any(profile.get("provider") == entry.get("provider") for profile in profiles):
            profiles.append(_profile(entry))

    for d in provider_catalog():
        if d.tab != "keys":
            continue
        # API-key vars: the first is the primary (password) field; aliases are
        # kept as additional password fields so users can clear them too.
        for index, env_var in enumerate(d.api_key_env_vars):
            if env_var in _non_provider_keys:
                continue  # don't hijack a shared tool/messaging credential
            entry = _provider_card(
                d, d.description, d.signup_url or None, is_password=True, advanced=False,
            )
            entry["provider_primary"] = index == 0
            _add_provider_env(
                env_var,
                entry,
            )
        # Base-URL override is an advanced, non-secret field for the same card.
        if d.base_url_env_var:
            meta.setdefault(
                d.base_url_env_var,
                _provider_card(d, f"{d.label} base URL override", None, is_password=False, advanced=True),
            )

        # Providers without api_key_env_vars would otherwise be invisible on
        # the Keys tab: AWS-SDK providers (Bedrock) authenticate via the AWS
        # credential chain, Vertex via OAuth2 (service-account JSON path or
        # ADC — a path, not a secret). Tag their env vars to the card.
        for var, describe in _AUTH_TYPE_ENV_VARS.get(d.auth_type, ()):
            existing = meta.get(var, {})
            meta[var] = _provider_card(
                d, existing.get("description") or describe(d, var), existing.get("url"),
                is_password=False, advanced=existing.get("advanced", True),
            )
    return meta


@router.get("/api/env")
async def get_env_vars(profile: Optional[str] = None):
    # _profile_scope takes _SKILLS_PROFILE_LOCK and load_env()/catalog
    # discovery read from disk — keep the whole build off the event loop.
    return await asyncio.to_thread(_get_env_vars_sync, profile)


def _get_env_vars_sync(profile: Optional[str] = None):
    with _profile_scope(profile):
        env_on_disk = load_env()
    channel_keys = _channel_managed_env_keys()
    catalog_meta = _catalog_provider_env_metadata()

    def _row(var_name: str, info: dict, *, custom: bool = False) -> dict:
        value = env_on_disk.get(var_name)
        cat_meta = catalog_meta.get(var_name) or {}
        # Hand OPTIONAL_ENV_VARS prose wins where present; the catalog fills any
        # gaps (description/url) and always supplies provider grouping hints.
        return {
            "is_set": bool(value),
            "redacted_value": redacted_credential_preview(value),
            "description": info.get("description") or cat_meta.get("description", ""),
            "url": info.get("url") if info.get("url") is not None else cat_meta.get("url"),
            "category": info.get("category") or cat_meta.get("category", ""),
            "is_password": info.get("password", cat_meta.get("is_password", False)),
            "tools": info.get("tools", []),
            "advanced": info.get("advanced", cat_meta.get("advanced", False)),
            # Messaging-platform credential owned by a Channels page card; the
            # Keys/Env page hides it rather than duplicate the richer UI.
            "channel_managed": var_name in channel_keys,
            # Provider grouping from the unified catalog, so the desktop groups
            # by the SAME provider identity the CLI `hermes model` picker uses.
            "provider": cat_meta.get("provider", ""),
            "provider_label": cat_meta.get("provider_label", ""),
            # One credential can intentionally serve multiple built-in routes.
            # Preserve those identities so Desktop can render distinct cards
            # that edit the same underlying env var.
            "provider_profiles": cat_meta.get("provider_profiles", []),
            # The provider's own index-0 credential flag. Desktop picks a card's
            # main "Paste key" field from this FIRST, so a shared alias that a
            # peer profile contributes (DASHSCOPE_API_KEY for the CN Coding /
            # Token Plan cards) can never re-point the card's primary field.
            "provider_primary": bool(cat_meta.get("provider_primary", False)),
            # True for a .env key in no catalog at all — an arbitrary/custom var
            # the user added directly, listed so the Keys page can manage it.
            "custom": custom,
        }

    result = {}
    for var_name, info in OPTIONAL_ENV_VARS.items():
        result[var_name] = _row(var_name, info)
    # Catalog provider env vars with no hand entry in OPTIONAL_ENV_VARS.
    for var_name in catalog_meta:
        if var_name not in result:
            result[var_name] = _row(var_name, {})
    # Custom keys from .env: always "set" (on disk), treated as secrets by
    # default (is_password=True -> redacted, reveal-gated) since an
    # unrecognised key could hold anything. Channel-managed credentials belong
    # to the Channels page. A key added via "add a custom key" round-trips here.
    for var_name in env_on_disk:
        if var_name in result or var_name in channel_keys:
            continue
        row = _row(var_name, {}, custom=True)
        row["category"] = "custom"
        row["is_password"] = True
        result[var_name] = row
    return result


@router.put("/api/env")
async def set_env_var(body: EnvVarUpdate, profile: Optional[str] = None):
    # Unified credential lifecycle: writes .env AND reconciles any config.yaml
    # mirror still holding the previous value of this var (model.api_key /
    # auxiliary.*.api_key / custom_providers[*]), so a rotation can't leave a
    # stale higher-precedence copy that keeps authenticating with the old key.
    # Display-only previews (sentinel or legacy mask) must never gain write authority.
    if is_redacted_credential_preview(body.value):
        raise HTTPException(status_code=400, detail=REDACTED_CREDENTIAL_WRITE_DETAIL)
    with _env_write_errors("PUT /api/env failed"):
        return await scoped_to_thread(body.profile or profile, lambda: _save_env_credential(body.key, body.value, body.provider_setup))


def _save_env_credential(key: str, value: str, provider_setup: bool = False) -> Any:
    """Save under the request's profile scope; a new provider API key also counts as a provider setup."""
    from hermes_cli.config import load_env
    from hermes_cli.credential_lifecycle import save_provider_env_credential
    from hermes_cli.observability.shared_metrics_setup import record_api_key_saved, web_setup_surface

    previous = load_env().get(key)
    result = save_provider_env_credential(key, value)
    record_api_key_saved(key, value, previous, web_setup_surface(), connecting=provider_setup)
    return result


# Live credential probes keyed by env var: (url, auth) where auth is "bearer"
# (Authorization header) or "query" (?key=). A cheap read-only call that 401s
# on a bad token — enough to catch a mistyped key before it's persisted.
# Providers absent here (or local endpoints) are not network-validated; the
# client treats those as "unknown".
_CREDENTIAL_PROBES: dict[str, tuple[str, str]] = {
    "OPENROUTER_API_KEY": ("https://openrouter.ai/api/v1/key", "bearer"),
    "OPENAI_API_KEY": ("https://api.openai.com/v1/models", "bearer"),
    "XAI_API_KEY": ("https://api.x.ai/v1/models", "bearer"),
    "GEMINI_API_KEY": ("https://generativelanguage.googleapis.com/v1beta/models", "query"),
}


def _custom_endpoint_id(raw: str, fallback: str = "custom") -> str:
    slug = re.sub(r"[^A-Za-z0-9_-]+", "-", coerce_provider_id(raw)).strip("-_").lower()
    return slug or fallback


def _resolve_custom_endpoint_entry(providers: Any, endpoint_id: str) -> Tuple[Any, Optional[Dict[str, Any]]]:
    """Resolve a custom endpoint id using the stored key first, then its legacy slug.

    The list route hands Desktop the literal ``providers.<key>`` (a v11→v12
    migration keeps dots/colons from the display name: ``local-127.0.0.1:8283``;
    hand-written keys keep their case), so that spelling must round-trip
    unchanged. Slugging is only the compatibility path for callers that still
    send an unslugged display name.
    """
    stored_key, entry = find_provider_entry(providers, endpoint_id)
    if entry is not None:
        return stored_key, entry
    normalized_key = _custom_endpoint_id(endpoint_id)
    if normalized_key == endpoint_id:
        return None, None
    return find_provider_entry(providers, normalized_key)


def _models_from_custom_endpoint_entry(entry: Dict[str, Any]) -> List[str]:
    models: List[str] = []
    raw_models = entry.get("models")
    if isinstance(raw_models, (dict, list)):
        models.extend(str(model).strip() for model in raw_models)

    default_model = str(entry.get("model") or entry.get("default_model") or "").strip()
    if default_model:
        models.insert(0, default_model)

    seen: set[str] = set()
    return [model for model in models if model and not (model in seen or seen.add(model))]


def _api_key_display(entry: Dict[str, Any]) -> Tuple[bool, Optional[str]]:
    """Return ``(has_api_key, preview)`` for a provider or model config block.

    Keys live in ``.env`` behind ``key_env``; only older entries still carry a
    plaintext ``api_key``. Checking both keeps the panel honest either way.

    See #69449.
    """
    plaintext = str(entry.get("api_key") or "").strip()
    if plaintext:
        return True, redacted_credential_preview(plaintext)
    key_env = str(entry.get("key_env") or "").strip()
    if key_env:
        return True, f"${{{key_env}}}"
    return False, None


def _raw_provider_api_key(endpoint_id: str) -> Any:
    """The on-disk (un-expanded) ``api_key`` of a providers entry, or ``None``."""
    _stored, entry = find_provider_entry(read_raw_config().get("providers"), endpoint_id)
    return entry.get("api_key") if isinstance(entry, dict) else None


def _config_api_key_is_env_ref(endpoint_id: str) -> bool:
    """True when this endpoint's on-disk ``api_key`` is a ``${VAR}`` template.

    ``load_config()`` expands env refs, so a hand-written ``api_key: ${MY_KEY}``
    is indistinguishable from a literal secret by the time it reaches us. Such
    an entry already keeps its secret out of config.yaml, so migrating it would
    only copy that secret into a second env var the user didn't ask for.
    """
    raw_key = _raw_provider_api_key(endpoint_id)
    return bool(isinstance(raw_key, str) and re.search(r"\$\{[^}]+\}", raw_key))


_DESKTOP_API_MODES = {"chat_completions", "codex_responses", "anthropic_messages"}


def _endpoint_api_mode(entry: Dict[str, Any]) -> str:
    """The transport a providers entry pins (``api_mode``, or the v12 migration's ``transport``
    spelling), canonicalized; ``""`` = runtime auto-detect. Mirrors the read order of
    ``runtime_provider_custom._get_named_custom_provider``."""
    raw = str(entry.get("api_mode") or entry.get("transport") or "")
    mode = _canonical_api_mode(raw).lower()
    return mode if mode in _DESKTOP_API_MODES else ""


def _endpoint_row(
    endpoint_id: str, name: str, base_url: str, model: str, models: List[str], context_length,
    discover_models: bool, key_entry: Dict[str, Any], is_current: bool, source: str,
) -> Dict[str, Any]:
    has_api_key, api_key_preview = _api_key_display(key_entry)
    return {
        "id": endpoint_id, "name": name, "base_url": base_url, "model": model, "models": models,
        "api_mode": _endpoint_api_mode(key_entry),
        "context_length": context_length, "discover_models": discover_models,
        "has_api_key": has_api_key, "api_key_preview": api_key_preview,
        "is_current": is_current, "source": source,
    }


def _model_names_provider(model_cfg: Dict[str, Any], provider_key: str, entry: Optional[Dict[str, Any]]) -> bool:
    """True when ``model.provider`` points at this ``providers`` entry.

    ``switch_model`` spells the active provider either as the stored key or as
    ``custom:<lowercased name>``; the list's ``is_current`` and delete's mirror
    detach must accept both, or a mixed-case key activates but never shows as
    active.
    """
    names = {coerce_provider_id(provider_key).lower()}
    if isinstance(entry, dict) and coerce_provider_id(entry.get("name")):
        names.add(coerce_provider_id(entry.get("name")).lower())
    current = str(model_cfg.get("provider") or "").strip().lower()
    return current.removeprefix("custom:") in names


def _custom_endpoint_response(cfg: Dict[str, Any]) -> Dict[str, Any]:
    model_cfg = cfg.get("model", {}) if isinstance(cfg.get("model"), dict) else {}
    current_provider = str(model_cfg.get("provider", "") or "")
    current_model = str(model_cfg.get("default", model_cfg.get("name", "")) or "")
    current_base_url = str(model_cfg.get("base_url", "") or "")

    endpoints: List[Dict[str, Any]] = []
    providers = cfg.get("providers")
    if isinstance(providers, dict):
        for provider_id, raw_entry in providers.items():
            if not isinstance(raw_entry, dict):
                continue
            base_url = str(raw_entry.get("base_url") or raw_entry.get("url") or raw_entry.get("api") or "").strip()
            if not base_url:
                continue
            endpoint_id = str(provider_id)
            models = _models_from_custom_endpoint_entry(raw_entry)
            endpoints.append(_endpoint_row(
                endpoint_id, str(raw_entry.get("name") or endpoint_id), base_url,
                str(raw_entry.get("model") or raw_entry.get("default_model") or (models[0] if models else "")),
                models, raw_entry.get("context_length"), bool(raw_entry.get("discover_models", True)),
                raw_entry, _model_names_provider(model_cfg, endpoint_id, raw_entry), "providers",
            ))

    # Legacy ``custom_providers:`` list entries the migration left behind are
    # still routed at runtime (get_compatible_custom_providers), so they need a
    # row too, or the panel hides an endpoint the agent can pick. Entries from
    # ``providers:`` carry ``provider_key``; the legacy ones do not. A bare
    # ``provider: custom`` main slot is "current" for the legacy row whose
    # base_url it points at.
    is_bare_custom = current_provider.lower() == "custom" and bool(current_base_url)
    seen_ids = {e["id"] for e in endpoints}
    for entry in get_compatible_custom_providers(cfg):
        if entry.get("provider_key"):
            continue
        endpoint_id = _custom_endpoint_id(entry["name"])
        if endpoint_id in seen_ids:
            continue
        seen_ids.add(endpoint_id)
        models = _models_from_custom_endpoint_entry(entry)
        is_current = is_bare_custom and entry["base_url"].rstrip("/") == current_base_url.rstrip("/")
        endpoints.append(_endpoint_row(
            endpoint_id, entry["name"], entry["base_url"],
            str(entry.get("model") or (models[0] if models else "")), models,
            entry.get("context_length"), bool(entry.get("discover_models", True)),
            entry, is_current, "custom_providers",
        ))

    if is_bare_custom and not any(e["id"] == "custom" or e["is_current"] for e in endpoints):
        endpoints.insert(0, _endpoint_row(
            "custom", "Custom", current_base_url, current_model, [current_model] if current_model else [],
            model_cfg.get("context_length"), True, model_cfg, True, "direct-config",
        ))

    return {
        "endpoints": endpoints,
        "current": {
            "provider": current_provider, "model": current_model, "base_url": current_base_url,
        },
    }


def _pop_legacy_custom_provider(cfg: Dict[str, Any], provider_key: str) -> Optional[Dict[str, Any]]:
    """Remove and return the legacy ``custom_providers:`` list entry whose name slugs to *provider_key*."""
    legacy = cfg.get("custom_providers")
    if not isinstance(legacy, list):
        return None
    for index, entry in enumerate(legacy):
        if isinstance(entry, dict) and _custom_endpoint_id(str(entry.get("name") or "")) == provider_key:
            return legacy.pop(index)
    return None


def _detach_main_model_from_provider(cfg: Dict[str, Any], provider_key: str, entry: Optional[Dict[str, Any]] = None) -> None:
    """Drop the main-slot mirror of a provider that no longer exists.

    ``activate_custom_endpoint`` copies the endpoint's ``base_url`` and
    ``api_key`` onto ``model``; that mirror outranks the environment at client
    construction, so deleting the endpoint without clearing it leaves the agent
    authenticating to the deleted host with the deleted key (and the key in
    config.yaml). Only touches ``model`` when it names the deleted provider —
    ``switch_model`` spells that either as the stored key or as
    ``custom:<lowercased name>``, so both spellings count.

    See #62269.
    """
    model_cfg = cfg.get("model")
    if not isinstance(model_cfg, dict) or not _model_names_provider(model_cfg, provider_key, entry):
        return
    for field in ("provider", "base_url", "api_key", "key_env"):
        model_cfg.pop(field, None)
    cfg["model"] = model_cfg


def _write_custom_endpoint(cfg: Dict[str, Any], body: CustomEndpointUpdate) -> Tuple[str, Dict[str, Any]]:
    name = (body.name or "").strip()
    base_url = (body.base_url or "").strip().rstrip("/")
    model = (body.model or "").strip()

    if not name:
        raise HTTPException(status_code=400, detail="name required")
    if not base_url:
        raise HTTPException(status_code=400, detail="base_url required")
    parsed = urllib.parse.urlparse(base_url)
    if not parsed.scheme or not parsed.netloc:
        raise HTTPException(status_code=400, detail="base_url must include scheme and host")
    if not model:
        raise HTTPException(status_code=400, detail="model required")

    # Deliver the bearer token through a named provider entry. A bare ``provider: custom`` cannot carry a
    # credential for this host: OPENAI_API_KEY is deliberately gated to openai.com (#28660), so the token
    # was dropped and requests went out as "no-key-required".
    providers = cfg.get("providers")
    if not isinstance(providers, dict):
        providers = {}
    # An edit payload carries the stored key verbatim; slugging it first would
    # miss the entry and fork a slugged twin next to the original.
    stored_key, existing = _resolve_custom_endpoint_entry(providers, body.id or body.name)
    endpoint_id = coerce_provider_id(stored_key) if existing is not None else _custom_endpoint_id(body.id or body.name)
    if existing is None:
        existing = {}

    # Merge onto the existing entry rather than replacing it: a providers.<name>
    # block can carry hand-written keys the dashboard has no field for
    # (``key_env``/``api_key_env``, ``extra_headers`` — possibly with
    # credentials — ``request_overrides``); rebuilding from scratch silently
    # dropped them on an unrelated edit.
    entry: Dict[str, Any] = dict(existing)
    entry.update({
        "name": name, "base_url": base_url, "model": model,
        "discover_models": bool(body.discover_models),
    })
    # A Responses-only or Anthropic-compatible host 404s on the runtime's
    # Chat Completions default, so the panel pins the transport the same way
    # ``hermes model`` does (``api_mode``; the runtime also reads the v12
    # ``transport`` spelling, so drop it rather than let the two disagree).
    # ``None`` = older UI payload: keep whatever is hand-written. See #93622.
    if body.api_mode is not None:
        entry.pop("transport", None)
        if body.api_mode:
            entry["api_mode"] = body.api_mode
        else:
            entry.pop("api_mode", None)
    # Same for the model map, so existing models keep their context lengths.
    # ``body.models`` is the catalogue the panel's Test button discovered;
    # without it only the hand-typed model survived Save. A payload with no
    # ``models`` (older UI) still ensures the named default is present.
    # See #69988.
    details = {d.id.strip(): d for d in (body.model_details or ()) if d.id.strip()}
    existing_models = entry.get("models")
    models_map: Dict[str, Any] = dict(existing_models) if isinstance(existing_models, dict) else {}
    for candidate in (*(body.models or ()), *details, model):
        model_id = str(candidate).strip()
        if not model_id:
            continue
        current = models_map.get(model_id)
        row = dict(current) if isinstance(current, dict) else {}
        detail = details.get(model_id)
        if detail is not None:
            # Keep the alias metadata ``/v1/models`` advertised so the catalogue
            # still says what ``gpt-5.6-sol-high`` stands for after Save.
            row.update({k: v.strip() for k, v in (("canonical_model", detail.canonical_model),
                                                  ("reasoning_effort", detail.reasoning_effort)) if v and v.strip()})
        models_map[model_id] = row
    entry["models"] = models_map
    # A reasoning alias is not a model the inference route accepts literally:
    # persist the canonical model and pin its effort through the one runtime
    # chokepoint (``agent.reasoning_overrides`` → ``resolve_reasoning_config``).
    alias = details.get(model)
    canonical = (alias.canonical_model or "").strip() if alias is not None else ""
    if canonical and canonical != model:
        from hermes_constants import parse_reasoning_effort
        effort = (alias.reasoning_effort or "").strip().lower()
        if parse_reasoning_effort(effort) is not None:
            agent_cfg = cfg.get("agent") if isinstance(cfg.get("agent"), dict) else {}
            overrides = agent_cfg.get("reasoning_overrides")
            overrides = dict(overrides) if isinstance(overrides, dict) else {}
            overrides[canonical] = effort
            agent_cfg["reasoning_overrides"] = overrides
            cfg["agent"] = agent_cfg
        model = canonical
        entry["model"] = model
        models_map.setdefault(model, {})
    if body.context_length and body.context_length > 0:
        entry["context_length"] = int(body.context_length)
        entry["models"][model]["context_length"] = int(body.context_length)

    # API keys never belong in config.yaml: write to .env and reference it via
    # ``key_env`` — the indirection built-in providers use and that
    # runtime_provider.py resolves at load time.
    # See #69449.
    env_var = custom_endpoint_key_env(endpoint_id)
    submitted_key = body.api_key.strip() if body.api_key is not None else None
    if submitted_key:
        # ``${KEY_ENV}`` is the GET display for key_env entries; the helper covers the
        # sentinel and legacy masks. Either one is display-only, current or stale.
        if _ENV_REF_RE.fullmatch(submitted_key) or is_redacted_credential_preview(submitted_key):
            raise HTTPException(status_code=400, detail=REDACTED_CREDENTIAL_WRITE_DETAIL)
        save_env_value(env_var, submitted_key)
        entry["key_env"] = env_var
        entry.pop("api_key", None)
    elif submitted_key is not None:
        # Blank field means "clear the key", not "leave it alone".
        remove_env_value(env_var)
        entry.pop("key_env", None)
        entry.pop("api_key", None)
    elif str(entry.get("api_key") or "").strip() and not _config_api_key_is_env_ref(endpoint_id):
        # Migrate a plaintext key an earlier release wrote, on the next save,
        # without the user having to re-enter it.
        save_env_value(env_var, entry["api_key"].strip())
        entry["key_env"] = env_var
        entry.pop("api_key", None)

    if stored_key is not None and stored_key != endpoint_id:
        providers.pop(stored_key, None)
    providers[endpoint_id] = entry
    cfg["providers"] = providers

    if body.make_default:
        result = _validated_main_model_selection(cfg, endpoint_id, model, base_url)
        cfg["model"] = _apply_main_model_assignment(cfg.get("model", {}), result)
        if entry.get("key_env") and isinstance(cfg["model"], dict):
            cfg["model"]["key_env"] = entry["key_env"]
            cfg["model"].pop("api_key", None)

    return endpoint_id, entry


@router.get("/api/providers/custom-endpoints")
def list_custom_endpoints(profile: Optional[str] = None):
    """Return configured OpenAI-compatible custom endpoints for Desktop.

    Scoped to the requested profile's config.yaml: the desktop settings UI
    targets the active profile, so read/write must resolve that profile's home
    rather than the process-level HERMES_HOME (mirrors ``/api/config``).
    """
    with http_failure("GET /api/providers/custom-endpoints failed", 500, detail="Failed to list custom endpoints"):
        with _config_profile_scope(profile):
            return _custom_endpoint_response(load_config())


@router.post("/api/providers/custom-endpoints")
def upsert_custom_endpoint(body: CustomEndpointUpdate, profile: Optional[str] = None):
    """Create or update a v12+ ``providers`` custom endpoint entry."""
    with http_failure("POST /api/providers/custom-endpoints failed", 500, detail="Failed to save custom endpoint"):
        # Sync-def endpoints run on worker threads: the load→mutate→save span
        # holds _CONFIG_MUTATION_LOCK so a concurrent config autosave cannot
        # drop this write (or vice versa).
        with _config_profile_scope(profile), _CONFIG_MUTATION_LOCK:
            cfg = load_config()
            providers = cfg.get("providers")
            created = _resolve_custom_endpoint_entry(
                providers if isinstance(providers, dict) else {}, body.id or body.name)[1] is None
            endpoint_id, _entry = _write_custom_endpoint(cfg, body)
            save_config(cfg)
            response = _custom_endpoint_response(cfg)
            from hermes_constants import get_hermes_home
            home = get_hermes_home()
        if created:  # editing an endpoint that exists sets nothing new up
            _record_custom_endpoint_setup(home)
        response["ok"] = True
        response["id"] = endpoint_id
        return response


def _record_custom_endpoint_setup(home: Any) -> None:
    """Counted after the config lock is released (a cold metrics runtime must not stall writers)."""
    from hermes_cli.observability.shared_metrics_setup import record_provider_setup_done, web_setup_surface

    record_provider_setup_done(web_setup_surface(), "custom", hermes_home=home, background=True)


@router.post("/api/providers/custom-endpoints/{endpoint_id}/activate")
def activate_custom_endpoint(endpoint_id: str, profile: Optional[str] = None):
    """Set a configured custom endpoint as the default model provider."""
    with http_failure(
        f"POST /api/providers/custom-endpoints/{endpoint_id}/activate failed", 500,
        detail="Failed to activate custom endpoint",
    ):
        with _config_profile_scope(profile), _CONFIG_MUTATION_LOCK:  # RMW span
            cfg = load_config()
            stored_key, entry = _resolve_custom_endpoint_entry(cfg.get("providers"), endpoint_id)
            if entry is None:
                # A legacy ``custom_providers:`` row: the main slot names providers by
                # key, so promote the entry to ``providers.<key>`` (the v12 shape the
                # migration would have written) before activating it.
                provider_key = _custom_endpoint_id(endpoint_id)
                legacy = _pop_legacy_custom_provider(cfg, provider_key)
                entry = _custom_provider_entry_to_provider_config(legacy, provider_key=provider_key) if legacy else None
                if entry is None:
                    raise HTTPException(status_code=404, detail="custom endpoint not found")
                providers = cfg.get("providers") if isinstance(cfg.get("providers"), dict) else {}
                providers[provider_key] = entry
                cfg["providers"] = providers
            else:
                provider_key = coerce_provider_id(stored_key)

            models = _models_from_custom_endpoint_entry(entry)
            model = str(entry.get("model") or entry.get("default_model") or (models[0] if models else "")).strip()
            base_url = str(entry.get("base_url") or entry.get("api") or "").strip()
            if not model or not base_url:
                raise HTTPException(status_code=400, detail="custom endpoint is incomplete")

            model_cfg = _apply_main_model_assignment(
                cfg.get("model", {}), _validated_main_model_selection(cfg, provider_key, model, base_url))
            if entry.get("key_env"):
                model_cfg["key_env"] = entry["key_env"]
                model_cfg.pop("api_key", None)
            elif entry.get("api_key"):
                # `cfg` is env-expanded, so a raw `${VAR}` api_key would land as
                # plaintext; copy the raw template when that's what's on disk.
                try:
                    _raw_key = str(_raw_provider_api_key(provider_key) or "").strip()
                except Exception:
                    _raw_key = ""
                if _raw_key.startswith("${") and _raw_key.endswith("}"):
                    model_cfg["api_key"] = _raw_key
                else:
                    model_cfg["api_key"] = entry["api_key"]
            cfg["model"] = model_cfg
            save_config(cfg)
        return {"ok": True, "provider": provider_key, "model": model}


@router.delete("/api/providers/custom-endpoints/{endpoint_id}")
def delete_custom_endpoint(endpoint_id: str, profile: Optional[str] = None):
    """Remove a configured custom endpoint from ``providers``."""
    with http_failure(
        f"DELETE /api/providers/custom-endpoints/{endpoint_id} failed", 500,
        detail="Failed to delete custom endpoint",
    ):
        with _config_profile_scope(profile), _CONFIG_MUTATION_LOCK:  # RMW span
            cfg = load_config()
            providers = cfg.get("providers")
            stored_key, entry = _resolve_custom_endpoint_entry(providers, endpoint_id)
            if entry is not None and isinstance(providers, dict):
                provider_key = coerce_provider_id(stored_key)
                providers.pop(stored_key, None)
                cfg["providers"] = providers
            else:
                # A legacy ``custom_providers:`` row is addressed by its slug.
                provider_key = _custom_endpoint_id(endpoint_id)
                if _pop_legacy_custom_provider(cfg, provider_key) is None:
                    raise HTTPException(status_code=404, detail="custom endpoint not found")
            _detach_main_model_from_provider(cfg, provider_key, entry)
            remove_env_value(custom_endpoint_key_env(provider_key))
            save_config(cfg)
            response = _custom_endpoint_response(cfg)
        response["ok"] = True
        return response


@router.post("/api/providers/custom-endpoints/validate")
async def validate_custom_endpoint(body: CustomEndpointUpdate):
    """Probe a custom endpoint by calling its OpenAI-compatible /models URL."""
    base_url = (body.base_url or "").strip().rstrip("/")
    if not base_url:
        return {"ok": False, "reachable": True, "message": "Enter an endpoint URL first.", "models": []}

    headers = {"Accept": "application/json"}
    if body.api_key and body.api_key.strip():
        headers["Authorization"] = f"Bearer {body.api_key.strip()}"

    resolved, resp = await _probe_openai_compatible_models(base_url, headers)
    if resp is None:
        return {"ok": False, "reachable": False, "message": f"Could not reach {base_url}/models.", "models": []}
    if resp.status_code in (401, 403):
        return {"ok": False, "reachable": True, "message": "The endpoint rejected the API key.", "models": []}
    if not resp.is_success:
        return {"ok": False, "reachable": True, "message": f"Endpoint returned HTTP {resp.status_code}.", "models": []}
    # ``models`` stays the bare id list older clients read; ``model_details`` keeps the
    # alias metadata (``canonical_model`` / ``reasoning_effort``) the id list flattens.
    entries = _parse_model_entries(resp)
    ids = [e["id"] for e in entries]
    # /models answering proves nothing about the transport the runtime will POST to:
    # a Responses-only host lists models fine and 404s every /chat/completions (#93622).
    # Probe the route the saved mode (or the runtime's URL auto-detect) actually uses, on the
    # base that actually served /models (#65488) — that is the URL the runtime will persist.
    mode = _canonical_api_mode(body.api_mode or "").lower() or _auto_api_mode(resolved)
    probe_model = (body.model or "").strip() or (ids[0] if ids else "")
    try:
        async with _endpoint_probe_client(resolved, 8.0) as client:
            missing = await _probe_transport_route(client, resolved, mode, probe_model, headers)
    except Exception:
        missing = ""  # inconclusive (see _probe_transport_route): never block on a transport error

    result = {"ok": True, "reachable": True, "message": "", "models": ids, "model_details": entries,
              "transport_checked": mode, "resolved_base_url": resolved}
    if missing:
        result.update(ok=False, message=missing)
    return result

async def _probe_openai_compatible_models(base_url: str, headers: Optional[dict]) -> Tuple[str, Any]:
    """GET ``{base}/models``, then ``{base}/v1/models`` (or the ``/v1``-stripped variant) when the
    first answers a non-success. Returns ``(resolved_base_url, response)`` — the base that served the
    model list is what the caller must PERSIST: the runtime appends ``/chat/completions`` to the saved
    URL verbatim, so a bare host root that only "detected" via ``/v1/models`` would 404 every chat
    (#65488). ``response`` is None when no candidate could be reached at all."""
    base = base_url.rstrip("/")
    alternate = base[:-3].rstrip("/") if base.lower().endswith("/v1") else base + "/v1"
    resolved, resp = base, None
    async with _endpoint_probe_client(base, 8.0) as client:
        for candidate in (base, alternate):
            try:
                candidate_resp = await client.get(candidate + "/models", headers=headers)
            except Exception:
                continue
            # Keep the most telling failure: a 401/403 from the /v1 alternate says "server is
            # there, key rejected", which beats the typed root's 404 (wrong path).
            if resp is None or candidate_resp.is_success or resp.status_code == 404:
                resolved, resp = candidate, candidate_resp
            if candidate_resp.is_success:
                break
    return resolved, resp


_TRANSPORT_ROUTES = {"chat_completions": "/chat/completions", "codex_responses": "/responses",
                     "anthropic_messages": "/messages"}
_TRANSPORT_LABELS = {"chat_completions": "Chat Completions", "codex_responses": "Responses API",
                     "anthropic_messages": "Anthropic Messages"}


def _auto_api_mode(base_url: str) -> str:
    """The transport the runtime falls back to for an endpoint without a pinned ``api_mode``
    (same resolver as ``runtime_provider_custom._custom_runtime``)."""
    from hermes_cli.runtime_provider import _detect_api_mode_for_url
    return _detect_api_mode_for_url(base_url) or "chat_completions"


async def _probe_transport_route(client, base_url: str, mode: str, model: str, headers: Dict[str, str]) -> str:
    """POST a 1-token request to ``mode``'s route; return a failure message when the host does
    not serve it (404/405/501), ``""`` otherwise. Any other status — 200, 400 (bad body), 401,
    422, 429 — means the route exists, which is all the check needs to know; a network error or
    timeout (a local server still loading the model) is inconclusive and does not block."""
    route = _TRANSPORT_ROUTES.get(mode)
    if route is None:
        return ""
    if mode == "anthropic_messages":
        payload = {"model": model, "max_tokens": 1, "messages": [{"role": "user", "content": "hi"}]}
        token = headers.get("Authorization", "").removeprefix("Bearer ")
        headers = {**headers, "anthropic-version": "2023-06-01", **({"x-api-key": token} if token else {})}
    elif mode == "codex_responses":
        payload = {"model": model, "input": "hi", "max_output_tokens": 16}
    else:
        payload = {"model": model, "max_tokens": 1, "messages": [{"role": "user", "content": "hi"}]}
    try:
        resp = await client.post(base_url + route, json=payload, headers=headers)
    except Exception:
        return ""
    if resp.status_code not in (404, 405, 501):
        return ""
    return (f"{base_url}/models answered, but POST {route} returned HTTP {resp.status_code}: this host "
            f"does not serve the {_TRANSPORT_LABELS[mode]} API. Pick the API mode it does serve.")


def _endpoint_probe_client(url: str, timeout: float):
    """httpx client for a user-entered endpoint probe. Local endpoints (loopback, LAN, Tailscale)
    ignore ``HTTP(S)_PROXY``: httpx honours the env/system proxy but not its bypass list, so a
    system proxy (Clash on Windows, corporate) answered the ``127.0.0.1`` probe with its own error
    page and the GUI reported "advertised no models" while the CLI saw the model (#63472)."""
    import httpx
    from agent.model_metadata import is_local_endpoint
    return httpx.AsyncClient(timeout=httpx.Timeout(timeout), trust_env=not is_local_endpoint(url))


@router.post("/api/providers/validate")
async def validate_provider_credential(body: EnvVarUpdate, request: Request):
    """Live-probe a provider credential before it's saved.

    Returns {ok, reachable, message}. ok=True means the provider accepted the
    key; ok=False + reachable=True means the key is bad (caller should block);
    reachable=False means the network probe couldn't run (caller may save with
    a warning rather than hard-blocking offline users).
    """
    _require_token(request)
    import httpx

    key = (body.key or "").strip()
    value = (body.value or "").strip()
    if not value:
        return {"ok": False, "reachable": True, "message": "Enter a value first."}

    # Local / custom endpoint: validate connectivity, not auth — any HTTP
    # response (even 401) proves the endpoint is up. Also surface the model ids
    # it advertises (OpenAI ``/v1/models`` shape) so the GUI can auto-pick a
    # default. The optional API key is sent so servers that require auth on
    # ``/v1/models`` still enumerate instead of returning an empty list.
    if key == "OPENAI_BASE_URL":
        api_key = (body.api_key or "").strip()
        headers = {"Authorization": f"Bearer {api_key}"} if api_key else None
        resolved, resp = await _probe_openai_compatible_models(value, headers)
        url = resolved + "/models"
        if resp is None:
            return {"ok": False, "reachable": False, "message": f"Could not reach {url}."}
        entries = _parse_model_entries(resp)
        models = [e["id"] for e in entries]
        if not models and not resp.is_success:
            # A proxy/gateway error page parses as "no models"; name the status instead so the
            # GUI does not tell the user to "start a model" on a server that answered.
            return {"ok": False, "reachable": True, "message": f"{url} answered HTTP {resp.status_code}.", "models": []}
        return {"ok": True, "reachable": True, "message": "", "models": models, "model_details": entries,
                "resolved_base_url": resolved}

    probe = _CREDENTIAL_PROBES.get(key)
    if not probe:
        # No probe for this provider — can't validate, don't block.
        return {"ok": True, "reachable": False, "message": ""}

    url, auth = probe
    if key == "GEMINI_API_KEY":
        from agent.gemini_native_adapter import normalize_gemini_base_url
        # Normalize guarantees the version segment; the key itself never decides the surface —
        # AQ. keys exist for both AI Studio and Vertex express mode (#115306).
        url = normalize_gemini_base_url(url.rsplit("/models", 1)[0]) + "/models"
    headers = {"Accept": "application/json"}
    params = {}
    if auth == "bearer":
        headers["Authorization"] = f"Bearer {value}"
    else:
        params["key"] = value

    try:
        async with httpx.AsyncClient(timeout=httpx.Timeout(10.0)) as client:
            resp = await client.get(url, headers=headers, params=params)
    except Exception:
        return {"ok": False, "reachable": False, "message": "Could not reach the provider to verify the key."}

    if resp.status_code in (401, 403):
        return {"ok": False, "reachable": True, "message": "That API key was rejected. Double-check it and try again."}
    if resp.status_code == 429 or resp.is_success:
        # 429 = key is valid but rate-limited; success = valid.
        return {"ok": True, "reachable": True, "message": ""}
    return {"ok": False, "reachable": True, "message": f"Provider returned HTTP {resp.status_code} for this key."}


@router.delete("/api/env")
async def remove_env_var(body: EnvVarDelete, profile: Optional[str] = None):
    # Unified credential lifecycle: clears the .env entry AND every mirror of
    # the credential — env-seeded credential_pool entries in auth.json (stale
    # ones kept providers alive in the model picker), the affected providers'
    # model-cache rows, and value-matched config.yaml api_key mirrors.
    # OAuth/device-code/manual pool entries for the same provider are preserved.
    with _env_write_errors("DELETE /api/env failed"):
        from hermes_cli.credential_lifecycle import remove_provider_env_credential

        result = await scoped_to_thread(
            body.profile or profile, lambda: remove_provider_env_credential(body.key)
        )
        if not result.get("found"):
            raise HTTPException(status_code=404, detail=f"{body.key} not found in .env")
        return result


@router.post("/api/env/reveal")
async def reveal_env_var(
    body: EnvVarReveal, request: Request, profile: Optional[str] = None
):
    """Return the real (unredacted) value of a single env var.

    Protected by the ephemeral session token (per server start, injected into
    the SPA), rate limiting (max 5 reveals per 30s window) and audit logging.
    """
    _require_token(request)

    now = time.time()
    cutoff = now - _REVEAL_WINDOW_SECONDS
    _reveal_timestamps[:] = [t for t in _reveal_timestamps if t > cutoff]
    if len(_reveal_timestamps) >= _REVEAL_MAX_PER_WINDOW:
        raise HTTPException(status_code=429, detail="Too many reveal requests. Try again shortly.")
    _reveal_timestamps.append(now)

    env_on_disk = await scoped_to_thread(body.profile or profile, load_env)
    value = env_on_disk.get(body.key)
    if value is None:
        raise HTTPException(status_code=404, detail=f"{body.key} not found in .env")

    _log.info("env/reveal: %s", body.key)
    return {"key": body.key, "value": value}
