"""Regression tests for issue #30963 — partial-stream stub finish_reason.

Pins the contract:

- text-only partial stream → stub.finish_reason == "length" so the
  conversation loop's existing length-continuation path can keep the
  agent moving against an unfinished goal.
- partial mid-tool-call → stub.finish_reason == "length" so the loop
  triggers continuation machinery with targeted chunking guidance
  instead of ending the turn immediately.
- conversation_loop's length-continuation prompt distinguishes a real
  output-length truncation from a partial-stream-stub network error
  via response.id.
"""

from __future__ import annotations

from types import SimpleNamespace
from unittest.mock import MagicMock, patch

import pytest

from hermes_constants import PARTIAL_STREAM_STUB_ID, FINISH_REASON_LENGTH
from agent.conversation_loop import _join_truncated_parts


# ── Helpers (mirrors test_streaming.py) ────────────────────────────────────

def _make_stream_chunk(content=None, tool_calls=None, finish_reason=None):
    delta = SimpleNamespace(
        content=content, tool_calls=tool_calls,
        reasoning_content=None, reasoning=None,
    )
    choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason)
    return SimpleNamespace(choices=[choice], model=None, usage=None)


def _make_tool_call_delta(index=0, tc_id=None, name=None, arguments=None):
    func = SimpleNamespace(name=name, arguments=arguments)
    return SimpleNamespace(index=index, id=tc_id, function=func)


def _make_agent():
    from run_agent import AIAgent
    agent = AIAgent(
        api_key="test-key",
        base_url="https://example.com/v1",
        model="test/model",
        quiet_mode=True,
        skip_context_files=True,
        skip_memory=True,
    )
    agent.api_mode = "chat_completions"
    agent._interrupt_requested = False
    return agent


# ── Stub finish_reason ────────────────────────────────────────────────────

class TestPartialStreamStubFinishReason:
    """The stub returned by interruptible_streaming_api_call when the
    upstream connection dies mid-flight."""

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_text_only_partial_returns_length(self, _mock_close, mock_create, monkeypatch):
        """#30963: text-only partials must classify as length so the loop
        keeps continuing instead of exiting with budget remaining."""

        def _stalling_stream():
            yield _make_stream_chunk(content="Here's my answer so far")
            raise RuntimeError("simulated upstream stall")

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = lambda *a, **kw: _stalling_stream()
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._current_streamed_assistant_text = "Here's my answer so far"

        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH, (
            "Text-only partial streams must use finish_reason=length so the "
            "conversation loop continues from where the network died "
            "(issue #30963)."
        )
        assert response.choices[0].message.content == "Here's my answer so far"
        assert response.choices[0].message.tool_calls is None
        assert getattr(response, "_clean_eof", False) is False, (
            "A stub built after a real transport exception (RuntimeError "
            "here) must NOT be tagged _clean_eof — that tag is reserved "
            "for a stream that ended with no exception and no "
            "finish_reason, a distinct failure class (#102766)."
        )


class TestTerminalChunkFenceException:
    """A superseded writer must still accept the provider's terminal
    finish_reason chunk. Fending that chunk leaves finish_reason None
    after real text was delivered, which the drop-guard mislabels as a
    mid-stream drop even though the provider completed the stream.
    """

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_superseded_writer_accepts_finish_reason_chunk(
        self, _mock_close, mock_create, monkeypatch,
    ):
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        agent_box = {}

        class SupersedeBeforeFinish:
            response = SimpleNamespace(headers={})

            def __iter__(self):
                yield _make_stream_chunk(content="Long prose that is complete.")
                agent_box["agent"]._claim_stream_writer()
                # Marker-only terminal chunk (empty delta), as vLLM emits.
                yield _make_stream_chunk(finish_reason="stop")

        mock_client = MagicMock()
        mock_client.chat.completions.create.return_value = SupersedeBeforeFinish()
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent_box["agent"] = agent
        response = agent._interruptible_streaming_api_call({})

        assert response.id != PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == "stop"
        assert response.choices[0].message.content == "Long prose that is complete."

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_superseded_writer_still_fences_further_content(
        self, _mock_close, mock_create, monkeypatch,
    ):
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        agent_box = {}

        class SupersedeBeforeMoreText:
            response = SimpleNamespace(headers={})

            def __iter__(self):
                yield _make_stream_chunk(content="kept ")
                agent_box["agent"]._claim_stream_writer()
                # A False accept_chunk ends consumption; this text must
                # never reach the accumulator, and the later finish chunk
                # is never seen (the fence still stops *further* content).
                yield _make_stream_chunk(content="must-not-append")
                yield _make_stream_chunk(finish_reason="stop")

        mock_client = MagicMock()
        mock_client.chat.completions.create.return_value = SupersedeBeforeMoreText()
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent_box["agent"] = agent
        response = agent._interruptible_streaming_api_call({})

        content = response.choices[0].message.content or ""
        assert "must-not-append" not in content
        assert "kept" in content
        assert response.id == PARTIAL_STREAM_STUB_ID

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_genuine_truncation_without_finish_still_drops(
        self, _mock_close, mock_create, monkeypatch,
    ):
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")

        def _truncated():
            yield _make_stream_chunk(content="cut off with no terminal chunk")

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = lambda *a, **kw: _truncated()
        mock_create.return_value = mock_client

        agent = _make_agent()
        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
        assert response._clean_eof is True, (
            "The stream ended with no exception and no finish_reason — "
            "that is the clean-EOF class the fix for #102766 must tag "
            "distinctly from a genuine transport drop."
        )


# ── Clean stream-end mid-tool-call (no exception, no finish_reason) ─────────

class TestCleanStreamEndMidToolCall:
    """The upstream closes the SSE stream cleanly after delivering a tool
    name + the opening '{' of its arguments — NO exception, NO finish_reason,
    NO [DONE].  Observed live on NVIDIA Nemotron Ultra via the Nous dedicated
    endpoint: it stalls/drops during large tool-arg generation.

    The mock-builder must NOT stamp this as finish_reason='length' (which
    routes it through the max_tokens-boost truncation path and finally
    reports the misleading 'Response truncated due to output length limit').
    It must route through the partial-stream-stub path so the loop reports
    an honest mid-tool-call drop and asks the model to chunk its output.
    """

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_no_finish_reason_partial_tool_args_routes_to_stub(
        self, _mock_close, mock_create, monkeypatch,
    ):
        def _clean_ending_stream():
            # Reasoning + tool name + the lone opening brace, then the
            # generator simply RETURNS (StopIteration) — no raise, no
            # finish_reason chunk, no [DONE].
            yield _make_stream_chunk(content="\n")
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, tc_id="call_x", name="execute_code"),
            ])
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, arguments="{"),
            ])
            # falls off the end — clean close, no terminator

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _clean_ending_stream()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._fire_stream_delta = lambda text: None

        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID, (
            "A clean stream-end mid tool-call (no finish_reason) must be "
            "tagged as a partial-stream stub, not a 'stream-<uuid>' "
            "truncation — otherwise the loop reports the false 'output "
            "length limit' error."
        )
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
        assert response.choices[0].message.tool_calls is None, (
            "Incomplete tool args must never auto-execute."
        )
        assert getattr(response, "_dropped_tool_names", None) == ["execute_code"]
        assert response._clean_eof is True, (
            "No exception was raised and no finish_reason ever arrived — "
            "this is the clean-EOF class the fix for #102766 must tag "
            "distinctly from a genuine transport drop."
        )


# ── Clean stream-end before any argument byte arrives (#80498) ─────────────

class TestCleanStreamEndBeforeAnyToolArgs:
    """The upstream closes the SSE stream cleanly right after delivering the
    tool NAME — not a single byte of the arguments delta ever arrived, no
    exception, no finish_reason, no [DONE].

    Before the fix, an empty ``arguments`` string skipped the
    truncated-JSON check entirely (it only ran when ``arguments and
    arguments.strip()``), so ``has_truncated_tool_args`` stayed False. With
    no other guard catching this shape, the stub-builder fell through to
    ``effective_finish_reason = finish_reason or "stop"`` and returned a
    normal "stop" turn carrying a tool call with ``arguments=""`` — which
    the dispatch boundary silently coerces to "{}" and executes with no
    retry (#80498, e.g. ``write_file`` running with no arguments).
    """

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_empty_tool_args_routes_to_stub_not_silent_empty_object(
        self, _mock_close, mock_create, monkeypatch,
    ):
        def _clean_ending_stream():
            # Tool name arrives, then the generator simply RETURNS
            # (StopIteration) before any arguments delta chunk — no raise,
            # no finish_reason chunk, no [DONE].
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, tc_id="call_x", name="write_file"),
            ])
            # falls off the end — clean close, no terminator, zero args bytes

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _clean_ending_stream()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._fire_stream_delta = lambda text: None

        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID, (
            "A tool call whose arguments never started streaming before a "
            "clean stream end must be tagged as a partial-stream stub, not "
            "silently returned as a completed 'stop' turn with empty "
            "arguments (#80498)."
        )
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
        assert response.choices[0].message.tool_calls is None, (
            "A tool call with zero argument bytes delivered must never "
            "auto-execute with a silently substituted empty object."
        )
        assert getattr(response, "_dropped_tool_names", None) == ["write_file"]


# ── Mixed response: one complete call, one dropped (#80498) ─────────────────

class TestMixedToolCallsOneDroppedOneComplete:
    """A single response can carry more than one tool call in parallel. If
    the stream dies before one call's arguments ever start while a sibling
    call in the SAME response already completed with valid, parseable
    arguments, the whole response is still discarded as one partial-stream
    stub — ``_build_partial_stream_stub`` unconditionally sets
    ``tool_calls=None``, so the valid sibling is not selectively kept.

    This all-or-nothing behavior predates #80498 (it already applied to the
    partial-but-nonempty-JSON case); this test just locks it in now that the
    zero-byte case routes through the same stub path, so a future change
    towards per-tool-call granularity doesn't silently regress either case.
    """

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_valid_sibling_tool_call_is_discarded_with_dropped_one(
        self, _mock_close, mock_create, monkeypatch,
    ):
        def _clean_ending_stream():
            # Tool call 0: complete, valid JSON arguments.
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, tc_id="call_ok", name="read_file"),
            ])
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, arguments='{"path": "a.txt"}'),
            ])
            # Tool call 1: name arrives, then the stream dies before a
            # single argument byte — the #80498 zero-byte case.
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=1, tc_id="call_x", name="write_file"),
            ])
            # falls off the end — clean close, no terminator

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _clean_ending_stream()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._fire_stream_delta = lambda text: None

        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
        assert response.choices[0].message.tool_calls is None, (
            "The valid 'read_file' call must not be selectively kept — the "
            "whole response is discarded as one stub so the retry "
            "regenerates every call in it together."
        )
        assert getattr(response, "_dropped_tool_names", None) == [
            "read_file", "write_file",
        ], (
            "_dropped_tool_names lists every tool call in the response, "
            "not just the one that actually failed to complete."
        )


# ── Length-continuation prompt branching ──────────────────────────────────

class TestLengthContinuationAssembly:
    def test_distinct_continuation_is_preserved(self):
        assert _join_truncated_parts([
            ("The first half ends here", True),
            ("and the second half adds new information.", False),
        ]) == (
            "The first half ends here\n"
            "and the second half adds new information."
        )

    def test_repeated_tail_is_trimmed_without_losing_new_text(self):
        repeated = "A sufficiently long repeated transition sentence ends here."

        assert _join_truncated_parts([
            (f"Existing answer. {repeated}", True),
            (f"{repeated} New inbound details remain visible.", False),
        ]) == (
            f"Existing answer. {repeated} New inbound details remain visible."
        )



# ── Integration: live conversation loop ───────────────────────────────────

@pytest.fixture()
def loop_agent():
    """AIAgent with a mocked OpenAI client (mirrors test_run_agent's fixture)
    so we can stage a stub + continuation pair on .chat.completions.create."""
    from run_agent import AIAgent
    with (
        patch("model_tools.get_tool_definitions", return_value=[]),
        patch("model_tools.check_toolset_requirements", return_value={}),
        patch("agent.process_bootstrap.OpenAI"),
    ):
        a = AIAgent(
            api_key="test-key-1234567890",
            base_url="https://openrouter.ai/api/v1",
            quiet_mode=True,
            skip_context_files=True,
            skip_memory=True,
        )
        a.client = MagicMock()
        a._cached_system_prompt = "You are helpful."
        a._use_prompt_caching = False
        a.compression_enabled = False
        a.save_trajectories = False
        return a


class TestConversationLoopPartialStreamContinuation:
    """End-to-end: a partial-stream stub feeds the loop and the loop
    asks for continuation instead of exiting with finish_reason=stop."""

    def test_partial_stream_stub_does_not_exit_loop_immediately(self, loop_agent):
        """The stub from chat_completion_helpers used to exit the loop with
        text_response(finish_reason=stop). Now finish_reason=length routes
        through length_continue_retries — the loop persists the partial
        content and asks the model to continue."""

        from tests.agent.test_run_agent import _mock_response, _mock_assistant_msg

        # First API call: the partial-stream stub (length on partial-stream-stub id).
        repeated_tail = (
            "**Conclusion**: the project remains on standby until the stable "
            "release. Nothing changes."
        )
        partial_stub = SimpleNamespace(
            id=PARTIAL_STREAM_STUB_ID,
            model="test/model",
            choices=[SimpleNamespace(
                index=0,
                message=_mock_assistant_msg(
                    content=f"The first half of the answer is forty-two.\n\n{repeated_tail}"
                ),
                finish_reason=FINISH_REASON_LENGTH,
            )],
            usage=None,
        )
        # Second API call: the model restarts from the complete-looking tail
        # even though the nudge said not to repeat it, then stops cleanly.
        continuation = _mock_response(
            content=repeated_tail, finish_reason="stop",
        )

        loop_agent.client.chat.completions.create.side_effect = [
            partial_stub, continuation,
        ]

        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            result = loop_agent.run_conversation("ask me something")

        # The loop made TWO API calls (stub + continuation), not one.
        assert loop_agent.client.chat.completions.create.call_count == 2, (
            "Partial-stream-stub must trigger a continuation API call, not "
            "exit the loop after one call."
        )
        # The continuation prompt the loop appended must be the network-error
        # variant, not the "output length limit" lie — otherwise the model
        # no-ops with "I wasn't truncated, I'm done."
        # We assert it indirectly by inspecting the second-call kwargs.
        second_call_kwargs = loop_agent.client.chat.completions.create.call_args_list[1]
        msgs = second_call_kwargs.kwargs.get("messages") or second_call_kwargs.args[0].get("messages")
        last_user = next(
            (m for m in reversed(msgs) if m.get("role") == "user"), None,
        )
        assert last_user is not None
        assert "network error mid-stream" in (last_user.get("content") or ""), (
            "Continuation prompt for partial-stream-stub must mention the "
            "network error, not the 'output length limit'."
        )

        # And the final response stitches both halves together.
        assert "first half of" in result["final_response"]
        assert "forty-two" in result["final_response"]
        assert result["final_response"].count(repeated_tail) == 1

    def test_output_limit_continuation_preserves_intentional_repetition(self, loop_agent):
        from tests.agent.test_run_agent import _mock_response

        repeated = "This intentionally repeated sentence is longer than thirty-two characters."
        first = _mock_response(
            content=f"First copy: {repeated}", finish_reason=FINISH_REASON_LENGTH,
        )
        continuation = _mock_response(content=repeated, finish_reason="stop")
        loop_agent.client.chat.completions.create.side_effect = [first, continuation]

        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            result = loop_agent.run_conversation("repeat this sentence twice")

        assert loop_agent.client.chat.completions.create.call_count == 2
        assert result["final_response"].count(repeated) == 2


class TestContentFilterStallActivatesFallback:
    """Regression for #32421: a provider output-layer content safety filter
    (e.g. MiniMax ``output new_sensitive (1027)``) terminates a streaming
    response mid-delivery.  The raw error is swallowed into a
    finish_reason=length partial-stream stub, so before the fix the loop
    burned 3 continuation retries against the SAME primary (re-hitting the
    content-deterministic filter every time) and gave up with
    ``"Response remained truncated after 3 continuation attempts"`` — the
    configured fallback chain was never consulted.

    The fix has three layers:
      1. error_classifier classifies ``new_sensitive`` as
         ``content_policy_blocked``.
      2. interruptible_streaming_api_call runs the swallowed error through
         that classifier and stamps the stub ``_content_filter_terminated``.
      3. the conversation loop reads the tag and activates fallback BEFORE
         burning any continuation retries.
    """

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_streaming_call_tags_content_filter_stub(
        self, _mock_close, mock_create, monkeypatch,
    ):
        """Layer 2: the real streaming path stamps _content_filter_terminated
        when the swallowed error matches a content-filter pattern."""

        def _minimax_stall():
            yield _make_stream_chunk(content="Writing the file: ")
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, tc_id="call_1", name="write_file"),
            ])
            yield _make_stream_chunk(tool_calls=[
                _make_tool_call_delta(index=0, arguments='{"path": "/tmp/x", '),
            ])
            raise RuntimeError("output new_sensitive (1027) [MiniMax-M2.7]")

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _minimax_stall()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._fire_stream_delta = lambda text: None
        agent._current_streamed_assistant_text = "Writing the file: "

        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID
        assert getattr(response, "_content_filter_terminated", False) is True, (
            "MiniMax new_sensitive stream stall must tag the stub so the loop "
            "can route to fallback (#32421)."
        )


    def test_tagged_stub_activates_fallback_first_pass(self, loop_agent):
        """Layer 3: a tagged stub activates fallback on the FIRST pass, with
        zero continuation retries burned, and the fallback provider then
        completes the turn."""
        from tests.agent.test_run_agent import _mock_assistant_msg, _mock_response

        def _filter_stub():
            return SimpleNamespace(
                id=PARTIAL_STREAM_STUB_ID,
                model="minimax/MiniMax-M2.7",
                choices=[SimpleNamespace(
                    index=0,
                    message=_mock_assistant_msg(content="Writing the file..."),
                    finish_reason=FINISH_REASON_LENGTH,
                )],
                usage=None,
                _dropped_tool_names=["write_file"],
                _content_filter_terminated=True,
            )

        recovery = _mock_response(
            content="Done on the fallback provider.", finish_reason="stop",
        )
        loop_agent.client.chat.completions.create.side_effect = [
            _filter_stub(), recovery,
        ]
        loop_agent._fallback_chain = [
            {"provider": "openrouter", "model": "anthropic/claude-sonnet-4.7"},
        ]
        loop_agent._fallback_index = 0
        fb_calls = {"n": 0}

        def _fake_activate(reason=None):
            fb_calls["n"] += 1
            loop_agent._fallback_index = len(loop_agent._fallback_chain)
            return True

        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
            patch.object(loop_agent, "_try_activate_fallback",
                         side_effect=_fake_activate),
        ):
            result = loop_agent.run_conversation("write me a long file")

        assert fb_calls["n"] == 1, (
            "Content-filter-tagged stub must activate fallback exactly once, "
            "on the first pass — not after exhausting continuation retries."
        )
        assert result["final_response"] == "Done on the fallback provider."
        assert result["completed"] is True


class TestEmptyPartialStreamStubNotPersisted:
    """Regression for the session-poisoning bug hit with moonshotai/kimi-k3
    via OpenRouter (2026-07-20): a stream dropped mid-``write_file`` tool
    call before ANY text was delivered.  The partial-stream-stub carries
    ``content=""`` and ``tool_calls=None``, so the loop's truncation path
    took the "no tool calls" branch and appended
    ``{"role": "assistant", "content": ""}`` to history before the
    continuation user-message.  Moonshot rejects empty assistant content
    ("the message at position N with role 'assistant' must not be empty")
    with HTTP 400 on the very next replay — and since the message is
    persisted, EVERY subsequent turn re-fails: session unrecoverable.

    Fix layer 1 (conversation_loop): an empty partial-stream stub must not
    be appended as an interim assistant message — only the continuation
    user-message is.
    """

    def test_empty_stub_only_appends_continuation_user_message(self, loop_agent):
        from tests.agent.test_run_agent import _mock_response, _mock_assistant_msg

        # First API call: empty partial-stream stub — stream died mid
        # tool-call args with zero text delivered.
        empty_stub = SimpleNamespace(
            id=PARTIAL_STREAM_STUB_ID,
            model="test/model",
            choices=[SimpleNamespace(
                index=0,
                message=_mock_assistant_msg(content=""),
                finish_reason=FINISH_REASON_LENGTH,
            )],
            usage=None,
            _dropped_tool_names=["write_file"],
        )
        # Second API call: the model answers normally after the nudge.
        recovery = _mock_response(content="Done — wrote it in chunks.",
                                  finish_reason="stop")

        loop_agent.client.chat.completions.create.side_effect = [
            empty_stub, recovery,
        ]

        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            result = loop_agent.run_conversation("make me a webpage")

        assert loop_agent.client.chat.completions.create.call_count == 2

        # Inspect the history replayed on the SECOND call: there must be NO
        # empty-content assistant message anywhere — that is the exact shape
        # Moonshot 400s on.
        second_call_kwargs = loop_agent.client.chat.completions.create.call_args_list[1]
        msgs = second_call_kwargs.kwargs.get("messages") or second_call_kwargs.args[0].get("messages")
        empty_assistants = [
            m for m in msgs
            if m.get("role") == "assistant" and not m.get("content")
        ]
        assert empty_assistants == [], (
            "Empty partial-stream stub must not be persisted as an "
            "empty-content assistant message — strict providers (Moonshot/"
            "Kimi) reject the replay with HTTP 400 and poison the session."
        )

        # The continuation nudge is still appended as a user message, and
        # it's the chunking variant (dropped tool call), not the length lie.
        last_user = next(
            (m for m in reversed(msgs) if m.get("role") == "user"), None,
        )
        assert last_user is not None
        assert "too large" in (last_user.get("content") or "")
        assert "output length limit" not in (last_user.get("content") or "")

        assert result["completed"] is True


class TestBuildAssistantMessageEmptyContentPad:
    """Layer 2 was consolidated into the class owner: the builder stores
    textless turns AS-IS (no write-time pad — a pad here broke codex
    commentary turns and forked the concept).  Wire safety is owned by
    ``repair_empty_non_final_messages`` inside ``sanitize_api_messages``.
    These tests pin the builder's store-as-is contract."""

    def _agent_for_builder(self):
        from run_agent import AIAgent
        with (
            patch("model_tools.get_tool_definitions", return_value=[]),
            patch("model_tools.check_toolset_requirements", return_value={}),
            patch("agent.process_bootstrap.OpenAI"),
        ):
            a = AIAgent(
                api_key="test-key-1234567890",
                base_url="https://openrouter.ai/api/v1",
                quiet_mode=True,
                skip_context_files=True,
                skip_memory=True,
            )
        return a

    def test_empty_content_stored_as_is(self):
        from agent.chat_completion_helpers import build_assistant_message
        from tests.agent.test_run_agent import _mock_assistant_msg

        agent = self._agent_for_builder()
        msg = build_assistant_message(agent, _mock_assistant_msg(content=""), "stop")
        assert msg["content"] == "", (
            "Builder must store textless turns as-is — wire repair is owned "
            "by repair_empty_non_final_messages at the send boundary."
        )
        assert isinstance(msg["timestamp"], float)


    def test_tool_call_turn_content_left_empty(self):
        from agent.chat_completion_helpers import build_assistant_message
        from tests.agent.test_run_agent import _mock_assistant_msg, _mock_tool_call

        agent = self._agent_for_builder()
        msg = build_assistant_message(
            agent,
            _mock_assistant_msg(content="", tool_calls=[_mock_tool_call()]),
            "tool_calls",
        )
        assert msg["content"] == ""
        assert msg["tool_calls"]
        assert isinstance(msg["timestamp"], float)


class TestSendTimeEmptyAssistantPad:
    """Durable repair for ALREADY-poisoned persisted sessions: a partial
    -stream-stub row written by an older build (content:'' ,
    finish_reason:'length') is rebuilt to content:'' on every reload —
    ``_rows_to_conversation`` strips whitespace, so a DB-side pad cannot
    survive.  The class owner ``repair_empty_non_final_messages`` (inside
    ``sanitize_api_messages``, the pre-send chokepoint) must repair the
    empty textless assistant turn at the serialization boundary, so a
    RESUMED poisoned session replays cleanly against strict providers
    (Moonshot/Kimi HTTP 400 "message ... with role 'assistant' must not
    be empty" / Anthropic "all messages must have non-empty content")."""

    def _run_one_turn_with_history(self, loop_agent, history):
        from tests.agent.test_run_agent import _mock_response
        loop_agent.client.chat.completions.create.return_value = _mock_response(
            content="ok", finish_reason="stop",
        )
        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            loop_agent.run_conversation(
                "continue", conversation_history=history,
            )
        kwargs = loop_agent.client.chat.completions.create.call_args_list[0]
        return kwargs.kwargs.get("messages") or kwargs.args[0].get("messages")

    def test_poisoned_resumed_history_repaired_on_send(self, loop_agent):
        # Byte-shape of a persisted poisoned session:
        # user -> assistant('' , finish_reason='length', NO tool_calls) -> user.
        poisoned = [
            {"role": "user", "content": "make me a webpage"},
            {"role": "assistant", "content": "", "finish_reason": "length"},
            {"role": "user", "content": "please proceed"},
        ]
        sent = self._run_one_turn_with_history(loop_agent, poisoned)
        empties = [
            m for m in sent
            if m.get("role") == "assistant"
            and not m.get("tool_calls")
            and not (m.get("content") or "").strip()
        ]
        assert empties == [], (
            "A resumed session carrying a persisted empty partial-stream "
            "stub must be repaired at the send boundary — strict providers "
            "reject the replay with HTTP 400 otherwise."
        )
        stub = next(
            (m for m in sent if m.get("role") == "assistant"
             and not m.get("tool_calls")),
            None,
        )
        assert stub is not None and stub["content"] == "[response interrupted]"

    def test_tool_call_turn_not_padded_on_send(self, loop_agent):
        history = [
            {"role": "user", "content": "search something"},
            {
                "role": "assistant",
                "content": "",
                "tool_calls": [{
                    "id": "call_1", "type": "function",
                    "function": {"name": "web_search", "arguments": "{}"},
                }],
            },
            {"role": "tool", "tool_call_id": "call_1", "content": "result"},
            {"role": "user", "content": "and now?"},
        ]
        sent = self._run_one_turn_with_history(loop_agent, history)
        tc_turn = next(
            (m for m in sent if m.get("role") == "assistant" and m.get("tool_calls")),
            None,
        )
        assert tc_turn is not None
        assert tc_turn["content"] == "", (
            "Tool-call turns are exempt from the pad: content:'' alongside "
            "tool_calls is accepted by every provider and normalizing it "
            "would alter prompt-cache keys."
        )


class TestSendTimePadMultimodalSafety:
    """Regression: the send-time repair must skip non-string (list) assistant
    content instead of crashing — a forked session whose new user turn
    attaches an image hit AttributeError: 'list' object has no attribute
    'strip' inside an earlier pad loop.

    The repair is now owned by ``repair_empty_non_final_messages``, whose
    ``_msg_has_payload`` treats a list with any typed block as payload —
    multimodal turns are never rewritten.  This test drives a multimodal
    history through the loop and asserts (a) no crash, and (b) the
    assistant turn's text is neither dropped nor replaced.
    """

    def test_multimodal_assistant_content_not_touched(self, loop_agent):
        from tests.agent.test_run_agent import _mock_response
        multimodal = [
            {"role": "user", "content": "look at this"},
            {"role": "assistant", "content": [
                {"type": "text", "text": "I see an image"},
            ]},
            {"role": "user", "content": [
                {"type": "text", "text": "animate it"},
                {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
            ]},
        ]
        loop_agent.client.chat.completions.create.return_value = _mock_response(
            content="ok", finish_reason="stop",
        )
        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            result = loop_agent.run_conversation(
                "animate it", conversation_history=multimodal,
            )
        assert result["completed"] is True
        kwargs = loop_agent.client.chat.completions.create.call_args_list[0]
        sent = kwargs.kwargs.get("messages") or kwargs.args[0].get("messages")
        # The assistant turn survives with its text intact — regardless of
        # whether upstream passes flattened it to a str or kept the list.
        mm = next(
            m for m in sent
            if m.get("role") == "assistant" and not m.get("tool_calls")
        )
        c = mm["content"]
        if isinstance(c, list):
            assert c == [{"type": "text", "text": "I see an image"}], (
                "Multimodal assistant list content must pass through untouched."
            )
        else:
            assert "I see an image" in (c or ""), (
                "Flattened multimodal assistant text must survive the repair."
            )

    def test_repair_owner_skips_list_content_directly(self):
        """Unit-shape check against the REAL owner: multimodal list content
        (the exact AttributeError shape) passes through untouched; a textless
        str turn is repaired; tool-call turns are exempt."""
        from agent.agent_runtime_helpers import repair_empty_non_final_messages
        api_messages = [
            {"role": "assistant", "content": [{"type": "text", "text": "hi"}]},
            {"role": "assistant", "content": ""},
            {"role": "assistant", "content": "", "tool_calls": [{"id": "c1"}]},
            {"role": "user", "content": "trailing turn keeps the above non-final"},
        ]
        out = repair_empty_non_final_messages(api_messages)
        assert out[0]["content"] == [{"type": "text", "text": "hi"}]
        assert out[1]["content"] == "[response interrupted]"
        assert out[2]["content"] == ""
        # input list untouched (repair is copy-on-write)
        assert api_messages[1]["content"] == ""


class TestPortalLastOneWithoutDone:
    """#90848: complete Portal streams can end with lastOne=true and no
    finish_reason / [DONE]. That is a clean terminal, not a drop."""

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_last_one_usage_frame_is_a_clean_stop(self, _mock_close, mock_create):
        def _portal_stream():
            yield _make_stream_chunk(content="LONGCAT_OK")
            yield SimpleNamespace(
                choices=[],
                model="meituan/longcat-2.0:free",
                usage=SimpleNamespace(prompt_tokens=1, completion_tokens=1),
                lastOne=True,
            )

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _portal_stream()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._fire_stream_delta = lambda text: None
        response = agent._interruptible_streaming_api_call({})

        assert getattr(response, "id", None) != PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == "stop"
        assert response.choices[0].message.content == "LONGCAT_OK"


# ── Trailing usage chunk with no finish_reason (#91373) ───────────────────

class TestStreamIncludeUsageFinalChunk:
    """OpenAI / vLLM streaming with stream_options={'include_usage': True} emits
    a final usage-only chunk with empty choices (choices=[]) and no finish_reason.
    Hermes must recognize that the presence of usage proves the stream completed
    cleanly and must NOT misclassify it as a mid-stream network drop (#91373).
    """

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_text_stream_with_final_usage_chunk_completes_as_stop(
        self, _mock_close, mock_create, monkeypatch,
    ):
        """When text is streamed and the final chunk is a usage-only chunk
        without finish_reason, the completion must complete with finish_reason='stop'
        instead of returning PARTIAL_STREAM_STUB_ID."""
        def _vllm_stream():
            # Chunks 1 and 2: text deltas with finish_reason=None
            yield _make_stream_chunk(content="The final answer is 42.")
            # Chunk 3: trailing usage-only chunk (choices=[], usage present)
            usage = SimpleNamespace(prompt_tokens=100, completion_tokens=10, total_tokens=110)
            yield SimpleNamespace(choices=[], model="vllm/test-model", usage=usage)

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = lambda *a, **kw: _vllm_stream()
        mock_create.return_value = mock_client

        agent = _make_agent()
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id != PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == "stop"
        assert response.choices[0].message.content == "The final answer is 42."
        assert response.usage is not None
        assert response.usage.prompt_tokens == 100
        assert response.usage.completion_tokens == 10

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_text_stream_abrupt_drop_without_usage_still_returns_stub(
        self, _mock_close, mock_create, monkeypatch,
    ):
        """When text is streamed but the connection is severed before any usage
        chunk arrives (usage is None and finish_reason is None), it remains
        correctly classified as a partial stream stub."""
        def _dropped_stream():
            yield _make_stream_chunk(content="Partial text before drop")
            # Stream ends abruptly without finish_reason and without usage

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = lambda *a, **kw: _dropped_stream()
        mock_create.return_value = mock_client

        agent = _make_agent()
        agent._current_streamed_assistant_text = "Partial text before drop"
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
        assert response.choices[0].message.content == "Partial text before drop"

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_reasoning_only_abrupt_drop_without_usage_still_returns_stub(
        self, _mock_close, mock_create, monkeypatch,
    ):
        """A stream severed while still reasoning (only ``delta.reasoning`` frames,
        no finish_reason, no usage) is a drop, not a clean stop: stamping "stop"
        would let the reasoning-only clean-stop promotion surface the truncated
        thought as the final answer instead of entering the continuation ladder."""
        def _dropped_stream():
            for text in ("Let me think about", " the question carefully, first"):
                chunk = _make_stream_chunk()
                chunk.choices[0].delta.reasoning = text
                yield chunk

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = lambda *a, **kw: _dropped_stream()
        mock_create.return_value = mock_client

        agent = _make_agent()
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id == PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
        assert response.choices[0].message.content is None
        assert response.choices[0].message.reasoning_content == (
            "Let me think about the question carefully, first"
        )


# ── Merged-finish content chunk swallowed by the SSE-echo guard (#94614) ──

class TestMergedFinishChunkSurvivesSSEGuard:
    """vLLM >= 0.1.dev20051 merges finish_reason into the final CONTENT
    chunk instead of a separate marker-only chunk. When the SSE-echo guard
    is engaged at that moment (GLM-family tokenizers emit standalone ':'
    tokens mid-prose), the guard's content-shape continue paths swallow the
    terminal chunk and finish_reason is never captured — a complete stream
    is misclassified as a mid-stream drop. Terminal fields must be
    extracted before any content-shape continue."""

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_merged_finish_chunk_after_sse_lookalike_prefix_completes(
        self, _mock_close, mock_create, monkeypatch,
    ):
        def _vllm_merged_finish_stream():
            # ':' then ' uniform' engage the SSE-echo guard (may_be_sse True)
            yield _make_stream_chunk(content=":")
            yield _make_stream_chunk(content=" uniform")
            # Merged-finish terminal content chunk, as the new vLLM emits
            yield _make_stream_chunk(content=".", finish_reason="stop")

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _vllm_merged_finish_stream()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id != PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == "stop"
        assert response.choices[0].message.content == ": uniform."

    @patch("run_agent.AIAgent._create_request_openai_client")
    @patch("run_agent.AIAgent._close_request_openai_client")
    def test_merged_finish_chunk_without_sse_trigger_still_completes(
        self, _mock_close, mock_create, monkeypatch,
    ):
        """Control: the same merged-finish shape without an SSE-lookalike
        prefix must keep completing (guard never engages)."""

        def _plain_merged_finish_stream():
            yield _make_stream_chunk(content="Hello")
            yield _make_stream_chunk(content=".", finish_reason="stop")

        mock_client = MagicMock()
        mock_client.chat.completions.create.side_effect = (
            lambda *a, **kw: _plain_merged_finish_stream()
        )
        mock_create.return_value = mock_client

        agent = _make_agent()
        monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
        response = agent._interruptible_streaming_api_call({})

        assert response.id != PARTIAL_STREAM_STUB_ID
        assert response.choices[0].finish_reason == "stop"
        assert response.choices[0].message.content == "Hello."


class TestRouterRewriteTruncationMessageIsHonest:
    """Regression for #91717: the invalid-JSON truncation detector (the
    secondary path that fires when the primary finish_reason='length' handler
    was bypassed) must NOT blame the output-length limit when the model never
    reported one.

    Scenario: after a transport timeout, OpenRouter's router delivers a retry
    whose finish_reason is rewritten from 'length' to 'tool_calls', and whose
    tool-call arguments are cut off mid-JSON. The old code hardcoded
    'Response truncated due to output length limit', misleading operators into
    tuning max_tokens / switching models when the real cause is transport /
    router corruption.
    """

    def _router_rewrite_response(self):
        # finish_reason='tool_calls' (NOT 'length') + truncated JSON args
        # ('{"cmd": "ls' — no closing brace), on a normal generation id (not
        # the partial-stream stub). This bypasses the length handler and lands
        # in the invalid-JSON truncation detector.
        msg = SimpleNamespace(
            role="assistant",
            content="",
            tool_calls=[SimpleNamespace(
                id="call_1", type="function",
                function=SimpleNamespace(name="terminal", arguments='{"cmd": "ls'),
            )],
            reasoning_content=None,
        )
        return SimpleNamespace(
            id="gen-router-rewrite-xyz",
            model="deepseek/deepseek-v4-flash",
            choices=[SimpleNamespace(
                index=0, message=msg, finish_reason="tool_calls",
            )],
            usage=None,
        )

    def test_router_rewrite_truncation_does_not_claim_output_length_limit(
        self, loop_agent,
    ):
        loop_agent.valid_tool_names = {"terminal"}
        loop_agent.client.chat.completions.create.side_effect = (
            lambda *a, **kw: self._router_rewrite_response()
        )

        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            result = loop_agent.run_conversation("list the files")

        final = result["final_response"] or ""
        error = result["error"] or ""

        # The misleading output-length-limit label must be gone: finish_reason
        # was 'tool_calls', never 'length'.
        assert "output length limit" not in final, (
            "Router-rewrite / transport truncation must not be reported as an "
            "output-length limit (#91717)."
        )
        assert "output length limit" not in error
        # The honest message names the real suspect (raw finish_reason stays in the diagnostic log).
        assert "finish_reason" not in final
        assert (
            "transport" in final or "router" in final
        ), "Honest message must point at transport/router corruption."
        # Recovery invariants preserved: incomplete tool args never execute.
        assert result["completed"] is False
        assert result["partial"] is True

    def _genuine_length_truncated_tool_call(self):
        # finish_reason='length' (the model truly hit its output cap) with a
        # tool call whose JSON args are cut off. On a normal generation id, so
        # it is NOT treated as a partial-stream network stub. After the loop
        # exhausts its truncated-tool-call retries this lands on the primary
        # length handler's tool-call path (scenario 1 in the issue's table).
        msg = SimpleNamespace(
            role="assistant",
            content="",
            tool_calls=[SimpleNamespace(
                id="call_1", type="function",
                function=SimpleNamespace(name="terminal", arguments='{"cmd": "ls'),
            )],
            reasoning_content=None,
        )
        return SimpleNamespace(
            id="gen-real-length",
            model="test/model",
            choices=[SimpleNamespace(index=0, message=msg, finish_reason="length")],
            usage=None,
        )

    def test_genuine_length_truncation_keeps_output_limit_wording(self, loop_agent):
        """The primary length handler (finish_reason='length', real
        output-cap exhaustion) must keep the accurate 'output length limit'
        wording — scenario 1 in the issue's table is correct and must not
        regress from the honest-message fix."""
        loop_agent.valid_tool_names = {"terminal"}
        loop_agent.client.chat.completions.create.side_effect = (
            lambda *a, **kw: self._genuine_length_truncated_tool_call()
        )

        with (
            patch.object(loop_agent, "_persist_session"),
            patch.object(loop_agent, "_save_trajectory"),
            patch.object(loop_agent, "_cleanup_task_resources"),
        ):
            result = loop_agent.run_conversation("run a long command")

        blob = (result.get("final_response") or "") + (result.get("error") or "")
        assert "output length limit" in blob, (
            "A genuine finish_reason='length' truncation must still reference "
            "the output length limit — the honest-message fix must not blank "
            "out the accurate case (#91717 scenario 1)."
        )
