"""The recommendation decision table — the reviewable matrix.

The recommendation itself is DERIVED (catalog.recommended_entry: best
quality among resident entries clearing the pleasant speed floor, else
fastest resident). Spilled models stay browseable but are never automatic
recommendations, so nobody hand-maintains per-hardware-class picks.
This table pins the model and reason across discrete and unified memory
classes so changes to the recommendation remain reviewable.

These are decision pins, not change-detectors: each cell is a choice a
human approved, exactly like a golden file. When a cell flips on
purpose, update it in the same commit and say why. When one flips by
surprise, that is the test doing its job.

Budgets mirror hardware.probe_budget's planning-mode shapes (margins,
UMA headroom) so the cells match what a real machine of that class
resolves.
"""

from __future__ import annotations

import pytest

from hermes_cli.local_runtime.catalog import (
    CATALOG,
    PLEASANT_FLOOR_TOK_S,
    predicted_decode_tok_s,
    recommended_entry,
    select_variant,
)
from hermes_cli.local_runtime.estimator import HardwareBudget

_GIB = 1 << 30


def _discrete(size_gb: int) -> HardwareBudget:
    total = size_gb * _GIB
    margin = max(2 * _GIB, int(total * 0.09))
    return HardwareBudget(usable_vram_bytes=max(0, total - margin),
                          total_device_bytes=total,
                          ram_available_bytes=64 * _GIB, uma=False)


def _unified(size_gb: int) -> HardwareBudget:
    total = size_gb * _GIB
    return HardwareBudget(usable_vram_bytes=int(total * 0.80),
                          total_device_bytes=total,
                          ram_available_bytes=0, uma=True)


# The decision table. Cells were generated by the resolver and then
# reviewed as editorial decisions:
#
#   VRAM | discrete                | unified
#   -----+-------------------------+------------------------
#     8  | (no recommendation)      | (none fits)
#    16  | (no recommendation)      | (none fits)
#    24  | qwen3.8-27b             | (none fits)
#    32  | qwen3.8-27b             | qwen3.6-35b-a3b
#    48  | qwen3.8-27b             | qwen3.6-35b-a3b
#    96  | qwen3.8-27b             | qwen3.6-35b-a3b
#   128  | qwen3.8-flash-next      | qwen3.6-35b-a3b
#   256  | qwen3.8-flash-next      | qwen3.8-flash-next
#   512  | qwen3.8-flash-next      | qwen3.8-flash-next
#
# Reading guide for reviewers:
# - Discrete <=16 GB: nothing runs resident; no automatic recommendation.
#   Browse remains available for explicit spill choices.
# - Discrete 24-96 GB: the 27B is the flagship experience — dense reads
#   at ~1 TB/s clear the floor easily, so quality decides.
# - Discrete/unified where Flash Next fits resident (128 GB discrete,
#   256+ GB unified): the frontier model is the pick — highest quality,
#   and its sparse decode clears the floor even at UMA bandwidth
#   (~24 tok/s predicted at 210 GB/s).
# - Unified 32-128 GB — the Spark class, the reason this resolver
#   exists: the dense 27B predicts ~13 tok/s at UMA bandwidth (below
#   the pleasant floor), so the 35B-A3B (~60 tok/s) wins.
# - Unified <=24 GB: no entry passes the physics check inside the UMA
#   budget (spilling is impossible on UMA by construction — the pool IS
#   the RAM). The pane's browse flow is the path for those machines
#   until a small catalog entry lands (revisit when one does).
DECISION_TABLE = [
    (8, "discrete", None, None),
    (8, "unified", None, None),
    (16, "discrete", None, None),
    (16, "unified", None, None),
    (24, "discrete", "qwen3.8-27b", "best-quality-resident"),
    (24, "unified", None, None),
    (32, "discrete", "qwen3.8-27b", "best-quality-resident"),
    (32, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
    (48, "discrete", "qwen3.8-27b", "best-quality-resident"),
    (48, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
    (96, "discrete", "qwen3.8-27b", "best-quality-resident"),
    (96, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
    (128, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
    (128, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
    (256, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
    (256, "unified", "qwen3.8-flash-next", "best-quality-resident"),
    (512, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
    (512, "unified", "qwen3.8-flash-next", "best-quality-resident"),
]


@pytest.mark.parametrize(
    ("size_gb", "kind", "expected", "expected_reason"),
    DECISION_TABLE,
    ids=[f"{s}GB-{k}" for s, k, _, _ in DECISION_TABLE])
def test_recommendation_decision_table(size_gb, kind, expected, expected_reason):
    """Pins the pick AND its reason per cell: the reason is user-facing
    (the Recommended badge's tooltip), so a cell whose rationale flips
    without the pick flipping is still a review-worthy change."""
    budget = _discrete(size_gb) if kind == "discrete" else _unified(size_gb)
    picked = recommended_entry(budget)
    if expected is None:
        assert picked is None
    else:
        assert picked is not None
        assert (picked[0].id, picked[1]) == (expected, expected_reason)


@pytest.mark.parametrize("gpu_name", [
    "NVIDIA RTX Spark N1X",
    "NVIDIA RTX Spark N1X (5120-core Blackwell RTX GPU)",
    "NVIDIA RTX Spark N1X (updated device description)",
])
@pytest.mark.parametrize("capacity", [24, 48, 256])
def test_measured_n1x_profile_changes_speed_eligibility_not_fit_or_quality(monkeypatch, capacity, gpu_name):
    import subprocess
    import urllib.request

    def no_io(*args, **kwargs):
        raise AssertionError("calibration must use shipped data, not a runtime benchmark")

    monkeypatch.setattr(subprocess, "run", no_io)
    monkeypatch.setattr(urllib.request, "urlopen", no_io)
    budget = _unified(capacity)
    budget.gpu_name = gpu_name
    budget.platform = "win32"
    entry = next(e for e in CATALOG if e.id == "qwen3.8-27b")
    assert predicted_decode_tok_s(entry, entry.variants[0], budget) >= PLEASANT_FLOOR_TOK_S
    picked = recommended_entry(budget)
    expected = {24: None, 48: "qwen3.8-27b", 256: "qwen3.8-flash-next"}[capacity]
    assert (picked[0].id if picked else None) == expected


@pytest.mark.parametrize("budget_changes, entry_changes, quant, backend, spilled", [
    ({"gpu_name": ""}, {}, "UD-Q4_K_M", "cuda", False),
    ({"gpu_name": "NVIDIA Other Device"}, {}, "UD-Q4_K_M", "cuda", False),
    ({"gpu_name": "NVIDIA RTX Spark N1X2"}, {}, "UD-Q4_K_M", "cuda", False),
    ({"gpu_name": "NVIDIA RTX Spark N1X Pro"}, {}, "UD-Q4_K_M", "cuda", False),
    ({"platform": "linux"}, {}, "UD-Q4_K_M", "cuda", False),
    ({"uma": False}, {}, "UD-Q4_K_M", "cuda", False),
    ({}, {}, "UD-Q4_K_M", "vulkan", False),
    ({}, {}, "UD-Q4_K_M", "cpu", False),
    ({}, {"mtp": False}, "UD-Q4_K_M", "cuda", False),
    ({}, {"mtp_draft_depth": 3}, "UD-Q4_K_M", "cuda", False),
    ({}, {}, "Q8_0", "cuda", False),
    ({}, {}, "UD-Q4_K_M", "cuda", True),
])
def test_unmatched_or_spilled_profiles_keep_the_existing_estimate(
        monkeypatch, budget_changes, entry_changes, quant, backend, spilled):
    from dataclasses import replace
    import subprocess
    import urllib.request

    budget = replace(_unified(48), gpu_name="NVIDIA RTX Spark N1X (5120-core Blackwell RTX GPU)",
                     platform="win32")
    budget = replace(budget, **budget_changes)
    entry = replace(next(e for e in CATALOG if e.id == "qwen3.8-27b"), **entry_changes)
    variant = replace(entry.variants[0], quant=quant)

    def no_io(*args, **kwargs):
        raise AssertionError("speed selection must be an offline lookup and arithmetic")

    monkeypatch.setattr(subprocess, "run", no_io)
    monkeypatch.setattr(urllib.request, "urlopen", no_io)
    fallback = predicted_decode_tok_s(entry, variant, replace(budget, gpu_name=""),
                                      spilled=spilled, backend=backend)
    assert predicted_decode_tok_s(entry, variant, budget, spilled=spilled, backend=backend) == fallback


# ── invariants behind the table (survive catalog changes) ──


def test_every_entry_carries_the_recommendation_axes():
    """quality and decode_fraction are authoring requirements: an entry
    without them silently loses every quality comparison (quality=0) or
    prices as dense (decode_fraction=1.0)."""
    for entry in CATALOG:
        assert entry.quality > 0, f"{entry.id} has no quality ordering"
        assert 0.0 < entry.decode_fraction <= 1.0, entry.id
        if not entry.moe:
            assert entry.decode_fraction == 1.0, (
                f"{entry.id} is dense — it reads every weight per token")


def test_unified_never_recommends_a_below_floor_dense_model():
    """The Spark rule, as an invariant: whatever the catalog holds, a
    unified-memory machine must not be told to run a model whose
    predicted decode is below the pleasant floor while a resident
    alternative clears it."""
    budget = _unified(128)
    pick = recommended_entry(budget)[0].id
    assert pick is not None
    entry = next(e for e in CATALOG if e.id == pick)
    choice = select_variant(entry, budget)
    assert choice is not None and choice.zero_spill
    clears = [
        e for e in CATALOG
        if (c := select_variant(e, budget)) is not None and c.zero_spill
        and predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
    ]
    if clears:
        assert predicted_decode_tok_s(entry, choice.variant, budget) >= PLEASANT_FLOOR_TOK_S


def test_quality_decides_where_speed_permits():
    """On big discrete hardware every resident entry clears the floor, so
    the pick must be the highest-quality fitting entry — the axis that
    justifies carrying an editorial field at all."""
    budget = _discrete(512)
    pick = recommended_entry(budget)[0].id
    resident = [
        e for e in CATALOG
        if (c := select_variant(e, budget)) is not None and c.zero_spill
    ]
    assert pick == max(resident, key=lambda e: e.quality).id
