mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-12 23:12:23 -06:00
68b22adfa3
Hoist the compat-lane injection into _protocol.merge_reasoning_template_kwargs (next to ModelCapabilities — one implementation for both local-server lanes) and retire OpenAIChatCompletionsProvider._apply_thinking_mode in its favor: _finalize_extra_body now receives the session effort knob, so thinking_mode manual/adaptive maps knob "none" to an explicit thinking_param false (previously the toggle was unconditionally true) and caps.effort_param carries the graded effort key on chat completions too. Operator server_compat pins still win; the Responses surface is untouched (native reasoning handles effort itself). The admin Models form grows an "Effort param" field that round-trips like thinking_param: lifted out of the raw capabilities JSON on edit-load, re-added on save, cleared by emptying the field. Verified live against qwen3.6-27b /v1/chat/completions: knob medium streams reasoning_content, knob none suppresses it.
350 lines
15 KiB
Python
350 lines
15 KiB
Python
"""Tests for turnstone.core.server_compat — profile suggestion and merging."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
from unittest.mock import MagicMock
|
|
|
|
from turnstone.core.providers._openai_chat import OpenAIChatCompletionsProvider
|
|
from turnstone.core.providers._protocol import ModelCapabilities
|
|
from turnstone.core.server_compat import merge_server_compat, suggest_profile
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# suggest_profile
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestSuggestProfile:
|
|
def test_vllm_gemma4(self) -> None:
|
|
p = suggest_profile("vllm", "google/gemma-4-31B-it")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
assert p["capabilities"]["thinking_param"] == "enable_thinking"
|
|
# No bug-workaround extra_body — gemma-4 needs only the thinking param.
|
|
assert "extra_body" not in p["server_compat"]
|
|
|
|
def test_vllm_gemma3(self) -> None:
|
|
p = suggest_profile("vllm", "google/gemma-3-27b-it")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
|
|
def test_vllm_qwen3(self) -> None:
|
|
p = suggest_profile("vllm", "Qwen/Qwen3-8B")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
assert p["capabilities"]["thinking_param"] == "enable_thinking"
|
|
# Qwen doesn't need skip_special_tokens workaround
|
|
assert "extra_body" not in p.get("server_compat", {})
|
|
|
|
def test_vllm_qwq(self) -> None:
|
|
p = suggest_profile("vllm", "Qwen/QwQ-32B")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
|
|
def test_vllm_granite(self) -> None:
|
|
p = suggest_profile("vllm", "ibm-granite/granite-3.2-2b-instruct")
|
|
assert p["capabilities"]["thinking_param"] == "thinking"
|
|
|
|
def test_vllm_deepseek_r1(self) -> None:
|
|
p = suggest_profile("vllm", "deepseek-ai/DeepSeek-R1-Distill-Qwen-7B")
|
|
assert p["capabilities"]["thinking_param"] == "thinking"
|
|
|
|
def test_vllm_deepseek_v3_no_thinking(self) -> None:
|
|
"""DeepSeek-V3 is a chat model, not a reasoning model — no thinking profile."""
|
|
p = suggest_profile("vllm", "deepseek-ai/DeepSeek-V3-0324")
|
|
assert "capabilities" not in p
|
|
assert p["server_compat"]["server_type"] == "vllm"
|
|
|
|
def test_vllm_non_thinking_model(self) -> None:
|
|
p = suggest_profile("vllm", "meta-llama/Llama-3-70B-Instruct")
|
|
assert "capabilities" not in p
|
|
assert p["server_compat"]["server_type"] == "vllm"
|
|
|
|
def test_llama_cpp_non_thinking(self) -> None:
|
|
p = suggest_profile("llama.cpp", "some-model")
|
|
assert p["server_compat"]["server_type"] == "llama.cpp"
|
|
assert "capabilities" not in p
|
|
|
|
def test_llama_cpp_gemma_thinking(self) -> None:
|
|
"""llama.cpp with Gemma model gets thinking profile with reasoning_format."""
|
|
p = suggest_profile("llama.cpp", "gemma-4-E4B-it.gguf")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
assert p["server_compat"]["extra_body"]["reasoning_format"] == "auto"
|
|
|
|
def test_llama_cpp_qwen_thinking(self) -> None:
|
|
p = suggest_profile("llama.cpp", "Qwen3-8B-Q4_K_M.gguf")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
|
|
def test_sglang(self) -> None:
|
|
p = suggest_profile("sglang", "some-model")
|
|
assert p["server_compat"]["server_type"] == "sglang"
|
|
|
|
def test_unknown_server(self) -> None:
|
|
assert suggest_profile("unknown", "foo") == {}
|
|
|
|
def test_empty_inputs(self) -> None:
|
|
assert suggest_profile("", "") == {}
|
|
|
|
def test_openai_compatible_fallback(self) -> None:
|
|
"""Generic openai-compatible without a specific profile."""
|
|
assert suggest_profile("openai-compatible", "some-local-model") == {}
|
|
|
|
def test_case_insensitive_model_match(self) -> None:
|
|
"""Model matching should be case-insensitive."""
|
|
p = suggest_profile("vllm", "Google/GEMMA-4-31B-IT")
|
|
assert p["capabilities"]["thinking_mode"] == "manual"
|
|
|
|
def test_vllm_mistral_medium_not_auto_suggested(self) -> None:
|
|
"""Mistral medium falls back to the generic vLLM profile.
|
|
|
|
We don't auto-suggest the Responses surface for Mistral medium because
|
|
vLLM's Responses API tool-call parser isn't wired up for it yet —
|
|
operators who want per-request reasoning effort must pick "Responses
|
|
API" manually in the admin UI and accept the tool-calling limitation.
|
|
"""
|
|
p = suggest_profile("vllm", "mistralai/Mistral-Medium-3-Instruct")
|
|
assert p["server_compat"]["server_type"] == "vllm"
|
|
assert "api_surface" not in p["server_compat"]
|
|
assert "capabilities" not in p
|
|
|
|
def test_vllm_mistral_medium_profile_still_available(self) -> None:
|
|
"""The vllm-mistral-medium profile remains in _PROFILES so an operator
|
|
who explicitly opts in via the admin UI gets the Responses surface."""
|
|
from turnstone.core.server_compat import _PROFILES
|
|
|
|
assert "vllm-mistral-medium" in _PROFILES
|
|
assert _PROFILES["vllm-mistral-medium"]["server_compat"]["api_surface"] == "responses"
|
|
|
|
def test_holo_requires_holo2(self) -> None:
|
|
"""Short 'holo' prefix shouldn't false-match; 'holo2' should match."""
|
|
p_short = suggest_profile("vllm", "some-org/hologram-7b")
|
|
assert "capabilities" not in p_short
|
|
p_long = suggest_profile("vllm", "some-org/Holo2-14B")
|
|
assert p_long["capabilities"]["thinking_mode"] == "manual"
|
|
|
|
def test_suggest_returns_deep_copy(self) -> None:
|
|
"""Mutating the returned profile should not affect future calls."""
|
|
p1 = suggest_profile("vllm", "google/gemma-4-31B-it")
|
|
p1["capabilities"]["thinking_mode"] = "none"
|
|
p2 = suggest_profile("vllm", "google/gemma-4-31B-it")
|
|
assert p2["capabilities"]["thinking_mode"] == "manual"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# merge_server_compat
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestMergeServerCompat:
|
|
def test_empty_base_and_compat_is_empty(self) -> None:
|
|
"""No base, no compat → no extra_body needed."""
|
|
assert merge_server_compat(None, {}) == {}
|
|
assert merge_server_compat({}, {}) == {}
|
|
|
|
def test_explicit_base_passes_through(self) -> None:
|
|
"""Explicit chat_template_kwargs base is forwarded as-is."""
|
|
base = {"reasoning_effort": "medium"}
|
|
result = merge_server_compat(base, {})
|
|
assert result == {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
|
|
|
def test_extra_body_merged_top_level_no_base(self) -> None:
|
|
"""Server-level overrides forward without a chat_template_kwargs wrapper."""
|
|
result = merge_server_compat(None, {"extra_body": {"skip_special_tokens": False}})
|
|
assert result == {"skip_special_tokens": False}
|
|
|
|
def test_full_server_compat_extra_body_no_base(self) -> None:
|
|
"""A server workaround (e.g. llama.cpp reasoning_format) forwards on its own."""
|
|
compat = {
|
|
"server_type": "llama.cpp",
|
|
"extra_body": {"reasoning_format": "auto"},
|
|
}
|
|
result = merge_server_compat(None, compat)
|
|
assert result == {"reasoning_format": "auto"}
|
|
|
|
def test_operator_chat_template_kwargs_only(self) -> None:
|
|
"""Operator can set chat_template_kwargs explicitly without seeding the base."""
|
|
compat = {
|
|
"extra_body": {
|
|
"chat_template_kwargs": {"reasoning_effort": "high"},
|
|
"skip_special_tokens": False,
|
|
},
|
|
}
|
|
result = merge_server_compat(None, compat)
|
|
assert result == {
|
|
"chat_template_kwargs": {"reasoning_effort": "high"},
|
|
"skip_special_tokens": False,
|
|
}
|
|
|
|
def test_extra_body_chat_template_kwargs_deep_merged_with_base(self) -> None:
|
|
"""Operator chat_template_kwargs deep-merges over the seeded base."""
|
|
base = {"reasoning_effort": "medium"}
|
|
compat = {
|
|
"extra_body": {
|
|
"chat_template_kwargs": {"custom_flag": True, "reasoning_effort": "high"},
|
|
"skip_special_tokens": False,
|
|
},
|
|
}
|
|
result = merge_server_compat(base, compat)
|
|
assert result["chat_template_kwargs"]["custom_flag"] is True
|
|
# Operator value wins over seeded base
|
|
assert result["chat_template_kwargs"]["reasoning_effort"] == "high"
|
|
assert result["skip_special_tokens"] is False
|
|
|
|
def test_extra_body_chat_template_kwargs_non_dict_ignored(self) -> None:
|
|
"""Non-dict chat_template_kwargs in extra_body is safely ignored."""
|
|
compat = {"extra_body": {"chat_template_kwargs": "bad"}}
|
|
assert merge_server_compat(None, compat) == {}
|
|
|
|
def test_base_not_mutated(self) -> None:
|
|
base = {"reasoning_effort": "medium"}
|
|
compat = {"extra_body": {"skip_special_tokens": False}}
|
|
merge_server_compat(base, compat)
|
|
assert "skip_special_tokens" not in base
|
|
|
|
def test_non_dict_extra_body_ignored(self) -> None:
|
|
"""Gracefully handle malformed server_compat."""
|
|
assert merge_server_compat(None, {"extra_body": 42}) == {}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# End-to-end: session merge + provider thinking mode
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestEndToEndRequestShaping:
|
|
"""Compose both layers — session builds extra_params, provider applies reasoning."""
|
|
|
|
provider = OpenAIChatCompletionsProvider()
|
|
|
|
def test_vllm_gemma_full_flow(self) -> None:
|
|
"""Gemma now needs only the thinking param — no server workaround."""
|
|
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
|
server_compat = {"server_type": "vllm"}
|
|
# Step 1: session forwards (no auto-injection of reasoning_effort).
|
|
extra_params = merge_server_compat(None, server_compat)
|
|
# Step 2: provider folds the effort knob into chat_template_kwargs.
|
|
extra_body = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
|
|
|
assert extra_body == {"chat_template_kwargs": {"enable_thinking": True}}
|
|
|
|
def test_knob_none_disables_toggle(self) -> None:
|
|
"""Session effort "none" turns the template toggle off dynamically."""
|
|
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
|
extra_body = self.provider._finalize_extra_body(
|
|
merge_server_compat(None, {"server_type": "vllm"}), caps, "none"
|
|
)
|
|
|
|
assert extra_body == {"chat_template_kwargs": {"enable_thinking": False}}
|
|
|
|
def test_server_workaround_composes_with_thinking(self) -> None:
|
|
"""A top-level server workaround forwards alongside the injected thinking param."""
|
|
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
|
compat = {"server_type": "llama.cpp", "extra_body": {"reasoning_format": "auto"}}
|
|
extra_body = self.provider._finalize_extra_body(
|
|
merge_server_compat(None, compat), caps, "medium"
|
|
)
|
|
|
|
assert extra_body == {
|
|
"chat_template_kwargs": {"enable_thinking": True},
|
|
"reasoning_format": "auto",
|
|
}
|
|
|
|
def test_granite_thinking_key(self) -> None:
|
|
"""Granite uses 'thinking' instead of 'enable_thinking'."""
|
|
caps = ModelCapabilities(thinking_mode="manual", thinking_param="thinking")
|
|
extra_body = self.provider._finalize_extra_body(
|
|
merge_server_compat(None, {}), caps, "medium"
|
|
)
|
|
|
|
assert extra_body == {"chat_template_kwargs": {"thinking": True}}
|
|
|
|
def test_non_thinking_model_no_injection(self) -> None:
|
|
"""Non-thinking model gets no extra_body at all."""
|
|
caps = ModelCapabilities() # thinking_mode="none"
|
|
extra_body = self.provider._finalize_extra_body(
|
|
merge_server_compat(None, {}), caps, "medium"
|
|
)
|
|
|
|
assert extra_body is None
|
|
|
|
def test_operator_reasoning_effort_passthrough(self) -> None:
|
|
"""Operator-supplied reasoning_effort under chat_template_kwargs is preserved."""
|
|
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
|
compat = {
|
|
"server_type": "vllm",
|
|
"extra_body": {"chat_template_kwargs": {"reasoning_effort": "high"}},
|
|
}
|
|
extra_body = self.provider._finalize_extra_body(
|
|
merge_server_compat(None, compat), caps, "medium"
|
|
)
|
|
|
|
assert extra_body == {
|
|
"chat_template_kwargs": {
|
|
"reasoning_effort": "high",
|
|
"enable_thinking": True,
|
|
},
|
|
}
|
|
|
|
def test_operator_pin_beats_effort_param(self) -> None:
|
|
"""A pinned effort key wins over the knob mapping (setdefault)."""
|
|
caps = ModelCapabilities(
|
|
thinking_mode="none",
|
|
effort_param="reasoning_effort",
|
|
)
|
|
compat = {"extra_body": {"chat_template_kwargs": {"reasoning_effort": "low"}}}
|
|
extra_body = self.provider._finalize_extra_body(
|
|
merge_server_compat(None, compat), caps, "high"
|
|
)
|
|
|
|
assert extra_body == {"chat_template_kwargs": {"reasoning_effort": "low"}}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Probe integration: suggest_profile called from _detect_openai_compat
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestProbeIntegration:
|
|
def test_detect_vllm_gemma_suggests_profile(self) -> None:
|
|
"""_detect_openai_compat returns suggested_capabilities and suggested_server_compat."""
|
|
from turnstone.core.model_registry import _detect_openai_compat
|
|
|
|
result: dict[str, Any] = {
|
|
"reachable": True,
|
|
"model_found": True,
|
|
"available_models": ["google/gemma-4-31B-it"],
|
|
"context_window": None,
|
|
"server_type": None,
|
|
"error": None,
|
|
}
|
|
model_obj = MagicMock()
|
|
model_obj.model_dump.return_value = {"owned_by": "vllm"}
|
|
|
|
_detect_openai_compat(
|
|
result, model_obj, "google/gemma-4-31B-it", "http://localhost:8000/v1"
|
|
)
|
|
|
|
assert result["server_type"] == "vllm"
|
|
assert result["suggested_capabilities"]["thinking_mode"] == "manual"
|
|
assert result["suggested_capabilities"]["thinking_param"] == "enable_thinking"
|
|
assert "extra_body" not in result["suggested_server_compat"]
|
|
|
|
def test_detect_non_thinking_no_suggested_capabilities(self) -> None:
|
|
"""Non-thinking vLLM model gets server_compat but no capabilities suggestion."""
|
|
from turnstone.core.model_registry import _detect_openai_compat
|
|
|
|
result: dict[str, Any] = {
|
|
"reachable": True,
|
|
"model_found": True,
|
|
"available_models": ["meta-llama/Llama-3-70B"],
|
|
"context_window": None,
|
|
"server_type": None,
|
|
"error": None,
|
|
}
|
|
model_obj = MagicMock()
|
|
model_obj.model_dump.return_value = {"owned_by": "vllm"}
|
|
|
|
_detect_openai_compat(
|
|
result, model_obj, "meta-llama/Llama-3-70B", "http://localhost:8000/v1"
|
|
)
|
|
|
|
assert result["server_type"] == "vllm"
|
|
assert "suggested_capabilities" not in result
|
|
assert result["suggested_server_compat"]["server_type"] == "vllm"
|