diff --git a/tests/test_judge.py b/tests/test_judge.py index 904eb6ec..4c2a8738 100644 --- a/tests/test_judge.py +++ b/tests/test_judge.py @@ -949,6 +949,29 @@ class TestModelAliasResolution: # alias's configured value, inherited through the lane. assert alias_provider.create_completion.call_args.kwargs["temperature"] == 0.3 + def test_constructor_resolves_from_one_config_fetch(self): + """The constructor consumes the ModelConfig that registry.resolve() + already returned (the ``cfg=`` pass-through) — ZERO independent + get_config fetches, so a registry hot-reload between two lookups + cannot bind the resolved client/window to a different capability + generation.""" + alias_provider = _make_mock_provider(response_content=_good_verdict_json()) + registry = self._make_alias_registry( + "judge-mini", + alias_provider, + MagicMock(base_url="https://a/v1", api_key="k"), + "local-9b", + ) + IntentJudge( + config=JudgeConfig(enabled=True, model="judge-mini"), + session_provider=_make_mock_provider(), + session_client=MagicMock(base_url="https://s/v1", api_key="s"), + session_model="session-model", + session_capabilities=MagicMock(context_window=100_000), + model_registry=registry, + ) + assert registry.get_config.call_count == 0 + def test_fallback_threads_session_capabilities_to_wire(self): """No judge alias → the judge inherits the session model AND the session's resolved capabilities, threaded to ``create_completion``.""" diff --git a/tests/test_model_turn.py b/tests/test_model_turn.py index d7486e12..0cdd9da7 100644 --- a/tests/test_model_turn.py +++ b/tests/test_model_turn.py @@ -368,12 +368,50 @@ def test_temperature_caller_value_wins_over_lane() -> None: assert provider.calls[0]["temperature"] == 0.9 -def test_temperature_omitted_when_unresolved() -> None: - # No caller value, no lane value → the kwarg is omitted entirely and - # the provider default applies (house rule: code never pins one). +def test_temperature_unresolved_passes_none_and_wire_omits_it() -> None: + # No caller value, no lane value → model_turn passes temperature=None, + # and the PROVIDER layer omits the field from the wire so the server + # default applies (house rule: code never pins one). Both halves are + # pinned: a Python-signature default of 0.5 anywhere on this path is a + # hidden universal pin — the exact bug the second xhigh review caught. provider = _FakeProvider([CompletionResult(content="")]) model_turn(_lane(provider), [Turn.user("x")]) - assert "temperature" not in provider.calls[0] + assert provider.calls[0]["temperature"] is None + + from turnstone.core.providers._openai_common import apply_temperature + + kwargs: dict[str, Any] = {} + apply_temperature(kwargs, ModelCapabilities(), None, "medium") + assert "temperature" not in kwargs # None never reaches the wire + apply_temperature(kwargs, ModelCapabilities(), 1.0, "medium") + assert kwargs["temperature"] == 1.0 # a real value still does + + +def test_resolve_lane_global_config_store_rung() -> None: + # ModelConfig.temperature=None means "use the global default from + # ConfigStore" (the documented ladder) — resolve_lane climbs it. + provider = _FakeProvider([]) + registry = _fake_registry(temperature=None) + store = SimpleNamespace(get=lambda key: 1.0 if key == "model.temperature" else None) + lane = resolve_lane(provider, object(), "m", alias="ali", registry=registry, config_store=store) + assert lane.temperature == 1.0 + # The per-model value wins over the global rung. + registry2 = _fake_registry(temperature=0.3) + lane2 = resolve_lane( + provider, object(), "m", alias="ali", registry=registry2, config_store=store + ) + assert lane2.temperature == 0.3 + + +def test_model_turn_fetches_config_once_per_call() -> None: + # ONE get_config per plant call feeds both live flags (replay + vLLM + # attach) — a hot-reload between them cannot mix config generations + # within a single request. + registry = _fake_registry(replay=True) + provider = _FakeProvider([CompletionResult(content="")]) + lane = _lane(provider, alias="ali", registry=registry) + model_turn(lane, [Turn.user("x")]) + assert registry.get_config.call_count == 1 def test_resolve_lane_inherits_config_temperature() -> None: diff --git a/tests/test_session.py b/tests/test_session.py index b8e7fe77..41da4122 100644 --- a/tests/test_session.py +++ b/tests/test_session.py @@ -7715,6 +7715,21 @@ def test_web_fetch_extraction_caps_max_tokens_to_window_reserve(): assert kw["max_tokens"] == 2048 # context_window // 4, not the 16384 session value +def test_resolve_capabilities_raises_loudly_on_registry_failure(): + """The session lane must NOT silently cache degraded static-table caps: + a get_config failure on the session's own alias PROPAGATES (pre-#827 + semantics) — the never-crash defensive fetch is a judge-constructor + property, and applying it here would let one transient registry hiccup + pin wrong capabilities (window, thinking mode, token param) onto the + session cache for its whole lifetime.""" + session = _make_session() + session._registry = MagicMock() + session._registry.get_config.side_effect = ValueError("Unknown model alias") + session._model_alias = "primary" + with pytest.raises(ValueError): + session._get_capabilities() + + def test_record_aux_usage_skips_when_usage_missing(): """A provider that reports no usage object must not emit a phantom zero-token row.""" diff --git a/tests/test_session_chat_reasoning_replay.py b/tests/test_session_chat_reasoning_replay.py index b3cd5106..40d6267e 100644 --- a/tests/test_session_chat_reasoning_replay.py +++ b/tests/test_session_chat_reasoning_replay.py @@ -27,6 +27,7 @@ import httpx import pytest from tests._session_helpers import make_session as _make_session +from tests._session_helpers import mock_completion_result from turnstone.core.providers._anthropic import AnthropicProvider from turnstone.core.providers._openai_chat import OpenAIChatCompletionsProvider from turnstone.core.providers._openai_responses import OpenAIResponsesProvider @@ -403,14 +404,7 @@ class TestCallSitesInvokeMaybeAttach: def capture_completion(**kwargs: Any) -> Any: captured.update(kwargs) - return SimpleNamespace( - content="", - tool_calls=None, - finish_reason="stop", - usage=None, - provider_blocks=[], - reasoning="", - ) + return mock_completion_result("") provider = OpenAIChatCompletionsProvider() provider.create_completion = capture_completion # type: ignore[method-assign] diff --git a/tests/test_session_replay_reasoning.py b/tests/test_session_replay_reasoning.py index 91e5fcde..39a96a69 100644 --- a/tests/test_session_replay_reasoning.py +++ b/tests/test_session_replay_reasoning.py @@ -27,6 +27,7 @@ from typing import Any from unittest.mock import MagicMock, patch from tests._session_helpers import make_session as _make_session +from tests._session_helpers import mock_completion_result from turnstone.core.trajectory import Turn @@ -627,14 +628,7 @@ class TestUtilityCompletionPassesFlag: def capture_completion(**kwargs: Any) -> Any: captured.update(kwargs) - return SimpleNamespace( - content="title", - tool_calls=None, - finish_reason="stop", - usage=None, - provider_blocks=[], - reasoning="", - ) + return mock_completion_result("title") mock_provider = MagicMock() mock_provider.create_completion = capture_completion diff --git a/tests/test_session_synth_reasoning_block.py b/tests/test_session_synth_reasoning_block.py index a6ca7784..7f0cd656 100644 --- a/tests/test_session_synth_reasoning_block.py +++ b/tests/test_session_synth_reasoning_block.py @@ -1,14 +1,16 @@ -"""Tests for ChatSession synthetic ``reasoning_text`` block stamping (Phase 3 path 3). +"""Tests for the synthetic ``reasoning_text`` block stamping (Phase 3 path 3). Path 3 covers OpenAI Chat Completions endpoints — vLLM with ``--reasoning-parser``, llama.cpp with ``reasoning_format``, Gemini's ``/v1beta/openai/`` endpoint, and any other server that surfaces ``delta.reasoning_content`` Pydantic extras. These have no native -provider_blocks shape on the wire, so ``ChatSession._stream_response`` -captures the streamed reasoning text into ``reasoning_parts`` and -``_maybe_synth_reasoning_block`` stamps it onto ``_provider_content`` -as a synthetic ``{type: "reasoning_text"}`` block at the end of the -turn. +provider_blocks shape on the wire, so the captured reasoning text is +stamped onto ``_provider_content`` as a synthetic +``{type: "reasoning_text"}`` block at the end of the turn by +``model_turn.synth_reasoning_block`` — the one synthesizer every lane +runs (the main loop reaches it through ``ChatSession._stream_response`` +→ ``_finalize_provider_blocks``; agents and judges through +``model_turn``). These tests pin: 1. The synthesizer fires only when no native blocks were emitted AND diff --git a/turnstone/core/judge.py b/turnstone/core/judge.py index 3540b082..c0bc1ec2 100644 --- a/turnstone/core/judge.py +++ b/turnstone/core/judge.py @@ -977,12 +977,15 @@ class IntentJudge: rule_registry: Any | None = None, model_registry: Any | None = None, session_model_alias: str = "", + config_store: Any | None = None, ) -> None: self._config = config self._rule_registry = rule_registry - # Carried into the per-evaluation ModelLane so extra_params and the - # live operator flags resolve from the registry like every other lane. + # Carried into the per-evaluation ModelLane so extra_params, the + # live operator flags, and the temperature ladder (per-model value → + # global ``model.temperature``) resolve like every other lane. self._model_registry = model_registry + self._config_store = config_store # The caller (ChatSession) resolves the session model's real caps from # _get_capabilities (config/registry-aware) and passes them in; they are # this judge's wire capabilities and window when it inherits the session @@ -1022,8 +1025,12 @@ class IntentJudge: # sized into the judge's window budget right below (the # static caps table reports 200000 for local models, which # would silently over-budget them). + # ``cfg=model_cfg`` reuses the config resolve() already + # fetched — one lookup, one generation; a hot-reload + # between two fetches cannot mix client/window with + # foreign capability overrides. self._capabilities = resolve_capabilities( - self._provider, self._model, config.model, model_registry + self._provider, self._model, config.model, model_registry, cfg=model_cfg ) # Use the registry's per-model context window, NOT # ``provider.get_capabilities().context_window``: the static @@ -1329,8 +1336,14 @@ class IntentJudge: tools = list(_JUDGE_TOOL_SCHEMAS) # The judge's resolved lane for this evaluation: fresh client per run - # (thread isolation), constructor-resolved capabilities, extra_params - # and live operator flags from the registry like every other lane. + # (thread isolation), extra_params / live flags / temperature ladder + # from the registry like every other lane. Capabilities are + # DELIBERATELY the constructor-frozen set (not re-resolved here): + # the judge's window budget was sized against them at construction, + # and the session swaps the whole judge on model/credential change — + # an in-place capabilities edit to the same alias applies on the + # next judge swap, keeping caps and window from ever disagreeing + # within one judge lifetime. lane = resolve_lane( self._provider, client, @@ -1338,6 +1351,7 @@ class IntentJudge: alias=self._alias, registry=self._model_registry, capabilities=self._capabilities, + config_store=self._config_store, ) # Multi-turn judge loop diff --git a/turnstone/core/model_turn.py b/turnstone/core/model_turn.py index 61162352..d56e5168 100644 --- a/turnstone/core/model_turn.py +++ b/turnstone/core/model_turn.py @@ -98,7 +98,7 @@ def resolve_capabilities( alias: str, registry: ModelRegistry | None, *, - cfg: Any | None = None, + cfg: Any | EllipsisType = ..., ) -> ModelCapabilities: """Provider static capabilities, merged with registry alias overrides. @@ -109,14 +109,16 @@ def resolve_capabilities( judges' old mirror — capability resolution must never crash a judge turn). *cfg* accepts a pre-fetched ModelConfig so a caller resolving several lane facets reads ONE config generation (see - :func:`resolve_lane`); omitted, the config is fetched defensively. + :func:`resolve_lane`); ``None`` means "fetched and missed" (no second + lookup), and the ``...`` sentinel default means "fetch defensively for + me". NOTE the window landmine documented on #826: ``ModelConfig.context_window`` is a separate top-level column and is deliberately NOT merged here; callers that need the operator window must read it off the config themselves. """ caps = provider.get_capabilities(model) - if cfg is None: + if cfg is ...: cfg = _get_config_or_none(registry, alias) if cfg is not None: overrides_raw = getattr(cfg, "capabilities", None) @@ -133,7 +135,7 @@ def provider_extra_params( registry: ModelRegistry | None, alias: str, *, - cfg: Any | None = None, + cfg: Any | EllipsisType = ..., ) -> dict[str, Any] | None: """Operator ``server_compat["extra_body"]`` pins for the OpenAI-shaped lanes (and the anthropic-compatible lane, whose SDK also takes @@ -147,7 +149,7 @@ def provider_extra_params( if provider.provider_name not in ("openai", "openai-compatible", "anthropic-compatible"): return None - if cfg is None: + if cfg is ...: cfg = _get_config_or_none(registry, alias) server_compat = getattr(cfg, "server_compat", None) if cfg is not None else None extra = merge_server_compat(None, server_compat if isinstance(server_compat, dict) else {}) @@ -184,6 +186,7 @@ def resolve_replay_reasoning_to_model( alias: str, *, caps: ModelCapabilities | None = None, + cfg: Any | EllipsisType = ..., ) -> bool: """Operator ``ModelConfig.replay_reasoning_to_model`` for an alias. @@ -197,13 +200,15 @@ def resolve_replay_reasoning_to_model( ``caps.supports_reasoning_replay`` (mirrors the gate in ``OpenAIResponsesProvider._build_kwargs``); omitted, the operator flag passes through unchanged for callers that haven't threaded caps. + *cfg* follows the shared sentinel contract (``...`` = fetch for me, + ``None`` = fetched-and-missed) so ``model_turn`` reads ONE config + generation per call across both per-call flags. """ - if not registry or not alias: - return False - try: - operator_on = bool(registry.get_config(alias).replay_reasoning_to_model) - except Exception: + if cfg is ...: + cfg = _get_config_or_none(registry, alias) + if cfg is None: return False + operator_on = bool(getattr(cfg, "replay_reasoning_to_model", False)) if caps is None: return operator_on return operator_on and bool(caps.supports_reasoning_replay) @@ -214,6 +219,8 @@ def maybe_attach_vllm_chat_reasoning( provider: LLMProvider, registry: ModelRegistry | None, alias: str, + *, + cfg: Any | EllipsisType = ..., ) -> list[dict[str, Any]]: """Phase 5 of reasoning persistence: attach vLLM's non-standard ``reasoning`` field to outgoing assistant messages so a vLLM-served @@ -235,7 +242,8 @@ def maybe_attach_vllm_chat_reasoning( if not isinstance(provider, OpenAIChatCompletionsProvider): return messages - cfg = _get_config_or_none(registry, alias) + if cfg is ...: + cfg = _get_config_or_none(registry, alias) if cfg is None: return messages # Both gate fields read off the single ``cfg`` fetch (no second @@ -289,6 +297,7 @@ def resolve_lane( registry: ModelRegistry | None = None, capabilities: ModelCapabilities | None = None, extra_params: dict[str, Any] | None | EllipsisType = ..., + config_store: Any | None = None, ) -> ModelLane: """Build a :class:`ModelLane`, resolving what the caller didn't supply. @@ -298,6 +307,15 @@ def resolve_lane( pass. ``...`` (the sentinel default) means "resolve for me" — ``None`` is a valid resolved value for *extra_params*. + The lane temperature climbs the documented ladder + (``ModelConfig.temperature`` docstring: "None = use global default + from ConfigStore"): the alias's per-model value, else the operator's + global ``model.temperature`` when *config_store* is supplied, else + ``None`` — which the providers translate to omitting the field so the + SERVER default applies. This mirrors the session_factory / ``/model`` + switch resolution so a judge or single-shot lane samples exactly like + the main loop on the same model. + All resolved facets read ONE defensively-fetched ModelConfig, so a registry hot-reload mid-resolution cannot mix config generations, and an alias that raced away degrades every facet to its miss behavior @@ -310,6 +328,15 @@ def resolve_lane( if extra_params is ... else extra_params ) + temperature = getattr(cfg, "temperature", None) if cfg is not None else None + if temperature is None and config_store is not None: + try: + temperature = config_store.get("model.temperature") + except Exception: + # Best-effort global rung — a broken store degrades to the + # server default, never crashes lane resolution. + log.debug("config_store model.temperature lookup failed", exc_info=True) + temperature = None return ModelLane( provider=provider, client=client, @@ -318,7 +345,7 @@ def resolve_lane( capabilities=caps, extra_params=extra, registry=registry, - temperature=getattr(cfg, "temperature", None) if cfg is not None else None, + temperature=temperature, ) @@ -496,6 +523,21 @@ class ModelTurnResult: return self.turn.text +def cap_tool_calls(result: ModelTurnResult, max_calls: int) -> tuple[list[dict[str, Any]], Turn]: + """Degenerate-repetition guard shared by the bounded tool loops (eval, + optimizer analyst): cap the mirror at *max_calls* and return the turn to + append. A capped turn is rebuilt WITHOUT its native lane — a full lane + beside a truncated mirror would replay orphan native tool blocks the + mirror no longer carries (the same native↔mirror rule + :func:`finalize_provider_blocks` enforces). + """ + capped = result.tool_calls[:max_calls] + turn = result.turn + if len(result.tool_calls) > len(capped): + turn = Turn.assistant(result.content, tool_calls=turn.tool_calls[: len(capped)]) + return capped, turn + + def model_turn( lane: ModelLane, turns: Sequence[Turn], @@ -553,29 +595,34 @@ def model_turn( "model_turn: mint requires wire_id_map — minted ids are " "unrestorable on the wire without the recovery map" ) + # ONE config fetch per plant call feeds both live per-call flags — a + # registry hot-reload cannot hand the replay gate and the attach gate + # different config generations within a single request. + cfg = _get_config_or_none(lane.registry, lane.alias) wire = restore_provider_tool_ids( sanitize_tool_call_arguments(dicts_from_turns(list(turns))), wire_id_map if wire_id_map is not None else {}, ) - wire = maybe_attach_vllm_chat_reasoning(wire, lane.provider, lane.registry, lane.alias) + wire = maybe_attach_vllm_chat_reasoning(wire, lane.provider, lane.registry, lane.alias, cfg=cfg) call_kwargs: dict[str, Any] = { "client": lane.client, "model": lane.model, "messages": wire, "tools": tools, "max_tokens": max_tokens, + # None temperature flows through to the providers, which OMIT the + # field from the wire so the server default applies (house rule: a + # Python-level constant anywhere on this path is a hidden pin). + "temperature": temperature if temperature is not None else lane.temperature, "reasoning_effort": reasoning_effort, "extra_params": lane.extra_params, "capabilities": lane.capabilities, "replay_reasoning_to_model": resolve_replay_reasoning_to_model( - lane.registry, lane.alias, caps=lane.capabilities + lane.registry, lane.alias, caps=lane.capabilities, cfg=cfg ), } if resolve_attachments is not None: call_kwargs["resolve_attachments"] = resolve_attachments - effective_temperature = temperature if temperature is not None else lane.temperature - if effective_temperature is not None: - call_kwargs["temperature"] = effective_temperature result = lane.provider.create_completion(**call_kwargs) raw_calls: list[dict[str, Any]] = list(result.tool_calls or []) diff --git a/turnstone/core/output_guard_judge.py b/turnstone/core/output_guard_judge.py index 40093ad8..aa212462 100644 --- a/turnstone/core/output_guard_judge.py +++ b/turnstone/core/output_guard_judge.py @@ -270,11 +270,14 @@ class OutputGuardJudge: model_registry: Any | None = None, session_capabilities: ModelCapabilities | None = None, session_model_alias: str = "", + config_store: Any | None = None, ) -> None: self._config = config - # Carried into the per-evaluation ModelLane so extra_params and the - # live operator flags resolve from the registry like every other lane. + # Carried into the per-evaluation ModelLane so extra_params, the + # live operator flags, and the temperature ladder resolve from the + # registry like every other lane. self._model_registry = model_registry + self._config_store = config_store # Caller's resolved session-model caps (config/registry-aware): the wire # capabilities + window when this judge inherits the session model, and # the alias path's window fallback. The window comes ONLY from these @@ -308,10 +311,17 @@ class OutputGuardJudge: ) self._model = model_name self._judge_model_alias = config.output_guard_model + self._lane_alias = config.output_guard_model # Shared lane resolver (model_turn); ModelConfig.context_window # stays separate and is sized into the guard window below. + # ``cfg=model_cfg`` reuses the config resolve() already + # fetched — one lookup, one generation. self._capabilities = resolve_capabilities( - self._provider, self._model, config.output_guard_model, model_registry + self._provider, + self._model, + config.output_guard_model, + model_registry, + cfg=model_cfg, ) self._judge_context_window = _positive_window( getattr(model_cfg, "context_window", None), @@ -338,10 +348,18 @@ class OutputGuardJudge: session_client, session_provider.provider_name ) self._model = session_model - # Inherit the session's registry alias so the lane resolves - # extra_params / replay flag exactly like every other lane on - # the same model (see IntentJudge's fallback for the rationale). - self._judge_model_alias = session_model_alias + # AUDIT label keeps its pre-#827 fallback semantics: "" here so + # recorded verdicts show ``judge_model = self._model`` (the raw + # model id), not the session alias — threading the alias into + # this field would silently change recorded judge_model values + # across the upgrade on default-config installs. + self._judge_model_alias = "" + # LANE alias inherits the session's registry alias so the lane + # resolves extra_params / replay flag / temperature exactly like + # every other lane on the same model (see IntentJudge's fallback + # for the rationale). Lane resolution and audit labeling are + # different roles — hence two fields. + self._lane_alias = session_model_alias # Wire caps: the caller's resolved session caps, or the provider's # static table as a last resort for degraded / legacy callers. self._capabilities = ( @@ -508,15 +526,18 @@ class OutputGuardJudge: # worker is non-daemon, and concurrent.futures joins it from an atexit # hook regardless of shutdown(wait=False) — so a wedged upstream call # would otherwise hang shutdown.) - # Single-shot lane: constructor-resolved capabilities, extra_params - # and live operator flags from the registry like every other lane. + # Single-shot lane: constructor-frozen capabilities (window-coupled, + # refreshed on judge swap — see IntentJudge's lane note), extra_params + # / live flags / temperature ladder from the registry like every + # other lane. ``_lane_alias``, not the audit label. lane = resolve_lane( self._provider, client, self._model, - alias=self._judge_model_alias, + alias=self._lane_alias, registry=self._model_registry, capabilities=self._capabilities, + config_store=self._config_store, ) # Temperature deliberately not pinned (house rule) — the lane # inherits the guard model's configured temperature. diff --git a/turnstone/core/perception.py b/turnstone/core/perception.py index 920a1e87..2237bd41 100644 --- a/turnstone/core/perception.py +++ b/turnstone/core/perception.py @@ -75,6 +75,9 @@ def describe( model: str, parts: list[dict[str, Any]], prompt: str = _DESCRIBE_PROMPT, + alias: str = "", + registry: Any | None = None, + config_store: Any | None = None, ) -> str: """Perceive ``parts`` via the perception model, returning the text. @@ -82,10 +85,16 @@ def describe( image/PDF-page perception, ``input_audio`` for audio. The trajectory carries them by reference; ``model_turn`` hands the resolver to the provider translator, which materializes the placeholder into these exact - parts (one ref may expand to many, e.g. a rasterized PDF). Temperature - is not pinned (house rule) — the perception model's own configuration - governs sampling. Raises :class:`PerceptionBackendError` if the backend - call fails. Never caches — see :func:`describe_cached`. + parts (one ref may expand to many, e.g. a rasterized PDF). + + *alias* / *registry* / *config_store* make the perception lane a real + lane: the alias's capability overrides, ``server_compat.extra_body`` + pins, and the temperature ladder (per-model → global + ``model.temperature`` → server default; house rule: no code pins) all + reach the wire, so an operator can actually remediate a degraded, + memoized description from the Models tab. Raises + :class:`PerceptionBackendError` if the backend call fails. Never + caches — see :func:`describe_cached`. """ if not parts: return "" @@ -99,7 +108,9 @@ def describe( ), ) ] - lane = resolve_lane(provider, client, model) + lane = resolve_lane( + provider, client, model, alias=alias, registry=registry, config_store=config_store + ) try: result = model_turn( lane, @@ -136,6 +147,8 @@ def describe_cached( content_hash: str, parts: list[dict[str, Any]], prompt: str = _DESCRIBE_PROMPT, + registry: Any | None = None, + config_store: Any | None = None, ) -> str: """Memoized, non-raising :func:`describe` for the wire fallback. @@ -154,6 +167,9 @@ def describe_cached( model=model, parts=parts, prompt=prompt, + alias=alias, + registry=registry, + config_store=config_store, ) except PerceptionBackendError as exc: log.warning("perception fallback failed (alias=%s): %s", alias, exc) diff --git a/turnstone/core/providers/_anthropic.py b/turnstone/core/providers/_anthropic.py index f39ca8d6..2107b320 100644 --- a/turnstone/core/providers/_anthropic.py +++ b/turnstone/core/providers/_anthropic.py @@ -407,7 +407,7 @@ class AnthropicProvider: reasoning_effort: str, extra_params: dict[str, Any] | None, max_tokens: int, - temperature: float, + temperature: float | None, converted_msgs: list[dict[str, Any]], system_prompt: str, model: str, @@ -453,7 +453,10 @@ class AnthropicProvider: # 90% input cost reduction on cache hits; 1.25x write on first turn. "cache_control": {"type": "ephemeral"}, } - if caps.supports_temperature: + # None temperature is never written — the request omits the field so + # the server default applies (house rule: no code pins). The + # thinking branches above still force 1.0 where the API requires it. + if caps.supports_temperature and temperature is not None: kwargs["temperature"] = temperature if system_prompt: kwargs["system"] = system_prompt @@ -1075,7 +1078,7 @@ class AnthropicProvider: messages: list[dict[str, Any]], tools: list[dict[str, Any]] | None = None, max_tokens: int = 4096, - temperature: float = 0.5, + temperature: float | None = None, reasoning_effort: str = "medium", extra_params: dict[str, Any] | None = None, deferred_names: frozenset[str] | None = None, diff --git a/turnstone/core/providers/_openai_chat.py b/turnstone/core/providers/_openai_chat.py index 547a4266..79724910 100644 --- a/turnstone/core/providers/_openai_chat.py +++ b/turnstone/core/providers/_openai_chat.py @@ -312,7 +312,7 @@ class OpenAIChatCompletionsProvider: messages: list[dict[str, Any]], tools: list[dict[str, Any]] | None = None, max_tokens: int = 4096, - temperature: float = 0.5, + temperature: float | None = None, reasoning_effort: str = "medium", extra_params: dict[str, Any] | None = None, deferred_names: frozenset[str] | None = None, diff --git a/turnstone/core/providers/_openai_common.py b/turnstone/core/providers/_openai_common.py index 567a599b..dd982aaa 100644 --- a/turnstone/core/providers/_openai_common.py +++ b/turnstone/core/providers/_openai_common.py @@ -361,16 +361,22 @@ def lookup_openai_capabilities(model: str) -> ModelCapabilities: def apply_temperature( kwargs: dict[str, Any], caps: ModelCapabilities, - temperature: float, + temperature: float | None, reasoning_effort: str, ) -> None: """Conditionally add temperature to *kwargs*. + - ``None`` temperature is never written: the request omits the field + so the SERVER default applies. House rule: code never pins a + temperature — a Python-level constant here would silently re-pin + every lane that deliberately left it unresolved. - Models with ``supports_temperature=False`` (GPT-5 base, O-series) never receive temperature. - Models that list ``"none"`` in their effort values (GPT-5.1/5.2) only receive temperature when reasoning is inactive. """ + if temperature is None: + return if not caps.supports_temperature: return if "none" in caps.reasoning_effort_values and reasoning_effort not in ("none", ""): @@ -381,7 +387,7 @@ def apply_temperature( def apply_temperature_and_effort( kwargs: dict[str, Any], caps: ModelCapabilities, - temperature: float, + temperature: float | None, reasoning_effort: str, ) -> None: """Conditionally add temperature and reasoning_effort to *kwargs*. diff --git a/turnstone/core/providers/_openai_responses.py b/turnstone/core/providers/_openai_responses.py index 5a2421fe..58d71739 100644 --- a/turnstone/core/providers/_openai_responses.py +++ b/turnstone/core/providers/_openai_responses.py @@ -354,7 +354,7 @@ class OpenAIResponsesProvider: messages: list[dict[str, Any]], tools: list[dict[str, Any]] | None, max_tokens: int, - temperature: float, + temperature: float | None, reasoning_effort: str, deferred_names: frozenset[str] | None, capabilities: ModelCapabilities | None = None, @@ -676,7 +676,7 @@ class OpenAIResponsesProvider: messages: list[dict[str, Any]], tools: list[dict[str, Any]] | None = None, max_tokens: int = 4096, - temperature: float = 0.5, + temperature: float | None = None, reasoning_effort: str = "medium", extra_params: dict[str, Any] | None = None, deferred_names: frozenset[str] | None = None, diff --git a/turnstone/core/providers/_protocol.py b/turnstone/core/providers/_protocol.py index 09204e33..7865220a 100644 --- a/turnstone/core/providers/_protocol.py +++ b/turnstone/core/providers/_protocol.py @@ -469,7 +469,7 @@ class LLMProvider(Protocol): messages: list[dict[str, Any]], tools: list[dict[str, Any]] | None = None, max_tokens: int = 4096, - temperature: float = 0.5, + temperature: float | None = None, reasoning_effort: str = "medium", extra_params: dict[str, Any] | None = None, deferred_names: frozenset[str] | None = None, @@ -480,6 +480,12 @@ class LLMProvider(Protocol): ) -> CompletionResult: """Create a non-streaming request, returning a normalized result. + ``temperature=None`` (the default) means the field is OMITTED from + the wire request and the server's own default applies — it must + never be replaced by a Python-level constant (house rule: code + never pins a temperature; ``model_turn`` resolves the operator's + ladder and passes ``None`` when nothing is configured). + ``replay_reasoning_to_model`` mirrors the per-model ``model_definitions`` operator flag. Anthropic uses it to gate the verbatim ``_provider_content`` replay (Phase 2); diff --git a/turnstone/core/providers/_xai.py b/turnstone/core/providers/_xai.py index 8c7882fa..afa38ecd 100644 --- a/turnstone/core/providers/_xai.py +++ b/turnstone/core/providers/_xai.py @@ -164,7 +164,7 @@ class XAIProvider(OpenAIResponsesProvider): messages: list[dict[str, Any]], tools: list[dict[str, Any]] | None, max_tokens: int, - temperature: float, + temperature: float | None, reasoning_effort: str, deferred_names: frozenset[str] | None, capabilities: ModelCapabilities | None = None, diff --git a/turnstone/core/session.py b/turnstone/core/session.py index d5aa875d..77de21cd 100644 --- a/turnstone/core/session.py +++ b/turnstone/core/session.py @@ -124,7 +124,6 @@ from turnstone.core.metacognition import ( should_nudge, ) from turnstone.core.model_turn import ( - ModelLane, ModelTurnResult, ensure_tool_call_ids, finalize_provider_blocks, @@ -132,6 +131,7 @@ from turnstone.core.model_turn import ( model_turn, provider_extra_params, resolve_capabilities, + resolve_lane, resolve_replay_reasoning_to_model, ) from turnstone.core.nudge_queue import ( @@ -2043,9 +2043,14 @@ class ChatSession: """Get model capabilities, applying config.toml overrides if present. Delegates to :func:`turnstone.core.model_turn.resolve_capabilities` — - the one resolution path every lane shares (#827). + the one resolution path every lane shares (#827) — but fetches the + config ITSELF, uncaught: a registry failure on the session's own + alias must raise loudly (pre-#827 semantics), never silently cache + degraded static-table caps for the session lifetime. The defensive + never-crash fetch is a judge-constructor property, not a session one. """ - return resolve_capabilities(provider, model, alias or "", self._registry) + cfg = self._registry.get_config(alias) if (self._registry and alias) else None + return resolve_capabilities(provider, model, alias or "", self._registry, cfg=cfg) def _get_capabilities(self, provider: Any = None, model: str = "") -> ModelCapabilities: """Get capabilities for a model. Cached for the primary session model.""" @@ -4234,6 +4239,14 @@ class ChatSession: alias=alias, content_hash=content_hash, parts=parts, + # Thread the registry + config store so the perception lane + # resolves the alias's extra_params / capability overrides / + # temperature ladder like every other lane — without these, + # operator settings on the perception alias never reach the + # wire and there is no remediation path for a degraded, + # memoized description. + registry=self._registry, + config_store=self._config_store, ) if not text: return None @@ -4762,14 +4775,15 @@ class ChatSession: """ caps = self._get_capabilities() clamped = min(max_tokens, caps.max_output_tokens) if caps.max_output_tokens else max_tokens - lane = ModelLane( - provider=self._provider, - client=self.client, - model=self.model, + lane = resolve_lane( + self._provider, + self.client, + self.model, alias=self._model_alias or "", + registry=self._registry, capabilities=caps, extra_params=self._provider_extra_params(), - registry=self._registry, + config_store=self._config_store, ) result = model_turn( lane, @@ -7754,6 +7768,8 @@ class ChatSession: # model — thread its alias too, so the lane resolves # extra_params / live flags like every other session lane. session_model_alias=self._model_alias or "", + # For the temperature ladder's global rung (model.temperature). + config_store=self._config_store, ) except Exception: log.warning("judge.init_failed", exc_info=True) @@ -7789,6 +7805,8 @@ class ChatSession: # source IntentJudge gets via _ensure_judge. session_capabilities=self._get_capabilities(), session_model_alias=self._model_alias or "", + # For the temperature ladder's global rung (model.temperature). + config_store=self._config_store, ) except Exception: log.warning("output_guard_judge.init_failed", exc_info=True) @@ -15064,18 +15082,24 @@ class ChatSession: # operator flags (replay-reasoning, Phase 5 vLLM attach) re-resolve # inside ``model_turn`` through the carried registry, so mid-session # admin toggles keep applying exactly as they did pre-extraction. + # Temperature follows the AGENT model's own ladder (its ModelConfig, + # else the global ``model.temperature``) — relaying the session + # model's value here would make a task alias's configured + # temperature unreachable (house rule: the model's configuration is + # the source of truth). # Agent trajectories stay excluded from the persistence/replay # contract — history is in-memory, rebuilt per ``_run_agent`` # invocation; the native lane carried here serves the WITHIN-RUN # reasoning continuity of the agent's own tool loop. - lane = ModelLane( - provider=agent_provider, - client=agent_client, - model=agent_model, + lane = resolve_lane( + agent_provider, + agent_client, + agent_model, alias=agent_alias or "", + registry=self._registry, capabilities=agent_caps, extra_params=agent_extra, - registry=self._registry, + config_store=self._config_store, ) def _api_call( @@ -15098,7 +15122,6 @@ class ChatSession: turns, tools=_tools, max_tokens=self.max_tokens, - temperature=self.temperature, reasoning_effort=reasoning_effort or self.reasoning_effort, mint=mint, wire_id_map=wire_id_map, diff --git a/turnstone/eval/core.py b/turnstone/eval/core.py index b2a0df69..edfa5c31 100644 --- a/turnstone/eval/core.py +++ b/turnstone/eval/core.py @@ -30,7 +30,7 @@ from typing import Any from openai import OpenAI -from turnstone.core.model_turn import ModelLane, model_turn +from turnstone.core.model_turn import cap_tool_calls, model_turn, resolve_lane from turnstone.core.providers import LLMProvider, create_client, create_provider from turnstone.core.session import ChatSession from turnstone.core.storage import get_storage, init_storage, reset_storage @@ -306,14 +306,14 @@ class HeadlessSession(ChatSession): # The eval lane — resolved once per run, like the sub-agent seam. # ``temperature`` relays the harness's operator-resolved knob per # call below (house rule: relay, never pin). - lane = ModelLane( - provider=self._provider, - client=self.client, - model=self.model, + lane = resolve_lane( + self._provider, + self.client, + self.model, alias=self._model_alias or "", + registry=self._registry, capabilities=self._get_capabilities(), extra_params=self._provider_extra_params(), - registry=self._registry, ) for turn in range(max_turns): @@ -345,18 +345,10 @@ class HeadlessSession(ChatSession): self._total_usage["prompt"] += result.usage.prompt_tokens self._total_usage["completion"] += result.usage.completion_tokens - # Cap parallel tool calls to prevent degenerate repetition. The - # cap applies to the mirror AND the appended turn; a capped turn - # drops its native lane rather than replaying orphan native - # tool_use blocks the mirror no longer carries. - capped_calls = (result.tool_calls or [])[:10] - assistant_turn = result.turn - if len(result.tool_calls) > len(capped_calls): - assistant_turn = Turn( - role=Role.ASSISTANT, - content=assistant_turn.content, - tool_calls=assistant_turn.tool_calls[: len(capped_calls)], - ) + # Cap parallel tool calls to prevent degenerate repetition + # (shared guard — see model_turn.cap_tool_calls for the + # native-lane-drop rationale on capped turns). + capped_calls, assistant_turn = cap_tool_calls(result, 10) self.messages.append(assistant_turn) msg_len = len(result.content or "") diff --git a/turnstone/optimizer.py b/turnstone/optimizer.py index 3d1b9fb8..9c34ba4c 100644 --- a/turnstone/optimizer.py +++ b/turnstone/optimizer.py @@ -27,7 +27,7 @@ from typing import Any from openai import OpenAI -from turnstone.core.model_turn import model_turn, resolve_lane +from turnstone.core.model_turn import cap_tool_calls, model_turn, resolve_lane from turnstone.core.providers import LLMProvider, create_provider from turnstone.core.session import ChatSession from turnstone.core.trajectory import Role, Turn @@ -776,15 +776,9 @@ def _run_analyst( reasoning_effort="medium", ) - # Same degenerate-repetition cap as before; a capped turn drops its - # native lane rather than replay orphan native tool blocks the - # mirror no longer carries. - capped = mtr.tool_calls[:5] - assistant_turn = mtr.turn - if len(mtr.tool_calls) > len(capped): - assistant_turn = Turn.assistant( - mtr.content, tool_calls=mtr.turn.tool_calls[: len(capped)] - ) + # Same degenerate-repetition cap as before (shared guard — see + # model_turn.cap_tool_calls for the native-lane-drop rationale). + capped, assistant_turn = cap_tool_calls(mtr, 5) turns.append(assistant_turn) if not capped: