mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-12 23:12:23 -06:00
bc3fa60011
Passthrough servers (parserless vLLM/llama.cpp, LM Studio, bare gateways) emit reasoning as literal <think>/<reasoning> blocks inside content, and only three of nine drained lanes stripped them: web_fetch tool results persisted raw think blocks into every following turn (#940), judge verdicts parsed through tag noise, and a draft verdict inside a think block could shadow the real one at the output guard. One rule at the seam now. drain_stream accumulates content in RUNS bounded by interleaving signals (provider-parsed reasoning deltas, tool-call deltas) with the interactive consumer's within-chunk ordering — reasoning, then content, then the tool-call close — and splits each run through split_inline_reasoning, the one-shot form of the interactive lane's ThinkTagSplitter: a pure raw split, exactly equivalent to the streaming form on every catalog case. One trim policy exists and the drain owns it: blank edge lines are trimmed once over the joined runs when a tag was consumed, so tag residue dies at the edges while genuine inter-run paragraph separators survive. Extracted text is appended to result.reasoning after any server-parsed reasoning with a blank-line boundary and rides the native lane as the reasoning_text synth block. Orphan CLOSE tags deliberately pass through byte-identical: a close whose open never arrived is indistinguishable from prose QUOTING the tag, and drained lanes routinely quote third-party text — reclassifying would let a malicious page containing the literal tag destroy the extraction that cites it. The title lane keeps a local rfind peel as display-string formatting. The citations footer folds only onto non-blank content — sourcing for an answer that does not exist is dropped rather than handed to emptiness checks as a footer-only "answer". Every private strip is deleted: the title lane's strip, the summarizer strip, _strip_reasoning itself, and the optimizer's five regexes (_strip_markdown_fence is now the one fence rule, applied to normalized model output only, never to or-fallback values). Think-only and whitespace-only responses drain to blank content, and every lane's no-answer fallback gates on blankness: web_fetch returns an honest extraction-error card, the intent judge takes the empty-retry ladder, the task-agent synthesis reports "(no output)", and the optimizer keeps the current observer system and prompt verbatim on no-answer passes. Final-say reads (optimizer analyst, eval final_content, the notify hook) use trajectory.final_assistant_text — the last assistant turn only, never an earlier narration presented as the conclusion — while last_assistant_text is the salvage walk (task_agent partial-work recovery), skipping tool-call-only, all-reasoning, and whitespace-only turns. Perception memoizes every completed description immediately, including an empty one — one perceive per key, ever — under a commit-lock guard so an empty result never overwrites a concurrently memoized real description; an all-reasoning perception model pins the placeholder until restart, and the remediation is server-side (a reasoning parser or the template thinking toggle on the perception alias). A true double-reasoning shape (inline-extracted text alongside a native reasoning block) logs chars-only at the drain, where it is distinguishable from the routine reasoning_delta mirror. The dialect's semantics are pinned as one table (tests/_reasoning_dialect.py) driven through shared fixtures (think_tag_stream, seam_provider): one-shot conformance, the exact one-shot/streaming equivalence property, the drain seam rules including quoted-tag safety, run-boundary and separator-preservation pins, per-lane pins for all nine lanes, and the empty-content assistant wire shape. Closes #965. Closes #940.
342 lines
12 KiB
Python
342 lines
12 KiB
Python
"""dict ↔ Turn adapter fidelity (the strangler bridge for the Turn migration).
|
|
|
|
The contract is byte-identical round-tripping — ``turn_to_dict(turn_from_dict(d))
|
|
== d`` — for every message-dict shape ``reconstruct_messages`` and the wire path
|
|
produce. These tests pin that, plus the field mapping (``_``-side channels →
|
|
typed ``Turn`` fields).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from turnstone.core.trajectory import (
|
|
AttachmentRef,
|
|
ProviderNative,
|
|
Role,
|
|
TextBlock,
|
|
ToolCall,
|
|
Turn,
|
|
final_assistant_text,
|
|
last_assistant_text,
|
|
materialize_attachments,
|
|
resolve_attachment_parts,
|
|
turn_from_dict,
|
|
turn_to_dict,
|
|
turns_from_dicts,
|
|
)
|
|
|
|
# Every shape reconstruct / the wire path emits, as the round-trip corpus.
|
|
_ROUNDTRIP: list[dict[str, Any]] = [
|
|
{"role": "user", "content": "hello"},
|
|
{"role": "user", "content": "with source", "_source": "user_interjection"},
|
|
{"role": "user", "content": "ev", "_event_id": 42},
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": "what's this?"},
|
|
{"type": "image", "attachment_id": "sha256:aaaa"},
|
|
],
|
|
"_attachments_meta": [{"kind": "image", "filename": "x.png", "mime_type": "image/png"}],
|
|
},
|
|
{
|
|
# All-text multipart list (the unreadable-attachment placeholder path)
|
|
# stays a list — must not collapse to a joined string.
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": "read this"},
|
|
{"type": "text", "text": "[unreadable attachment: bad.bin]"},
|
|
],
|
|
},
|
|
{
|
|
# By-reference attachments: the canonical content form (id, never bytes).
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": "what's this?"},
|
|
{"type": "image", "attachment_id": "sha256:abc"},
|
|
{"type": "document", "attachment_id": "sha256:def"},
|
|
],
|
|
},
|
|
{"role": "assistant", "content": "hi there"},
|
|
{"role": "assistant", "content": ""}, # empty assistant (no text, no tools)
|
|
{
|
|
"role": "assistant",
|
|
"content": "",
|
|
"tool_calls": [
|
|
{"id": "c1", "type": "function", "function": {"name": "bash", "arguments": '{"x":1}'}},
|
|
],
|
|
},
|
|
{
|
|
"role": "assistant",
|
|
"content": "let me think",
|
|
"_provider_content": [{"type": "thinking", "thinking": "hmm", "signature": "s"}],
|
|
"_producer": "anthropic",
|
|
},
|
|
{
|
|
# native lane WITHOUT a producer tag (the legacy bare-list path).
|
|
"role": "assistant",
|
|
"content": "x",
|
|
"_provider_content": [{"type": "reasoning_text", "text": "r"}],
|
|
},
|
|
{"role": "tool", "tool_call_id": "c1", "content": "result"},
|
|
{"role": "tool", "tool_call_id": "c2", "content": "boom", "is_error": True},
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": "c3",
|
|
"content": [
|
|
{"type": "text", "text": "saw an image"},
|
|
{"type": "image", "attachment_id": "sha256:bbbb"},
|
|
],
|
|
},
|
|
{"role": "system", "content": "guard note", "_source": "output_guard"},
|
|
{"role": "system", "content": "you are an assistant"}, # base prompt, no _source
|
|
# Operator turn with structured per-kind meta (the watch-result card source).
|
|
{
|
|
"role": "system",
|
|
"content": "ci failed",
|
|
"_source": "watch_triggered",
|
|
"_source_meta": {"watch_name": "ci", "command": "make test", "poll_count": 3},
|
|
},
|
|
]
|
|
|
|
|
|
@pytest.mark.parametrize("msg", _ROUNDTRIP, ids=range(len(_ROUNDTRIP)))
|
|
def test_dict_turn_dict_roundtrip(msg: dict[str, Any]) -> None:
|
|
assert turn_to_dict(turn_from_dict(msg)) == msg
|
|
|
|
|
|
def test_turns_from_dicts_preserves_order_and_count() -> None:
|
|
turns = turns_from_dicts(_ROUNDTRIP)
|
|
assert len(turns) == len(_ROUNDTRIP)
|
|
assert [t.role.value for t in turns] == [
|
|
m["role"] if m["role"] != "developer" else "system" for m in _ROUNDTRIP
|
|
]
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# Field mapping — the side channels become typed Turn fields.
|
|
# --------------------------------------------------------------------------- #
|
|
def test_source_meta_maps_to_meta_extra() -> None:
|
|
# The per-kind operator meta rides a single ``_source_meta`` dict and lands
|
|
# in ``Turn.meta.extra["source_meta"]`` (and round-trips back out).
|
|
t = turn_from_dict(
|
|
{
|
|
"role": "system",
|
|
"content": "ci failed",
|
|
"_source": "watch_triggered",
|
|
"_source_meta": {"watch_name": "ci", "poll_count": 3},
|
|
}
|
|
)
|
|
assert t.meta.extra["source_meta"] == {"watch_name": "ci", "poll_count": 3}
|
|
assert turn_to_dict(t)["_source_meta"] == {"watch_name": "ci", "poll_count": 3}
|
|
|
|
|
|
def test_empty_source_meta_omitted() -> None:
|
|
# An empty meta dict is not carried (no key on the Turn, none re-emitted).
|
|
t = turn_from_dict(
|
|
{"role": "system", "content": "n", "_source": "correction", "_source_meta": {}}
|
|
)
|
|
assert "source_meta" not in t.meta.extra
|
|
assert "_source_meta" not in turn_to_dict(t)
|
|
|
|
|
|
def test_source_maps_to_source_field() -> None:
|
|
t = turn_from_dict({"role": "system", "content": "n", "_source": "tool_error"})
|
|
assert t.role is Role.SYSTEM
|
|
assert t.source == "tool_error"
|
|
|
|
|
|
def test_provider_content_maps_to_native_lane() -> None:
|
|
t = turn_from_dict(
|
|
{
|
|
"role": "assistant",
|
|
"content": "x",
|
|
"_provider_content": [{"type": "thinking", "thinking": "z"}],
|
|
"_producer": "anthropic",
|
|
}
|
|
)
|
|
assert isinstance(t.native, ProviderNative)
|
|
assert t.native.producer == "anthropic"
|
|
assert t.native.blocks == ({"type": "thinking", "thinking": "z"},)
|
|
|
|
|
|
def test_native_without_producer_defaults_empty_and_omits_on_emit() -> None:
|
|
t = turn_from_dict(
|
|
{"role": "assistant", "content": "x", "_provider_content": [{"type": "reasoning_text"}]}
|
|
)
|
|
assert t.native is not None and t.native.producer == ""
|
|
# Empty producer is not re-emitted (matches the bare-list legacy dict).
|
|
assert "_producer" not in turn_to_dict(t)
|
|
|
|
|
|
def test_tool_call_id_and_is_error_map_to_tool_fields() -> None:
|
|
t = turn_from_dict({"role": "tool", "tool_call_id": "c1", "content": "e", "is_error": True})
|
|
assert t.role is Role.TOOL
|
|
assert t.tool_call_id == "c1"
|
|
assert t.is_error is True
|
|
|
|
|
|
def test_event_id_maps_to_meta() -> None:
|
|
t = turn_from_dict({"role": "user", "content": "x", "_event_id": 7})
|
|
assert t.meta.event_id == 7
|
|
|
|
|
|
def test_tool_calls_map_to_typed_toolcalls() -> None:
|
|
t = turn_from_dict(
|
|
{
|
|
"role": "assistant",
|
|
"content": "",
|
|
"tool_calls": [
|
|
{"id": "c1", "type": "function", "function": {"name": "f", "arguments": "{}"}},
|
|
],
|
|
}
|
|
)
|
|
assert t.tool_calls == (ToolCall(id="c1", name="f", arguments="{}"),)
|
|
|
|
|
|
def test_multipart_text_textblock_placeholder_attachmentref() -> None:
|
|
t = turn_from_dict(
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": "q"},
|
|
{"type": "image", "attachment_id": "sha256:abc"},
|
|
],
|
|
}
|
|
)
|
|
assert isinstance(t.content[0], TextBlock) and t.content[0].text == "q"
|
|
assert isinstance(t.content[1], AttachmentRef)
|
|
assert t.text == "q" # only the text part contributes to FTS
|
|
|
|
|
|
def test_developer_collapses_to_system() -> None:
|
|
t = turn_from_dict({"role": "developer", "content": "d"})
|
|
assert t.role is Role.SYSTEM
|
|
# Re-emits as system (wire-identical: providers treat system/developer alike).
|
|
assert turn_to_dict(t)["role"] == "system"
|
|
|
|
|
|
def test_empty_content_roundtrips_to_empty_string() -> None:
|
|
assert turn_to_dict(Turn(Role.ASSISTANT)) == {"role": "assistant", "content": ""}
|
|
|
|
|
|
def test_attachment_ref_is_the_canonical_non_text_form() -> None:
|
|
# By-reference placeholders ``{type: image|document, attachment_id}`` are the
|
|
# canonical non-text content; a resolved inline ``image_url`` (no id) never
|
|
# reaches turn_from_dict on the canonical path and is dropped if it does.
|
|
t = turn_from_dict(
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": "q"},
|
|
{"type": "image", "attachment_id": "sha256:abc"},
|
|
{"type": "document", "attachment_id": "sha256:def"},
|
|
{"type": "image_url", "image_url": {"url": "data:..."}},
|
|
],
|
|
}
|
|
)
|
|
assert [type(b).__name__ for b in t.content] == ["TextBlock", "AttachmentRef", "AttachmentRef"]
|
|
assert t.content[1].attachment_id == "sha256:abc" # type: ignore[union-attr]
|
|
assert t.content[2].kind == "document" # type: ignore[union-attr]
|
|
|
|
|
|
def test_attachment_ref_emits_placeholder() -> None:
|
|
t = Turn(Role.USER, (AttachmentRef(attachment_id="abc", kind="image"),))
|
|
assert turn_to_dict(t)["content"] == [{"type": "image", "attachment_id": "abc"}]
|
|
|
|
|
|
def test_resolve_attachment_parts_materializes_and_drops_missing() -> None:
|
|
# The dict-side resolver: placeholders → inline parts; a pruned id is dropped.
|
|
part = {"type": "image_url", "image_url": {"url": "data:img"}}
|
|
messages = [
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "text", "text": "look"},
|
|
{"type": "image", "attachment_id": "a1"},
|
|
{"type": "image", "attachment_id": "gone"},
|
|
],
|
|
}
|
|
]
|
|
out = resolve_attachment_parts(messages, {"a1": part})
|
|
assert out[0]["content"] == [{"type": "text", "text": "look"}, part]
|
|
|
|
|
|
def test_resolve_attachment_parts_identity_when_no_placeholders() -> None:
|
|
messages = [{"role": "user", "content": "hi"}]
|
|
assert resolve_attachment_parts(messages, {}) is messages
|
|
|
|
|
|
def test_materialize_attachments_collects_ids_and_substitutes() -> None:
|
|
part = {"type": "image_url", "image_url": {"url": "data:img"}}
|
|
seen_ids: list[list[str]] = []
|
|
|
|
def _resolve(ids: list[str]) -> dict[str, dict[str, object]]:
|
|
seen_ids.append(ids)
|
|
return {"a1": part}
|
|
|
|
messages = [
|
|
{"role": "user", "content": [{"type": "image", "attachment_id": "a1"}]},
|
|
]
|
|
out = materialize_attachments(messages, _resolve)
|
|
assert seen_ids == [["a1"]] # collected the placeholder id
|
|
assert out[0]["content"] == [part]
|
|
|
|
|
|
def test_materialize_attachments_noop_without_resolver_or_placeholders() -> None:
|
|
messages = [{"role": "user", "content": "hi"}]
|
|
assert materialize_attachments(messages, None) is messages
|
|
assert materialize_attachments(messages, lambda ids: {}) is messages
|
|
|
|
|
|
def test_last_assistant_text_picks_most_recent_substantive() -> None:
|
|
turns = [
|
|
Turn.user("q"),
|
|
Turn.assistant("early answer"),
|
|
Turn.assistant("", tool_calls=(ToolCall(id="c1", name="bash", arguments="{}"),)),
|
|
Turn.tool("c1", "out"),
|
|
Turn.assistant("final answer"),
|
|
]
|
|
assert last_assistant_text(turns) == "final answer"
|
|
|
|
|
|
def test_last_assistant_text_skips_textless_finals() -> None:
|
|
# Tool-call-only and all-reasoning (empty-text) finals are skipped —
|
|
# the walk falls back to the last SUBSTANTIVE assistant text.
|
|
turns = [
|
|
Turn.assistant("substantive"),
|
|
Turn.assistant("", tool_calls=(ToolCall(id="c2", name="bash", arguments="{}"),)),
|
|
Turn.assistant(""),
|
|
]
|
|
assert last_assistant_text(turns) == "substantive"
|
|
|
|
|
|
def test_last_assistant_text_none_when_no_assistant_text() -> None:
|
|
assert last_assistant_text([]) is None
|
|
assert last_assistant_text([Turn.user("q"), Turn.assistant("")]) is None
|
|
|
|
|
|
def test_last_assistant_text_skips_whitespace_only_turns() -> None:
|
|
# "Substantive" means non-blank, not merely truthy: a whitespace-only
|
|
# final turn must not be salvaged as partial work.
|
|
turns = [Turn.assistant("real work"), Turn.assistant(" \n")]
|
|
assert last_assistant_text(turns) == "real work"
|
|
assert last_assistant_text([Turn.assistant(" \n")]) is None
|
|
|
|
|
|
def test_final_assistant_text_reads_last_turn_only() -> None:
|
|
# The final-say read: no walk-back — an empty final say reports empty,
|
|
# never an earlier turn's narration.
|
|
substantive_then_empty = [Turn.assistant("mid-loop narration"), Turn.assistant("")]
|
|
assert final_assistant_text(substantive_then_empty) == ""
|
|
tool_final = [
|
|
Turn.assistant("narration"),
|
|
Turn.assistant("", tool_calls=(ToolCall(id="c1", name="bash", arguments="{}"),)),
|
|
]
|
|
assert final_assistant_text(tool_final) == ""
|
|
assert final_assistant_text([Turn.assistant(" the answer\n")]) == "the answer"
|
|
assert final_assistant_text([Turn.user("q")]) == ""
|
|
assert final_assistant_text([]) == ""
|