mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-13 07:22:24 -06:00
Compare commits
74 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 78f7b12c2d | |||
| 647939fe4d | |||
| 457b01737a | |||
| a40ff249ec | |||
| 36419a9809 | |||
| 0c2c534c86 | |||
| 5fded65b82 | |||
| 7da731cbe1 | |||
| 8ed86ae7ab | |||
| ed30e4f0bf | |||
| 62f62ae624 | |||
| 5ae6c2316f | |||
| d793adb24c | |||
| 3cf94dd80f | |||
| 0422f9214a | |||
| 4428e185e5 | |||
| c5ff3147ce | |||
| bfcfb0c791 | |||
| cbf5c5f3b6 | |||
| 31a1d5c3ee | |||
| 93a7486cc2 | |||
| 56624f9597 | |||
| 16a68ae6d6 | |||
| 2d4cb6fea9 | |||
| 357d00400e | |||
| 3615f98c19 | |||
| 801d5dfb59 | |||
| a352b20786 | |||
| 6f8efaa44e | |||
| 9c90fe2722 | |||
| 1035fe05eb | |||
| 530958e06b | |||
| e136237b63 | |||
| 41e9907803 | |||
| 7c34d859b4 | |||
| 16ee4e12ef | |||
| 976c07d047 | |||
| 8da5dc3f5a | |||
| 7f0e0406b3 | |||
| 6c94514106 | |||
| 0d0fe8dd71 | |||
| 18c3301428 | |||
| 185dcc2960 | |||
| 0fe8e4106f | |||
| fd65a490dc | |||
| 3607517814 | |||
| a0e04a8588 | |||
| ffe8214cfe | |||
| 1f63f622c9 | |||
| f4701bf0f9 | |||
| 06cc184227 | |||
| 59a527f2f2 | |||
| d7941c88be | |||
| c64dc16319 | |||
| d564cee43d | |||
| 2cf23b6fe2 | |||
| 68b22adfa3 | |||
| 9289693730 | |||
| deff44bcea | |||
| 217d3a3a9b | |||
| bcf509a440 | |||
| 10f726f83d | |||
| acc262c405 | |||
| 62034378c6 | |||
| bcb8c5ab88 | |||
| fec5067fcd | |||
| b0ed67aa60 | |||
| c023272b16 | |||
| 3568a6db50 | |||
| 9bf8d5699b | |||
| c0ff00a1ff | |||
| 45010f5890 | |||
| 845df69031 | |||
| d47d528d9a |
@@ -0,0 +1,5 @@
|
||||
# Funding platforms for the GitHub "Sponsor" button.
|
||||
# https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/displaying-a-sponsor-button-in-your-repository
|
||||
|
||||
github: [eous]
|
||||
custom: ["https://paypal.me/eousphoros"]
|
||||
@@ -152,7 +152,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- run: uv lock --check
|
||||
@@ -161,7 +161,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@6c0083bb7289c31716797a039b6367b3079cc46e # v1
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
allowed_bots: 'renovate[bot]' # let Renovate PRs get reviewed
|
||||
|
||||
@@ -45,7 +45,7 @@ jobs:
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@6c0083bb7289c31716797a039b6367b3079cc46e # v1
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
|
||||
|
||||
@@ -54,7 +54,7 @@ jobs:
|
||||
|
||||
- name: Log in to GHCR
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v4
|
||||
uses: docker/login-action@af1e73f918a031802d376d3c8bbc3fe56130a9b0 # v4
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
@@ -78,7 +78,7 @@ jobs:
|
||||
fi
|
||||
echo "tags=${TAGS}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4
|
||||
- uses: docker/setup-buildx-action@bb05f3f5519dd87d3ba754cc423b652a5edd6d2c # v4
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
|
||||
- name: Build and push
|
||||
|
||||
@@ -28,3 +28,4 @@ tools/skill_audit_analysis/data/
|
||||
tools/skill_audit_analysis/output/
|
||||
design_ideas/
|
||||
.claude/
|
||||
docs/design/
|
||||
|
||||
+131
-4
@@ -13,7 +13,28 @@ stable, and the experimental line:
|
||||
- **`stable/1.6`** — patch-only (`v1.6.x`)
|
||||
- **`main`** — experimental (next major)
|
||||
|
||||
## [Unreleased]
|
||||
## [1.7.0]
|
||||
|
||||
The headline of the 1.7 line is **Personas** — operator-authored control
|
||||
over how each workstream composes its system message and capability
|
||||
envelope. The rest of the release hardens the pieces a persona leans on:
|
||||
concurrent approvals, cross-provider reasoning-effort control, cooperative
|
||||
compaction, multi-user session safety, and MCP resilience for unattended
|
||||
work.
|
||||
|
||||
> **⚠️ Before upgrading:** 1.7.0 adds Alembic migrations `062`–`065`,
|
||||
> applied automatically on first start (projects, personas, and two
|
||||
> smaller schema tidy-ups). Migration `063` creates the `personas` table
|
||||
> with its six seed personas and converts existing `creative_mode`
|
||||
> workstreams to the `writer` persona in place. The changes are additive
|
||||
> to your conversation data, but — as always — back up your storage before
|
||||
> upgrading (`pg_dump` for PostgreSQL; copy the database file for SQLite).
|
||||
|
||||
**Breaking changes at a glance** (details in the sections below): the
|
||||
`/creative` REPL toggle removed (replaced by the `writer` persona), the
|
||||
`turnstone-bootstrap` entry point renamed to `turnstone-doctor`, and the
|
||||
approval-status API/SDK field `pending_approval_details` changed from a
|
||||
single object to a list (one entry per concurrent approval cycle).
|
||||
|
||||
### Added
|
||||
|
||||
@@ -31,12 +52,106 @@ stable, and the experimental line:
|
||||
`turnstone --persona <name>`); authored in the console's new
|
||||
Governance → Personas tab (`persona.{create,read,write}` perms,
|
||||
archive-only lifecycle). See `docs/personas.md`.
|
||||
- **Projects — governed resource containers** (#724) — group workstreams
|
||||
and their resources under a project (migration `062`), with
|
||||
project-scoped memory, a per-project resources view, a project column on
|
||||
the saved list, and server-enforced private-project workstream
|
||||
visibility.
|
||||
- **Task-agent sub-harness** (#732) — a spawned task agent now runs on its
|
||||
own Turn-IR sub-harness with parent-tagged step events: its sub-tool
|
||||
steps nest inside an expandable card in the parent trajectory, its
|
||||
sub-trajectory is recallable, and each agent gets read isolation from
|
||||
its siblings.
|
||||
- **MCP static-server autonomous reconnect** (#768) — statically
|
||||
configured MCP servers are now kept live by a health loop
|
||||
(capped-jittered backoff, ping-based liveness) instead of silently
|
||||
staying dead after the first transport drop.
|
||||
- **Attachments — capability-gated client-side fallback** — when the
|
||||
active model can't natively handle an attachment, the client degrades
|
||||
gracefully (PDF → extracted text, audio → transcript) instead of
|
||||
failing the turn.
|
||||
- **Eval measurement / optimizer split** (#763, #765) — `turnstone-eval`
|
||||
is now a measure-only substrate with the prompt optimizer factored out,
|
||||
plus a new skill-adherence measurement mode.
|
||||
- **Deployment examples** — a vLLM + LiteLLM unified-memory inference
|
||||
example showing a 3-model co-resident stack with an HF loader (#686,
|
||||
#688), and an Altair + `vl-convert-python` visualization stack (#685).
|
||||
- **Concurrent approvals and a long-session frontend overhaul** (#754,
|
||||
#755, #773, #775) — the live-session frontend was reworked for long
|
||||
runs (the pipeline is wedge-proofed and its hot paths de-O(N)'d), and on
|
||||
top of it a workstream can now hold more than one tool call awaiting
|
||||
approval at a time. Each parallel batch gets its own approval cycle,
|
||||
with one card per pending call in the interactive and coordinator UIs,
|
||||
cycle-keyed tracking in Slack and Discord, and cycle-routed resolution
|
||||
across the server/console/SDK APIs; sub-agent tool gates run the
|
||||
intent-judge pipeline as their own generation. The send button no longer
|
||||
sticks disabled after a batch resolves — orphaned approval cycles are
|
||||
pruned and the app is the sole owner of the button state.
|
||||
*(BREAKING: the `pending_approval_details` field is now a list, oldest
|
||||
first.)*
|
||||
- **Reasoning-effort control on every provider lane** (#771, #774) — the
|
||||
session effort knob now reaches local backends too: it drives
|
||||
`chat_template_kwargs` on the anthropic-compatible and openai-compatible
|
||||
lanes and threads through to Gemini and xAI, alongside the commercial
|
||||
providers that handle effort natively. The console surfaces each model's
|
||||
effective effort ladder in plain words and adds an always-on
|
||||
thinking-mode option to the model form. Effort snapping is ordinal —
|
||||
it rounds up and caps at the model's ceiling rather than silently
|
||||
dropping.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Skills are capability-context, not identity** (#762) — a task agent's
|
||||
identity now comes from its persona; an applied skill's body is demoted
|
||||
to capability context and moved out of the identity system message.
|
||||
Skill-body substitution is unified across every invocation context so
|
||||
the same skill renders identically whether loaded interactively, by the
|
||||
model, or inside a sub-agent.
|
||||
- **`turnstone-doctor` replaces `turnstone-bootstrap`** (#718)
|
||||
*(BREAKING)* — the setup/diagnostics entry point is renamed; update any
|
||||
scripts or service units that invoke `turnstone-bootstrap`.
|
||||
- **Honest cancellation dispositions** — cancelled or timed-out
|
||||
side-effecting tools now report an `UNKNOWN` disposition rather than a
|
||||
flat failure, tool dispositions are typed (not just prose), and a
|
||||
coordinator cancel propagates down the sub-tree.
|
||||
- **Multi-user shared-workstream context** (#750) — in a shared
|
||||
workstream, send is gated to the acting participant while a turn is in
|
||||
flight (both the interactive and coordinator surfaces), cross-user
|
||||
mid-turn interjections are blocked, and shared-workstream state plus
|
||||
fork sender attribution are now durable.
|
||||
- **Cooperative compaction** (#730) — the context budget is anchored to
|
||||
the provider's true capacity, the summary call is chunked so it can't
|
||||
overflow, and the active plan and the outstanding ask are carried across
|
||||
compaction verbatim. The `recall` tool is scoped to the compacted-away
|
||||
past.
|
||||
- **Intent judge sees the full tool arguments** (#760) — the judge's
|
||||
argument projection is no longer narrowed, so it stops issuing confident
|
||||
false denials on a partial view. The output-guard judge sources its real
|
||||
context window, and `context_window = 0` in `config.toml` now means
|
||||
auto-detect.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Compaction resume hardening** (#731) — checkpoint markers are
|
||||
persisted so resume rehydration is bounded, context-overflow on resume
|
||||
is recovered across providers, and a recognized rate-limit is no longer
|
||||
misclassified as context overflow.
|
||||
- **MCP unattended-work resilience** (#706, #742, #767) — dead-transport
|
||||
handling is completed, consented OAuth (OBO) tokens are refreshed
|
||||
proactively so autonomous runs don't strand on an expired grant, the
|
||||
Entra ID on-behalf-of impersonation flow blockers are closed (migration
|
||||
`065` adds the OIDC `oid`), and OAuth refresh failures are classified so
|
||||
a transient blip never revokes consent nor a dead grant strands the
|
||||
user.
|
||||
- **Memory writes** (#735) — save/update is a single atomic upsert, and
|
||||
writing a memory no longer recomposes the system prefix mid-session.
|
||||
|
||||
### Removed
|
||||
|
||||
- **`/creative` removed** *(BREAKING)* — the REPL toggle (and its tab
|
||||
completion) is gone; the `writer` seed persona replaces it — start a
|
||||
session with `turnstone --persona writer` or pick *Writer* in the web
|
||||
- **`/creative` removed** *(BREAKING)* — subsumed by the Personas feature
|
||||
above: the REPL toggle (and its tab completion) is gone, and the
|
||||
`writer` seed persona replaces it — start a session with
|
||||
`turnstone --persona writer` or pick *Writer* in the web
|
||||
pickers. Unlike the old fork, the writer persona composes the full
|
||||
system message, so session context and mandatory prompt policies now
|
||||
apply to prose-only sessions too. The `creative_mode` key in
|
||||
@@ -45,6 +160,18 @@ stable, and the experimental line:
|
||||
automatically, so they resume as writing sessions rather than as
|
||||
legacy defaults.
|
||||
|
||||
### Security
|
||||
|
||||
- **High-risk skill activation is gated** (#762) — a model-initiated load
|
||||
of a `high`- or `critical`-risk skill is gated and fails closed when the
|
||||
backing storage is unavailable, so an untrusted turn can't silently
|
||||
pull in a dangerous capability.
|
||||
- **Dependency security floors** — `cryptography` and `starlette` are
|
||||
pinned to security-fixed minimums.
|
||||
- **CI publish hardening** — the vendored-JS dispatch path refuses fork
|
||||
PRs, and `workflow_run` publishing is gated to same-repo tag pushes, so
|
||||
a fork can't trigger a release build.
|
||||
|
||||
## [1.6.0]
|
||||
|
||||
The first stable release of the 1.6 line — and the first under Apache 2.0.
|
||||
|
||||
+1
-1
@@ -8,7 +8,7 @@ FROM python:3.14-slim
|
||||
LABEL org.opencontainers.image.title="turnstone" \
|
||||
org.opencontainers.image.description="Multi-node AI orchestration platform"
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.26 /uv /usr/local/bin/uv
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.27 /uv /usr/local/bin/uv
|
||||
|
||||
# Remove the slim image's man page exclusion so man-db has actual content
|
||||
RUN rm -f /etc/dpkg/dpkg.cfg.d/docker
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
[](https://pypi.org/project/turnstone/)
|
||||
[](LICENSE)
|
||||
[](https://discord.gg/Nh3bWMacaq)
|
||||
[](https://github.com/sponsors/eous)
|
||||
|
||||
Self-hosted, local-first orchestration for tool-using AI agents. Give LLMs real tools — shell, files, search, web — and run them across your own cluster with direct HTTP routing and interactive interfaces. Your code, your models, your data stay on hardware you control: no telemetry, no phone-home.
|
||||
|
||||
@@ -171,6 +172,14 @@ UML diagrams in [`docs/diagrams/`](docs/diagrams/):
|
||||
- Optional: Discord / Slack channel integrations (`pip install turnstone[discord,slack]`)
|
||||
- [Git LFS](https://git-lfs.com/) for cloning (diagram PNGs)
|
||||
|
||||
## Support
|
||||
|
||||
Turnstone is free, Apache-2.0, and self-hosted — no paid tier, no telemetry, no upsell. If it saves you time or you'd like to help keep development moving, you can sponsor the project:
|
||||
|
||||
**[❤ Sponsor Turnstone →](https://github.com/sponsors/eous)** · one-off via **[PayPal](https://paypal.me/eousphoros)**
|
||||
|
||||
Sponsorship is entirely optional and funds maintenance, new features, and infrastructure. Prefer to contribute in other ways? Filing issues, improving docs, and [pull requests](CONTRIBUTING.md) help just as much.
|
||||
|
||||
## Community
|
||||
|
||||
Questions, ideas, or want to show what you're building? Join us on Discord:
|
||||
|
||||
+108
-9
@@ -634,8 +634,17 @@ function tool (the model always searches). Citations from `url_citation`
|
||||
annotations are formatted as footnotes. Extended prompt cache retention
|
||||
(`prompt_cache_retention: "24h"`) is enabled for GPT-5.x models at no
|
||||
additional cost. Cached token counts are extracted from
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models (local servers) get
|
||||
permissive defaults with `supports_vision=False` and use SearxNG for web search.
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models get permissive
|
||||
defaults with `supports_vision=False` and use SearxNG for web search. The
|
||||
`openai-compatible` lane never consults this table at all — on either API
|
||||
surface (the responses pin is served by a compat-mode
|
||||
`OpenAIResponsesProvider`, mirroring `AnthropicProvider(compat=True)`): a
|
||||
local server serves whatever the operator named it (vLLM
|
||||
`--served-model-name` is a free string), so a prefix collision with a cloud
|
||||
model id must not inherit that model's sampling/effort contract — every
|
||||
local model gets the plain defaults, and anything beyond them is declared on
|
||||
the model definition (capabilities JSON + `server_compat`), matching the
|
||||
`anthropic-compatible` lane.
|
||||
|
||||
**AnthropicProvider** (`_anthropic.py`): converts OpenAI-format messages to
|
||||
Anthropic content blocks, maps `system`/`developer` roles to the `system`
|
||||
@@ -798,15 +807,105 @@ model = "deepseek-ai/DeepSeek-V4-Flash"
|
||||
supports_vision = true # multimodal checkpoints only
|
||||
supports_mid_conversation_system = true # template-dependent
|
||||
context_window = 131072
|
||||
thinking_mode = "manual" # session effort knob drives the template toggle
|
||||
thinking_param = "enable_thinking" # Qwen/Gemma key; "thinking" for Granite/DeepSeek
|
||||
```
|
||||
|
||||
The reasoning toggle does NOT use Anthropic's `thinking` request param.
|
||||
Toggle it through the chat template instead: set `{"chat_template_kwargs":
|
||||
{"thinking": false}}` as extra body params in the admin Models
|
||||
server-compat section (for this provider the section shows only the
|
||||
extra-body field — server type, API surface, and thinking mode are
|
||||
openai-compatible-only knobs); the provider forwards it via the SDK's
|
||||
`extra_body`.
|
||||
Reasoning control does NOT use Anthropic's `thinking` request param —
|
||||
the levers live in the chat template, reached through
|
||||
`chat_template_kwargs` in the request body. Two channels, dynamic first:
|
||||
|
||||
* **Session effort knob (dynamic).** Set the model's thinking mode to
|
||||
"Effort-knob controlled" in the admin Models form (or
|
||||
`thinking_mode = "manual"` + `thinking_param` under
|
||||
`[models.*.capabilities]`) and the provider maps the session's
|
||||
reasoning-effort knob onto the template toggle per-request: effort
|
||||
`none` sends `{<thinking_param>: false}`, any other level sends
|
||||
`true` — the same contract as the real lane's manual mode. ("Always
|
||||
on" / `thinking_mode = "adaptive"` instead always sends `true`: the
|
||||
model self-regulates, so the knob never force-disables — mirroring
|
||||
the native adaptive branch.) The graded effort value always rides
|
||||
alongside the toggle: under `effort_param` when the operator names
|
||||
the template's key, else under the conventional fallback key
|
||||
(`reasoning_effort`) on the anthropic-compatible lane — the user's
|
||||
effort setting always reaches the wire, and a template that doesn't
|
||||
reference the kwarg ignores it. On the openai-compatible lane the
|
||||
undeclared-key case rides the flat top-level `reasoning_effort`
|
||||
param instead (the documented compat field), forwarded verbatim.
|
||||
Optional `reasoning_effort_values` / `default_reasoning_effort`
|
||||
validate the knob before it reaches the server; without declared
|
||||
values the knob is forwarded as-is. The knob is ordinal, and validation
|
||||
respects that: an off-list knob value rounds UP onto the declared
|
||||
list and a value above the ceiling rides the ceiling
|
||||
(`snap_reasoning_effort`) — asking for more effort than the model
|
||||
declares never falls back to a lower default tier. The knob's
|
||||
`none` position is forwarded verbatim when the model declares an
|
||||
explicit `none` level (gpt-5.1+, grok-4.3) — omitting it there would
|
||||
leave a reasoning-on server default (e.g. gpt-5.5's `medium`) in
|
||||
charge of a knob that promises off — and omitted otherwise; `none`
|
||||
is never a snap target for other positions.
|
||||
`default_reasoning_effort` only catches values the ordinal snap
|
||||
cannot rank (custom strings). Declare values that match the
|
||||
template's documented vocabulary: for DeepSeek-V4, which officially
|
||||
accepts `high`/`max` (Think High is the default thinking tier;
|
||||
`low`/`medium` alias to `high`, `xhigh` to `max`), a
|
||||
`("high", "max")` values list reproduces the official aliasing
|
||||
exactly — `low`/`medium` round up to `high`, `xhigh` to `max` —
|
||||
and freeform passthrough matches it too. To map an undocumented
|
||||
template, probe with per-request `chat_template_kwargs` and compare
|
||||
`input_tokens`. Setting `effort_param` also suppresses the
|
||||
flat top-level `reasoning_effort` request param on the
|
||||
openai-compatible lane — the template channel replaces it, never
|
||||
doubles it. With the default `thinking_mode = "none"` nothing is
|
||||
injected and the server's template default decides.
|
||||
|
||||
Upgrade note: before 1.7.0a7 the openai-compatible lane sent the
|
||||
toggle unconditionally `true` whenever thinking mode was enabled. A
|
||||
stored per-model `reasoning_effort = "none"` now disables thinking
|
||||
on such models — pick any real level (or clear the override) to keep
|
||||
it on. Also since 1.7.0a7 the effort level itself always reaches the
|
||||
wire on the local lanes (previously dropped unless
|
||||
`reasoning_effort_values` was declared): flat `reasoning_effort` on
|
||||
openai-compatible, the `effort_param`-or-fallback template key on
|
||||
anthropic-compatible when reasoning control is engaged.
|
||||
* **Operator pin (static).** Entries under `{"chat_template_kwargs":
|
||||
...}` in the admin Models extra-body field ride the SDK's
|
||||
`extra_body` unconditionally and win over the knob mapping on key
|
||||
collision — e.g. pin `{"enable_thinking": true}` to keep thinking on
|
||||
regardless of the session knob. (Server type and API surface remain
|
||||
openai-compatible-only knobs and stay hidden for this provider.)
|
||||
|
||||
The same knob mapping drives the `openai-compatible` lane's Chat
|
||||
Completions requests — `merge_reasoning_template_kwargs` is shared by
|
||||
both local-server lanes, so `thinking_mode`/`thinking_param`/
|
||||
`effort_param` mean the same thing whichever endpoint serves the model.
|
||||
Only the Responses API surface (native reasoning) ignores it.
|
||||
|
||||
The console surfaces this projection as an *effective effort ladder*:
|
||||
the admin model form's per-model effort select and the skill
|
||||
launch-config effort select annotate each position with what the
|
||||
request will carry, in plain words — a position whose delivered level
|
||||
matches its name stays plain ("Max"), a snapped position says so
|
||||
("Low — sends high"), the adaptive lanes' none position warns
|
||||
"thinking stays on", and budget detail lives in the tooltip. A
|
||||
position is never labeled after a sibling that shares its wire (that
|
||||
rendered "Max (= minimal)", implying a downgrade the wire doesn't
|
||||
contain). Computed server-side by `providers/effort_ladder.py` from
|
||||
the same mapping functions the providers use at request time and
|
||||
shipped on `/v1/api/models` rows (every row carries `effort_ladder`,
|
||||
empty when the capabilities column fails to parse) and
|
||||
`POST /v1/api/admin/models/effort-ladder`. The ladder describes what
|
||||
Turnstone sends — a server-side template may alias further (DeepSeek-V4
|
||||
folds `low`/`medium` into its default `high` tier).
|
||||
|
||||
The `anthropic-compatible` lane never sends Anthropic's native
|
||||
`thinking`/`output_config` params — they are not in vLLM's request
|
||||
schema. The real `anthropic` provider is unaffected: official Claude
|
||||
models keep native thinking, budget mapping, and `output_config`
|
||||
effort. A gateway fronting *real* Claude on a Messages-shaped URL
|
||||
(e.g. a LiteLLM `anthropic/` route to the Claude API) should use
|
||||
`provider = "anthropic"` with a custom `base_url`, which keeps the
|
||||
native thinking params.
|
||||
|
||||
Verified quirks of vLLM's Anthropic endpoint:
|
||||
|
||||
|
||||
+37
-4
@@ -118,9 +118,37 @@ seeded):
|
||||
`persona` argument, validated when the coordinator prepares the spawn
|
||||
and re-checked by the node that creates the child (children are always
|
||||
interactive-kind). Omitted means the interactive **default** — a child
|
||||
never inherits its parent coordinator's persona. Sub-agents spawned via
|
||||
`task_agent` have no persona parameter at all; they keep their own
|
||||
identity and envelope.
|
||||
never inherits its parent coordinator's persona.
|
||||
- **Sub-agents**: `task_agent` takes a `persona` argument setting the
|
||||
sub-agent's identity and capability envelope (resolved against
|
||||
interactive-kind personas, frozen into the task at prep). Omitted keeps
|
||||
the default autonomous task-agent identity — never the parent's persona.
|
||||
|
||||
## How agents discover personas
|
||||
|
||||
Agents are told, not expected to guess: the live persona list (enabled,
|
||||
interactive-kind — children and sub-agents are always interactive) is
|
||||
injected into the `persona` parameter description of `task_agent`,
|
||||
`spawn_workstream`, and `spawn_batch` whenever the session's tool surface
|
||||
is rendered — session start, MCP catalog change, model-registry reload.
|
||||
Each entry carries the name, the default marker, and the persona's
|
||||
one-line description so the model can pick by purpose (descriptions drop
|
||||
out past 25 personas; the name list always enumerates completely).
|
||||
|
||||
A persona created after that render is still reachable — pass its name.
|
||||
Every resolve failure enumerates the names currently valid for the kind,
|
||||
so a stale list (or a typo) self-corrects on the next attempt.
|
||||
|
||||
Resolution is forgiving on all surfaces (they share one rule):
|
||||
|
||||
- names match case-insensitively (`Writer` resolves `writer`);
|
||||
- an input that uniquely matches a persona's **display name**
|
||||
(case-insensitive, among the kind's enabled personas — display names are
|
||||
not unique, and a same-label persona of another kind neither blocks nor
|
||||
wins) resolves to that persona; an ambiguous match errors, listing the
|
||||
candidate slugs;
|
||||
- whatever variant matched, the stamped identity, approval chrome, and
|
||||
wire always carry the canonical `name` slug.
|
||||
|
||||
## Authoring (console)
|
||||
|
||||
@@ -128,7 +156,12 @@ Personas are managed in the console's **Manage → Governance → Personas**
|
||||
tab. The admin shelf exposes exactly the four levers plus the kind
|
||||
list, the default marker, and archive. Rules:
|
||||
|
||||
- `name` is an immutable lowercase slug; edit `display_name` instead.
|
||||
- `name` is an immutable lowercase slug — and the identifier agents and
|
||||
the CLI launch the persona by (`persona=` on the spawn tools,
|
||||
`--persona` on the CLI); the create shelf says so under **Name**.
|
||||
`display_name` is a list label, editable any time, and deliberately
|
||||
not an identifier (a unique display name happens to resolve, as a
|
||||
forgiveness fallback — don't design workflows around it).
|
||||
- Exactly one default per kind, storage-enforced: flipping the flag on a
|
||||
successor demotes the incumbent atomically, defaults are single-kind,
|
||||
and a default cannot be archived.
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
{
|
||||
"defaults": {
|
||||
"n_runs": 3
|
||||
},
|
||||
"cases": [
|
||||
{
|
||||
"id": "search-first",
|
||||
"skill": {
|
||||
"name": "search-first",
|
||||
"content": "# Search First\n\nBefore answering ANY question about where something lives in the codebase, you MUST call the `search` tool first. Never answer from memory."
|
||||
},
|
||||
"user_prompt": "Where is JWT token validation implemented in this project?",
|
||||
"expected_actions": [{ "tool": "search" }],
|
||||
"match_mode": "ordered_subset",
|
||||
"max_turns": 4
|
||||
},
|
||||
{
|
||||
"id": "test-after-edit",
|
||||
"skill": {
|
||||
"name": "test-after-edit",
|
||||
"content": "# Test After Edit\n\nAfter editing or writing ANY file, you MUST run the test suite with `python -m pytest` via bash before you finish. Do not report done until tests have run."
|
||||
},
|
||||
"user_prompt": "Add a function `clamp(x, lo, hi)` that clamps x to [lo, hi] in utils.py.",
|
||||
"setup": {
|
||||
"files": {
|
||||
"utils.py": ""
|
||||
}
|
||||
},
|
||||
"expected_actions": [
|
||||
{ "tool": "write_file" },
|
||||
{ "tool": "bash", "args_pattern": { "command": "pytest" } }
|
||||
],
|
||||
"match_mode": "ordered_subset",
|
||||
"max_turns": 8
|
||||
},
|
||||
{
|
||||
"id": "changelog-update",
|
||||
"skill": {
|
||||
"name": "changelog-update",
|
||||
"content": "# Changelog Discipline\n\nWhenever you modify a file, you MUST also append a one-line entry to CHANGELOG.md describing the change in the same task."
|
||||
},
|
||||
"user_prompt": "Fix the off-by-one so pager.py shows the last page. Edit pager.py.",
|
||||
"setup": {
|
||||
"files": {
|
||||
"pager.py": "def last_page(total_items, per_page):\n # off-by-one: drops the final partial page\n return total_items // per_page\n",
|
||||
"CHANGELOG.md": "# Changelog\n"
|
||||
}
|
||||
},
|
||||
"expected_actions": [
|
||||
{ "tool": "edit_file", "args_pattern": { "path": "CHANGELOG.md" } }
|
||||
],
|
||||
"match_mode": "subset",
|
||||
"max_turns": 8
|
||||
}
|
||||
]
|
||||
}
|
||||
+1
-1
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "turnstone"
|
||||
version = "1.7.0a6"
|
||||
version = "1.7.0rc1"
|
||||
description = "Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
|
||||
@@ -11201,6 +11201,7 @@
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline BASE override \u2014 required. Every persona must name a prompt source; built-in file-backed personas are seeded by migration, not created here, so an operator-created persona must supply base_prompt.",
|
||||
"title": "Base Prompt"
|
||||
},
|
||||
"tool_allowlist": {
|
||||
@@ -11259,7 +11260,7 @@
|
||||
"type": "object"
|
||||
},
|
||||
"UpdatePersonaRequest": {
|
||||
"description": "PATCH body \u2014 absent fields are left unchanged.\n\nExplicit ``null`` is meaningful only on the two resettable fields:\n``base_prompt: null`` clears the override back to the kind's stock\nBASE, and ``tool_allowlist: null`` resets to unrestricted. ``null``\non the boolean flags or ``applies_to_kinds`` is ignored (treated as\nabsent), so a client serializing unset optionals as null cannot\narchive a persona or flip levers by accident.\n\nArchive = ``{\"enabled\": false}``; default flip = ``{\"is_default\": true}``\non the successor (storage demotes the incumbent atomically). ``name``\nis immutable; existing workstreams are never affected by edits.",
|
||||
"description": "PATCH body \u2014 absent fields are left unchanged.\n\nExplicit ``null`` resets ``tool_allowlist`` to unrestricted, and \u2014 on a\nBUILT-IN persona only \u2014 clears ``base_prompt`` (the operator override),\nreverting to that persona's file-backed prompt. An OPERATOR persona has no\nfallback source, so ``base_prompt: null`` on one is rejected: every persona\nmust name a prompt source. ``null`` on the boolean flags or\n``applies_to_kinds`` is ignored (treated as absent), so a client serializing\nunset optionals as null cannot archive a persona or flip levers by accident.\n\nArchive = ``{\"enabled\": false}``; default flip = ``{\"is_default\": true}``\non the successor (storage demotes the incumbent atomically). ``name``\nis immutable; existing workstreams are never affected by edits.",
|
||||
"properties": {
|
||||
"display_name": {
|
||||
"anyOf": [
|
||||
@@ -13305,21 +13306,17 @@
|
||||
},
|
||||
"pending_approval": {
|
||||
"default": false,
|
||||
"description": "True when the workstream is parked on ``_approval_event`` awaiting an operator approve/deny. Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"description": "True when at least one approval cycle is live (a gate thread parked awaiting an operator approve/deny). Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"title": "Pending Approval",
|
||||
"type": "boolean"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload \u2014 same shape as ``DashboardWorkstream.pending_approval_detail``. ``None`` when no approval is pending. Lets a reload paint the action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payloads, one per live cycle, oldest first \u2014 same shape as ``DashboardWorkstream.pending_approval_details``. Empty when no approval is pending. Lets a reload paint every action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -13332,8 +13329,14 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalDetail": {
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nSet when a workstream's ``approve_tools`` is parked on\n``_approval_event``; ``None`` (omitted) otherwise. Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nOne entry per live approval CYCLE \u2014 a gate thread parked in\n``approve_tools`` awaiting the operator. Parallel task agents run\nconcurrent gates, so a workstream can have several of these at\nonce (``pending_approval_details``, oldest first). Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"properties": {
|
||||
"cycle_id": {
|
||||
"default": "",
|
||||
"description": "Identity of this approval cycle. Echo it back on ``POST /v1/api/workstreams/{ws_id}/approve`` to resolve exactly this round \u2014 required for correctness when several cycles are live (parallel task agents).",
|
||||
"title": "Cycle Id",
|
||||
"type": "string"
|
||||
},
|
||||
"call_id": {
|
||||
"default": "",
|
||||
"description": "Primary call_id \u2014 first non-empty call_id in items list order. Matches the 409 ``current_call_id`` response from ``POST /v1/api/workstreams/{ws_id}/approve`` so the UI can render the same identifier the server reports as current.",
|
||||
|
||||
@@ -2692,21 +2692,17 @@
|
||||
},
|
||||
"pending_approval": {
|
||||
"default": false,
|
||||
"description": "True when the workstream is parked on ``_approval_event`` awaiting an operator approve/deny. Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"description": "True when at least one approval cycle is live (a gate thread parked awaiting an operator approve/deny). Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"title": "Pending Approval",
|
||||
"type": "boolean"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload \u2014 same shape as ``DashboardWorkstream.pending_approval_detail``. ``None`` when no approval is pending. Lets a reload paint the action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payloads, one per live cycle, oldest first \u2014 same shape as ``DashboardWorkstream.pending_approval_details``. Empty when no approval is pending. Lets a reload paint every action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -2719,8 +2715,14 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalDetail": {
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nSet when a workstream's ``approve_tools`` is parked on\n``_approval_event``; ``None`` (omitted) otherwise. Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nOne entry per live approval CYCLE \u2014 a gate thread parked in\n``approve_tools`` awaiting the operator. Parallel task agents run\nconcurrent gates, so a workstream can have several of these at\nonce (``pending_approval_details``, oldest first). Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"properties": {
|
||||
"cycle_id": {
|
||||
"default": "",
|
||||
"description": "Identity of this approval cycle. Echo it back on ``POST /v1/api/workstreams/{ws_id}/approve`` to resolve exactly this round \u2014 required for correctness when several cycles are live (parallel task agents).",
|
||||
"title": "Cycle Id",
|
||||
"type": "string"
|
||||
},
|
||||
"call_id": {
|
||||
"default": "",
|
||||
"description": "Primary call_id \u2014 first non-empty call_id in items list order. Matches the 409 ``current_call_id`` response from ``POST /v1/api/workstreams/{ws_id}/approve`` so the UI can render the same identifier the server reports as current.",
|
||||
@@ -3004,17 +3006,13 @@
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload for the coordinator children-tree UI. Carries the merged ``_pending_approval`` items list + per-call_id LLM verdict cache so a coord can render approve/deny buttons + judge pill without a separate per-child round-trip. ``None`` when no approval is pending. Also surfaced (verbatim) on ``GET /v1/api/cluster/ws/live`` via the ``_CLUSTER_WS_LIVE_KEYS`` projection."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payload for the coordinator children-tree UI: EVERY live approval cycle, oldest first \u2014 parallel task agents gate concurrently, so a workstream can hold several prompts at once. Each entry carries the cycle's items + per-call_id LLM verdict cache so a coord can render approve/deny buttons + judge pill without a separate per-child round-trip; resolve each with its ``cycle_id``. Empty when no approval is pending. Also surfaced (verbatim) on ``GET /v1/api/cluster/ws/live`` via the ``_CLUSTER_WS_LIVE_KEYS`` projection. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
},
|
||||
"recent_auto_approvals": {
|
||||
"description": "Per-ws ring buffer (cap 10) of recent tool calls that bypassed the operator approval gate. Surfaces ``WebUI._recent_auto_approvals`` so the coord-tree row can render an 'auto-approved by ...' pill when the child's skill / blanket / admin-policy rules silently let a tool through. Also projected onto ``GET /v1/api/cluster/ws/live`` via ``_CLUSTER_WS_LIVE_KEYS``.",
|
||||
|
||||
Generated
+50
-50
@@ -409,16 +409,16 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@vitest/expect": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.9.tgz",
|
||||
"integrity": "sha512-vl/rYsUKcBr3SnQn166+XR5ZQcgMx3DQhFWdfli/cWpLnLUmbxZvyrJZotLFUryib+LtArYMSTJ5RbQ57ZqrlA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.10.tgz",
|
||||
"integrity": "sha512-YsCn+qAk1GWjQOWFEsEcL2gNQ0zmVmQu3T03qP6UyjhtmdtwtbuI+DASn/7iQB3HGTXkdBwGddzxPlmiql5vlA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@standard-schema/spec": "^1.1.0",
|
||||
"@types/chai": "^5.2.2",
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"chai": "^6.2.2",
|
||||
"tinyrainbow": "^3.1.0"
|
||||
},
|
||||
@@ -427,13 +427,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/mocker": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.9.tgz",
|
||||
"integrity": "sha512-EVkXzBjrPGM+cK8/ANWgBrkUCfJfb38/EfTSO8h7pWvKkyPkpWxvR7BkD2MyItMF62C97zAEoqdpUixwR/e+Rw==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.10.tgz",
|
||||
"integrity": "sha512-v0xaezt+DKEmKfaxg133ldzADrwLGd7Ze1MfQQTYfvs8OqZIwbxyxaYURivwV7sWy5fqn3rH5uOrSp07bp44Ow==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"estree-walker": "^3.0.3",
|
||||
"magic-string": "^0.30.21"
|
||||
},
|
||||
@@ -454,9 +454,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/pretty-format": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.9.tgz",
|
||||
"integrity": "sha512-s0iufns3iIFitdgm+YR7g1whCAaGtXz459VS9/PqyKDEEFgYIhsHOQmXgIgDuYCt7DeQmiZT0Qe2OA2p4ZPu5A==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.10.tgz",
|
||||
"integrity": "sha512-W1HsjSH4MXQ9YfmmhLAoIYf1HRfekQCGngeIgcei6MP5QQGWUe0gkopdZQaVCFO+JDJMrAJGwa5pRpNpvy4P8Q==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -467,13 +467,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/runner": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.9.tgz",
|
||||
"integrity": "sha512-KXLMDtc7oe70+3mJfGrPUWPesswH+3sTxAMAMl8DG7I8IUQT4XW718dY5ID3vPUcmlu27CcKfY4P3h3I29SLJg==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.10.tgz",
|
||||
"integrity": "sha512-IKI6kpIH+LmpROplyLwBBaCfMgOZOMsygVa6BARD6ahA04VRuJSa6OaVG7kRvSEMD870Vd91rSSw0eegtWyLGg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"pathe": "^2.0.3"
|
||||
},
|
||||
"funding": {
|
||||
@@ -481,14 +481,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/snapshot": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.9.tgz",
|
||||
"integrity": "sha512-Jc7RKGNBo8Z28WYIm0Niej4xdSPByRf6mU58VpHQkd6Zh05rlnA+twjbK5HyeIGHxrzsc3mJgS43uM0CZKzaIA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.10.tgz",
|
||||
"integrity": "sha512-xRkfOT1qpTAi/Ti4Y1LtfRc3kEuqxGw59eN2jN9pRWMtS/XDevekhcFSqvQqjUNGksfjMJu3Y+oJ+4Ypn2OaJw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"magic-string": "^0.30.21",
|
||||
"pathe": "^2.0.3"
|
||||
},
|
||||
@@ -497,9 +497,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/spy": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.9.tgz",
|
||||
"integrity": "sha512-fHpsS6mIi+PiEW+vcRVOMkX1oSaPKne3VOclSFICPcGOmfKgXPU5iAah+wcNcj2xPrCCmfq99IDGf+EojhhvhA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.10.tgz",
|
||||
"integrity": "sha512-PLf/Ugvoq5wO/b4rwYCR1h2PSIdXz7wnkQFMiUpLdtM7l6pqVFcQIBEHyT1+l+cj7mNwAfZHzqXqDyjvOuwbDw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
@@ -507,13 +507,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/utils": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.9.tgz",
|
||||
"integrity": "sha512-A51o8ymO5PpqlWNnBP9ZHPXDIpuMtTLlGSjN7la4US+LJzoUMyhwjA5QXlm39JexgwHKW4Xjs8Z2d3dLCXOeuA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.10.tgz",
|
||||
"integrity": "sha512-fy9am/HWxbaGt/Sawrp90vt6Y6jQwf1RX77cz3uwoJwJVMli/e1IEwRPnMNJ7vKfPTwo0diXifkpPvwH9v7nGA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"convert-source-map": "^2.0.0",
|
||||
"tinyrainbow": "^3.1.0"
|
||||
},
|
||||
@@ -949,9 +949,9 @@
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/picomatch": {
|
||||
"version": "4.0.4",
|
||||
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz",
|
||||
"integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==",
|
||||
"version": "4.0.5",
|
||||
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.5.tgz",
|
||||
"integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -1122,9 +1122,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vite": {
|
||||
"version": "8.1.2",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.2.tgz",
|
||||
"integrity": "sha512-6YYPbRXTxx6bRXmOn7XdnQAy5DQNHhDgtjhDHI13oe4pY93kkcdGJWxpGwOm++/Wh0QpQhDrpIoVMrmrsI5AGQ==",
|
||||
"version": "8.1.3",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.3.tgz",
|
||||
"integrity": "sha512-Ds+gBRbj0lwRO2Y5hwnUBdxSwlAve9LeRyU4sNnAr0ewW0gWF0n5bgXgUzbgZ49MV9BVUAQUFYVcDUcilUExMA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -1200,19 +1200,19 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vitest": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.9.tgz",
|
||||
"integrity": "sha512-nE3/LEyc0z87uHYLZebqCUOaJr2hdtuPp7BQ4BosVFnfltxgAvMG08NyrSGlPpOUWvR27c5flSmYFTNr78L9GQ==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.10.tgz",
|
||||
"integrity": "sha512-R9jUTe5S4Qb0HCd4TNqpC7oGcrMssMRGXLW80ubjWsW9VH5GF8y1Y0SFLY9AbqSk6nt0PnOx4H4WNJYZ13GUPw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/expect": "4.1.9",
|
||||
"@vitest/mocker": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/runner": "4.1.9",
|
||||
"@vitest/snapshot": "4.1.9",
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/expect": "4.1.10",
|
||||
"@vitest/mocker": "4.1.10",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"@vitest/runner": "4.1.10",
|
||||
"@vitest/snapshot": "4.1.10",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"es-module-lexer": "^2.0.0",
|
||||
"expect-type": "^1.3.0",
|
||||
"magic-string": "^0.30.21",
|
||||
@@ -1240,12 +1240,12 @@
|
||||
"@edge-runtime/vm": "*",
|
||||
"@opentelemetry/api": "^1.9.0",
|
||||
"@types/node": "^20.0.0 || ^22.0.0 || >=24.0.0",
|
||||
"@vitest/browser-playwright": "4.1.9",
|
||||
"@vitest/browser-preview": "4.1.9",
|
||||
"@vitest/browser-webdriverio": "4.1.9",
|
||||
"@vitest/coverage-istanbul": "4.1.9",
|
||||
"@vitest/coverage-v8": "4.1.9",
|
||||
"@vitest/ui": "4.1.9",
|
||||
"@vitest/browser-playwright": "4.1.10",
|
||||
"@vitest/browser-preview": "4.1.10",
|
||||
"@vitest/browser-webdriverio": "4.1.10",
|
||||
"@vitest/coverage-istanbul": "4.1.10",
|
||||
"@vitest/coverage-v8": "4.1.10",
|
||||
"@vitest/ui": "4.1.10",
|
||||
"happy-dom": "*",
|
||||
"jsdom": "*",
|
||||
"vite": "^6.0.0 || ^7.0.0 || ^8.0.0"
|
||||
|
||||
@@ -75,15 +75,35 @@ export interface ToolInfoEvent {
|
||||
items: Array<Record<string, unknown>>;
|
||||
}
|
||||
|
||||
/** One approval CYCLE awaiting the operator. Several can be outstanding
|
||||
* at once (parallel task agents each gate their own tool calls) — key
|
||||
* prompt UI by `cycle_id` and echo it back on the approve POST.
|
||||
*
|
||||
* `cycle_id` is optional because it was added in 1.7: a pre-1.7 server
|
||||
* omits it on the wire, so a current SDK talking to an older node sees
|
||||
* `undefined`. Resolve those the legacy way (no selector → oldest
|
||||
* cycle). A current server always sends it. */
|
||||
export interface ApproveRequestEvent {
|
||||
type: "approve_request";
|
||||
cycle_id?: string;
|
||||
items: Array<Record<string, unknown>>;
|
||||
judge_pending?: boolean;
|
||||
}
|
||||
|
||||
/** A specific approval cycle resolved; `cycle_id`/`call_ids` identify
|
||||
* which prompt to dismiss.
|
||||
*
|
||||
* Both are optional for the same reason as `ApproveRequestEvent.cycle_id`
|
||||
* — a pre-1.7 server emits neither, so a bare "something resolved"
|
||||
* dismisses the sole tracked prompt (the legacy fallback the UI and
|
||||
* channel adapters keep). A current server always sends both. */
|
||||
export interface ApprovalResolvedEvent {
|
||||
type: "approval_resolved";
|
||||
approved: boolean;
|
||||
feedback: string;
|
||||
always?: boolean;
|
||||
cycle_id?: string;
|
||||
call_ids?: string[];
|
||||
}
|
||||
|
||||
export interface ToolResultEvent {
|
||||
|
||||
@@ -166,6 +166,13 @@ export class TurnstoneServer extends BaseClient {
|
||||
approved?: boolean;
|
||||
feedback?: string | null;
|
||||
always?: boolean;
|
||||
/** Resolve exactly this approval cycle (from ApproveRequestEvent.cycle_id).
|
||||
* Omitting it resolves the OLDEST live cycle — ambiguous when parallel
|
||||
* task agents have several prompts outstanding, so pass it whenever the
|
||||
* triggering event is known. */
|
||||
cycleId?: string;
|
||||
/** Alternative selector: any call_id inside the target cycle. */
|
||||
callId?: string;
|
||||
}): Promise<StatusResponse> {
|
||||
return this.request(
|
||||
"POST",
|
||||
@@ -175,6 +182,8 @@ export class TurnstoneServer extends BaseClient {
|
||||
approved: opts.approved ?? true,
|
||||
feedback: opts.feedback,
|
||||
always: opts.always,
|
||||
cycle_id: opts.cycleId,
|
||||
call_id: opts.callId,
|
||||
},
|
||||
},
|
||||
);
|
||||
|
||||
@@ -104,7 +104,6 @@ describe("TurnstoneServer attachments", () => {
|
||||
const [, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(JSON.parse(init.body)).toEqual({
|
||||
message: "hi",
|
||||
ws_id: "ws-X",
|
||||
attachment_ids: ["a1", "a2"],
|
||||
});
|
||||
});
|
||||
@@ -117,7 +116,7 @@ describe("TurnstoneServer attachments", () => {
|
||||
});
|
||||
await client.send("hi", "ws-X");
|
||||
const [, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "hi", ws_id: "ws-X" });
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "hi" });
|
||||
});
|
||||
|
||||
it("createWorkstream with attachments sends multipart and auto-generates ws_id", async () => {
|
||||
|
||||
@@ -74,8 +74,8 @@ describe("TurnstoneServer", () => {
|
||||
await client.send("Hello", "ws1");
|
||||
|
||||
const [url, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(url).toBe("http://test/v1/api/send");
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "Hello", ws_id: "ws1" });
|
||||
expect(url).toBe("http://test/v1/api/workstreams/ws1/send");
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "Hello" });
|
||||
});
|
||||
|
||||
it("injects auth header when token provided", async () => {
|
||||
|
||||
@@ -51,6 +51,12 @@ def make_replay_mocks(
|
||||
ui._ws_messages = 0
|
||||
for key, value in ui_overrides.items():
|
||||
setattr(ui, key, value)
|
||||
# Both replay paths read cycle cards via ``pending_approval_cards()``
|
||||
# (one card per concurrent approval cycle). Model it from the
|
||||
# single-slot ``_pending_approval`` override so tests keep seeding
|
||||
# the one field; a bare MagicMock here would iterate empty and
|
||||
# silently drop the approve_request from the replay.
|
||||
ui.pending_approval_cards = lambda: [ui._pending_approval] if ui._pending_approval else []
|
||||
ws = MagicMock()
|
||||
ws.session = session
|
||||
request = MagicMock()
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Recording fake SDK client — captures the kwargs at each provider's seam.
|
||||
|
||||
Every provider's ``create_streaming`` assembles its kwargs and calls the
|
||||
SDK *eagerly* before returning the stream iterator (Anthropic
|
||||
``client.messages.stream``, OpenAI ``client.chat.completions.create``,
|
||||
Responses ``client.responses.create/stream``), so driving a provider
|
||||
against a :class:`RecordingClient` captures the full composed request
|
||||
payload without a network round-trip.
|
||||
|
||||
Shared by the wire-payload golden harness (``test_wire_payload_golden``)
|
||||
and the effort-ladder parity harness (``test_effort_ladder_wire_parity``)
|
||||
so both assert against the same capture seam.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
|
||||
class _EmptyStream:
|
||||
"""Stand-in for an SDK stream / stream-manager: empty iterable AND no-op CM."""
|
||||
|
||||
def __iter__(self) -> Iterator[Any]:
|
||||
return iter(())
|
||||
|
||||
def __enter__(self) -> _EmptyStream:
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc: object) -> None:
|
||||
return None
|
||||
|
||||
|
||||
class _Seam:
|
||||
"""Records the kwargs of a single SDK call, returns an empty stream stub."""
|
||||
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self._sink = sink
|
||||
|
||||
def __call__(self, **kwargs: Any) -> _EmptyStream:
|
||||
# Last write wins; only one seam is exercised per provider call.
|
||||
self._sink["payload"] = kwargs
|
||||
return _EmptyStream()
|
||||
|
||||
|
||||
class _Completions:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
|
||||
|
||||
class _Chat:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.completions = _Completions(sink)
|
||||
|
||||
|
||||
class _Messages:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class _Responses:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class RecordingClient:
|
||||
"""Fake SDK client exposing every provider's call seam, recording kwargs."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.captured: dict[str, Any] = {}
|
||||
self.messages = _Messages(self.captured)
|
||||
self.chat = _Chat(self.captured)
|
||||
self.responses = _Responses(self.captured)
|
||||
+69
-1
@@ -52,8 +52,76 @@ def serve_until_exit(server: Any) -> None:
|
||||
loop.close()
|
||||
|
||||
|
||||
class _PendingResolver:
|
||||
"""Race-free drop-in for ``threading.Timer(delay, ui.resolve_approval)``.
|
||||
|
||||
``approve_tools`` runs ``_approval_event.clear()`` -> register
|
||||
``_pending_approval`` -> ``_approval_event.wait(_APPROVAL_WAIT_TIMEOUT)``
|
||||
(3600s). A *fixed-delay* timer can fire ``resolve_approval``
|
||||
(``_approval_event.set()``) BEFORE that ``.clear()`` on a slow/loaded
|
||||
runner, so the set is wiped by the clear and ``approve_tools`` blocks the
|
||||
full hour -- surfacing as a CI hang. This instead waits until the approval
|
||||
is actually registered (which happens *after* the clear), then resolves, so
|
||||
the wakeup can never be lost. ``start()`` / ``cancel()`` mirror
|
||||
``threading.Timer`` so it drops into existing scaffolding. ``cancel()``
|
||||
signals the worker to stop and joins it, so a test that errors *before* the
|
||||
approval registers can't leak the thread or resolve late into a finished
|
||||
test. ``before`` runs just before resolving -- e.g. to snapshot
|
||||
pending-state fields the test asserts on.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
ui: Any,
|
||||
*args: Any,
|
||||
before: Callable[[], None] | None = None,
|
||||
deadline: float = 10.0,
|
||||
**kwargs: Any,
|
||||
) -> None:
|
||||
self._ui = ui
|
||||
self._args = args
|
||||
self._kwargs = kwargs
|
||||
self._before = before
|
||||
self._deadline = deadline
|
||||
self._cancelled = threading.Event()
|
||||
self._started = False
|
||||
self._thread = threading.Thread(target=self._run, name="resolve-when-pending", daemon=True)
|
||||
|
||||
def _run(self) -> None:
|
||||
end = time.monotonic() + self._deadline
|
||||
while time.monotonic() < end:
|
||||
if self._cancelled.is_set():
|
||||
return
|
||||
# getattr (not a bare read) so a UI without _pending_approval can't
|
||||
# crash the worker into a silent death that leaves approve_tools
|
||||
# blocked for the full _APPROVAL_WAIT_TIMEOUT.
|
||||
if getattr(self._ui, "_pending_approval", None) is not None:
|
||||
if self._before is not None:
|
||||
self._before()
|
||||
self._ui.resolve_approval(*self._args, **self._kwargs)
|
||||
return
|
||||
time.sleep(0.001)
|
||||
# Deadline without registration: approve_tools isn't parked on the
|
||||
# approval event (returned early, or never reached it) -- don't resolve
|
||||
# into an unknown state; let the test's own assertions speak.
|
||||
|
||||
def start(self) -> None:
|
||||
self._started = True
|
||||
self._thread.start()
|
||||
|
||||
def cancel(self) -> None:
|
||||
self._cancelled.set()
|
||||
if self._started:
|
||||
self._thread.join(timeout=5)
|
||||
|
||||
|
||||
def resolve_when_pending(ui: Any, *args: Any, **kwargs: Any) -> _PendingResolver:
|
||||
"""Build a race-free approval resolver (see :class:`_PendingResolver`)."""
|
||||
return _PendingResolver(ui, *args, **kwargs)
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
from collections.abc import Callable, Iterator
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager, StaticServerState
|
||||
from turnstone.core.mcp_crypto import MCPTokenCipher
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris and London?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
},
|
||||
{
|
||||
"id": "call_2",
|
||||
"input": {
|
||||
"city": "London"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_2",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"text": "Actually, never mind London.",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "What's in this image?",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"source": {
|
||||
"data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==",
|
||||
"media_type": "image/png",
|
||||
"type": "base64"
|
||||
},
|
||||
"type": "image"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Think about the weather.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"signature": "sig-abc",
|
||||
"thinking": "The user wants weather.",
|
||||
"type": "thinking"
|
||||
},
|
||||
{
|
||||
"text": "Let me check.",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Think about the weather.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"signature": "sig-abc",
|
||||
"thinking": "The user wants weather.",
|
||||
"type": "thinking"
|
||||
},
|
||||
{
|
||||
"text": "Let me check.",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Run the deploy.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {},
|
||||
"name": "deploy",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "deployed",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"text": "Great, what's next?",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Hi there.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "Hello! How can I help?",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": "What's the weather in Paris?",
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "It's 18C and clear in Paris.",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -43,6 +43,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -43,6 +43,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
+170
-23
@@ -9,6 +9,8 @@ manual testing.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
@@ -17,6 +19,12 @@ import pytest
|
||||
|
||||
_APP_JS = Path(__file__).resolve().parent.parent / "turnstone/ui/static/app.js"
|
||||
_INTERACTIVE_JS = Path(__file__).resolve().parent.parent / "turnstone/shared_static/interactive.js"
|
||||
_SHELL_JS = Path(__file__).resolve().parent.parent / "turnstone/shared_static/shell.js"
|
||||
_REDACT_CREDENTIALS_JS = (
|
||||
Path(__file__).resolve().parent.parent / "turnstone/shared_static/redact_credentials.js"
|
||||
)
|
||||
_CONSOLE_APP_JS = Path(__file__).resolve().parent.parent / "turnstone/console/static/app.js"
|
||||
_CONSOLE_INDEX = Path(__file__).resolve().parent.parent / "turnstone/console/static/index.html"
|
||||
|
||||
|
||||
def _pane_method_offset(body: str, name: str) -> int:
|
||||
@@ -555,7 +563,6 @@ _CONSOLE_ADMIN_JS = Path(__file__).resolve().parent.parent / "turnstone/console/
|
||||
_CONSOLE_GOVERNANCE_JS = (
|
||||
Path(__file__).resolve().parent.parent / "turnstone/console/static/governance.js"
|
||||
)
|
||||
_CONSOLE_INTERACTIVE_JS = Path(__file__).resolve().parent.parent / "turnstone/console/static/app.js"
|
||||
|
||||
|
||||
_UNSAFE_CODE_SINK_LINT_TARGETS = [
|
||||
@@ -566,7 +573,7 @@ _UNSAFE_CODE_SINK_LINT_TARGETS = [
|
||||
("turnstone/console/static/coordinator/coordinator.js", _COORD_JS),
|
||||
("turnstone/console/static/admin.js", _CONSOLE_ADMIN_JS),
|
||||
("turnstone/console/static/governance.js", _CONSOLE_GOVERNANCE_JS),
|
||||
("turnstone/console/static/app.js", _CONSOLE_INTERACTIVE_JS),
|
||||
("turnstone/console/static/app.js", _CONSOLE_APP_JS),
|
||||
]
|
||||
|
||||
|
||||
@@ -955,6 +962,7 @@ _CONST_GUARD_BUNDLES = _SWEPT_BUNDLES + [
|
||||
_REPO_ROOT / "turnstone/shared_static/rail.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/interactive.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/conversation.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/redact_credentials.js",
|
||||
]
|
||||
|
||||
|
||||
@@ -1267,41 +1275,76 @@ def test_swept_bundle_has_no_const_reassign(bundle: Path) -> None:
|
||||
)
|
||||
|
||||
|
||||
def test_redact_api_keys_runtime_smoke() -> None:
|
||||
"""Runtime smoke for ``_redactApiKeys``. The function is pure — no
|
||||
DOM dependency — so it transplants cleanly into a standalone
|
||||
``node -e`` invocation. This is the bit that would have caught
|
||||
the original ``const redacted`` bug (which ``node --check`` and a
|
||||
pure-static keyword scan both miss; the ``TypeError`` only fires
|
||||
at call-time)."""
|
||||
body = _INTERACTIVE_JS.read_text(encoding="utf-8")
|
||||
m = re.search(
|
||||
r"function _redactApiKeys\(text\) \{.*?\n\}\n",
|
||||
body,
|
||||
re.DOTALL,
|
||||
)
|
||||
assert m is not None, "_redactApiKeys not found in app.js"
|
||||
fn = m.group(0)
|
||||
script = (
|
||||
fn
|
||||
+ "\nconst q = _redactApiKeys('https://x?api_key=abc&u=foo');\n"
|
||||
def test_redact_credentials_runtime_smoke() -> None:
|
||||
"""Runtime smoke for ``redactCredentials`` via a temp harness file.
|
||||
The function is pure (no DOM dependency). Tests the shared module
|
||||
directly via ESM import (replaces the legacy ``_redactApiKeys`` test
|
||||
which now delegates to this).
|
||||
|
||||
The tempfile is written with a ``.mjs`` extension so Node forces ESM
|
||||
parsing regardless of any ``package.json`` ``type`` field in parent
|
||||
directories. The ``redact_credentials.js`` source file is imported
|
||||
by absolute path so resolution is unambiguous.
|
||||
"""
|
||||
import tempfile
|
||||
|
||||
mod_path = _REDACT_CREDENTIALS_JS.resolve()
|
||||
harness = (
|
||||
"import { redactCredentials } from "
|
||||
+ json.dumps(str(mod_path))
|
||||
+ ";\n"
|
||||
+ "const q = redactCredentials('https://x?api_key=abc&u=foo');\n"
|
||||
+ 'if (q !== "https://x?api_key=***&u=foo") '
|
||||
+ "throw new Error('query-string redact failed: ' + q);\n"
|
||||
+ 'const j = _redactApiKeys(\'{"api_key":"abc"}\');\n'
|
||||
+ 'const j = redactCredentials(\'{"api_key":"abc"}\');\n'
|
||||
+ 'if (j !== \'{"api_key":"***"}\') '
|
||||
+ "throw new Error('json-style redact failed: ' + j);\n"
|
||||
+ "// Bearer token redaction (raw input)\n"
|
||||
+ "const b = redactCredentials('Authorization: Bearer eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjMifQ.test-token_here');\n"
|
||||
+ "if (!b.includes('[REDACTED:api_key]')) "
|
||||
+ "throw new Error('bearer redact failed: ' + b);\n"
|
||||
+ "// Connection string redaction (raw input)\n"
|
||||
+ "const c = redactCredentials('postgresql://user:supersecret@localhost/db');\n"
|
||||
+ "if (!c.includes('[REDACTED:password]')) "
|
||||
+ "throw new Error('conn-string redact failed: ' + c);\n"
|
||||
+ "// Authorization JSON key redaction (step 6 comprehensive)\n"
|
||||
+ 'const a = redactCredentials(\'{"Authorization": "Bearer canstillseethis"}\');\n'
|
||||
+ "if (!a.includes('[REDACTED:secret]')) "
|
||||
+ "throw new Error('authorization JSON redact failed: ' + a);\n"
|
||||
+ "// Single-quote JSON (Python dict repr / JS object literal)\n"
|
||||
+ "const sq = redactCredentials(\"{'Authorization': 'Bearer canstillseethis'}\");\n"
|
||||
+ "if (!sq.includes('[REDACTED:secret]')) "
|
||||
+ "throw new Error('single-quote authorization redact failed: ' + sq);\n"
|
||||
+ "// mongodb+srv connection string (Atlas SRV)\n"
|
||||
+ "const ms = redactCredentials('mongodb+srv://u:s3cretpw@cluster.mongodb.net/db');\n"
|
||||
+ "if (!ms.includes('[REDACTED:password]')) "
|
||||
+ "throw new Error('mongodb+srv redact failed: ' + ms);\n"
|
||||
+ "// lowercase bearer scheme (RFC 7235 case-insensitive)\n"
|
||||
+ "const lb = redactCredentials('authorization: bearer "
|
||||
+ "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.sig12345');\n"
|
||||
+ "if (!lb.includes('[REDACTED:api_key]')) "
|
||||
+ "throw new Error('lowercase bearer redact failed: ' + lb);\n"
|
||||
+ "// api_key= assignment redacts the whole token, not a garbled api_[REDACTED\n"
|
||||
+ "const ak = redactCredentials('api_key=abcdefghijklmnopqrstuvwxyz');\n"
|
||||
+ "if (ak !== '[REDACTED:api_key]') "
|
||||
+ "throw new Error('api_key= clean redact failed: ' + ak);\n"
|
||||
)
|
||||
with tempfile.NamedTemporaryFile(mode="w", suffix=".mjs", delete=False) as f:
|
||||
f.write(harness)
|
||||
tmp = f.name
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["node", "-e", script],
|
||||
["node", tmp],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=15,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
pytest.skip("node binary not available on PATH")
|
||||
finally:
|
||||
os.unlink(tmp)
|
||||
assert proc.returncode == 0, (
|
||||
f"_redactApiKeys runtime smoke failed. stdout={proc.stdout!r} stderr={proc.stderr!r}"
|
||||
f"redactCredentials runtime smoke failed. stdout={proc.stdout!r} stderr={proc.stderr!r}"
|
||||
)
|
||||
|
||||
|
||||
@@ -1757,3 +1800,107 @@ def test_global_stream_recovery_floor_and_render_coalescing() -> None:
|
||||
assert "requestAnimationFrame(" in body[fire : fire + 700], (
|
||||
"fireRender must coalesce subscriber repaints to one per frame"
|
||||
)
|
||||
|
||||
|
||||
def test_server_global_accels_are_platform_aware_and_scoped() -> None:
|
||||
"""The standalone's keydown handler owns only the GLOBAL accels — new
|
||||
workstream, switch, dashboard. They pick the modifier per platform (Ctrl on
|
||||
macOS where the browser owns Cmd, Alt elsewhere) so Ctrl+T/1-9 aren't eaten
|
||||
by the browser off macOS. The per-pane verbs (edit/refresh/fork/delete/
|
||||
close) moved to shell.js, so the handler must not invoke them itself."""
|
||||
body = _APP_JS.read_text(encoding="utf-8")
|
||||
assert "const IS_MAC" in body and 'navigator.platform.indexOf("Mac")' in body, (
|
||||
"the accelerators need a platform check to choose Ctrl vs Alt"
|
||||
)
|
||||
handler = body[body.index('document.addEventListener("keydown"') :]
|
||||
assert "const paneMod" in handler, (
|
||||
"global accels must gate on the platform-aware paneMod, not raw ctrlKey"
|
||||
)
|
||||
assert 'e.ctrlKey && e.key === "t"' not in handler, (
|
||||
"Ctrl+T is browser-reserved off macOS — new workstream must bind via paneMod"
|
||||
)
|
||||
assert "newWorkstream()" in handler and "switchTab(" in handler, (
|
||||
"the standalone handler still owns new + switch"
|
||||
)
|
||||
# macOS Ctrl+T / Ctrl+D are the Cocoa transpose / delete-forward text
|
||||
# bindings; the creation/dashboard chords must yield while typing, through
|
||||
# the shared TS_SHELL.inEditable guard (not a per-file copy).
|
||||
assert "TS_SHELL.inEditable(" in handler, (
|
||||
"new + dashboard must yield to text editing (macOS Ctrl+T / Ctrl+D)"
|
||||
)
|
||||
# The per-pane verbs are shell.js's job now — the standalone handler must not
|
||||
# double-bind them (shell.js drives them off the active pane's menu).
|
||||
for verb in ("editWorkstreamTitle()", "forkWorkstream()", "confirmDeleteWorkstream()"):
|
||||
assert verb not in handler, (
|
||||
f"{verb} moved to shell.js — the app.js handler must not also bind it"
|
||||
)
|
||||
|
||||
|
||||
def test_shortcut_overlay_labels_match_the_platform_modifier() -> None:
|
||||
"""The '?' help overlay must advertise the same modifier the handler
|
||||
listens for — Ctrl on macOS, Alt on Windows/Linux — instead of a hardcoded
|
||||
Ctrl that is wrong (and non-functional) off macOS."""
|
||||
index = _INDEX_HTML.read_text(encoding="utf-8")
|
||||
assert "const PANE_MOD" in index and 'navigator.platform.indexOf("Mac")' in index, (
|
||||
"the overlay must compute its modifier label per platform"
|
||||
)
|
||||
assert "${PANE_MOD}+T" in index, "the New-workstream badge must render through PANE_MOD"
|
||||
assert '<span class="kb-key">Ctrl+T</span>' not in index, (
|
||||
"the New-workstream badge must not hardcode Ctrl (wrong off macOS)"
|
||||
)
|
||||
|
||||
|
||||
def test_pane_menu_accels_are_shared_and_platform_aware() -> None:
|
||||
"""shell.js is the single source of truth for the per-pane tab-menu
|
||||
shortcuts: the badge string and the keydown handler come from ONE registry,
|
||||
so a badge can't advertise a chord the handler ignores. Badges must be
|
||||
platform-aware (no hardcoded Ctrl), and the shared handler must drive the
|
||||
ACTIVE pane's own menu so each surface contributes only what it supports."""
|
||||
shell = _SHELL_JS.read_text(encoding="utf-8")
|
||||
assert "PANE_MENU_ACCELS" in shell and "function paneAccelBadge" in shell, (
|
||||
"shell.js must own the accel registry + badge builder"
|
||||
)
|
||||
assert "const PANE_MOD_LABEL" in shell and 'navigator.platform.indexOf("Mac")' in shell, (
|
||||
"the shared badge must be platform-aware (Ctrl on macOS, Alt elsewhere)"
|
||||
)
|
||||
# The tab-menu items carry a stable accel + a computed badge, NOT a hardcoded
|
||||
# Ctrl string that would lie on Windows/Linux.
|
||||
for accel in ("close-pane", "edit-title", "refresh-title", "delete"):
|
||||
assert f'accel: "{accel}"' in shell, f"tab menu must tag the {accel} item"
|
||||
assert 'key: "Ctrl+Shift+E"' not in shell and 'key: "Ctrl+W"' not in shell, (
|
||||
"tab-menu badges must go through paneAccelBadge, not hardcoded Ctrl"
|
||||
)
|
||||
# The shared handler resolves the active pane and runs its menu item by accel.
|
||||
assert "paneAccelFor(e)" in shell and "pane.tabMenu()" in shell, (
|
||||
"the shared keydown handler must drive the active pane's menu by accel"
|
||||
)
|
||||
# The typing guard is shared (TS_SHELL.inEditable), not copied per surface.
|
||||
assert "function inEditable(" in shell and "inEditable," in shell, (
|
||||
"shell.js must define + expose the shared inEditable guard on TS_SHELL"
|
||||
)
|
||||
ui = _APP_JS.read_text(encoding="utf-8")
|
||||
console = _CONSOLE_APP_JS.read_text(encoding="utf-8")
|
||||
assert "_inEditable" not in ui and "_consoleInEditable" not in console, (
|
||||
"surfaces must use TS_SHELL.inEditable, not a per-file copy of the guard"
|
||||
)
|
||||
|
||||
|
||||
def test_console_has_matching_pane_hotkeys() -> None:
|
||||
"""The console regained pane hotkeys to match the standalone: a keydown
|
||||
handler for switch (Mod+1-9) + dashboard (Ctrl+D), and a '?' overlay that
|
||||
advertises them platform-aware. New workstream and Fork are intentionally
|
||||
omitted (no console fork / blank-new surface)."""
|
||||
app = _CONSOLE_APP_JS.read_text(encoding="utf-8")
|
||||
assert (
|
||||
"_CONSOLE_IS_MAC" in app and "statefulTabs()" in app and 'openPane("dashboard")' in app
|
||||
), "the console must wire switch (statefulTabs) + dashboard hotkeys"
|
||||
index = _CONSOLE_INDEX.read_text(encoding="utf-8")
|
||||
assert "const PANE_MOD" in index and '"Panes"' in index, (
|
||||
"the console '?' overlay needs a platform-aware Panes section"
|
||||
)
|
||||
assert "${PANE_MOD}+W" in index and "${PANE_MOD}+Shift+E" in index, (
|
||||
"console badges must render through PANE_MOD"
|
||||
)
|
||||
assert '"Fork"' not in index and "New workstream" not in index, (
|
||||
"Fork + New are intentionally omitted on the console"
|
||||
)
|
||||
|
||||
@@ -41,6 +41,11 @@ def _bind_ws_event_handlers(bot, cls):
|
||||
attr = getattr(cls, name)
|
||||
if callable(attr):
|
||||
setattr(bot, name, attr.__get__(bot, cls))
|
||||
# ``_handle_stream_end`` delegates the all-cycles sweep to
|
||||
# ``_pop_ws_approvals``; bind the real method too so dispatcher
|
||||
# tests observe the pop instead of a spec'd AsyncMock no-op.
|
||||
if hasattr(cls, "_pop_ws_approvals"):
|
||||
bot._pop_ws_approvals = cls._pop_ws_approvals.__get__(bot, cls)
|
||||
|
||||
|
||||
def _make_message(*, bot=False, guild=True, content="hello", channel=None, reference=None):
|
||||
@@ -537,7 +542,7 @@ class TestApprovalVerdictDisplay:
|
||||
},
|
||||
}
|
||||
]
|
||||
event = ApproveRequestEvent(ws_id="ws-1", items=items)
|
||||
event = ApproveRequestEvent(ws_id="ws-1", cycle_id="cyc-1", items=items)
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
# thread.send was called with an embed containing a verdict field
|
||||
@@ -551,8 +556,8 @@ class TestApprovalVerdictDisplay:
|
||||
assert "HIGH" in field.value
|
||||
assert "85%" in field.value
|
||||
|
||||
# Pending approval message tracked
|
||||
assert "ws-1" in bot._pending_approval_msgs
|
||||
# Pending approval message tracked under (ws_id, cycle_id).
|
||||
assert ("ws-1", "cyc-1") in bot._pending_approval_msgs
|
||||
|
||||
def test_approval_without_verdict(self):
|
||||
"""ApproveRequestEvent items without verdict still work normally."""
|
||||
@@ -585,10 +590,11 @@ class TestApprovalVerdictDisplay:
|
||||
embed = MagicMock()
|
||||
msg.embeds = [embed]
|
||||
msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (msg, frozenset({"c-1"}))
|
||||
|
||||
event = IntentVerdictEvent(
|
||||
ws_id="ws-1",
|
||||
call_id="c-1",
|
||||
func_name="bash",
|
||||
risk_level="high",
|
||||
recommendation="deny",
|
||||
@@ -628,7 +634,10 @@ class TestApprovalVerdictDisplay:
|
||||
bot._streaming = {}
|
||||
bot._thinking_msgs = {}
|
||||
bot._tool_info_msgs = {}
|
||||
bot._pending_approval_msgs = {"ws-1": MagicMock()}
|
||||
bot._pending_approval_msgs = {
|
||||
("ws-1", "cyc-1"): (MagicMock(), frozenset()),
|
||||
("ws-1", "cyc-2"): (MagicMock(), frozenset()),
|
||||
}
|
||||
bot._notify_reply_channels = {}
|
||||
_bind_ws_event_handlers(bot, TurnstoneBot)
|
||||
|
||||
@@ -636,7 +645,8 @@ class TestApprovalVerdictDisplay:
|
||||
event = StreamEndEvent(ws_id="ws-1")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
assert "ws-1" not in bot._pending_approval_msgs
|
||||
# ALL of the ws's cycles are swept, not just one entry.
|
||||
assert not bot._pending_approval_msgs
|
||||
|
||||
|
||||
class TestStreamEndBehavior:
|
||||
@@ -1657,19 +1667,21 @@ class TestApprovalResolved:
|
||||
bot = self._make_bot()
|
||||
thread = AsyncMock()
|
||||
|
||||
# Set up a pending approval message with components.
|
||||
# Set up a pending approval message with components. The event
|
||||
# below carries no cycle_id (pre-multi-cycle server) — the
|
||||
# legacy fallback clears the ws's single tracked entry.
|
||||
approval_msg = MagicMock()
|
||||
approval_msg.embeds = [MagicMock()]
|
||||
approval_msg.components = []
|
||||
approval_msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = approval_msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (approval_msg, frozenset())
|
||||
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=False, feedback="timeout")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
approval_msg.edit.assert_awaited_once()
|
||||
# Pending approval message should be removed.
|
||||
assert "ws-1" not in bot._pending_approval_msgs
|
||||
assert not bot._pending_approval_msgs
|
||||
|
||||
def test_disables_buttons_on_approved(self):
|
||||
from turnstone.sdk.events import ApprovalResolvedEvent
|
||||
@@ -1681,9 +1693,11 @@ class TestApprovalResolved:
|
||||
approval_msg.embeds = [MagicMock()]
|
||||
approval_msg.components = []
|
||||
approval_msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = approval_msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (approval_msg, frozenset())
|
||||
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True)
|
||||
# Cycle-routed resolution: the event's cycle_id selects exactly
|
||||
# this tracked message.
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True, cycle_id="cyc-1")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
approval_msg.edit.assert_awaited_once()
|
||||
|
||||
@@ -87,7 +87,7 @@ class TestSendApproval:
|
||||
monkeypatch.setattr(router._server, "approve", mock_approve)
|
||||
await router.send_approval("ws-1", "corr-abc", approved=True, feedback="ok")
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=True, feedback="ok", always=False
|
||||
ws_id="ws-1", approved=True, feedback="ok", always=False, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
@pytest.mark.anyio
|
||||
@@ -99,7 +99,7 @@ class TestSendApproval:
|
||||
monkeypatch.setattr(router._server, "approve", mock_approve)
|
||||
await router.send_approval("ws-1", "corr-abc", approved=False)
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=False, feedback=None, always=False
|
||||
ws_id="ws-1", approved=False, feedback=None, always=False, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
@pytest.mark.anyio
|
||||
@@ -110,7 +110,9 @@ class TestSendApproval:
|
||||
mock_approve = AsyncMock()
|
||||
monkeypatch.setattr(console_router._console, "route_approve", mock_approve)
|
||||
await console_router.send_approval("ws-1", "corr-abc", approved=True, always=True)
|
||||
mock_approve.assert_awaited_once_with(ws_id="ws-1", approved=True, feedback="", always=True)
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=True, feedback="", always=True, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteRoute:
|
||||
|
||||
@@ -576,10 +576,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -598,10 +599,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -620,10 +622,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -776,7 +779,9 @@ class TestWsEventDispatch:
|
||||
bot, client = self._make_ws_bot()
|
||||
|
||||
event = ApproveRequestEvent(
|
||||
ws_id="ws-1", items=[{"func_name": "bash", "needs_approval": True}]
|
||||
ws_id="ws-1",
|
||||
cycle_id="cyc-1",
|
||||
items=[{"call_id": "c-1", "func_name": "bash", "needs_approval": True}],
|
||||
)
|
||||
route = SlackRoute(channel="C1", user_id="U12345", thread_ts="123.456")
|
||||
_run(bot._on_ws_event("ws-1", route, event)) # type: ignore[attr-defined]
|
||||
@@ -784,8 +789,12 @@ class TestWsEventDispatch:
|
||||
client.chat_postMessage.assert_awaited_once()
|
||||
call_kwargs = client.chat_postMessage.call_args[1]
|
||||
assert "blocks" in call_kwargs
|
||||
assert "ws-1" in bot._pending_approval # type: ignore[attr-defined]
|
||||
assert bot._pending_approval["ws-1"].owner_user_id == "U12345" # type: ignore[attr-defined]
|
||||
# Tracked under (ws_id, cycle_id) so concurrent cycles each get
|
||||
# their own Slack message.
|
||||
entry = bot._pending_approval[("ws-1", "cyc-1")] # type: ignore[attr-defined]
|
||||
assert entry.owner_user_id == "U12345"
|
||||
assert entry.cycle_id == "cyc-1"
|
||||
assert entry.call_ids == frozenset({"c-1"})
|
||||
|
||||
def test_intent_verdict_updates_approval_message(self) -> None:
|
||||
from turnstone.channels.slack.bot import PendingApproval
|
||||
@@ -797,14 +806,17 @@ class TestWsEventDispatch:
|
||||
return_value={"ok": True, "messages": [{"blocks": []}]}
|
||||
)
|
||||
|
||||
bot._pending_approval["ws-1"] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[("ws-1", "cyc-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C1",
|
||||
message_ts="999.000",
|
||||
owner_user_id="U12345",
|
||||
cycle_id="cyc-1",
|
||||
call_ids=frozenset({"c-1"}),
|
||||
)
|
||||
|
||||
event = IntentVerdictEvent(
|
||||
ws_id="ws-1",
|
||||
call_id="c-1",
|
||||
func_name="bash",
|
||||
risk_level="high",
|
||||
confidence=0.9,
|
||||
@@ -821,17 +833,20 @@ class TestWsEventDispatch:
|
||||
from turnstone.sdk.events import ApprovalResolvedEvent
|
||||
|
||||
bot, client = self._make_ws_bot()
|
||||
bot._pending_approval["ws-1"] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[("ws-1", "cyc-9")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C1",
|
||||
message_ts="999.000",
|
||||
owner_user_id="U12345",
|
||||
cycle_id="cyc-9",
|
||||
)
|
||||
|
||||
# Event WITHOUT a cycle_id (pre-multi-cycle server): the legacy
|
||||
# fallback clears the ws's single tracked entry, as before.
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True)
|
||||
route = SlackRoute(channel="C1", user_id="U12345", thread_ts="123.456")
|
||||
_run(bot._on_ws_event("ws-1", route, event)) # type: ignore[attr-defined]
|
||||
|
||||
assert "ws-1" not in bot._pending_approval # type: ignore[attr-defined]
|
||||
assert not bot._pending_approval # type: ignore[attr-defined]
|
||||
client.chat_update.assert_awaited_once()
|
||||
|
||||
def test_link_prefix_does_not_hijack_regular_prompt(self) -> None:
|
||||
|
||||
@@ -1600,6 +1600,82 @@ class TestConsoleProxy:
|
||||
# browser's interactive UI 403-loops on every retry.
|
||||
assert sse_mock.await_args.kwargs.get("use_service_auth") is True
|
||||
|
||||
def test_proxy_events_global_403_without_cluster_inspect(self, mock_collector):
|
||||
"""A plain authenticated user (no service scope, no
|
||||
admin.cluster.inspect) cannot reach the node's cross-tenant
|
||||
firehose through the proxy: elevating to the console's service
|
||||
identity would bypass per-user filtering, so the path is
|
||||
operator-gated. _proxy_sse must NOT be reached."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from starlette.responses import Response
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from turnstone.console.server import _load_static, create_app
|
||||
from turnstone.core.auth import JWT_AUD_CONSOLE, create_jwt
|
||||
|
||||
_load_static()
|
||||
app = create_app(collector=mock_collector, jwt_secret=_TEST_JWT_SECRET)
|
||||
user_jwt = create_jwt(
|
||||
user_id="plain-user",
|
||||
scopes=frozenset({"read"}),
|
||||
source="test",
|
||||
secret=_TEST_JWT_SECRET,
|
||||
audience=JWT_AUD_CONSOLE,
|
||||
permissions=frozenset(),
|
||||
)
|
||||
user_client = TestClient(
|
||||
app,
|
||||
raise_server_exceptions=False,
|
||||
headers={"Authorization": f"Bearer {user_jwt}"},
|
||||
)
|
||||
with patch(
|
||||
"turnstone.console.server._proxy_sse",
|
||||
new_callable=AsyncMock,
|
||||
return_value=Response("ok", status_code=200),
|
||||
) as sse_mock:
|
||||
resp = user_client.get("/node/node-a/v1/api/events/global")
|
||||
assert resp.status_code == 403
|
||||
assert sse_mock.await_count == 0
|
||||
user_client.close()
|
||||
|
||||
def test_proxy_events_global_allows_cluster_inspect(self, mock_collector):
|
||||
"""An operator holding admin.cluster.inspect passes the gate and
|
||||
reaches the SSE proxy with the service token."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from starlette.responses import Response
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from turnstone.console.server import _load_static, create_app
|
||||
from turnstone.core.auth import JWT_AUD_CONSOLE, create_jwt
|
||||
|
||||
_load_static()
|
||||
app = create_app(collector=mock_collector, jwt_secret=_TEST_JWT_SECRET)
|
||||
op_jwt = create_jwt(
|
||||
user_id="operator",
|
||||
scopes=frozenset({"read"}),
|
||||
source="test",
|
||||
secret=_TEST_JWT_SECRET,
|
||||
audience=JWT_AUD_CONSOLE,
|
||||
permissions=frozenset({"admin.cluster.inspect"}),
|
||||
)
|
||||
op_client = TestClient(
|
||||
app,
|
||||
raise_server_exceptions=False,
|
||||
headers={"Authorization": f"Bearer {op_jwt}"},
|
||||
)
|
||||
with patch(
|
||||
"turnstone.console.server._proxy_sse",
|
||||
new_callable=AsyncMock,
|
||||
return_value=Response("ok", status_code=200),
|
||||
) as sse_mock:
|
||||
resp = op_client.get("/node/node-a/v1/api/events/global")
|
||||
assert resp.status_code == 200
|
||||
assert sse_mock.await_count == 1
|
||||
assert sse_mock.await_args.kwargs.get("use_service_auth") is True
|
||||
op_client.close()
|
||||
|
||||
def test_proxy_api_per_ws_events_uses_user_auth_not_service(self, client, mock_collector):
|
||||
"""Per-ws events route uses the user's re-minted JWT, not the
|
||||
service token — the upstream per-ws SSE handler scopes by
|
||||
|
||||
@@ -336,10 +336,88 @@ def test_channel_default_alias_blanked_when_disabled(
|
||||
|
||||
|
||||
def test_models_payload_strips_secret_fields(storage: SQLiteBackend) -> None:
|
||||
"""Regression guard: only alias/model/provider land in the response,
|
||||
never api_key / base_url / context_window / capabilities."""
|
||||
"""Regression guard: only alias/model/provider (+ the derived
|
||||
effort_ladder) land in the response, never api_key / base_url /
|
||||
context_window / raw capabilities."""
|
||||
_seed_model(storage, definition_id="m1", alias="primary")
|
||||
body = _get_models(_make_client(storage))
|
||||
assert body["models"] == [
|
||||
{"alias": "primary", "model": "model-x", "provider": "openai-compatible"}
|
||||
]
|
||||
assert len(body["models"]) == 1
|
||||
entry = body["models"][0]
|
||||
assert set(entry) == {"alias", "model", "provider", "effort_ladder"}
|
||||
assert entry["alias"] == "primary"
|
||||
assert entry["model"] == "model-x"
|
||||
assert entry["provider"] == "openai-compatible"
|
||||
|
||||
|
||||
def test_effort_ladder_parses_string_capabilities(storage: SQLiteBackend) -> None:
|
||||
"""The capabilities column is a JSON STRING — the ladder must survive
|
||||
the parse (regression: .items() on the raw string threw and the
|
||||
guard silently dropped the field from every row)."""
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="qwen",
|
||||
model="qwen3.6-27b",
|
||||
provider="anthropic-compatible",
|
||||
base_url="http://localhost:8000",
|
||||
api_key="dummy",
|
||||
context_window=262144,
|
||||
capabilities='{"thinking_mode": "manual", "thinking_param": "enable_thinking"}',
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
ladder = {r["value"]: r["effective"] for r in body["models"][0]["effort_ladder"]}
|
||||
assert ladder["none"] == "off"
|
||||
assert ladder["medium"] == "on+medium"
|
||||
assert ladder["max"] == "on+max"
|
||||
|
||||
|
||||
def test_effort_ladder_key_survives_malformed_capabilities(
|
||||
storage: SQLiteBackend,
|
||||
) -> None:
|
||||
"""A capabilities column that fails to parse must not drop the key —
|
||||
every row carries ``effort_ladder`` (empty on failure) so clients can
|
||||
index it unconditionally instead of null-checking per row."""
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="broken",
|
||||
model="model-x",
|
||||
provider="openai-compatible",
|
||||
base_url="http://localhost:8000/v1",
|
||||
api_key="dummy",
|
||||
context_window=131072,
|
||||
capabilities="{not valid json",
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
entry = body["models"][0]
|
||||
assert set(entry) == {"alias", "model", "provider", "effort_ladder"}
|
||||
assert entry["effort_ladder"] == []
|
||||
|
||||
|
||||
def test_effort_ladder_honors_responses_api_surface(storage: SQLiteBackend) -> None:
|
||||
"""server_compat.api_surface (namespaced inside the capabilities JSON)
|
||||
switches the projection to the flat-param path — no template toggle."""
|
||||
caps = (
|
||||
'{"thinking_mode": "manual", "thinking_param": "enable_thinking",'
|
||||
' "reasoning_effort_values": ["low", "medium", "high"],'
|
||||
' "server_compat": {"api_surface": "responses"}}'
|
||||
)
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="mistral",
|
||||
model="mistral-medium",
|
||||
provider="openai-compatible",
|
||||
base_url="http://localhost:8000/v1",
|
||||
api_key="dummy",
|
||||
context_window=131072,
|
||||
capabilities=caps,
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
ladder = {r["value"]: r["effective"] for r in body["models"][0]["effort_ladder"]}
|
||||
# Responses surface: flat param only — no "on+"/"off" toggle tokens.
|
||||
assert ladder["medium"] == "medium"
|
||||
assert ladder["none"] == "default"
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
"""``POST /v1/api/admin/models/effort-ladder`` — live modal projection.
|
||||
|
||||
Pure computation over (provider, model, unsaved capability overrides,
|
||||
api_surface); every malformed input must land as a 400, never a 500 —
|
||||
the body is operator-typed form state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from starlette.applications import Starlette
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.routing import Route
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from tests._coord_test_helpers import _AuthMiddleware
|
||||
from turnstone.console.server import admin_effort_ladder
|
||||
|
||||
|
||||
def _make_client() -> TestClient:
|
||||
app = Starlette(
|
||||
routes=[Route("/v1/api/admin/models/effort-ladder", admin_effort_ladder, methods=["POST"])],
|
||||
middleware=[Middleware(_AuthMiddleware)],
|
||||
)
|
||||
client = TestClient(app)
|
||||
client.headers.update({"X-Test-User": "admin", "X-Test-Perms": "admin.models"})
|
||||
return client
|
||||
|
||||
|
||||
def _post(client: TestClient, body: Any) -> Any:
|
||||
return client.post("/v1/api/admin/models/effort-ladder", json=body)
|
||||
|
||||
|
||||
def test_valid_request_returns_ladder() -> None:
|
||||
resp = _post(
|
||||
_make_client(),
|
||||
{
|
||||
"provider": "anthropic-compatible",
|
||||
"model": "qwen3.6-27b",
|
||||
"capabilities": {"thinking_mode": "manual", "thinking_param": "enable_thinking"},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
ladder = {r["value"]: r["effective"] for r in resp.json()["ladder"]}
|
||||
assert ladder["none"] == "off"
|
||||
assert ladder["high"] == "on+high"
|
||||
|
||||
|
||||
def test_api_surface_switches_projection() -> None:
|
||||
body = {
|
||||
"provider": "openai-compatible",
|
||||
"model": "m",
|
||||
"capabilities": {
|
||||
"thinking_mode": "manual",
|
||||
"reasoning_effort_values": ["low", "medium", "high"],
|
||||
},
|
||||
}
|
||||
client = _make_client()
|
||||
chat = {r["value"]: r["effective"] for r in _post(client, body).json()["ladder"]}
|
||||
body["api_surface"] = "responses"
|
||||
responses = {r["value"]: r["effective"] for r in _post(client, body).json()["ladder"]}
|
||||
assert chat["medium"] == "on+medium" # toggle + flat on the chat surface
|
||||
assert responses["medium"] == "medium" # flat only on the responses surface
|
||||
|
||||
|
||||
def test_non_dict_json_body_is_400_not_500() -> None:
|
||||
client = _make_client()
|
||||
for body in (None, [], "x", 7):
|
||||
resp = _post(client, body)
|
||||
assert resp.status_code == 400, (body, resp.status_code, resp.text)
|
||||
|
||||
|
||||
def test_unknown_provider_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "nope", "model": "m"})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_missing_model_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "openai", "model": ""})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_non_dict_capabilities_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "openai", "model": "m", "capabilities": [1]})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_garbage_capability_value_types_are_400() -> None:
|
||||
"""Wrong-typed override values raise inside the resolver → clean 400."""
|
||||
resp = _post(
|
||||
_make_client(),
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-fable-5",
|
||||
"capabilities": {"supports_effort": True, "effort_levels": 5},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_requires_admin_models_permission() -> None:
|
||||
client = _make_client()
|
||||
client.headers.update({"X-Test-Perms": "read"})
|
||||
resp = _post(client, {"provider": "openai", "model": "m"})
|
||||
assert resp.status_code in (401, 403)
|
||||
@@ -16,10 +16,10 @@ to ``SessionUIBase`` automatically enables:
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from tests.conftest import resolve_when_pending
|
||||
from turnstone.console.coordinator_ui import ConsoleCoordinatorUI
|
||||
|
||||
|
||||
@@ -153,7 +153,7 @@ def test_coord_heuristic_verdict_persists_to_storage() -> None:
|
||||
items[0]["_heuristic_verdict"] = hv
|
||||
|
||||
storage = MagicMock()
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(storage):
|
||||
@@ -246,9 +246,8 @@ def test_coord_pending_approval_sets_activity_tag() -> None:
|
||||
def _capture_activity() -> None:
|
||||
captured["activity"] = ui._ws_current_activity
|
||||
captured["state"] = ui._ws_activity_state
|
||||
ui.resolve_approval(False)
|
||||
|
||||
timer = threading.Timer(0.05, _capture_activity)
|
||||
timer = resolve_when_pending(ui, False, before=_capture_activity)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -292,7 +291,7 @@ def test_coord_judge_pending_flag_dynamic_when_heuristic_present() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -338,7 +337,7 @@ def test_coord_judge_pending_false_when_no_heuristic_verdict() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -410,7 +409,7 @@ def test_coord_budget_override_prompts_even_under_blanket_auto_approve() -> None
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(True))
|
||||
timer = resolve_when_pending(ui, True)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -453,7 +452,7 @@ def test_coord_budget_override_survives_wildcard_allow_policy() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(True))
|
||||
timer = resolve_when_pending(ui, True)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()), _patch_policies({"__budget_override__": "allow"}):
|
||||
@@ -526,12 +525,16 @@ class TestBroadcastApprovalResolved:
|
||||
collector = MagicMock()
|
||||
ConsoleCoordinatorUI._collector = collector
|
||||
try:
|
||||
ui._broadcast_approval_resolved(True, "lgtm", always=True)
|
||||
ui._broadcast_approval_resolved(
|
||||
True, "lgtm", always=True, cycle_id="cyc-1", call_ids=("c-1", "c-2")
|
||||
)
|
||||
collector.emit_console_ws_approval_resolved.assert_called_once_with(
|
||||
"coord-a",
|
||||
approved=True,
|
||||
feedback="lgtm",
|
||||
always=True,
|
||||
cycle_id="cyc-1",
|
||||
call_ids=["c-1", "c-2"],
|
||||
)
|
||||
finally:
|
||||
ConsoleCoordinatorUI._collector = None
|
||||
@@ -547,6 +550,8 @@ class TestBroadcastApprovalResolved:
|
||||
approved=False,
|
||||
feedback="",
|
||||
always=False,
|
||||
cycle_id="",
|
||||
call_ids=[],
|
||||
)
|
||||
finally:
|
||||
ConsoleCoordinatorUI._collector = None
|
||||
|
||||
@@ -195,6 +195,21 @@ def test_emit_tolerates_collector_exception() -> None:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_cleanup_ui_sweeps_all_approval_cycles_on_registry_uis() -> None:
|
||||
"""The real ConsoleCoordinatorUI carries the approval-cycle
|
||||
registry: cleanup denies + wakes EVERY parked gate via
|
||||
``resolve_all_approvals`` (parallel task agents can hold several),
|
||||
not the pre-cycle single-slot kick."""
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
ws.ui.resolve_all_approvals = MagicMock(return_value=2) # type: ignore[attr-defined]
|
||||
adapter.cleanup_ui(ws)
|
||||
ws.ui.resolve_all_approvals.assert_called_once_with( # type: ignore[attr-defined]
|
||||
False, "Workstream closed"
|
||||
)
|
||||
assert ws.ui._fg_event.is_set() # type: ignore[attr-defined]
|
||||
|
||||
|
||||
def test_cleanup_ui_unblocks_events_and_broadcasts_to_listeners() -> None:
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
|
||||
@@ -42,6 +42,7 @@ from turnstone.console.server import (
|
||||
_coord_create_post_install,
|
||||
_coord_create_validate_request,
|
||||
_coord_saved_loaded_lookup,
|
||||
_coordinator_tenant_check,
|
||||
_require_admin_coordinator,
|
||||
_require_coord_mgr,
|
||||
cluster_ws_detail,
|
||||
@@ -83,15 +84,24 @@ def _coord_attach_owner(request, ws_id, mgr):
|
||||
|
||||
Kind-strict — coord attachments can only be accessed for
|
||||
workstreams currently held by ``coord_mgr``; no storage fallback
|
||||
so cross-kind ws_ids 404 instead of leaking through storage.
|
||||
so cross-kind ws_ids 404 instead of leaking through storage. Also
|
||||
project-tenancy-strict: mirrors ``_coord_attachment_owner`` so a
|
||||
private-project coordinator's attachments 404-mask non-members.
|
||||
"""
|
||||
from starlette.responses import JSONResponse
|
||||
|
||||
from turnstone.core.auth import WorkstreamProjectVisibility
|
||||
from turnstone.core.web_helpers import auth_user_id
|
||||
|
||||
ws = mgr.get(ws_id)
|
||||
if ws is None:
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
storage = getattr(request.app.state, "auth_storage", None)
|
||||
if storage is None:
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
visibility = WorkstreamProjectVisibility.for_request(request, storage=storage)
|
||||
if not visibility.ws_visible(getattr(ws, "project_id", "") or "", ws_owner=ws.user_id or ""):
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
return ws.user_id or auth_user_id(request), None
|
||||
|
||||
|
||||
@@ -101,7 +111,7 @@ def _coord_attach_owner(request, ws_id, mgr):
|
||||
_coord_endpoint_config = SessionEndpointConfig(
|
||||
permission_gate=_require_admin_coordinator,
|
||||
manager_lookup=_require_coord_mgr,
|
||||
tenant_check=None,
|
||||
tenant_check=_coordinator_tenant_check,
|
||||
not_found_label="coordinator not found",
|
||||
audit_action_prefix="coordinator",
|
||||
supports_attachments=True,
|
||||
@@ -1103,18 +1113,7 @@ def test_approve_resolves_ui_event(storage):
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
assert isinstance(ws.ui, ConsoleCoordinatorUI)
|
||||
ws.ui._pending_approval = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
],
|
||||
}
|
||||
ws.ui._approval_event.clear()
|
||||
cycle = _seed_pending(ws, "c-1")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1122,34 +1121,46 @@ def test_approve_resolves_ui_event(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert ws.ui._approval_result == (True, None)
|
||||
assert resp.json()["cycle_id"] == cycle.cycle_id
|
||||
assert cycle.event.is_set()
|
||||
assert cycle.result == (True, None)
|
||||
assert "spawn_workstream" in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
def _seed_pending(ws, *call_ids: str) -> None:
|
||||
ws.ui._pending_approval = {
|
||||
def _seed_pending(ws, *call_ids: str, func_name: str = "spawn_workstream"):
|
||||
"""Register a live ApprovalCycle on the coord UI the way its
|
||||
``approve_tools`` gate does, returning the cycle for direct
|
||||
event/result assertions (the pre-cycle singleton
|
||||
``_approval_event`` / ``_approval_result`` slots are gone)."""
|
||||
from turnstone.core.session_ui_base import ApprovalCycle
|
||||
|
||||
items = [
|
||||
{
|
||||
"call_id": cid,
|
||||
"func_name": func_name,
|
||||
"approval_label": func_name,
|
||||
"needs_approval": True,
|
||||
}
|
||||
for cid in call_ids
|
||||
]
|
||||
card = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": cid,
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
for cid in call_ids
|
||||
],
|
||||
"cycle_id": f"cyc-{'-'.join(call_ids)}",
|
||||
"items": ws.ui._serialize_approval_items(items),
|
||||
"judge_pending": False,
|
||||
}
|
||||
ws.ui._approval_event.clear()
|
||||
cycle = ApprovalCycle(items, card, None)
|
||||
ws.ui._register_approval_cycle(cycle)
|
||||
return cycle
|
||||
|
||||
|
||||
def test_approve_409_on_stale_call_id(storage):
|
||||
"""Body call_id doesn't match any pending item → 409 with the
|
||||
current primary call_id so the UI can re-render against the
|
||||
new round."""
|
||||
current primary call_id + cycle_id so the UI can re-render
|
||||
against the new round."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-current")
|
||||
cycle = _seed_pending(ws, "c-current")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1160,17 +1171,17 @@ def test_approve_409_on_stale_call_id(storage):
|
||||
body = resp.json()
|
||||
assert body["error"] == "stale call_id"
|
||||
assert body["current_call_id"] == "c-current"
|
||||
# Approval event must NOT be set — no resolve_approval ran.
|
||||
assert not ws.ui._approval_event.is_set()
|
||||
assert body["current_cycle_id"] == cycle.cycle_id
|
||||
# The live cycle must NOT have been resolved.
|
||||
assert not cycle.event.is_set()
|
||||
|
||||
|
||||
def test_approve_409_when_no_pending_and_call_id_sent(storage):
|
||||
"""Body sends a call_id but the UI has no pending approval —
|
||||
409 with current_call_id=None so the UI knows to clear the row."""
|
||||
"""Body sends a call_id but the UI has no live cycle — 409 with
|
||||
current_call_id=None so the UI knows to clear the row."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
# No _pending_approval seeded → ui._pending_approval is None.
|
||||
ws.ui._approval_event.clear()
|
||||
# No cycle registered.
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1179,18 +1190,18 @@ def test_approve_409_when_no_pending_and_call_id_sent(storage):
|
||||
)
|
||||
assert resp.status_code == 409
|
||||
body = resp.json()
|
||||
assert body["error"] == "no pending approval"
|
||||
assert body["error"] == "stale call_id"
|
||||
assert body["current_call_id"] is None
|
||||
assert not ws.ui._approval_event.is_set()
|
||||
assert body["current_cycle_id"] is None
|
||||
|
||||
|
||||
def test_approve_no_call_id_preserves_backward_compat(storage):
|
||||
"""Existing clients (CLI, channel adapters) that omit call_id
|
||||
must still resolve approvals — the guard only kicks in when
|
||||
call_id is present in the body."""
|
||||
must still resolve approvals — a selector-less body lands on the
|
||||
oldest live cycle."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-1")
|
||||
cycle = _seed_pending(ws, "c-1")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1198,18 +1209,18 @@ def test_approve_no_call_id_preserves_backward_compat(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert resp.json()["cycle_id"] == cycle.cycle_id
|
||||
assert cycle.event.is_set()
|
||||
|
||||
|
||||
def test_approve_no_call_id_no_pending_falls_through(storage):
|
||||
"""Legacy clients (no call_id) calling approve when pending is
|
||||
None hit the existing resolve_approval no-op path — the new
|
||||
guard must not change that behavior. Regression guard for the
|
||||
legacy code path that the call_id check intentionally bypasses."""
|
||||
def test_approve_no_call_id_no_pending_resolves_nothing(storage):
|
||||
"""Legacy clients (no call_id) calling approve with no live cycle:
|
||||
200 with ``cycle_id: null`` — the handler resolves NOTHING rather
|
||||
than racing a cycle that registers between its lookup and its
|
||||
resolve (the client can't have been looking at one)."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
ws.ui._approval_event.clear()
|
||||
# No _pending_approval seeded.
|
||||
# No cycle registered.
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1217,7 +1228,7 @@ def test_approve_no_call_id_no_pending_falls_through(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert resp.json()["cycle_id"] is None
|
||||
|
||||
|
||||
def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
@@ -1226,7 +1237,7 @@ def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
one-boolean semantics of resolve_approval."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-1", "c-2", "c-3")
|
||||
cycle = _seed_pending(ws, "c-1", "c-2", "c-3")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1234,7 +1245,61 @@ def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert cycle.event.is_set()
|
||||
|
||||
|
||||
def test_selectorless_always_whitelists_only_the_resolved_oldest_cycle(storage):
|
||||
"""sweep-3 regression: with several live cycles, a selector-less
|
||||
"Approve + Always" must whitelist the tools of the cycle it
|
||||
actually resolved (the oldest) — not a sibling's."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
oldest = _seed_pending(ws, "a-1", func_name="spawn_workstream")
|
||||
newer = _seed_pending(ws, "b-1", func_name="send_message")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
json={"approved": True, "always": True}, # no selector
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["cycle_id"] == oldest.cycle_id
|
||||
assert oldest.event.is_set()
|
||||
assert not newer.event.is_set()
|
||||
assert "spawn_workstream" in ws.ui.auto_approve_tools
|
||||
assert "send_message" not in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
def test_approve_always_skips_whitelist_when_pinned_cycle_lost_the_race(storage):
|
||||
"""sweep-3 regression: the handler collects always-names from the
|
||||
cycle its lookup pinned; if that cycle is resolved by someone else
|
||||
(gate timeout, peer tab) between lookup and resolve, the whitelist
|
||||
must NOT grow — approving a card that already resolved must not
|
||||
auto-approve anything."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "a-1", func_name="spawn_workstream")
|
||||
ui = ws.ui
|
||||
real_find = ui.find_approval_cycle
|
||||
|
||||
def racing_find(**kwargs):
|
||||
card = real_find(**kwargs)
|
||||
if card is not None:
|
||||
# A concurrent resolver wins the gap between the handler's
|
||||
# lookup and its (pinned) resolve.
|
||||
ui.resolve_approval(False, "raced", cycle_id=card["cycle_id"])
|
||||
return card
|
||||
|
||||
ui.find_approval_cycle = racing_find
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
json={"approved": True, "always": True},
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["cycle_id"] is None
|
||||
assert "spawn_workstream" not in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1353,6 +1418,110 @@ def test_history_any_admin_coordinator_caller_can_read(storage):
|
||||
assert resp.json()["ws_id"] == ws.id
|
||||
|
||||
|
||||
def test_history_private_project_hidden_from_non_member(storage):
|
||||
# admin.coordinator gates the surface, but a coordinator in a private
|
||||
# project the caller isn't a member of is 404-masked — the conversation
|
||||
# does not leak to a non-member operator.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/history",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_history_private_project_visible_to_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/history",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert any(m.get("content") == "secret plan" for m in resp.json()["messages"])
|
||||
|
||||
|
||||
def test_export_private_project_hidden_from_non_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/export",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_children_private_project_hidden_from_non_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/children",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_open_private_project_hidden_from_non_member(storage):
|
||||
# `open` rehydrates + returns the auto-titled name, so an ungated open is a
|
||||
# private-project existence/metadata oracle AND an unauthorized resurrection.
|
||||
# The tenant_check must fire before the already-loaded shortcut and mgr.open.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{'c' * 32}/open",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_coord_attachments_private_project_hidden_from_non_member(storage):
|
||||
# Attachment list/serve resolves the owner as the coord owner and only
|
||||
# enforced cross-kind before — a non-member operator could enumerate and
|
||||
# download the owner's staged blobs. Now 404-masked by project tenancy.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{ws.id}/attachments",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_coord_attachments_private_project_visible_to_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{ws.id}/attachments",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
|
||||
|
||||
def test_history_serves_storage_only_workstream(storage):
|
||||
"""Persisted-but-not-loaded coordinators (closed / evicted) are still
|
||||
readable via /history without rehydrating. Mirrors the pre-lift
|
||||
@@ -1521,15 +1690,19 @@ def test_export_404_when_kind_interactive(storage):
|
||||
|
||||
|
||||
def test_cancel_resolves_pending_approval(storage):
|
||||
"""Cancel addresses the workstream, not one batch — EVERY live
|
||||
cycle resolves (parallel task agents can hold several gates)."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
assert isinstance(ws.ui, ConsoleCoordinatorUI)
|
||||
ws.ui._pending_approval = {"type": "approve_request", "items": []}
|
||||
ws.ui._approval_event.clear()
|
||||
first = _seed_pending(ws, "c-1")
|
||||
second = _seed_pending(ws, "c-2")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(f"/v1/api/workstreams/{ws.id}/cancel", headers=_COORD_HEADERS)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert first.event.is_set()
|
||||
assert second.event.is_set()
|
||||
assert first.result == (False, "Cancelled by user")
|
||||
|
||||
|
||||
def test_cancel_response_always_includes_dropped_key(storage):
|
||||
@@ -2049,6 +2222,10 @@ def test_open_any_admin_coordinator_caller_succeeds_in_memory(storage):
|
||||
|
||||
def test_open_rehydrates_when_not_in_memory(storage, monkeypatch):
|
||||
mgr = _build_mgr(storage)
|
||||
# The tenancy gate resolves the row from storage before rehydrating, so a
|
||||
# legitimately-openable coordinator must exist there (it always does in
|
||||
# production — open rehydrates a persisted row).
|
||||
storage.register_workstream("coord-rehy", kind="coordinator", user_id="user-1")
|
||||
rehydrated = MagicMock()
|
||||
rehydrated.id = "coord-rehy"
|
||||
rehydrated.name = "rehydrated"
|
||||
@@ -2082,6 +2259,7 @@ def test_open_503_on_coord_mgr_unavailable(storage):
|
||||
|
||||
def test_open_correlation_id_on_factory_failure(storage, monkeypatch):
|
||||
mgr = _build_mgr(storage)
|
||||
storage.register_workstream("bad-ws", kind="coordinator", user_id="user-1")
|
||||
monkeypatch.setattr(mgr, "open", MagicMock(side_effect=RuntimeError("boom")))
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post("/v1/api/workstreams/bad-ws/open", headers=_COORD_HEADERS)
|
||||
@@ -2092,6 +2270,7 @@ def test_open_correlation_id_on_factory_failure(storage, monkeypatch):
|
||||
def test_open_503_when_open_raises_value_error(storage, monkeypatch):
|
||||
"""ValueError from the factory surfaces as 503 with the remediation text."""
|
||||
mgr = _build_mgr(storage)
|
||||
storage.register_workstream("bad-ws", kind="coordinator", user_id="user-1")
|
||||
monkeypatch.setattr(mgr, "open", MagicMock(side_effect=ValueError("coord registry missing")))
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post("/v1/api/workstreams/bad-ws/open", headers=_COORD_HEADERS)
|
||||
@@ -2257,7 +2436,8 @@ def test_cluster_inspect_invalid_ws_id_400(storage):
|
||||
|
||||
|
||||
def test_cluster_inspect_any_inspect_caller_sees_detail(storage):
|
||||
# Trusted-team visibility: admin.cluster.inspect sees every row.
|
||||
# A project-less workstream has no tenancy to enforce, so any
|
||||
# admin.cluster.inspect caller sees it (trusted-team default).
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="owner")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
@@ -2269,6 +2449,46 @@ def test_cluster_inspect_any_inspect_caller_sees_detail(storage):
|
||||
assert resp.json()["persisted"]["ws_id"] == ws.id
|
||||
|
||||
|
||||
def test_cluster_inspect_private_project_hidden_from_non_member(storage):
|
||||
# admin.cluster.inspect gates the surface, but a workstream in a
|
||||
# private project the caller isn't a member of is masked as 404 —
|
||||
# no private-project oracle even for a cluster admin.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/{'c' * 32}/detail",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_cluster_inspect_private_project_visible_to_member(storage):
|
||||
# A project member (even a non-owner) still sees the persisted row.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
storage.register_workstream(
|
||||
"c" * 32,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/{'c' * 32}/detail",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["persisted"]["ws_id"] == "c" * 32
|
||||
|
||||
|
||||
def test_cluster_inspect_coordinator_self_path(storage):
|
||||
"""A coordinator row returns live from the in-process manager."""
|
||||
mgr = _build_mgr(storage)
|
||||
@@ -2399,6 +2619,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
ws_id = "f0" * 16
|
||||
_seed_node_workstream(storage, ws_id=ws_id, node_id="node-a")
|
||||
detail = {
|
||||
"cycle_id": "cyc-bash",
|
||||
"call_id": "c-bash",
|
||||
"judge_pending": False,
|
||||
"items": [
|
||||
@@ -2427,7 +2648,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
"activity_state": "approval",
|
||||
"activity": "awaiting approval",
|
||||
"tokens": 100,
|
||||
"pending_approval_detail": detail,
|
||||
"pending_approval_details": [detail],
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -2438,7 +2659,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
assert resp.status_code == 200
|
||||
live = resp.json()["live"]
|
||||
assert live["pending_approval"] is True # derived bool, existing behavior
|
||||
assert live["pending_approval_detail"] == detail # full payload, new behavior
|
||||
assert live["pending_approval_details"] == [detail] # full payload passthrough
|
||||
|
||||
|
||||
def test_cluster_inspect_node_backed_pending_approval_synthesized(storage):
|
||||
|
||||
@@ -313,17 +313,17 @@ def test_coordinator_js_handle_child_state_no_longer_reads_sse_pending_approval_
|
||||
)
|
||||
|
||||
# The merge body must preserve BOTH pending_approval and
|
||||
# pending_approval_detail from prev — preserving only one would
|
||||
# pending_approval_details from prev — preserving only one would
|
||||
# render a row with a phantom badge but no buttons (or vice versa).
|
||||
merge_body = re.search(
|
||||
r"mergedLive\s*=\s*Object\.assign\(\s*\{\}\s*,\s*live\s*,\s*\{"
|
||||
r"[^}]*pending_approval:\s*prev\.live\.pending_approval[^}]*"
|
||||
r"pending_approval_detail:\s*prev\.live\.pending_approval_detail",
|
||||
r"pending_approval_details:\s*prev\.live\.pending_approval_details",
|
||||
body,
|
||||
)
|
||||
assert merge_body is not None, (
|
||||
"Merge body must preserve both pending_approval AND "
|
||||
"pending_approval_detail from prev.live — preserving only one "
|
||||
"pending_approval_details from prev.live — preserving only one "
|
||||
"creates a half-rendered approval row."
|
||||
)
|
||||
|
||||
|
||||
@@ -198,6 +198,23 @@ def test_spawn_prepare_needs_approval(coord_session):
|
||||
assert item["skill"] == "s"
|
||||
|
||||
|
||||
def test_spawn_prepare_denies_high_risk_skill(coord_session):
|
||||
"""Review fix: the high/critical-risk gate that blocks skills(load) also
|
||||
blocks spawn_workstream(skill=…), so a child spawn can't route around it."""
|
||||
sess, _coord, _ui = coord_session
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = {
|
||||
"name": "danger",
|
||||
"risk_level": "critical",
|
||||
}
|
||||
item = sess._prepare_tool(
|
||||
_tc("spawn_workstream", {"initial_message": "go", "skill": "danger"})
|
||||
)
|
||||
assert "error" in item
|
||||
assert "/skill danger" in item["error"]
|
||||
assert item.get("needs_approval") is not True
|
||||
|
||||
|
||||
def test_spawn_exec_calls_client_and_returns_summary(coord_session):
|
||||
sess, coord, _ui = coord_session
|
||||
coord.spawn.return_value = {
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
"""Tests for the effective effort-ladder projection.
|
||||
|
||||
The ladder must mirror the request-time mapping functions exactly —
|
||||
equal ``effective`` tokens promise byte-identical effort behavior on
|
||||
the wire, which is what the UI annotations lean on.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from turnstone.core.providers._protocol import ModelCapabilities
|
||||
from turnstone.core.providers.effort_ladder import (
|
||||
KNOB_VALUES,
|
||||
effort_ladder,
|
||||
effort_ladder_for_model,
|
||||
)
|
||||
|
||||
|
||||
def _as_map(ladder: list[dict[str, str]]) -> dict[str, str]:
|
||||
assert [r["value"] for r in ladder] == list(KNOB_VALUES)
|
||||
return {r["value"]: r["effective"] for r in ladder}
|
||||
|
||||
|
||||
class TestLocalLanes:
|
||||
def test_toggle_engaged_carries_graded_value_per_position(self) -> None:
|
||||
"""No declared effort key: the toggle rides the knob AND the graded
|
||||
value is forwarded under the fallback template key — the user's
|
||||
effort setting always reaches the wire (a template that doesn't
|
||||
reference the kwarg ignores it), so every position is distinct."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
eff = _as_map(effort_ladder("anthropic-compatible", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["minimal"] == "on+minimal"
|
||||
assert eff["max"] == "on+max"
|
||||
assert len({eff[k] for k in KNOB_VALUES}) == len(KNOB_VALUES)
|
||||
|
||||
def test_freeform_effort_param_forwards_each_value(self) -> None:
|
||||
"""deepseek-style config: toggle + verbatim effort per position."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic-compatible", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["low"] == "on+low"
|
||||
assert eff["max"] == "on+max"
|
||||
|
||||
def test_validated_effort_param_shows_snapping(self) -> None:
|
||||
"""Off-list positions round up onto the declared values; above the
|
||||
ceiling they ride the ceiling — never the (possibly lower) default."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["minimal"] == "on+low"
|
||||
assert eff["high"] == "on+high"
|
||||
assert eff["xhigh"] == "on+high"
|
||||
assert eff["max"] == "on+high"
|
||||
|
||||
def test_openai_compatible_flat_param_without_effort_param(self) -> None:
|
||||
caps = ModelCapabilities(
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == "high" # ceiling, not default
|
||||
|
||||
def test_adaptive_local_never_off(self) -> None:
|
||||
caps = ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking")
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["none"] == "on"
|
||||
assert eff["max"] == "on"
|
||||
|
||||
|
||||
class TestNativeAnthropicLane:
|
||||
def test_adaptive_with_effort_levels(self) -> None:
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="adaptive",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high", "xhigh", "max"),
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["none"] == "adaptive" # thinking on, model decides
|
||||
assert eff["minimal"] == "low" # rounds up onto the declared levels
|
||||
assert eff["low"] == "low"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_sonnet_5_registry_row(self) -> None:
|
||||
"""claude-sonnet-5: adaptive + full effort ladder incl. xhigh/max —
|
||||
every knob level above none is a distinct wire behavior."""
|
||||
eff = _as_map(effort_ladder_for_model("anthropic", "claude-sonnet-5", None))
|
||||
assert eff["none"] == "adaptive"
|
||||
assert eff["minimal"] == "low" # rounds up onto declared levels
|
||||
assert eff["low"] == "low"
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_sonnet_4_6_xhigh_rides_max(self) -> None:
|
||||
"""Sonnet 4.6 declares (low, medium, high, max) — no xhigh, so the
|
||||
knob's xhigh snaps up onto max rather than down onto high."""
|
||||
eff = _as_map(effort_ladder_for_model("anthropic", "claude-sonnet-4-6", None))
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == "max"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_manual_budget_ladder(self) -> None:
|
||||
"""Budgets are monotone over the whole knob domain."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["minimal"] == eff["low"] == "budget:1024" # 1024 = API floor
|
||||
assert eff["medium"] == "budget:4096"
|
||||
assert eff["high"] == "budget:16384"
|
||||
assert eff["xhigh"] == "budget:32768"
|
||||
assert eff["max"] == "budget:65536"
|
||||
|
||||
|
||||
class TestFlatParamLanes:
|
||||
def test_google_default_caps(self) -> None:
|
||||
eff = _as_map(effort_ladder_for_model("google", "gemini-3-flash", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "minimal"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_google_override_routes_through_chat_lane(self) -> None:
|
||||
"""GoogleProvider inherits _finalize_extra_body — a thinking_mode
|
||||
override changes real requests, and the ladder must mirror it."""
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"google",
|
||||
"gemini-3-flash",
|
||||
{"thinking_mode": "manual", "thinking_param": "enable_thinking"},
|
||||
)
|
||||
)
|
||||
assert eff["none"] == "off"
|
||||
assert eff["medium"] == "on+medium" # toggle + inherited flat param
|
||||
|
||||
def test_responses_surface_projects_flat_only(self) -> None:
|
||||
caps_overrides = {
|
||||
"thinking_mode": "manual",
|
||||
"reasoning_effort_values": ["low", "medium", "high"],
|
||||
}
|
||||
chat = _as_map(effort_ladder_for_model("openai-compatible", "m", caps_overrides))
|
||||
responses = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"openai-compatible", "m", caps_overrides, api_surface="responses"
|
||||
)
|
||||
)
|
||||
assert chat["medium"] == "on+medium"
|
||||
assert responses["medium"] == "medium"
|
||||
assert responses["none"] == "default"
|
||||
|
||||
def test_xai_projects_flat_only(self) -> None:
|
||||
"""grok-4.3 declares values (none/low/medium/high, default low);
|
||||
knob positions above the ceiling ride the ceiling (high). The
|
||||
declared "none" IS forwarded for the knob's off position (xAI
|
||||
documents it as disabling reasoning) but is never a snap target
|
||||
for other positions."""
|
||||
eff = _as_map(effort_ladder_for_model("xai", "grok-4.3", None))
|
||||
assert eff["none"] == "none" # explicit disable, declared by grok
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["low"] == "low"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_xai_ignores_template_overrides(self) -> None:
|
||||
"""XAIProvider subclasses OpenAIResponsesProvider, which drops
|
||||
extra_body — a thinking_mode/effort_param override cannot change
|
||||
an xai request, so it must not change the ladder either."""
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"xai",
|
||||
"grok-4.3",
|
||||
{
|
||||
"thinking_mode": "manual",
|
||||
"thinking_param": "enable_thinking",
|
||||
"effort_param": "reasoning_effort",
|
||||
},
|
||||
)
|
||||
)
|
||||
assert eff["none"] == "none" # flat channel, not an "off" toggle
|
||||
assert eff["medium"] == "medium"
|
||||
assert all("+" not in v and v not in ("on", "off") for v in eff.values())
|
||||
|
||||
def test_openai_gpt55_registry_row(self) -> None:
|
||||
"""gpt-5.5 declares none/low/medium/high/xhigh with default medium:
|
||||
knob none sends the explicit "none" level (server default is
|
||||
MEDIUM, so omission would not disable), max rides the xhigh
|
||||
ceiling, minimal rounds up to low."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.5", None))
|
||||
assert eff["none"] == "none"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_openai_o3_registry_row(self) -> None:
|
||||
"""o-series (except o1-mini) accept low/medium/high; no declared
|
||||
"none" level, so the knob's off position omits the param."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "o3", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["medium"] == "medium"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_openai_codex_max_has_xhigh(self) -> None:
|
||||
"""gpt-5.1-codex-max must not prefix-fall onto the gpt-5.1 row
|
||||
(which lacks xhigh) — xhigh reaches the wire verbatim."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.1-codex-max", None))
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_anthropic_effort_applies_even_with_thinking_mode_none(self) -> None:
|
||||
"""output_config gates on supports_effort alone at request time."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["high"] == "high"
|
||||
assert eff["none"] == "default"
|
||||
|
||||
def test_overrides_merge_and_unknown_keys_ignored(self) -> None:
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"google",
|
||||
"gemini-3-flash",
|
||||
{"reasoning_effort_values": [], "not_a_field": True},
|
||||
)
|
||||
)
|
||||
# Operator cleared the values → nothing effort-related is sent.
|
||||
assert set(eff.values()) == {"default"}
|
||||
@@ -0,0 +1,410 @@
|
||||
"""Ladder↔wire parity harness — the effort ladder must tell the truth.
|
||||
|
||||
``effort_ladder`` *projects* the session effort knob through the same
|
||||
mapping functions the providers use at request time. This suite proves
|
||||
that projection against the REAL request path: for every provider lane
|
||||
and capability shape, each knob position is driven through the actual
|
||||
provider ``create_streaming`` against a recording fake client (the same
|
||||
SDK-seam capture the wire-payload goldens use), the effort-relevant
|
||||
subset of the captured kwargs is extracted, and it must equal what the
|
||||
ladder token decodes to. Two invariants per shape:
|
||||
|
||||
1. **Semantics** — each ladder token decodes to an expected wire subset
|
||||
(``on``/``off`` ⇒ the chat-template toggle, ``budget:N`` ⇒ Anthropic
|
||||
thinking budget, a bare level ⇒ the lane's flat/effort channel) and
|
||||
the observed wire subset must match it exactly.
|
||||
2. **Grouping** — the ladder's core promise: two knob positions carry
|
||||
equal ``effective`` tokens if and only if they produce identical
|
||||
effort-relevant wire payloads.
|
||||
|
||||
A failure here means the UI annotates behavior the wire does not have —
|
||||
the bug class that shipped xai in the ladder's chat-lane set even though
|
||||
``XAIProvider`` rides the Responses surface, which drops ``extra_body``.
|
||||
|
||||
The harness goes through ``create_provider`` (not direct classes) so the
|
||||
provider ROUTING the ladder assumes — e.g. ``api_surface="responses"``
|
||||
selecting the Responses adapter — is itself under test.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import dataclasses
|
||||
import itertools
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._wire_capture import RecordingClient
|
||||
from turnstone.core.providers import create_provider
|
||||
from turnstone.core.providers._protocol import (
|
||||
EFFORT_TEMPLATE_FALLBACK_PARAM,
|
||||
ModelCapabilities,
|
||||
)
|
||||
from turnstone.core.providers.effort_ladder import KNOB_VALUES, effort_ladder
|
||||
|
||||
# Above the largest manual-mode thinking budget (max: 65536) so the
|
||||
# request path's budget<max_tokens clamp never fires — the ladder
|
||||
# documents budgets unclamped, so the capture must be too. (At small
|
||||
# per-request max_tokens the clamp can genuinely alias adjacent budget
|
||||
# tiers on the wire; that is the ladder's documented approximation, not
|
||||
# a parity break.)
|
||||
_MAX_TOKENS = 128_000
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
class Shape:
|
||||
"""One (provider lane, capability shape) point of the parity matrix."""
|
||||
|
||||
id: str
|
||||
provider: str
|
||||
caps: ModelCapabilities
|
||||
api_surface: str = ""
|
||||
model: str = "m"
|
||||
|
||||
|
||||
# Real registry rows for the lanes whose defaults carry effort values —
|
||||
# parity should cover what ships, not only synthetic shapes.
|
||||
_GEMINI_CAPS = create_provider("google").get_capabilities("gemini-3-flash")
|
||||
_GROK_CAPS = create_provider("xai").get_capabilities("grok-4.3")
|
||||
_GPT55_CAPS = create_provider("openai").get_capabilities("gpt-5.5")
|
||||
|
||||
SHAPES: tuple[Shape, ...] = (
|
||||
# -- anthropic-compatible (vLLM /v1/messages): template channel only --
|
||||
Shape(
|
||||
"compat-toggle-manual",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"compat-toggle-adaptive",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"compat-freeform-effort",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
# DeepSeek-V4 official contract: toggle + effort in {high, max}.
|
||||
"compat-validated-effort",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("high", "max"),
|
||||
default_reasoning_effort="high",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"compat-inert",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="none"),
|
||||
),
|
||||
# -- openai-compatible on the Chat Completions surface: both channels --
|
||||
Shape(
|
||||
"oc-toggle-only",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"oc-toggle-plus-flat",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-effort-param-suppresses-flat",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-flat-only",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-adaptive",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking"),
|
||||
),
|
||||
# -- openai-compatible pinned to the Responses surface: template caps
|
||||
# become inert and only the native flat channel remains --
|
||||
Shape(
|
||||
"oc-responses-surface",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
api_surface="responses",
|
||||
),
|
||||
# -- commercial flat lanes --
|
||||
Shape(
|
||||
# Real registry row: none/low/medium/high/xhigh, default medium.
|
||||
# Knob none must send the EXPLICIT "none" level (omission would
|
||||
# leave the server default medium reasoning on); knob max rides
|
||||
# the xhigh ceiling.
|
||||
"openai-gpt-5.5",
|
||||
"openai",
|
||||
_GPT55_CAPS,
|
||||
model="gpt-5.5",
|
||||
),
|
||||
Shape("google-default", "google", _GEMINI_CAPS, model="gemini-3-flash"),
|
||||
Shape(
|
||||
# GoogleProvider subclasses the chat provider, so a template
|
||||
# override DOES change real requests — hybrid toggle + flat.
|
||||
"google-manual-override",
|
||||
"google",
|
||||
dataclasses.replace(_GEMINI_CAPS, thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
model="gemini-3-flash",
|
||||
),
|
||||
Shape("xai-default", "xai", _GROK_CAPS, model="grok-4.3"),
|
||||
Shape(
|
||||
# XAIProvider rides the Responses surface: template overrides are
|
||||
# inert on the wire, and the ladder must not pretend otherwise.
|
||||
"xai-template-override-inert",
|
||||
"xai",
|
||||
dataclasses.replace(
|
||||
_GROK_CAPS,
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
),
|
||||
model="grok-4.3",
|
||||
),
|
||||
# -- native Anthropic --
|
||||
Shape(
|
||||
"anthropic-adaptive-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="adaptive",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high", "xhigh", "max"),
|
||||
),
|
||||
model="claude-fable-5",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-adaptive-plain",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="adaptive"),
|
||||
model="claude-fable-5",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-manual-budgets",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="manual"),
|
||||
model="claude-3-7-sonnet-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-manual-plus-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
),
|
||||
model="claude-3-7-sonnet-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-none-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
),
|
||||
model="claude-3-5-haiku-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-inert",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="none"),
|
||||
model="claude-3-5-haiku-latest",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Wire capture + effort-subset extraction
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _wire_payload(shape: Shape, knob: str) -> dict[str, Any]:
|
||||
"""Drive the real provider request path; return the captured SDK kwargs."""
|
||||
provider = create_provider(shape.provider, api_surface=shape.api_surface or None)
|
||||
client = RecordingClient()
|
||||
gen = provider.create_streaming(
|
||||
client=client,
|
||||
model=shape.model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=_MAX_TOKENS,
|
||||
reasoning_effort=knob,
|
||||
capabilities=shape.caps,
|
||||
)
|
||||
# kwargs are recorded eagerly during the call above; close the
|
||||
# unconsumed iterator so stream-manager cleanup runs on the stub.
|
||||
close = getattr(gen, "close", None)
|
||||
if callable(close):
|
||||
with contextlib.suppress(Exception):
|
||||
close()
|
||||
assert "payload" in client.captured, f"{shape.id}: provider made no SDK call"
|
||||
return dict(client.captured["payload"])
|
||||
|
||||
|
||||
def _effort_wire_subset(payload: dict[str, Any], shape: Shape) -> dict[str, Any]:
|
||||
"""Every effort-related lever in *payload*, normalized across lanes.
|
||||
|
||||
Keys: ``thinking`` (native Anthropic param), ``output_effort``
|
||||
(Anthropic ``output_config.effort``), ``flat`` (Chat Completions
|
||||
``reasoning_effort`` / Responses ``reasoning.effort``), ``toggle``
|
||||
and ``template_effort`` (``extra_body.chat_template_kwargs`` — the
|
||||
graded key is ``caps.effort_param``, else the fallback template key
|
||||
on the anthropic-compatible lane, whose only effort channel is the
|
||||
template).
|
||||
"""
|
||||
caps = shape.caps
|
||||
effort_key = caps.effort_param or (
|
||||
EFFORT_TEMPLATE_FALLBACK_PARAM if shape.provider == "anthropic-compatible" else ""
|
||||
)
|
||||
subset: dict[str, Any] = {}
|
||||
if "thinking" in payload:
|
||||
subset["thinking"] = payload["thinking"]
|
||||
output_config = payload.get("output_config")
|
||||
if isinstance(output_config, dict) and "effort" in output_config:
|
||||
subset["output_effort"] = output_config["effort"]
|
||||
if "reasoning_effort" in payload:
|
||||
subset["flat"] = payload["reasoning_effort"]
|
||||
reasoning = payload.get("reasoning")
|
||||
if isinstance(reasoning, dict) and "effort" in reasoning:
|
||||
subset["flat"] = reasoning["effort"]
|
||||
extra_body = payload.get("extra_body")
|
||||
ctk = extra_body.get("chat_template_kwargs") if isinstance(extra_body, dict) else None
|
||||
if isinstance(ctk, dict):
|
||||
known = {caps.thinking_param, effort_key} - {""}
|
||||
unexpected = set(ctk) - known
|
||||
assert not unexpected, f"unexpected chat_template_kwargs keys: {unexpected}"
|
||||
if caps.thinking_param in ctk:
|
||||
subset["toggle"] = ctk[caps.thinking_param]
|
||||
if effort_key and effort_key in ctk:
|
||||
subset["template_effort"] = ctk[effort_key]
|
||||
return subset
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Ladder-token decoding — the token grammar, made executable
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _decode_token(shape: Shape, token: str) -> dict[str, Any]:
|
||||
"""Expected effort wire subset for a ladder ``effective`` token."""
|
||||
caps = shape.caps
|
||||
if shape.provider == "anthropic":
|
||||
return _decode_native(caps, token)
|
||||
if shape.provider in ("openai", "xai") or shape.api_surface == "responses":
|
||||
return {} if token == "default" else {"flat": token}
|
||||
return _decode_template(shape.provider, caps, token)
|
||||
|
||||
|
||||
def _decode_native(caps: ModelCapabilities, token: str) -> dict[str, Any]:
|
||||
if caps.thinking_mode == "adaptive":
|
||||
# Thinking is unconditionally adaptive; a non-"adaptive" token is
|
||||
# the output_config effort level riding on top.
|
||||
expected: dict[str, Any] = {"thinking": {"type": "adaptive"}}
|
||||
if token != "adaptive":
|
||||
expected["output_effort"] = token
|
||||
return expected
|
||||
if token in ("default", "off"):
|
||||
return {}
|
||||
effort, sep, budget = token.partition("·budget:")
|
||||
if sep:
|
||||
return {
|
||||
"output_effort": effort,
|
||||
"thinking": {"type": "enabled", "budget_tokens": int(budget)},
|
||||
}
|
||||
if token.startswith("budget:"):
|
||||
budget_tokens = int(token.removeprefix("budget:"))
|
||||
return {"thinking": {"type": "enabled", "budget_tokens": budget_tokens}}
|
||||
return {"output_effort": token}
|
||||
|
||||
|
||||
def _decode_template(provider: str, caps: ModelCapabilities, token: str) -> dict[str, Any]:
|
||||
if token == "default":
|
||||
return {}
|
||||
parts = token.split("+")
|
||||
expected: dict[str, Any] = {}
|
||||
if parts[0] in ("on", "off"):
|
||||
expected["toggle"] = parts[0] == "on"
|
||||
parts = parts[1:]
|
||||
if parts:
|
||||
assert len(parts) == 1, f"unparseable ladder token: {token!r}"
|
||||
if caps.effort_param or provider == "anthropic-compatible":
|
||||
# Declared graded key, or the anthropic-compatible fallback
|
||||
# template key — that lane has no flat channel, so a graded
|
||||
# part there is always template-borne.
|
||||
expected["template_effort"] = parts[0]
|
||||
else:
|
||||
expected["flat"] = parts[0]
|
||||
return expected
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# The parity tests
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", SHAPES, ids=lambda s: s.id)
|
||||
def test_ladder_tokens_match_wire(shape: Shape) -> None:
|
||||
"""Invariant 1: each token's decoded meaning equals the captured wire."""
|
||||
ladder = effort_ladder(shape.provider, shape.caps, shape.api_surface)
|
||||
assert [row["value"] for row in ladder] == list(KNOB_VALUES)
|
||||
for row in ladder:
|
||||
knob, token = row["value"], row["effective"]
|
||||
observed = _effort_wire_subset(_wire_payload(shape, knob), shape)
|
||||
expected = _decode_token(shape, token)
|
||||
assert observed == expected, (
|
||||
f"{shape.id}/knob={knob}: ladder says {token!r} which decodes to "
|
||||
f"{expected}, but the wire carries {observed}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", SHAPES, ids=lambda s: s.id)
|
||||
def test_equal_tokens_iff_equal_wire(shape: Shape) -> None:
|
||||
"""Invariant 2: token equality ⇔ effort-wire equality, per shape."""
|
||||
tokens = {
|
||||
row["value"]: row["effective"]
|
||||
for row in effort_ladder(shape.provider, shape.caps, shape.api_surface)
|
||||
}
|
||||
subsets = {knob: _effort_wire_subset(_wire_payload(shape, knob), shape) for knob in KNOB_VALUES}
|
||||
for a, b in itertools.combinations(KNOB_VALUES, 2):
|
||||
same_token = tokens[a] == tokens[b]
|
||||
same_wire = subsets[a] == subsets[b]
|
||||
assert same_token == same_wire, (
|
||||
f"{shape.id}: knobs {a!r}/{b!r} have "
|
||||
f"{'equal' if same_token else 'distinct'} tokens "
|
||||
f"({tokens[a]!r} vs {tokens[b]!r}) but "
|
||||
f"{'identical' if same_wire else 'different'} wire subsets "
|
||||
f"({subsets[a]} vs {subsets[b]})"
|
||||
)
|
||||
@@ -409,3 +409,23 @@ def test_pane_handles_cross_user_409() -> None:
|
||||
assert "r.status === 409" in body
|
||||
assert 'status: "cross_user_interjection"' in body
|
||||
assert 'data.status === "cross_user_interjection"' in body
|
||||
|
||||
|
||||
def test_sync_approval_state_prunes_orphan_cycles() -> None:
|
||||
"""``_syncApprovalState`` prunes cycles whose block elements are no longer
|
||||
in the living DOM (``.isConnected === false``). This covers the rare case
|
||||
where an ``approve_request`` event is processed between a DOM wipe
|
||||
(``clear_ui`` / ``replay_truncated`` / ``replaceChildren``) and the
|
||||
refetch-restore — the cycle card lives in a detached subtree, the matching
|
||||
``approval_resolved`` never arrives, and the send button stays disabled
|
||||
forever without this guard. The pin guards against a future refactor that
|
||||
drops the orphan prune but doesn't otherwise break ``_syncApprovalState``."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
fn_start = body.index("_syncApprovalState() {")
|
||||
assert "entry.blockEls && !entry.blockEls.some((el) => el.isConnected)" in body, (
|
||||
"orphan pruning must check .isConnected on block elements"
|
||||
)
|
||||
tail = body[fn_start : body.index("_oldestCycleId()", fn_start)]
|
||||
assert "this.approvalCycles.delete(cid);" in tail, (
|
||||
"orphan pruning must delete the cycle from the Map"
|
||||
)
|
||||
|
||||
@@ -11,12 +11,16 @@ this pins the behaviour the old ``_anthropic`` ``pc_tool_ids`` /
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from turnstone.core.lowering import (
|
||||
CANCELLED_TOOL_RESULT,
|
||||
_find_orphaned_tool_calls,
|
||||
repair_wire_messages,
|
||||
sanitize_tool_call_arguments,
|
||||
tool_args_preview,
|
||||
wire_valid_arguments,
|
||||
)
|
||||
|
||||
|
||||
@@ -180,3 +184,159 @@ def test_repair_does_not_mutate_input() -> None:
|
||||
repair_wire_messages(msgs)
|
||||
assert len(msgs) == original_len # caller's list untouched
|
||||
assert "tool_calls" in msgs[0]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# wire_valid_arguments — the shared "is this renderable" predicate
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_wire_valid_arguments_accepts_json_objects() -> None:
|
||||
assert wire_valid_arguments("{}") is True
|
||||
assert wire_valid_arguments('{"command": "ls -la"}') is True
|
||||
assert wire_valid_arguments(' { "a": 1 }\n') is True # surrounding whitespace ok
|
||||
|
||||
|
||||
def test_wire_valid_arguments_rejects_unrenderable() -> None:
|
||||
assert wire_valid_arguments('{"command": "cat /va') is False # unterminated (the incident)
|
||||
assert wire_valid_arguments("") is False # empty (no-arg call) — json.loads raises
|
||||
assert wire_valid_arguments("[]") is False # array, not object
|
||||
assert wire_valid_arguments("5") is False # bare scalar
|
||||
assert wire_valid_arguments('"hi"') is False # bare string
|
||||
assert wire_valid_arguments(None) is False # missing
|
||||
assert wire_valid_arguments({"a": 1}) is False # raw dict — not a string on the wire
|
||||
|
||||
|
||||
def test_wire_valid_arguments_totals_on_deeply_nested_json() -> None:
|
||||
# Deeply-nested JSON makes json.loads raise RecursionError (not a ValueError);
|
||||
# the predicate must return False, not propagate and crash the send.
|
||||
deep = "[" * 5000 + "]" * 5000
|
||||
assert wire_valid_arguments(deep) is False
|
||||
|
||||
|
||||
def test_tool_args_preview_stringifies_and_caps() -> None:
|
||||
assert tool_args_preview("x" * 500) == "x" * 120
|
||||
assert tool_args_preview(None) == "None"
|
||||
assert tool_args_preview({"a": 1}) == "{'a': 1}"
|
||||
|
||||
|
||||
def test_tool_args_preview_redacts_credentials() -> None:
|
||||
# Secrets in tool args (bash commands, tokens) must not reach logs — the preview
|
||||
# runs output_guard.redact_credentials over the full value first (PR #778 review).
|
||||
out = tool_args_preview('{"command": "aws configure set key AKIAIOSFODNN7EXAMPLE"}')
|
||||
assert "AKIAIOSFODNN7EXAMPLE" not in out
|
||||
assert "[REDACTED:api_key]" in out
|
||||
|
||||
|
||||
def test_tool_args_preview_is_single_line() -> None:
|
||||
# Control chars (LF/CR/TAB) collapse to spaces so the preview stays one log line.
|
||||
raw = "line1" + chr(10) + "line2" + chr(13) + "end" + chr(9) + "z"
|
||||
out = tool_args_preview(raw)
|
||||
assert chr(10) not in out and chr(13) not in out and chr(9) not in out
|
||||
assert "line1" in out and "end" in out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# sanitize_tool_call_arguments — the legalize pass
|
||||
# --------------------------------------------------------------------------- #
|
||||
def _call(call_id: str, arguments: Any, name: str = "bash") -> dict[str, Any]:
|
||||
return {"id": call_id, "type": "function", "function": {"name": name, "arguments": arguments}}
|
||||
|
||||
|
||||
def _assistant_calls(*calls: dict[str, Any]) -> dict[str, Any]:
|
||||
return {"role": "assistant", "content": "", "tool_calls": list(calls)}
|
||||
|
||||
|
||||
def test_sanitize_identity_when_all_valid() -> None:
|
||||
msgs = [_assistant_calls(_call("c1", "{}"), _call("c2", '{"a": 1}')), _tool("c1"), _tool("c2")]
|
||||
# Every arguments already a JSON object → same object returned (allocation-free).
|
||||
assert sanitize_tool_call_arguments(msgs) is msgs
|
||||
|
||||
|
||||
def test_sanitize_identity_when_no_tool_calls() -> None:
|
||||
msgs = [{"role": "user", "content": "hi"}, {"role": "assistant", "content": "yo"}]
|
||||
assert sanitize_tool_call_arguments(msgs) is msgs
|
||||
|
||||
|
||||
def test_sanitize_legalizes_unterminated_arguments() -> None:
|
||||
# The production incident: deepseek-v4-flash emitted an unterminated args string
|
||||
# with a non-``length`` finish reason, so it was committed and replayed verbatim.
|
||||
msgs = [_assistant_calls(_call("c1", '{"command": "cat /va')), _tool("c1", "retry")]
|
||||
out = sanitize_tool_call_arguments(msgs)
|
||||
assert out is not msgs # copied on repair
|
||||
assert out[0]["tool_calls"][0]["function"]["arguments"] == "{}"
|
||||
assert json.loads(out[0]["tool_calls"][0]["function"]["arguments"]) == {}
|
||||
|
||||
|
||||
def test_sanitize_legalizes_empty_arguments() -> None:
|
||||
# A no-arg tool call sends ``""``; json.loads("") raises, so deepseek_v4 would 400.
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", ""))])
|
||||
assert out[0]["tool_calls"][0]["function"]["arguments"] == "{}"
|
||||
|
||||
|
||||
def test_sanitize_legalizes_non_object_json() -> None:
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", "[]"), _call("c2", "5"))])
|
||||
assert [tc["function"]["arguments"] for tc in out[0]["tool_calls"]] == ["{}", "{}"]
|
||||
|
||||
|
||||
def test_sanitize_serializes_raw_dict_arguments() -> None:
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", {"command": "ls"}))])
|
||||
got = out[0]["tool_calls"][0]["function"]["arguments"]
|
||||
assert isinstance(got, str) and json.loads(got) == {"command": "ls"}
|
||||
|
||||
|
||||
def test_sanitize_falls_back_when_dict_not_serializable() -> None:
|
||||
# Defensive branch: a dict arguments carrying a non-JSON-encodable value
|
||||
# (a set) makes json.dumps raise TypeError — it collapses to "{}", not a crash.
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", {"x": {1, 2, 3}}))])
|
||||
assert out[0]["tool_calls"][0]["function"]["arguments"] == "{}"
|
||||
|
||||
|
||||
def test_sanitize_touches_only_the_offending_call() -> None:
|
||||
good = _call("c1", '{"a": 1}')
|
||||
bad = _call("c2", "{oops")
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(good, bad)])
|
||||
# Valid sibling preserved by identity; only the bad call is rebuilt.
|
||||
assert out[0]["tool_calls"][0] is good
|
||||
assert out[0]["tool_calls"][1]["function"]["arguments"] == "{}"
|
||||
|
||||
|
||||
def test_sanitize_does_not_mutate_input() -> None:
|
||||
raw = '{"command": "cat /va'
|
||||
bad = _call("c1", raw)
|
||||
msgs = [_assistant_calls(bad)]
|
||||
sanitize_tool_call_arguments(msgs)
|
||||
assert bad["function"]["arguments"] == raw # caller's dict untouched
|
||||
assert msgs[0]["tool_calls"][0] is bad
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# legalize ∘ repair — the two send-time validity passes compose
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_legalize_then_repair_answered_call() -> None:
|
||||
# Malformed-but-answered (the poison-pill shape): args legalized, no orphan added.
|
||||
msgs = [_assistant_calls(_call("c1", "{bad")), _tool("c1", "retry with valid JSON")]
|
||||
out = repair_wire_messages(sanitize_tool_call_arguments(msgs))
|
||||
assert [m["role"] for m in out] == ["assistant", "tool"]
|
||||
assert json.loads(out[0]["tool_calls"][0]["function"]["arguments"]) == {}
|
||||
|
||||
|
||||
def test_legalize_then_repair_orphaned_call() -> None:
|
||||
# Malformed AND unanswered: legalized args + a synthesized cancellation result.
|
||||
msgs = [_assistant_calls(_call("c1", "{bad"))]
|
||||
out = repair_wire_messages(sanitize_tool_call_arguments(msgs))
|
||||
assert [m["role"] for m in out] == ["assistant", "tool"]
|
||||
assert json.loads(out[0]["tool_calls"][0]["function"]["arguments"]) == {}
|
||||
assert out[1]["content"] == CANCELLED_TOOL_RESULT
|
||||
|
||||
|
||||
def test_pipeline_every_emitted_arguments_is_a_json_object() -> None:
|
||||
# The end-state invariant a strict renderer relies on.
|
||||
msgs = [
|
||||
_assistant_calls(_call("c1", ""), _call("c2", "{oops"), _call("c3", '{"ok": true}')),
|
||||
_tool("c1"),
|
||||
_tool("c2"),
|
||||
_tool("c3"),
|
||||
]
|
||||
out = repair_wire_messages(sanitize_tool_call_arguments(msgs))
|
||||
for m in out:
|
||||
for tc in m.get("tool_calls", []):
|
||||
assert isinstance(json.loads(tc["function"]["arguments"]), dict)
|
||||
|
||||
+1145
-73
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,207 @@
|
||||
"""Live flaky-server smoke test: SIGKILL-flap a real MCP server, no CPU spin.
|
||||
|
||||
End-to-end regression for the flaky-server 100%-CPU incident: a real
|
||||
streamable-http MCP server (FastMCP, subprocess) is SIGKILLed and restarted
|
||||
several times underneath a real ``MCPClientManager`` with the health loop
|
||||
running on compressed timings. The production failure signature was armed
|
||||
anyio ``CancelScope``s — each one re-delivers cancellation via ``call_soon``
|
||||
every event-loop iteration, forever (~10^5+ callbacks/s), one more per flap
|
||||
cycle — so the pass criterion is structural: after the flaps settle, ZERO
|
||||
armed scopes exist on the mcp-loop, exactly one transport owner is alive, the
|
||||
health loop still runs, and a real tool call round-trips.
|
||||
|
||||
Self-contained (spawns its own server; no LLM backend, no network beyond
|
||||
127.0.0.1) — deliberately NOT marked ``live``. Wall clock ~10-15s.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import gc
|
||||
import signal
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import textwrap
|
||||
import time
|
||||
from typing import TYPE_CHECKING
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
SERVER_SRC = textwrap.dedent(
|
||||
'''
|
||||
"""Healthy streamable-http MCP server; the test SIGKILLs it to flap."""
|
||||
import sys
|
||||
|
||||
from mcp.server.fastmcp import FastMCP
|
||||
|
||||
port = int(sys.argv[1])
|
||||
mcp = FastMCP("flaky-victim", host="127.0.0.1", port=port)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def ping_me(x: int) -> int:
|
||||
"""Return x + 1."""
|
||||
return x + 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
mcp.run(transport="streamable-http")
|
||||
'''
|
||||
).lstrip()
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return int(s.getsockname()[1])
|
||||
|
||||
|
||||
def _wait_tcp_ready(port: int, timeout: float) -> bool:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
def _wait_session_live(mgr: MCPClientManager, name: str, timeout: float) -> bool:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
state = mgr._static_servers.get(name)
|
||||
if state is not None and state.session is not None:
|
||||
return True
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
async def _armed_scope_count() -> int:
|
||||
"""Armed scopes hosted on THIS (the mcp) loop — mirrors the production
|
||||
disarm sweep's scoping, and keeps an unrelated scope on another loop that
|
||||
is momentarily mid-cancellation from flaking the assertion."""
|
||||
import asyncio as _asyncio
|
||||
|
||||
from anyio._backends._asyncio import CancelScope
|
||||
|
||||
this_loop = _asyncio.get_running_loop()
|
||||
armed = 0
|
||||
for obj in gc.get_objects():
|
||||
if not isinstance(obj, CancelScope):
|
||||
continue
|
||||
if getattr(obj, "_cancel_handle", None) is None:
|
||||
continue
|
||||
host = getattr(obj, "_host_task", None)
|
||||
if host is not None and host.get_loop() is not this_loop:
|
||||
continue
|
||||
armed += 1
|
||||
return armed
|
||||
|
||||
|
||||
async def _live_owner_count() -> int:
|
||||
return sum(
|
||||
1
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-transport-owner:") and not t.done()
|
||||
)
|
||||
|
||||
|
||||
class TestFlakyServerNoSpin:
|
||||
def test_sigkill_flap_cycle_no_armed_scopes_and_recovers(
|
||||
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# The subprocess runs sys.executable, so importability HERE is a
|
||||
# faithful proxy for the server side. Environment gaps skip, not fail.
|
||||
pytest.importorskip("mcp.server.fastmcp")
|
||||
script = tmp_path / "flaky_srv.py"
|
||||
script.write_text(SERVER_SRC)
|
||||
port = _free_port()
|
||||
|
||||
# Compress recovery timings so 3 flap cycles fit a unit-test budget.
|
||||
monkeypatch.setattr(MCPClientManager, "_CONNECT_TIMEOUT", 3)
|
||||
monkeypatch.setattr(MCPClientManager, "_TCP_PROBE_TIMEOUT", 1)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_ATTEMPT_TIMEOUT_S", 5.0)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_CALLER_TIMEOUT_S", 6.0)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_BASE_S", 0.2)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_MAX_S", 0.8)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_HEALTH_PING_TIMEOUT_S", 1.5)
|
||||
|
||||
def _spawn_server(*, initial: bool = False) -> subprocess.Popen[bytes]:
|
||||
proc = subprocess.Popen(
|
||||
[sys.executable, str(script), str(port)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
if not _wait_tcp_ready(port, 10.0):
|
||||
proc.kill()
|
||||
proc.wait(timeout=5)
|
||||
if initial:
|
||||
# Environment gap (loaded CI runner, sandboxed sockets) —
|
||||
# not a regression signal. Mid-test respawns DO fail: the
|
||||
# server already bound once, so a vanishing rebind is real.
|
||||
pytest.skip("flaky-server subprocess did not come up")
|
||||
raise AssertionError("flaky server did not come back up mid-test")
|
||||
return proc
|
||||
|
||||
proc: subprocess.Popen[bytes] | None = None
|
||||
mgr: MCPClientManager | None = None
|
||||
try:
|
||||
proc = _spawn_server(initial=True)
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.load_config",
|
||||
return_value={"static_health_check_seconds": 0.4},
|
||||
):
|
||||
mgr = MCPClientManager(
|
||||
{"flaky": {"type": "http", "url": f"http://127.0.0.1:{port}/mcp"}}
|
||||
)
|
||||
mgr.start()
|
||||
assert _wait_session_live(mgr, "flaky", 8.0), "initial connect failed"
|
||||
|
||||
for _cycle in range(3):
|
||||
proc.send_signal(signal.SIGKILL)
|
||||
proc.wait()
|
||||
time.sleep(0.6) # dead window: health loop sees the corpse
|
||||
proc = _spawn_server()
|
||||
assert _wait_session_live(mgr, "flaky", 10.0), (
|
||||
f"no reconnect after flap cycle {_cycle}"
|
||||
)
|
||||
|
||||
# Let in-flight teardown/backoff machinery fully settle.
|
||||
time.sleep(1.5)
|
||||
|
||||
assert mgr._loop is not None
|
||||
armed = asyncio.run_coroutine_threadsafe(_armed_scope_count(), mgr._loop).result(
|
||||
timeout=10
|
||||
)
|
||||
owners = asyncio.run_coroutine_threadsafe(_live_owner_count(), mgr._loop).result(
|
||||
timeout=10
|
||||
)
|
||||
health = mgr._static_health_task
|
||||
|
||||
# The production failure signature: one armed scope per flap cycle.
|
||||
assert armed == 0, f"{armed} armed cancel scope(s) — the CPU-spin signature"
|
||||
# Exactly the current session's owner is alive; the flapped ones
|
||||
# all unwound instead of leaking.
|
||||
assert owners == 1
|
||||
# The recovery machinery itself survived every flap.
|
||||
assert health is not None and not health.done()
|
||||
# The structural fix did the work — the disarm backstop never ran.
|
||||
assert mgr._last_scope_disarm == 0.0
|
||||
|
||||
# And the recovered session actually dispatches.
|
||||
out = mgr.call_tool_sync("mcp__flaky__ping_me", {"x": 41}, timeout=10)
|
||||
assert "42" in out
|
||||
finally:
|
||||
if mgr is not None:
|
||||
mgr.shutdown()
|
||||
if proc is not None:
|
||||
proc.send_signal(signal.SIGKILL)
|
||||
proc.wait(timeout=5)
|
||||
@@ -630,8 +630,6 @@ class TestCallback:
|
||||
server_name="srv-oauth",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id="ws-1",
|
||||
last_tool_call_id="tool-1",
|
||||
now_iso="2026-05-11T12:00:00",
|
||||
)
|
||||
storage.upsert_mcp_pending_consent(
|
||||
@@ -639,8 +637,6 @@ class TestCallback:
|
||||
server_name="srv-oauth",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T12:00:00",
|
||||
)
|
||||
token_store = _make_token_store(storage)
|
||||
|
||||
@@ -408,6 +408,97 @@ class TestRefreshFailureClassification:
|
||||
assert ("user-1", "srv-oauth") not in state.mcp_oauth_refresh_locks
|
||||
|
||||
|
||||
class TestObserveOnlyLookup:
|
||||
"""``revoke_on_failure=False`` (the background token-freshness sweep): still
|
||||
refresh a healthy token, but on failure NEVER delete a token or mutate the
|
||||
shared streak — a timer must not destroy consent or move a foreground user's
|
||||
revoke threshold. A permanent rejection surfaces as ``refresh_failed`` with
|
||||
the row INTACT; an ambiguous one as transient with the streak untouched."""
|
||||
|
||||
def _lookup(self, state: SimpleNamespace) -> Any:
|
||||
from turnstone.core.mcp_oauth import get_user_access_token_classified
|
||||
|
||||
async def _run() -> Any:
|
||||
with _public_addr_patch():
|
||||
return await get_user_access_token_classified(
|
||||
app_state=state,
|
||||
user_id="user-1",
|
||||
server_name="srv-oauth",
|
||||
force_refresh=True,
|
||||
revoke_on_failure=False,
|
||||
)
|
||||
|
||||
return asyncio.run(_run())
|
||||
|
||||
def test_permanent_invalid_grant_does_not_revoke(self, storage: SQLiteBackend) -> None:
|
||||
"""The exact contrast to ``test_permanent_invalid_grant_revokes``: same
|
||||
dead-grant signal, but observe-only leaves the row for the lazy path."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
client.post = AsyncMock(return_value=_mk_response(400, {"error": "invalid_grant"}))
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000)
|
||||
|
||||
result = self._lookup(state)
|
||||
|
||||
assert result.kind == "refresh_failed"
|
||||
assert state.mcp_token_store.get_user_token("user-1", "srv-oauth") is not None
|
||||
|
||||
def test_ambiguous_does_not_touch_shared_streak(self, storage: SQLiteBackend) -> None:
|
||||
"""Repeated observe-mode ambiguous failures never bump the shared
|
||||
ambiguous_streak, so a later foreground dispatch is not pushed over the
|
||||
escalation edge by background activity (the finding this guards)."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
client.post = AsyncMock(return_value=_mk_response(400, None))
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000)
|
||||
|
||||
with patch("turnstone.core.mcp_oauth._AMBIGUOUS_ESCALATION_THRESHOLD", 2):
|
||||
for _ in range(5):
|
||||
assert self._lookup(state).kind == "refresh_failed_transient"
|
||||
|
||||
backoff = getattr(state, "mcp_oauth_refresh_backoff", {})
|
||||
entry = backoff.get(("user-1", "srv-oauth"))
|
||||
assert entry is None or entry.ambiguous_streak == 0
|
||||
assert state.mcp_token_store.get_user_token("user-1", "srv-oauth") is not None
|
||||
|
||||
def test_expired_no_refresh_does_not_revoke(self, storage: SQLiteBackend) -> None:
|
||||
"""An expired token with no refresh token surfaces as a dead grant but is
|
||||
NOT deleted on the observe path."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000, refresh=None)
|
||||
|
||||
result = self._lookup(state)
|
||||
|
||||
assert result.kind == "refresh_failed"
|
||||
assert state.mcp_token_store.get_user_token("user-1", "srv-oauth") is not None
|
||||
|
||||
def test_healthy_token_still_refreshes(self, storage: SQLiteBackend) -> None:
|
||||
"""Observe mode is not read-only: a near-expiry token is still refreshed
|
||||
(only the destructive failure paths change)."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
client.post = AsyncMock(
|
||||
return_value=_mk_response(
|
||||
200, {"access_token": "fresh-bbb", "expires_in": 3600, "token_type": "Bearer"}
|
||||
)
|
||||
)
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000)
|
||||
|
||||
result = self._lookup(state)
|
||||
|
||||
assert result.kind == "token"
|
||||
assert result.token == "fresh-bbb"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Happy paths
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -108,8 +108,6 @@ def _seed_pending(
|
||||
server_name=server_name,
|
||||
error_code=error_code,
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=now_iso,
|
||||
)
|
||||
|
||||
|
||||
@@ -24,8 +24,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required="read write",
|
||||
last_ws_id="ws-1",
|
||||
last_tool_call_id="tool-1",
|
||||
now_iso=_iso(),
|
||||
)
|
||||
rows = backend.list_mcp_pending_consent_by_user("user-a")
|
||||
@@ -35,8 +33,6 @@ class TestUpsertAndList:
|
||||
assert r["server_name"] == "srv-x"
|
||||
assert r["error_code"] == "mcp_consent_required"
|
||||
assert r["scopes_required"] == "read write"
|
||||
assert r["last_ws_id"] == "ws-1"
|
||||
assert r["last_tool_call_id"] == "tool-1"
|
||||
assert r["occurrence_count"] == 1
|
||||
assert r["first_seen_at"] == r["last_seen_at"]
|
||||
|
||||
@@ -46,8 +42,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T12:00:00",
|
||||
)
|
||||
backend.upsert_mcp_pending_consent(
|
||||
@@ -55,8 +49,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_insufficient_scope",
|
||||
scopes_required="read",
|
||||
last_ws_id="ws-2",
|
||||
last_tool_call_id="tool-2",
|
||||
now_iso="2026-05-11T13:00:00",
|
||||
)
|
||||
rows = backend.list_mcp_pending_consent_by_user("user-a")
|
||||
@@ -66,8 +58,6 @@ class TestUpsertAndList:
|
||||
assert r["occurrence_count"] == 2
|
||||
assert r["error_code"] == "mcp_insufficient_scope"
|
||||
assert r["scopes_required"] == "read"
|
||||
assert r["last_ws_id"] == "ws-2"
|
||||
assert r["last_tool_call_id"] == "tool-2"
|
||||
assert r["last_seen_at"] == "2026-05-11T13:00:00"
|
||||
# first_seen_at preserved — that's the load-bearing audit value.
|
||||
assert r["first_seen_at"] == "2026-05-11T12:00:00"
|
||||
@@ -78,8 +68,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-old",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T10:00:00",
|
||||
)
|
||||
backend.upsert_mcp_pending_consent(
|
||||
@@ -87,8 +75,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-new",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T11:00:00",
|
||||
)
|
||||
rows = backend.list_mcp_pending_consent_by_user("user-a")
|
||||
@@ -100,8 +86,6 @@ class TestUpsertAndList:
|
||||
server_name="srv",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
assert backend.list_mcp_pending_consent_by_user("user-b") == []
|
||||
@@ -114,8 +98,6 @@ class TestDelete:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
assert backend.delete_mcp_pending_consent("user-a", "srv-x") is True
|
||||
@@ -133,8 +115,6 @@ class TestDelete:
|
||||
server_name=name,
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
# Cross-user row that must NOT be touched.
|
||||
@@ -143,8 +123,6 @@ class TestDelete:
|
||||
server_name="srv-z",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
assert backend.delete_all_mcp_pending_consent_by_user("user-a") == 3
|
||||
|
||||
@@ -1033,20 +1033,22 @@ class TestStaticPathUnchanged:
|
||||
|
||||
from turnstone.core import mcp_client
|
||||
|
||||
source = inspect.getsource(mcp_client.MCPClientManager._connect_one)
|
||||
# The static path's streamablehttp_client call site lives in the
|
||||
# transport owner task (``_static_transport_owner``); ``_connect_one``
|
||||
# is a per-name-lock wrapper and ``_connect_one_locked`` only waits on
|
||||
# the owner's readiness.
|
||||
source = inspect.getsource(mcp_client.MCPClientManager._static_transport_owner)
|
||||
|
||||
# The static path's streamablehttp_client invocation should NOT
|
||||
# mention ``httpx_client_factory``. Pool path keeps it.
|
||||
# Find the streamablehttp_client(...) call inside _connect_one.
|
||||
assert "streamablehttp_client" in source
|
||||
# The call site in _connect_one is bare — no factory keyword.
|
||||
# We grep by line: the factory keyword must not appear in the
|
||||
# static-path source.
|
||||
# The call site in the owner is bare — no factory keyword. We grep by
|
||||
# line: the factory keyword must not appear in the static-path source.
|
||||
for line in source.splitlines():
|
||||
if "httpx_client_factory" in line:
|
||||
pytest.fail(
|
||||
"_connect_one (static path) passes httpx_client_factory to "
|
||||
"streamablehttp_client; hard invariant 1 violated."
|
||||
"_static_transport_owner (static path) passes httpx_client_factory "
|
||||
"to streamablehttp_client; hard invariant 1 violated."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,478 @@
|
||||
"""Pool transport owner-task lifecycle + anyio cancel-scope regressions.
|
||||
|
||||
The pool (auth_type=oauth_user) sibling of ``test_mcp_transport_owner.py``.
|
||||
Each ``(user, server)`` pool entry's transport + ``ClientSession`` cms are now
|
||||
entered, parked, and exited by ONE long-lived owner task
|
||||
(``_pool_transport_owner``) with a one-cancel close protocol, so a cancel scope
|
||||
whose host task has finished can never be left re-delivering cancellation in a
|
||||
``call_soon`` loop (the SDK #2147 100%-CPU spin). These fast mock-transport
|
||||
tests pin that protocol for the pool path; the real-server integration coverage
|
||||
lives in ``test_mcp_pool_auth_integration.py``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import threading
|
||||
import time
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager, PoolEntryState, _AuthCapture
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def running_loop_mgr():
|
||||
"""Background-loop fixture matching the pool-path test convention.
|
||||
|
||||
Teardown drains the eviction / sweep / health tasks AND any parked pool
|
||||
transport owner a successful connect left installed — the conftest fails
|
||||
leaked threads and an undrained owner is destroyed pending at GC.
|
||||
"""
|
||||
cfg: dict[str, Any] = {}
|
||||
mgr = MCPClientManager(cfg)
|
||||
loop = asyncio.new_event_loop()
|
||||
thread = threading.Thread(target=loop.run_forever, daemon=True, name="mcp-pool-owner-test-loop")
|
||||
thread.start()
|
||||
mgr._loop = loop
|
||||
try:
|
||||
yield mgr, loop, thread
|
||||
finally:
|
||||
|
||||
async def _drain(m: MCPClientManager) -> None:
|
||||
for attr in (
|
||||
"_user_pool_eviction_task",
|
||||
"_user_token_sweep_task",
|
||||
"_static_health_task",
|
||||
):
|
||||
task = getattr(m, attr)
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
await asyncio.gather(task, return_exceptions=True)
|
||||
setattr(m, attr, None)
|
||||
for entry in list(m._user_pool_entries.values()):
|
||||
owner = entry.owner_task
|
||||
if owner is not None and not owner.done():
|
||||
if entry.close_requested is not None:
|
||||
entry.close_requested.set()
|
||||
owner.cancel()
|
||||
await asyncio.gather(owner, return_exceptions=True)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
asyncio.run_coroutine_threadsafe(_drain(mgr), loop).result(timeout=5)
|
||||
loop.call_soon_threadsafe(loop.stop)
|
||||
thread.join(timeout=5)
|
||||
if not thread.is_alive():
|
||||
loop.close()
|
||||
|
||||
|
||||
def _run(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 5.0) -> Any:
|
||||
return asyncio.run_coroutine_threadsafe(coro, loop).result(timeout=timeout)
|
||||
|
||||
|
||||
def _http_cfg() -> dict[str, Any]:
|
||||
return {"type": "streamable-http", "url": "https://mcp.example.com/mcp", "headers": {}}
|
||||
|
||||
|
||||
def _make_pool_session_mock() -> AsyncMock:
|
||||
"""A ClientSession-shaped mock good enough for pool connect + discovery."""
|
||||
session = AsyncMock()
|
||||
session.initialize = AsyncMock()
|
||||
# None caps → resources/prompts discovery is skipped; only list_tools runs.
|
||||
session.get_server_capabilities = MagicMock(return_value=None)
|
||||
session.list_tools = AsyncMock(return_value=MagicMock(tools=[]))
|
||||
return session
|
||||
|
||||
|
||||
def _fake_transport_and_session(patches: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Build fake streamable-http transport + ClientSession cms.
|
||||
|
||||
Records enter/exit events and captures the kwargs that reach
|
||||
``streamablehttp_client`` (so the bearer-header / factory contract is
|
||||
observable).
|
||||
"""
|
||||
events: list[str] = []
|
||||
captured_kwargs: dict[str, Any] = {}
|
||||
session = _make_pool_session_mock()
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_streamablehttp_client(**kwargs: Any):
|
||||
captured_kwargs.clear()
|
||||
captured_kwargs.update(kwargs)
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
yield (AsyncMock(), AsyncMock(), lambda: None)
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_client_session_cm():
|
||||
events.append("session_enter")
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
events.append("session_exit")
|
||||
|
||||
def fake_client_session(_read: Any, _write: Any, message_handler: Any = None):
|
||||
return fake_client_session_cm()
|
||||
|
||||
patches["streamablehttp_client"] = fake_streamablehttp_client
|
||||
patches["ClientSession"] = fake_client_session
|
||||
return {"events": events, "session": session, "kwargs": captured_kwargs}
|
||||
|
||||
|
||||
async def _connect_under_lock(
|
||||
mgr: MCPClientManager, key: tuple[str, str], cfg: dict[str, Any], **kw: Any
|
||||
) -> PoolEntryState:
|
||||
"""Drive ``_connect_one_pool`` the way production does — under open_lock."""
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
async with entry.open_lock:
|
||||
return await mgr._connect_one_pool(key, cfg, "tok-aaa", **kw)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Owner lifecycle
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestPoolTransportOwnerLifecycle:
|
||||
def test_connect_installs_owner_and_teardown_closes_gracefully(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client", patches["streamablehttp_client"]
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
entry = _run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
assert entry.session is fake["session"]
|
||||
owner = entry.owner_task
|
||||
assert owner is not None and not owner.done()
|
||||
assert entry.close_requested is not None
|
||||
assert fake["events"] == ["transport_enter", "session_enter"]
|
||||
|
||||
_run(loop, mgr._teardown_pool_entry(key))
|
||||
|
||||
# Graceful close: the parked owner exits via the event — no cancel —
|
||||
# and unwinds BOTH cms in-task, inner-out (session before transport).
|
||||
assert owner.done() and not owner.cancelled()
|
||||
assert fake["events"] == [
|
||||
"transport_enter",
|
||||
"session_enter",
|
||||
"session_exit",
|
||||
"transport_exit",
|
||||
]
|
||||
assert entry.session is None
|
||||
assert entry.owner_task is None
|
||||
assert entry.close_requested is None
|
||||
# The entry itself is NOT popped — teardown leaves map/catalog cleanup
|
||||
# to callers.
|
||||
assert key in mgr._user_pool_entries
|
||||
|
||||
def test_owner_death_during_discovery_fails_fast(self, running_loop_mgr) -> None:
|
||||
"""The owner-died branch of ``_await_owner_discovery`` — the reason the
|
||||
helper exists: discovery runs in the caller while the transport is
|
||||
hosted by the owner, so a transport collapse mid-discovery cancels the
|
||||
OWNER and a bare await on the response stream would hang until the 30s
|
||||
phase timeout. The race must convert that into a PROMPT
|
||||
``ConnectionError``, reap the parked discovery future, and leave the
|
||||
entry torn down."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
discovery_parked = asyncio.Event()
|
||||
|
||||
async def _parked_list_tools() -> Any:
|
||||
discovery_parked.set()
|
||||
await asyncio.sleep(3600) # the transport never answers
|
||||
|
||||
fake["session"].list_tools = AsyncMock(side_effect=_parked_list_tools)
|
||||
|
||||
async def _drive() -> tuple[float, BaseException | None]:
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
|
||||
async def _collapse_owner_when_parked() -> None:
|
||||
await discovery_parked.wait()
|
||||
owner = entry.owner_task # installed before discovery begins
|
||||
assert owner is not None
|
||||
# The transport task group collapsing under live discovery
|
||||
# (e.g. an upstream 401) surfaces as the owner being cancelled.
|
||||
owner.cancel()
|
||||
|
||||
collapser = asyncio.create_task(_collapse_owner_when_parked())
|
||||
t0 = asyncio.get_running_loop().time()
|
||||
exc: BaseException | None = None
|
||||
try:
|
||||
async with entry.open_lock:
|
||||
await mgr._connect_one_pool(key, _http_cfg(), "tok-aaa")
|
||||
except Exception as e:
|
||||
# The expected ConnectionError; anything else (a cancel leak,
|
||||
# an interpreter exit) propagates and fails the test loudly.
|
||||
exc = e
|
||||
_ = await collapser # synchronization point; failures propagate
|
||||
return asyncio.get_running_loop().time() - t0, exc
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client", patches["streamablehttp_client"]
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
elapsed, exc = _run(loop, _drive(), timeout=15)
|
||||
|
||||
assert isinstance(exc, ConnectionError)
|
||||
assert "died during discovery" in str(exc)
|
||||
assert elapsed < 5.0 # prompt fail — not the 30s phase timeout
|
||||
entry = mgr._user_pool_entries[key]
|
||||
assert entry.session is None # discovery-failure teardown ran
|
||||
assert entry.owner_task is None
|
||||
|
||||
def test_cancelled_discovery_future_converts_to_connection_error(
|
||||
self, running_loop_mgr
|
||||
) -> None:
|
||||
"""A discovery future that completes CANCELLED without this race's own
|
||||
reap (an SDK-internal cancellation shape) is the transport-failure
|
||||
class, not the caller's cancellation — ``_await_owner_discovery`` must
|
||||
surface it as ``ConnectionError``, never a bare ``CancelledError`` the
|
||||
caller would misread as its own cancel."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
async def _drive() -> BaseException | None:
|
||||
parked = asyncio.Event()
|
||||
|
||||
async def _parked_owner() -> None:
|
||||
await parked.wait()
|
||||
|
||||
owner = asyncio.create_task(_parked_owner())
|
||||
await asyncio.sleep(0)
|
||||
|
||||
async def _self_cancelling_discovery() -> Any:
|
||||
# A coroutine raising CancelledError makes its wrapping task
|
||||
# complete CANCELLED — the shape of an SDK-internal cancel.
|
||||
raise asyncio.CancelledError
|
||||
|
||||
exc: BaseException | None = None
|
||||
try:
|
||||
await mgr._await_owner_discovery(owner, _self_cancelling_discovery())
|
||||
except (Exception, asyncio.CancelledError) as e:
|
||||
# Exception covers the expected ConnectionError; CancelledError
|
||||
# covers the exact regression this test guards (the bare cancel
|
||||
# leaking through instead of being converted).
|
||||
exc = e
|
||||
parked.set()
|
||||
_ = await owner # synchronization point; failures propagate
|
||||
return exc
|
||||
|
||||
exc = _run(loop, _drive())
|
||||
assert isinstance(exc, ConnectionError)
|
||||
assert "cancelled by transport failure" in str(exc)
|
||||
|
||||
def test_teardown_single_cancel_escalation(self, running_loop_mgr) -> None:
|
||||
"""A parked owner whose in-task unwind stalls past the graceful window
|
||||
gets EXACTLY ONE cancel — never a second (a second abandons an anyio
|
||||
scope exit mid-flight and mints the zombie the protocol prevents)."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._OWNER_CLOSE_GRACE_S = 0.1
|
||||
mgr._OWNER_CANCEL_GRACE_S = 1.0
|
||||
|
||||
events: list[str] = []
|
||||
cancels = {"n": 0}
|
||||
session = _make_pool_session_mock()
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_streamablehttp_client(**_kwargs: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
yield (AsyncMock(), AsyncMock(), lambda: None)
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_session_cm():
|
||||
events.append("session_enter")
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
# Stall the graceful unwind so teardown must escalate; count
|
||||
# each cancellation that reaches this in-task exit.
|
||||
try:
|
||||
await asyncio.sleep(3600)
|
||||
except asyncio.CancelledError:
|
||||
cancels["n"] += 1
|
||||
raise
|
||||
finally:
|
||||
events.append("session_exit")
|
||||
|
||||
def fake_session(_read: Any, _write: Any, message_handler: Any = None):
|
||||
return fake_session_cm()
|
||||
|
||||
key = ("user-1", "pool-srv")
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.streamablehttp_client", fake_streamablehttp_client),
|
||||
patch("turnstone.core.mcp_client.ClientSession", fake_session),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
entry = _run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
owner = entry.owner_task
|
||||
assert owner is not None
|
||||
_run(loop, mgr._teardown_pool_entry(key), timeout=10)
|
||||
|
||||
assert owner.done() and owner.cancelled()
|
||||
assert cancels["n"] == 1
|
||||
assert events[-1] == "transport_exit"
|
||||
assert entry.session is None and entry.owner_task is None
|
||||
|
||||
def test_owner_death_evicts_session_keeps_entry_and_catalog(self, running_loop_mgr) -> None:
|
||||
"""The transport collapsing under a live session (owner dies with no
|
||||
requested close) evicts the session via the done-callback but leaves the
|
||||
entry AND its discovered catalog in place for the next dispatch."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client", patches["streamablehttp_client"]
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
entry = _run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
owner = entry.owner_task
|
||||
assert owner is not None and entry.session is fake["session"]
|
||||
# Seed a catalog so we can prove the death-callback leaves it alone.
|
||||
entry.tools = [{"name": "mcp__pool-srv__ping", "server": "pool-srv"}]
|
||||
|
||||
# Simulate the transport task group collapsing: the owner gets a
|
||||
# stray cancellation (exactly what anyio's scope delivery does).
|
||||
loop.call_soon_threadsafe(owner.cancel)
|
||||
deadline = time.monotonic() + 5
|
||||
while time.monotonic() < deadline and entry.owner_task is not None:
|
||||
time.sleep(0.02)
|
||||
|
||||
assert owner.done()
|
||||
assert entry.session is None # evicted by the done-callback
|
||||
assert entry.owner_task is None
|
||||
assert key in mgr._user_pool_entries # entry kept
|
||||
assert entry.tools == [
|
||||
{"name": "mcp__pool-srv__ping", "server": "pool-srv"}
|
||||
] # catalog kept
|
||||
# The cms were still unwound in-task despite the stray cancel.
|
||||
assert fake["events"][-2:] == ["session_exit", "transport_exit"]
|
||||
|
||||
def test_caller_cancel_mid_connect_does_not_abandon_cms(self, running_loop_mgr) -> None:
|
||||
"""Cancelling the CONNECTING caller (an eviction giving up, shutdown, a
|
||||
sync boundary timing out) must close the owner via the one-cancel
|
||||
protocol — the transport cm still exits, in-task."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
events: list[str] = []
|
||||
entered = asyncio.Event()
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
@asynccontextmanager
|
||||
async def hanging_streamablehttp_client(**_kwargs: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
entered.set()
|
||||
await asyncio.sleep(3600) # server accepted, then stalled
|
||||
yield (AsyncMock(), AsyncMock(), lambda: None)
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
async def _drive() -> None:
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
|
||||
async def _connect() -> None:
|
||||
async with entry.open_lock:
|
||||
await mgr._connect_one_pool(key, _http_cfg(), "tok-aaa")
|
||||
|
||||
connect = asyncio.create_task(_connect())
|
||||
await asyncio.wait_for(entered.wait(), timeout=5)
|
||||
connect.cancel() # the attempt-timeout / shutdown shape
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
_ = await connect # only the expected cancel is absorbed
|
||||
# The owner must be closed (one cancel) and fully unwound.
|
||||
deadline = asyncio.get_running_loop().time() + 5
|
||||
while asyncio.get_running_loop().time() < deadline:
|
||||
owners = [
|
||||
t
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-pool-owner:") and not t.done()
|
||||
]
|
||||
if not owners:
|
||||
return
|
||||
await asyncio.sleep(0.02)
|
||||
raise AssertionError("owner task still alive after caller cancel")
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.streamablehttp_client", hanging_streamablehttp_client),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
_run(loop, _drive(), timeout=15)
|
||||
|
||||
assert events == ["transport_enter", "transport_exit"]
|
||||
assert mgr._user_pool_entries[key].session is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Client-kwargs contract (bearer header + auth-capture factory)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestPoolOwnerClientKwargs:
|
||||
def test_client_factory_present_iff_auth_capture(self, running_loop_mgr) -> None:
|
||||
"""The caller builds ``client_kwargs`` and the owner passes them to
|
||||
``streamablehttp_client`` verbatim: the auth-capture
|
||||
``httpx_client_factory`` is present exactly when a carrier is supplied,
|
||||
and the per-user bearer always reaches the wire."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
# With auth_capture → factory present.
|
||||
patches_a: dict[str, Any] = {}
|
||||
fake_a = _fake_transport_and_session(patches_a)
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client",
|
||||
patches_a["streamablehttp_client"],
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches_a["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
_run(loop, _connect_under_lock(mgr, key, _http_cfg(), auth_capture=_AuthCapture()))
|
||||
assert "httpx_client_factory" in fake_a["kwargs"]
|
||||
assert fake_a["kwargs"]["headers"]["Authorization"] == "Bearer tok-aaa"
|
||||
_run(loop, mgr._teardown_pool_entry(key))
|
||||
|
||||
# Without auth_capture → factory absent (but bearer still present).
|
||||
patches_b: dict[str, Any] = {}
|
||||
fake_b = _fake_transport_and_session(patches_b)
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client",
|
||||
patches_b["streamablehttp_client"],
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches_b["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
_run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
assert "httpx_client_factory" not in fake_b["kwargs"]
|
||||
assert fake_b["kwargs"]["headers"]["Authorization"] == "Bearer tok-aaa"
|
||||
_run(loop, mgr._teardown_pool_entry(key))
|
||||
@@ -0,0 +1,488 @@
|
||||
"""Transport owner-task lifecycle + anyio cancel-scope zombie regressions.
|
||||
|
||||
Covers the two bugs behind the flaky-MCP-server 100%-CPU incident:
|
||||
|
||||
* Bug 1 — an anyio cancel scope whose host task has finished can never be
|
||||
exited; once cancelled (SDK task-group child death, or a teardown racing a
|
||||
connect) anyio re-delivers cancellation to it via ``call_soon`` every loop
|
||||
iteration, forever. The fix routes every transport cm through a long-lived
|
||||
per-server OWNER task (enter, park, exit — all in one task) with a
|
||||
one-cancel close protocol; these tests pin the protocol's behavior.
|
||||
* Bug 2 — ``BaseExceptionGroup`` (BaseException-derived) escaping
|
||||
``except Exception`` killed ``_connect_all`` before the health/sweep loops
|
||||
were created, silently disabling all autonomous recovery.
|
||||
|
||||
The live end-to-end flap test (real server, SIGKILL cycle) lives in
|
||||
``test_mcp_live_flaky_server.py``; these are fast mock-transport unit tests.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import threading
|
||||
import time
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def running_loop_mgr():
|
||||
"""Background-loop fixture matching the static-path test convention."""
|
||||
cfg: dict[str, Any] = {"srv": {"type": "stdio", "command": "fake-cmd"}}
|
||||
mgr = MCPClientManager(cfg)
|
||||
loop = asyncio.new_event_loop()
|
||||
thread = threading.Thread(target=loop.run_forever, daemon=True, name="mcp-owner-test-loop")
|
||||
thread.start()
|
||||
mgr._loop = loop
|
||||
try:
|
||||
yield mgr, loop, thread
|
||||
finally:
|
||||
|
||||
async def _drain(m: MCPClientManager) -> None:
|
||||
for attr in (
|
||||
"_user_pool_eviction_task",
|
||||
"_user_token_sweep_task",
|
||||
"_static_health_task",
|
||||
):
|
||||
task = getattr(m, attr)
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
await asyncio.gather(task, return_exceptions=True)
|
||||
setattr(m, attr, None)
|
||||
for state in m._static_servers.values():
|
||||
owner = state.owner_task
|
||||
if owner is not None and not owner.done():
|
||||
if state.close_requested is not None:
|
||||
state.close_requested.set()
|
||||
owner.cancel()
|
||||
await asyncio.gather(owner, return_exceptions=True)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
asyncio.run_coroutine_threadsafe(_drain(mgr), loop).result(timeout=5)
|
||||
loop.call_soon_threadsafe(loop.stop)
|
||||
thread.join(timeout=5)
|
||||
if not thread.is_alive():
|
||||
loop.close()
|
||||
|
||||
|
||||
def _run(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 5.0) -> Any:
|
||||
return asyncio.run_coroutine_threadsafe(coro, loop).result(timeout=timeout)
|
||||
|
||||
|
||||
def _make_session_mock() -> AsyncMock:
|
||||
"""A ClientSession-shaped mock good enough for connect + discovery."""
|
||||
session = AsyncMock()
|
||||
session.initialize = AsyncMock()
|
||||
session.get_server_capabilities = MagicMock(return_value=None)
|
||||
session.list_tools = AsyncMock(return_value=MagicMock(tools=[]))
|
||||
return session
|
||||
|
||||
|
||||
def _fake_transport_and_session(mgr_module_patches: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Build fake stdio transport + ClientSession cms, recording enter/exit."""
|
||||
events: list[str] = []
|
||||
session = _make_session_mock()
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_stdio_client(_params: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
yield (AsyncMock(), AsyncMock())
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_client_session_cm():
|
||||
events.append("session_enter")
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
events.append("session_exit")
|
||||
|
||||
def fake_client_session(_read: Any, _write: Any, message_handler: Any = None):
|
||||
return fake_client_session_cm()
|
||||
|
||||
mgr_module_patches["stdio_client"] = fake_stdio_client
|
||||
mgr_module_patches["ClientSession"] = fake_client_session
|
||||
return {"events": events, "session": session}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Owner lifecycle
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestTransportOwnerLifecycle:
|
||||
def test_connect_installs_owner_and_teardown_closes_gracefully(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", patches["stdio_client"]),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
):
|
||||
_run(loop, mgr._connect_one_locked("srv", mgr._server_configs["srv"]))
|
||||
state = mgr._static_servers["srv"]
|
||||
assert state.session is fake["session"]
|
||||
owner = state.owner_task
|
||||
assert owner is not None and not owner.done()
|
||||
assert state.close_requested is not None
|
||||
assert fake["events"] == ["transport_enter", "session_enter"]
|
||||
|
||||
_run(loop, mgr._teardown_static_session("srv"))
|
||||
|
||||
# Graceful close: the parked owner exits via the event — no cancel —
|
||||
# and unwinds BOTH cms in-task, inner-out.
|
||||
assert owner.done() and not owner.cancelled()
|
||||
assert fake["events"] == [
|
||||
"transport_enter",
|
||||
"session_enter",
|
||||
"session_exit",
|
||||
"transport_exit",
|
||||
]
|
||||
assert state.session is None
|
||||
assert state.owner_task is None
|
||||
assert state.close_requested is None
|
||||
|
||||
def test_owner_death_evicts_session(self, running_loop_mgr) -> None:
|
||||
"""Trigger-A observer: the transport collapsing under a live session
|
||||
(owner task dies without a requested close) evicts the session so the
|
||||
health loop / next dispatch reconnects instead of probing a corpse."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", patches["stdio_client"]),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
):
|
||||
_run(loop, mgr._connect_one_locked("srv", mgr._server_configs["srv"]))
|
||||
state = mgr._static_servers["srv"]
|
||||
owner = state.owner_task
|
||||
assert owner is not None and state.session is fake["session"]
|
||||
|
||||
# Simulate the transport task group collapsing: the owner gets a
|
||||
# stray cancellation (exactly what anyio's scope delivery does).
|
||||
loop.call_soon_threadsafe(owner.cancel)
|
||||
deadline = time.monotonic() + 5
|
||||
while time.monotonic() < deadline and state.owner_task is not None:
|
||||
time.sleep(0.02)
|
||||
|
||||
assert owner.done()
|
||||
assert state.session is None # evicted by the done-callback
|
||||
assert state.owner_task is None
|
||||
# The cms were still unwound in-task despite the stray cancel.
|
||||
assert fake["events"][-2:] == ["session_exit", "transport_exit"]
|
||||
|
||||
def test_owner_death_during_discovery_fails_fast(self, running_loop_mgr) -> None:
|
||||
"""The static sibling of the pool's owner-death discovery race:
|
||||
discovery runs in the connecting caller while the transport is hosted
|
||||
by the owner, so a transport collapse mid-discovery cancels the OWNER
|
||||
and a bare await on the response stream would hang to the caller-side
|
||||
attempt timeout (~45s). ``_await_owner_discovery`` must convert it
|
||||
into a PROMPT ``ConnectionError`` and leave the state torn down."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
|
||||
discovery_parked = asyncio.Event()
|
||||
|
||||
async def _parked_list_tools() -> Any:
|
||||
discovery_parked.set()
|
||||
await asyncio.sleep(3600) # the transport never answers
|
||||
|
||||
fake["session"].list_tools = AsyncMock(side_effect=_parked_list_tools)
|
||||
|
||||
async def _drive() -> tuple[float, BaseException | None]:
|
||||
async def _collapse_owner_when_parked() -> None:
|
||||
await discovery_parked.wait()
|
||||
owner = mgr._static_servers["srv"].owner_task
|
||||
assert owner is not None
|
||||
owner.cancel() # the transport task group collapsing
|
||||
|
||||
collapser = asyncio.create_task(_collapse_owner_when_parked())
|
||||
t0 = asyncio.get_running_loop().time()
|
||||
exc: BaseException | None = None
|
||||
try:
|
||||
await mgr._connect_one_locked("srv", mgr._server_configs["srv"])
|
||||
except Exception as e:
|
||||
# The expected ConnectionError; anything else (a cancel leak,
|
||||
# an interpreter exit) propagates and fails the test loudly.
|
||||
exc = e
|
||||
_ = await collapser # synchronization point; failures propagate
|
||||
return asyncio.get_running_loop().time() - t0, exc
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", patches["stdio_client"]),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
):
|
||||
elapsed, exc = _run(loop, _drive(), timeout=15)
|
||||
|
||||
assert isinstance(exc, ConnectionError)
|
||||
assert "died during discovery" in str(exc)
|
||||
assert elapsed < 5.0 # prompt fail — not the attempt-timeout hang
|
||||
assert mgr._static_servers["srv"].session is None
|
||||
# The owner unwound its cms despite dying mid-discovery.
|
||||
assert fake["events"][-2:] == ["session_exit", "transport_exit"]
|
||||
|
||||
def test_base_exception_escape_resolves_waiter_and_propagates(self, running_loop_mgr) -> None:
|
||||
"""A BaseException-derived escape that is neither CancelledError nor
|
||||
Exception/group (a library control-flow escape; SystemExit and
|
||||
KeyboardInterrupt take the same path but additionally stop the loop —
|
||||
asyncio semantics, unobservable in-process) is NOT swallowed — it
|
||||
propagates from the owner task — but the waiter must still be resolved
|
||||
with a transport-failure error, or the connecting caller would block
|
||||
until its outer bound (and ``_connect_all``'s initial connect has
|
||||
none)."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
class _TransportLibraryEscape(BaseException):
|
||||
pass
|
||||
|
||||
@asynccontextmanager
|
||||
async def escaping_stdio_client(_params: Any):
|
||||
raise _TransportLibraryEscape("control-flow escape")
|
||||
yield # pragma: no cover
|
||||
|
||||
async def _drive() -> tuple[BaseException | None, BaseException | None]:
|
||||
ready: asyncio.Future[Any] = asyncio.get_running_loop().create_future()
|
||||
close_requested = asyncio.Event()
|
||||
owner = asyncio.create_task(
|
||||
mgr._static_transport_owner(
|
||||
"srv", mgr._server_configs["srv"], ready, close_requested
|
||||
)
|
||||
)
|
||||
waiter_exc: BaseException | None = None
|
||||
try:
|
||||
await ready
|
||||
except (Exception, _TransportLibraryEscape) as e:
|
||||
# Exception covers the expected ConnectionError; the escape
|
||||
# type covers the exact regression this test guards (the raw
|
||||
# escape leaking to the waiter instead of being converted).
|
||||
waiter_exc = e
|
||||
await asyncio.wait({owner}, timeout=5)
|
||||
owner_exc = owner.exception() if owner.done() and not owner.cancelled() else None
|
||||
return waiter_exc, owner_exc
|
||||
|
||||
with patch("turnstone.core.mcp_client.stdio_client", escaping_stdio_client):
|
||||
waiter_exc, owner_exc = _run(loop, _drive(), timeout=10)
|
||||
|
||||
assert isinstance(waiter_exc, ConnectionError) # waiter resolved, never hung
|
||||
assert isinstance(owner_exc, _TransportLibraryEscape) # propagated, unswallowed
|
||||
|
||||
def test_connect_failure_unwinds_owner_and_raises(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
@asynccontextmanager
|
||||
async def failing_stdio_client(_params: Any):
|
||||
raise ConnectionError("refused")
|
||||
yield # pragma: no cover
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", failing_stdio_client),
|
||||
pytest.raises(ConnectionError, match="refused"),
|
||||
):
|
||||
_run(loop, mgr._connect_one_locked("srv", mgr._server_configs["srv"]))
|
||||
|
||||
state = mgr._static_servers["srv"]
|
||||
assert state.session is None
|
||||
assert state.owner_task is None
|
||||
|
||||
async def _no_owner_tasks() -> int:
|
||||
return sum(
|
||||
1
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-transport-owner:") and not t.done()
|
||||
)
|
||||
|
||||
assert _run(loop, _no_owner_tasks()) == 0
|
||||
|
||||
def test_caller_cancel_mid_connect_does_not_abandon_cms(self, running_loop_mgr) -> None:
|
||||
"""Bug-1 core regression: cancelling the CONNECTING caller (attempt
|
||||
timeout, shutdown, sync boundary giving up) must close the owner via
|
||||
the one-cancel protocol — the transport cm still exits, in-task."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
events: list[str] = []
|
||||
entered = asyncio.Event()
|
||||
|
||||
@asynccontextmanager
|
||||
async def hanging_stdio_client(_params: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
entered.set()
|
||||
await asyncio.sleep(3600) # server accepted, then stalled
|
||||
yield (AsyncMock(), AsyncMock())
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
async def _drive() -> None:
|
||||
connect = asyncio.create_task(
|
||||
mgr._connect_one_locked("srv", mgr._server_configs["srv"])
|
||||
)
|
||||
await asyncio.wait_for(entered.wait(), timeout=5)
|
||||
connect.cancel() # the attempt-timeout / shutdown shape
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
_ = await connect # only the expected cancel is absorbed
|
||||
# The owner must be closed (one cancel) and fully unwound.
|
||||
deadline = asyncio.get_running_loop().time() + 5
|
||||
while asyncio.get_running_loop().time() < deadline:
|
||||
owners = [
|
||||
t
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-transport-owner:") and not t.done()
|
||||
]
|
||||
if not owners:
|
||||
return
|
||||
await asyncio.sleep(0.02)
|
||||
raise AssertionError("owner task still alive after caller cancel")
|
||||
|
||||
with patch("turnstone.core.mcp_client.stdio_client", hanging_stdio_client):
|
||||
_run(loop, _drive(), timeout=15)
|
||||
|
||||
assert events == ["transport_enter", "transport_exit"]
|
||||
assert mgr._static_servers["srv"].session is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bug 2: BaseExceptionGroup vs except Exception
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestBaseExceptionGroupHardening:
|
||||
def test_connect_all_survives_group_and_starts_loops(self, running_loop_mgr) -> None:
|
||||
"""A transport failure wrapped in BaseExceptionGroup (e.g. an
|
||||
accept-then-RST server collapsing the SDK task group with a stray
|
||||
CancelledError inside) must not kill ``_connect_all`` before the
|
||||
health/sweep loops are started — that silently disabled ALL
|
||||
autonomous recovery."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
# Pin the loop cadences: the assertions below require both loops to be
|
||||
# ENABLED, independent of whatever mcp config the environment carries.
|
||||
mgr._user_token_sweep_s = 240.0
|
||||
mgr._static_health_check_s = 30.0
|
||||
|
||||
async def _exploding_connect(name: str, _cfg: dict[str, Any]) -> None:
|
||||
raise BaseExceptionGroup("transport collapsed", [asyncio.CancelledError()])
|
||||
|
||||
with patch.object(mgr, "_connect_one", side_effect=_exploding_connect):
|
||||
_run(loop, mgr._connect_all())
|
||||
|
||||
assert mgr._connected.is_set()
|
||||
assert "srv" in mgr._last_error
|
||||
health = mgr._static_health_task
|
||||
sweep = mgr._user_token_sweep_task
|
||||
assert health is not None and not health.done()
|
||||
assert sweep is not None and not sweep.done()
|
||||
|
||||
def test_health_loop_survives_group(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
ticks: list[int] = []
|
||||
|
||||
async def _tick_then_group() -> float:
|
||||
ticks.append(1)
|
||||
if len(ticks) == 1:
|
||||
raise BaseExceptionGroup("boom", [asyncio.CancelledError()])
|
||||
return 3600.0
|
||||
|
||||
mgr._static_health_check_s = 0.05 # quick recovery sleep after the group
|
||||
with patch.object(mgr, "_static_health_tick", side_effect=_tick_then_group):
|
||||
|
||||
async def _drive() -> asyncio.Task[None]:
|
||||
task = asyncio.create_task(mgr._static_health_loop())
|
||||
deadline = asyncio.get_running_loop().time() + 5
|
||||
while asyncio.get_running_loop().time() < deadline and len(ticks) < 2:
|
||||
await asyncio.sleep(0.02)
|
||||
assert len(ticks) >= 2, "loop died on BaseExceptionGroup"
|
||||
assert not task.done()
|
||||
task.cancel()
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
_ = await task # only the expected cancel is absorbed
|
||||
return task
|
||||
|
||||
_run(loop, _drive(), timeout=10)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Orphaned-scope disarm backstop
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestScopeDisarmBackstop:
|
||||
def test_disarms_exactly_the_all_done_scope_on_this_loop(self, running_loop_mgr) -> None:
|
||||
"""One sweep over three armed scopes must touch EXACTLY the true
|
||||
orphan: the all-done-tasks scope hosted on the mcp-loop. The
|
||||
live-task scope (its task may still drain the scope) and the
|
||||
hostless scope (loop unknown — not ours to reach into) stay armed.
|
||||
Asserting ``disarmed == 1`` discriminates both failure directions:
|
||||
a no-op sweep and an over-eager one."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
async def _arm_and_sweep() -> dict[str, Any]:
|
||||
from anyio._backends._asyncio import CancelScope
|
||||
|
||||
this_loop = asyncio.get_running_loop()
|
||||
|
||||
async def _noop() -> None:
|
||||
return None
|
||||
|
||||
blocker = asyncio.Event()
|
||||
|
||||
async def _parked() -> None:
|
||||
await blocker.wait()
|
||||
|
||||
done_task = asyncio.create_task(_noop())
|
||||
_ = await done_task # synchronization point; failures propagate
|
||||
live_task = asyncio.create_task(_parked())
|
||||
await asyncio.sleep(0)
|
||||
|
||||
orphan = CancelScope()
|
||||
orphan._host_task = done_task
|
||||
orphan._tasks.add(done_task)
|
||||
orphan._cancel_handle = this_loop.call_soon(lambda: None)
|
||||
|
||||
live_scope = CancelScope()
|
||||
live_scope._host_task = live_task
|
||||
live_scope._tasks.add(live_task)
|
||||
live_scope._cancel_handle = this_loop.call_soon(lambda: None)
|
||||
|
||||
hostless = CancelScope()
|
||||
hostless._tasks.add(done_task)
|
||||
hostless._cancel_handle = this_loop.call_soon(lambda: None)
|
||||
|
||||
mgr._last_scope_disarm = 0.0
|
||||
disarmed = mgr._maybe_disarm_orphaned_scopes("unit test")
|
||||
results = {
|
||||
"disarmed": disarmed,
|
||||
"orphan_handle_cleared": orphan._cancel_handle is None,
|
||||
"orphan_tasks_cleared": len(orphan._tasks) == 0,
|
||||
"live_still_armed": live_scope._cancel_handle is not None,
|
||||
"live_task_kept": live_task in live_scope._tasks,
|
||||
"hostless_still_armed": hostless._cancel_handle is not None,
|
||||
"rate_limited_second": mgr._maybe_disarm_orphaned_scopes("again"),
|
||||
}
|
||||
for scope in (live_scope, hostless):
|
||||
if scope._cancel_handle is not None:
|
||||
scope._cancel_handle.cancel()
|
||||
scope._cancel_handle = None
|
||||
scope._tasks.clear()
|
||||
blocker.set()
|
||||
_ = await live_task # synchronization point; failures propagate
|
||||
return results
|
||||
|
||||
r = _run(loop, _arm_and_sweep())
|
||||
assert r["disarmed"] == 1
|
||||
assert r["orphan_handle_cleared"] and r["orphan_tasks_cleared"]
|
||||
assert r["live_still_armed"] and r["live_task_kept"]
|
||||
assert r["hostless_still_armed"]
|
||||
assert r["rate_limited_second"] == 0
|
||||
+448
-15
@@ -15,13 +15,13 @@ from __future__ import annotations
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from contextlib import AsyncExitStack
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -114,12 +114,31 @@ def running_loop_mgr():
|
||||
# handlers don't fire after pytest has torn its handlers down. Mirrors
|
||||
# the production ``shutdown()`` shape.
|
||||
async def _drain(m: MCPClientManager) -> None:
|
||||
task = m._user_pool_eviction_task
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
with contextlib.suppress(BaseException):
|
||||
await task
|
||||
m._user_pool_eviction_task = None
|
||||
# ``_static_health_task`` included: since the BaseExceptionGroup
|
||||
# hardening, ``_connect_all`` reliably starts (and keeps alive) the
|
||||
# health loop even when every configured connect fails — a test
|
||||
# that drives ``_connect_all`` must drain it like production
|
||||
# ``shutdown()`` does, or the task is destroyed pending at GC.
|
||||
for attr in (
|
||||
"_user_pool_eviction_task",
|
||||
"_user_token_sweep_task",
|
||||
"_static_health_task",
|
||||
):
|
||||
task = getattr(m, attr)
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
await asyncio.gather(task, return_exceptions=True)
|
||||
setattr(m, attr, None)
|
||||
# Close any parked pool transport owners a successful
|
||||
# ``_connect_one_pool`` left installed, mirroring production
|
||||
# ``shutdown()`` — an undrained owner is destroyed pending at GC.
|
||||
for entry in list(m._user_pool_entries.values()):
|
||||
owner = entry.owner_task
|
||||
if owner is not None and not owner.done():
|
||||
if entry.close_requested is not None:
|
||||
entry.close_requested.set()
|
||||
owner.cancel()
|
||||
await asyncio.gather(owner, return_exceptions=True)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
asyncio.run_coroutine_threadsafe(_drain(mgr), loop).result(timeout=2)
|
||||
@@ -341,25 +360,42 @@ class TestEviction:
|
||||
assert ("u4", "pool-srv") in mgr._user_pool_entries
|
||||
assert ("u3", "pool-srv") in mgr._user_pool_entries
|
||||
|
||||
def test_eviction_resilient_to_close_errors(self, running_loop_mgr) -> None:
|
||||
def test_eviction_resilient_to_owner_unwind_errors(self, running_loop_mgr) -> None:
|
||||
"""Owner-model successor to the old ``resilient_to_close_errors`` test.
|
||||
|
||||
Teardown reaps the entry's owner through a bounded ``asyncio.wait`` that
|
||||
never re-raises, so even an owner whose in-task unwind raises cannot
|
||||
break eviction. The old failure mode this guarded — a cross-task
|
||||
``stack.aclose()`` raising ``RuntimeError('...different task...')`` — is
|
||||
structurally impossible now: the transport cms live in, and unwind in,
|
||||
the owner task, never the evictor.
|
||||
"""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._user_pool_idle_ttl_s = 0.0
|
||||
|
||||
broken_stack = MagicMock(spec=AsyncExitStack)
|
||||
broken_stack.aclose = AsyncMock(side_effect=RuntimeError("close failed"))
|
||||
|
||||
async def _seed() -> None:
|
||||
for i in range(2):
|
||||
entry = await mgr._ensure_pool_entry((f"u{i}", "pool-srv"))
|
||||
key = (f"u{i}", "pool-srv")
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
event = asyncio.Event()
|
||||
|
||||
async def _owner(ev: asyncio.Event = event) -> None:
|
||||
await ev.wait()
|
||||
raise RuntimeError("unwind failed")
|
||||
|
||||
owner = asyncio.create_task(_owner(), name=f"mcp-pool-owner-test:{i}")
|
||||
# Retrieve the exception so the raising owner doesn't warn at GC.
|
||||
owner.add_done_callback(lambda t: None if t.cancelled() else t.exception())
|
||||
entry.session = MagicMock()
|
||||
entry.stack = broken_stack
|
||||
entry.owner_task = owner
|
||||
entry.close_requested = event
|
||||
|
||||
_run_on_loop(loop, _seed())
|
||||
|
||||
async def _evict() -> None:
|
||||
await mgr._evict_idle_pool_entries()
|
||||
|
||||
# Eviction must not raise even if close fails.
|
||||
# Eviction must not raise even if the owner's unwind raises.
|
||||
_run_on_loop(loop, _evict())
|
||||
# All entries removed from the dict regardless.
|
||||
assert mgr._user_pool_entries == {}
|
||||
@@ -969,3 +1005,400 @@ class TestUserIdThreadThrough:
|
||||
assert result == "static-output"
|
||||
# No pool entries were created.
|
||||
assert mgr._user_pool_entries == {}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Background token-freshness sweep (oauth_user keep-hot, no connection warming)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestUserTokenFreshnessSweep:
|
||||
"""The background sweep that keeps every consented ``oauth_user`` grant hot
|
||||
for unattended / autonomous work: refresh-on-expiry via the canonical path,
|
||||
proactive dead-grant badging, once-only surfacing, and — the load-bearing
|
||||
property — total invisibility to static / no-auth deployments."""
|
||||
|
||||
def _wire(self, mgr: MCPClientManager, storage: SQLiteBackend, cipher: Any) -> None:
|
||||
mgr.set_storage(storage)
|
||||
mgr.set_app_state(_make_app_state(storage, cipher=cipher))
|
||||
mgr._oauth_user_server_names = {"pool-srv"}
|
||||
|
||||
@staticmethod
|
||||
def _classified(kind: str, token: str | None = None):
|
||||
async def _fake(**kwargs: Any) -> Any:
|
||||
return SimpleNamespace(kind=kind, token=token)
|
||||
|
||||
return _fake
|
||||
|
||||
# -- no-auth / static safety: the sweep must be structurally invisible ----
|
||||
|
||||
def test_sweep_noop_without_oauth_servers(self, running_loop_mgr, storage) -> None:
|
||||
"""A static-only / no-auth deployment: the OBO gate returns before any
|
||||
DB scan or AS round-trip — the single most important property."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._oauth_user_server_names = set() # no oauth_user server configured
|
||||
storage.list_mcp_user_token_reconcile_targets = MagicMock(return_value=[]) # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=AsyncMock(),
|
||||
) as classified:
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
storage.list_mcp_user_token_reconcile_targets.assert_not_called() # no token-table scan
|
||||
classified.assert_not_awaited() # no AS round-trip
|
||||
|
||||
def test_sweep_noop_before_storage_wired(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._oauth_user_server_names = {"pool-srv"} # oauth configured but app not wired yet
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=AsyncMock(),
|
||||
) as classified:
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
classified.assert_not_awaited()
|
||||
|
||||
def test_sweep_skips_server_not_in_oauth_set(self, running_loop_mgr, storage) -> None:
|
||||
"""A token row lingering for a since-demoted / renamed server is not
|
||||
reconciled — only pairs whose server is currently ``oauth_user``."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="ghost-srv")
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=AsyncMock(),
|
||||
) as classified:
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
classified.assert_not_awaited() # ghost-srv is not in _oauth_user_server_names
|
||||
|
||||
# -- classification branches --------------------------------------------
|
||||
|
||||
def test_healthy_token_no_badge(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("token", token="access-aaa"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
storage.upsert_mcp_pending_consent.assert_not_called()
|
||||
assert ("u1", "pool-srv") not in mgr._token_sweep_warned
|
||||
|
||||
def test_dead_grant_badges_once_and_dedups(self, running_loop_mgr, storage, caplog) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed"),
|
||||
),
|
||||
caplog.at_level(logging.WARNING, logger="turnstone.core.mcp_client"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness()) # second tick: no re-badge
|
||||
|
||||
# Badge raised exactly once, proactively, with the dashboard's code.
|
||||
storage.upsert_mcp_pending_consent.assert_called_once()
|
||||
assert (
|
||||
storage.upsert_mcp_pending_consent.call_args.kwargs["error_code"]
|
||||
== "mcp_consent_required"
|
||||
)
|
||||
assert ("u1", "pool-srv") in mgr._token_sweep_warned
|
||||
escalations = [r for r in caplog.records if "needs re-consent" in r.getMessage()]
|
||||
assert len(escalations) == 1 # logged loud-once, not every tick
|
||||
|
||||
def test_decrypt_failure_warns_but_does_not_badge(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("decrypt_failure"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
# Operator-actionable (key unknown) — surfaced in the warned set, but NOT
|
||||
# a user-consent badge (outside the dashboard's scope).
|
||||
storage.upsert_mcp_pending_consent.assert_not_called()
|
||||
assert ("u1", "pool-srv") in mgr._token_sweep_warned
|
||||
|
||||
def test_transient_failure_is_silent(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed_transient"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
storage.upsert_mcp_pending_consent.assert_not_called()
|
||||
assert ("u1", "pool-srv") not in mgr._token_sweep_warned # retryable, not surfaced
|
||||
|
||||
def test_recovery_rearms_and_clears_badge(self, running_loop_mgr, storage) -> None:
|
||||
"""A dead grant that later returns healthy clears its warned pin AND drops
|
||||
the stale badge — the self-heal for a spurious invalid_grant that has
|
||||
since recovered. Production-reachable now that the observe-only sweep no
|
||||
longer deletes the row on refresh_failed, so the pair keeps enumerating."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.delete_mcp_pending_consent = MagicMock(return_value=True) # type: ignore[method-assign]
|
||||
key = ("u1", "pool-srv")
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert key in mgr._token_sweep_warned
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("token", token="access-aaa"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert key not in mgr._token_sweep_warned # recovered → re-armed
|
||||
storage.delete_mcp_pending_consent.assert_called_once_with("u1", "pool-srv")
|
||||
|
||||
def test_dead_grant_not_pinned_when_badge_persist_fails(
|
||||
self, running_loop_mgr, storage
|
||||
) -> None:
|
||||
"""If the badge write fails, the pair is NOT pinned, so the next tick
|
||||
retries — a single failed persist must not permanently lose the only
|
||||
proactive signal for a sweep-detected dead grant."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock( # type: ignore[method-assign]
|
||||
side_effect=RuntimeError("db down")
|
||||
)
|
||||
key = ("u1", "pool-srv")
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert key not in mgr._token_sweep_warned # not pinned — will retry
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
# Retried on the second tick rather than deduped away by a phantom pin.
|
||||
assert storage.upsert_mcp_pending_consent.call_count == 2
|
||||
|
||||
def test_sweep_uses_non_revoking_observe_mode(self, running_loop_mgr, storage) -> None:
|
||||
"""The background sweep MUST call the canonical lookup non-destructively:
|
||||
a timer may never delete a token or move a foreground user's revoke
|
||||
threshold."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
seen_kwargs: list[dict[str, Any]] = []
|
||||
|
||||
async def _spy(**kwargs: Any) -> Any:
|
||||
seen_kwargs.append(kwargs)
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_spy):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert seen_kwargs and seen_kwargs[0]["revoke_on_failure"] is False
|
||||
assert seen_kwargs[0]["revoke_ambiguous_escalation"] is False
|
||||
|
||||
# -- keepalive refresh (exercise the refresh token before it idles out) ---
|
||||
|
||||
def test_keepalive_refresh_due_logic(self) -> None:
|
||||
mgr = MCPClientManager({})
|
||||
mgr._user_token_refresh_keepalive_s = 3600.0
|
||||
old = (datetime.now(UTC) - timedelta(hours=2)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
recent = (datetime.now(UTC) - timedelta(minutes=1)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
assert mgr._keepalive_refresh_due(old) is True # past the window → force
|
||||
assert mgr._keepalive_refresh_due(recent) is False # still warm
|
||||
assert mgr._keepalive_refresh_due(None) is True # unknown → force once, safe
|
||||
assert mgr._keepalive_refresh_due("not-a-date") is True # unparseable → force
|
||||
mgr._user_token_refresh_keepalive_s = 0.0
|
||||
assert mgr._keepalive_refresh_due(old) is False # disabled → never force
|
||||
|
||||
def test_keepalive_due_forces_refresh(self, running_loop_mgr, storage) -> None:
|
||||
"""A grant whose refresh token has idled past the window is force-refreshed
|
||||
even though its access token may be fresh — the [6] fix: keep the refresh
|
||||
token alive so an unattended run never finds it aged out."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._user_token_refresh_keepalive_s = 1800.0
|
||||
stale = (datetime.now(UTC) - timedelta(hours=2)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
storage.list_mcp_user_token_reconcile_targets = MagicMock( # type: ignore[method-assign]
|
||||
return_value=[("u1", "pool-srv", stale)]
|
||||
)
|
||||
seen_kwargs: list[dict[str, Any]] = []
|
||||
|
||||
async def _spy(**kwargs: Any) -> Any:
|
||||
seen_kwargs.append(kwargs)
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_spy):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert seen_kwargs and seen_kwargs[0]["force_refresh"] is True
|
||||
|
||||
def test_keepalive_not_due_does_not_force(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._user_token_refresh_keepalive_s = 1800.0
|
||||
recent = (datetime.now(UTC) - timedelta(minutes=1)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
storage.list_mcp_user_token_reconcile_targets = MagicMock( # type: ignore[method-assign]
|
||||
return_value=[("u1", "pool-srv", recent)]
|
||||
)
|
||||
seen_kwargs: list[dict[str, Any]] = []
|
||||
|
||||
async def _spy(**kwargs: Any) -> Any:
|
||||
seen_kwargs.append(kwargs)
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_spy):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert seen_kwargs and seen_kwargs[0]["force_refresh"] is False # still warm
|
||||
|
||||
def test_warned_set_pruned_to_consented_pairs(self, running_loop_mgr, storage) -> None:
|
||||
"""A warned pair that is no longer consented (row gone) is dropped from
|
||||
the dedup set so it can't grow unbounded across transient dead grants."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
mgr._token_sweep_warned = {("gone-user", "pool-srv"), ("u1", "pool-srv")}
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("token", token="access-aaa"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert ("gone-user", "pool-srv") not in mgr._token_sweep_warned # pruned
|
||||
assert ("u1", "pool-srv") not in mgr._token_sweep_warned # healthy → cleared
|
||||
|
||||
def test_per_pair_failure_isolated(self, running_loop_mgr, storage) -> None:
|
||||
"""One pair raising must not starve the rest of the pass."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._oauth_user_server_names = {"pool-srv"}
|
||||
_seed_user_token(storage, cipher, user_id="u-bad", server_name="pool-srv")
|
||||
_seed_user_token(storage, cipher, user_id="u-ok", server_name="pool-srv")
|
||||
seen: list[str] = []
|
||||
|
||||
async def _flaky(**kwargs: Any) -> Any:
|
||||
uid = kwargs["user_id"]
|
||||
seen.append(uid)
|
||||
if uid == "u-bad":
|
||||
raise RuntimeError("boom")
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_flaky):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert {"u-bad", "u-ok"} <= set(seen) # both attempted despite one raising
|
||||
|
||||
def test_sweep_loop_cancel_returns_cleanly(self, running_loop_mgr) -> None:
|
||||
"""The loop body exits on cancellation without raising (mirrors the
|
||||
eviction loop's teardown contract)."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._user_token_sweep_s = 999.0 # park in the sleep
|
||||
|
||||
async def _spawn() -> asyncio.Task[None]:
|
||||
return asyncio.ensure_future(mgr._user_token_sweep_loop())
|
||||
|
||||
task = _run_on_loop(loop, _spawn())
|
||||
|
||||
async def _cancel() -> None:
|
||||
task.cancel()
|
||||
with contextlib.suppress(BaseException):
|
||||
await task
|
||||
|
||||
_run_on_loop(loop, _cancel())
|
||||
assert task.cancelled() or task.done()
|
||||
|
||||
def test_connect_all_starts_the_sweep_task(self, running_loop_mgr) -> None:
|
||||
"""Wiring guard: ``_connect_all`` must start the sweep once, even with no
|
||||
servers configured — otherwise the whole keep-hot mechanism is dead code."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
assert mgr._user_token_sweep_task is None
|
||||
|
||||
_run_on_loop(loop, mgr._connect_all())
|
||||
try:
|
||||
task = mgr._user_token_sweep_task
|
||||
assert task is not None and not task.done() # live, single instance
|
||||
finally:
|
||||
|
||||
async def _drain() -> None:
|
||||
t = mgr._user_token_sweep_task
|
||||
if t is not None:
|
||||
t.cancel()
|
||||
with contextlib.suppress(BaseException):
|
||||
await t
|
||||
mgr._user_token_sweep_task = None
|
||||
|
||||
_run_on_loop(loop, _drain())
|
||||
|
||||
def test_disabled_sweep_not_started_by_connect_all(self, running_loop_mgr) -> None:
|
||||
"""Cadence <= 0 disables the sweep entirely — no task is spawned."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._user_token_sweep_s = 0.0
|
||||
_run_on_loop(loop, mgr._connect_all())
|
||||
assert mgr._user_token_sweep_task is None
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("configured", "expected"),
|
||||
[
|
||||
(0, 0.0), # explicit disable
|
||||
(-5, 0.0), # negative disables (no busy-loop)
|
||||
(1, 30.0), # tiny positive floored to _MIN_USER_TOKEN_SWEEP_S
|
||||
(600, 600.0), # normal value passes through
|
||||
],
|
||||
)
|
||||
def test_cadence_clamped_or_disabled(self, configured, expected) -> None:
|
||||
"""The config cadence is floored (positive) or disabled (<= 0) so an
|
||||
``asyncio.sleep(0)`` busy-loop is unreachable."""
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.load_config",
|
||||
return_value={"user_token_sweep_seconds": configured},
|
||||
):
|
||||
mgr = MCPClientManager({})
|
||||
assert mgr._user_token_sweep_s == expected
|
||||
|
||||
# -- storage enumerator --------------------------------------------------
|
||||
|
||||
def test_reconcile_targets_pairs_expiry_unfiltered_with_last_exercised(self, storage) -> None:
|
||||
cipher = make_mcp_token_cipher()
|
||||
# alice consents to two servers → two rows.
|
||||
_seed_user_token(storage, cipher, user_id="alice", server_name="srv-a")
|
||||
_seed_user_token(storage, cipher, user_id="alice", server_name="srv-b")
|
||||
# bob's access token is expired but the refresh token is live — still a
|
||||
# consented, reconcilable grant, so bob must be enumerated.
|
||||
_seed_user_token(
|
||||
storage, cipher, user_id="bob", server_name="srv-a", expires_in_seconds=-999
|
||||
)
|
||||
targets = storage.list_mcp_user_token_reconcile_targets()
|
||||
# (user, server) identity, all three grants present regardless of expiry.
|
||||
assert sorted((u, s) for u, s, _ in targets) == [
|
||||
("alice", "srv-a"),
|
||||
("alice", "srv-b"),
|
||||
("bob", "srv-a"),
|
||||
]
|
||||
# last_exercised = COALESCE(last_refreshed, created); never-refreshed rows
|
||||
# fall back to created, so it is always populated (drives the keepalive).
|
||||
assert all(last_exercised for _, _, last_exercised in targets)
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Tests for alembic migration 065 (capture Entra oid/tid on oidc_identities).
|
||||
|
||||
Drives ``command.upgrade``/``downgrade`` against an isolated SQLite database per
|
||||
test (the 060/062/063 harness pattern), then asserts:
|
||||
|
||||
* upgrade adds the ``oid``/``tid`` columns and the ``idx_oidc_identities_oid``
|
||||
index;
|
||||
* a pre-065 row migrates cleanly, gaining ``""`` for the new columns;
|
||||
* downgrade removes the columns + index, returning ``oidc_identities`` to its
|
||||
exact pre-065 shape — this pins the **clean-rollback** guarantee (the change
|
||||
can be backed out with no orphaned state if the upstream PR is rejected).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import command
|
||||
from alembic.config import Config
|
||||
|
||||
_MIGRATIONS_DIR = str(
|
||||
Path(__file__).resolve().parent.parent / "turnstone" / "core" / "storage" / "migrations"
|
||||
)
|
||||
|
||||
|
||||
def _alembic_cfg(db_path: Path) -> Config:
|
||||
cfg = Config()
|
||||
cfg.set_main_option("script_location", _MIGRATIONS_DIR)
|
||||
cfg.set_main_option("sqlalchemy.url", f"sqlite:///{db_path}")
|
||||
return cfg
|
||||
|
||||
|
||||
class TestMigration065:
|
||||
def test_upgrade_adds_oid_tid_and_index(self, tmp_path: Path) -> None:
|
||||
db_path = tmp_path / "065-up.db"
|
||||
command.upgrade(_alembic_cfg(db_path), "065")
|
||||
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
insp = sa.inspect(engine)
|
||||
cols = {c["name"] for c in insp.get_columns("oidc_identities")}
|
||||
assert {"oid", "tid"} <= cols
|
||||
idx = {i["name"] for i in insp.get_indexes("oidc_identities")}
|
||||
assert "idx_oidc_identities_oid" in idx
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
def test_preexisting_row_migrates_with_empty_default(self, tmp_path: Path) -> None:
|
||||
db_path = tmp_path / "065-default.db"
|
||||
cfg = _alembic_cfg(db_path)
|
||||
# Stop at 064, insert a pre-065 identity, THEN upgrade to 065.
|
||||
command.upgrade(cfg, "064")
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
with engine.begin() as conn:
|
||||
conn.execute(
|
||||
sa.text(
|
||||
"INSERT INTO oidc_identities "
|
||||
"(issuer, subject, user_id, email, created, last_login) "
|
||||
"VALUES ('iss', 'sub', 'u1', '', "
|
||||
"'2026-01-01T00:00:00', '2026-01-01T00:00:00')"
|
||||
)
|
||||
)
|
||||
command.upgrade(cfg, "065")
|
||||
with engine.connect() as conn:
|
||||
row = conn.execute(
|
||||
sa.text("SELECT oid, tid FROM oidc_identities WHERE subject = 'sub'")
|
||||
).fetchone()
|
||||
assert row is not None
|
||||
assert row[0] == "" and row[1] == ""
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
def test_downgrade_removes_oid_tid_and_index(self, tmp_path: Path) -> None:
|
||||
db_path = tmp_path / "065-down.db"
|
||||
cfg = _alembic_cfg(db_path)
|
||||
command.upgrade(cfg, "065")
|
||||
command.downgrade(cfg, "064")
|
||||
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
insp = sa.inspect(engine)
|
||||
cols = {c["name"] for c in insp.get_columns("oidc_identities")}
|
||||
assert "oid" not in cols and "tid" not in cols
|
||||
idx = {i["name"] for i in insp.get_indexes("oidc_identities")}
|
||||
assert "idx_oidc_identities_oid" not in idx
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
def test_downgrade_then_upgrade_round_trip(self, tmp_path: Path) -> None:
|
||||
"""up -> down -> up must land cleanly (no leftover column/index conflict)."""
|
||||
db_path = tmp_path / "065-roundtrip.db"
|
||||
cfg = _alembic_cfg(db_path)
|
||||
command.upgrade(cfg, "065")
|
||||
command.downgrade(cfg, "064")
|
||||
command.upgrade(cfg, "065")
|
||||
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
cols = {c["name"] for c in sa.inspect(engine).get_columns("oidc_identities")}
|
||||
assert {"oid", "tid"} <= cols
|
||||
finally:
|
||||
engine.dispose()
|
||||
@@ -1643,6 +1643,65 @@ class TestProvisionOIDCUser:
|
||||
|
||||
storage.assign_role.assert_not_called()
|
||||
|
||||
def test_provision_oidc_user_null_oid_tid_collapse_to_empty(self):
|
||||
"""A present-but-null oid/tid claim must store "" — never the string "None".
|
||||
|
||||
`claims.get("oid", "")` returns None (not the "" default) when the key is
|
||||
present with a JSON null, and str(None) == "None" would slip past both the
|
||||
server_default and the truthy backfill guard, storing a bogus non-empty
|
||||
sentinel that collides across every null-emitting user. New-user path.
|
||||
"""
|
||||
config = _make_config()
|
||||
storage = _mock_storage()
|
||||
storage.get_user.return_value = {
|
||||
"user_id": "u-new",
|
||||
"username": "bob",
|
||||
"display_name": "Bob",
|
||||
"password_hash": "!oidc",
|
||||
}
|
||||
|
||||
claims = {"sub": "sub-null", "preferred_username": "bob", "oid": None, "tid": None}
|
||||
with patch("turnstone.core.oidc.uuid") as mock_uuid:
|
||||
mock_uuid.uuid4.return_value = MagicMock(hex="u-new-hex-00000000000000000000")
|
||||
provision_oidc_user(storage, config, claims)
|
||||
|
||||
kwargs = storage.create_oidc_user.call_args.kwargs
|
||||
assert kwargs["oid"] == ""
|
||||
assert kwargs["tid"] == ""
|
||||
|
||||
def test_provision_oidc_user_null_oid_tid_not_backfilled_existing(self):
|
||||
"""Existing-identity path: null oid/tid claims must not backfill "None".
|
||||
|
||||
The truthy guard in update_oidc_identity_login only protects against ""; a
|
||||
"None" produced by str(None) is truthy and would be written, clobbering a
|
||||
real value captured on an earlier login.
|
||||
"""
|
||||
config = _make_config()
|
||||
existing_user = {
|
||||
"user_id": "u1",
|
||||
"username": "alice",
|
||||
"display_name": "Alice",
|
||||
"password_hash": "!oidc",
|
||||
}
|
||||
existing_identity = {
|
||||
"issuer": "https://idp.example.com",
|
||||
"subject": "sub-123",
|
||||
"user_id": "u1",
|
||||
"email": "alice@example.com",
|
||||
"created": "2024-01-01T00:00:00",
|
||||
"last_login": "2024-01-01T00:00:00",
|
||||
"oid": "obj-real",
|
||||
"tid": "ten-real",
|
||||
}
|
||||
storage = _mock_storage(identity=existing_identity, user=existing_user)
|
||||
|
||||
claims = {"sub": "sub-123", "email": "alice@example.com", "oid": None, "tid": None}
|
||||
provision_oidc_user(storage, config, claims)
|
||||
|
||||
kwargs = storage.update_oidc_identity_login.call_args.kwargs
|
||||
assert kwargs["oid"] == ""
|
||||
assert kwargs["tid"] == ""
|
||||
|
||||
def test_existing_identity_self_heals_zero_roles(self):
|
||||
"""Existing identity user with zero roles -> safety-net assigns builtin-viewer.
|
||||
|
||||
|
||||
@@ -84,6 +84,40 @@ class TestCreateOIDCUser:
|
||||
assert identity is not None
|
||||
assert identity["user_id"] == "u-other"
|
||||
|
||||
def test_create_oidc_user_captures_oid_tid(self, db):
|
||||
"""Entra oid/tid are persisted and returned on the identity."""
|
||||
db.create_oidc_user(
|
||||
user_id="u-oid",
|
||||
username="carol",
|
||||
display_name="Carol",
|
||||
password_hash="!oidc",
|
||||
issuer="https://idp.example.com",
|
||||
subject="sub-oid",
|
||||
email="carol@example.com",
|
||||
oid="obj-123",
|
||||
tid="tenant-abc",
|
||||
)
|
||||
identity = db.get_oidc_identity("https://idp.example.com", "sub-oid")
|
||||
assert identity is not None
|
||||
assert identity["oid"] == "obj-123"
|
||||
assert identity["tid"] == "tenant-abc"
|
||||
|
||||
def test_create_oidc_user_oid_tid_default_empty(self, db):
|
||||
"""Omitting oid/tid (non-Entra IdP) stores "" — never NULL."""
|
||||
db.create_oidc_user(
|
||||
user_id="u-noid",
|
||||
username="dave",
|
||||
display_name="Dave",
|
||||
password_hash="!oidc",
|
||||
issuer="https://idp.example.com",
|
||||
subject="sub-noid",
|
||||
email="dave@example.com",
|
||||
)
|
||||
identity = db.get_oidc_identity("https://idp.example.com", "sub-noid")
|
||||
assert identity is not None
|
||||
assert identity["oid"] == ""
|
||||
assert identity["tid"] == ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# OIDC Identity CRUD
|
||||
@@ -137,6 +171,33 @@ class TestOIDCIdentityCRUD:
|
||||
result = db.update_oidc_identity_login("https://idp.example.com", "sub-999")
|
||||
assert result is False
|
||||
|
||||
def test_update_oidc_identity_login_backfills_oid_tid(self, db):
|
||||
"""A login carrying oid/tid backfills them onto a pre-existing row."""
|
||||
db.create_oidc_identity("https://idp.example.com", "sub-bf", "u1", "a@example.com")
|
||||
before = db.get_oidc_identity("https://idp.example.com", "sub-bf")
|
||||
assert before is not None and before["oid"] == ""
|
||||
|
||||
db.update_oidc_identity_login("https://idp.example.com", "sub-bf", oid="obj-9", tid="ten-9")
|
||||
|
||||
after = db.get_oidc_identity("https://idp.example.com", "sub-bf")
|
||||
assert after is not None
|
||||
assert after["oid"] == "obj-9"
|
||||
assert after["tid"] == "ten-9"
|
||||
|
||||
def test_update_oidc_identity_login_omitted_does_not_clobber_oid_tid(self, db):
|
||||
"""A later login WITHOUT oid/tid must not wipe previously-captured values."""
|
||||
db.create_oidc_identity("https://idp.example.com", "sub-keep", "u1", "a@example.com")
|
||||
db.update_oidc_identity_login(
|
||||
"https://idp.example.com", "sub-keep", oid="obj-keep", tid="ten-keep"
|
||||
)
|
||||
# Simulate a subsequent login where the token omitted oid/tid.
|
||||
db.update_oidc_identity_login("https://idp.example.com", "sub-keep")
|
||||
|
||||
identity = db.get_oidc_identity("https://idp.example.com", "sub-keep")
|
||||
assert identity is not None
|
||||
assert identity["oid"] == "obj-keep"
|
||||
assert identity["tid"] == "ten-keep"
|
||||
|
||||
def test_list_oidc_identities_for_user(self, db):
|
||||
"""Two identities for same user, list returns both."""
|
||||
db.create_oidc_identity("https://idp1.example.com", "sub-A", "u1", "alice@idp1.com")
|
||||
|
||||
@@ -8,6 +8,7 @@ capability-gated emission in ``ChatSession._init_system_messages``.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
@@ -330,3 +331,38 @@ class TestEmptyUserTurnDrop:
|
||||
assert len(user_turns) == 1
|
||||
assert f"[start system-reminder_{nonce}]" in user_turns[0]["content"]
|
||||
assert "child done" in user_turns[0]["content"]
|
||||
|
||||
|
||||
class TestToolArgumentLegalization:
|
||||
"""``_prepare_wire_messages`` legalizes malformed tool-call ``arguments`` so a
|
||||
strict renderer (vLLM ``deepseek_v4``) can ``json.loads`` every arguments string
|
||||
— the sibling send-time validity pass to orphan repair."""
|
||||
|
||||
def test_unterminated_arguments_legalized_on_the_wire(self) -> None:
|
||||
s = make_session()
|
||||
msgs = [
|
||||
{"role": "user", "content": "go"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "c1",
|
||||
"type": "function",
|
||||
"function": {"name": "bash", "arguments": '{"command": "cat /va'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "retry with valid JSON"},
|
||||
]
|
||||
out = s._prepare_wire_messages(msgs)
|
||||
emitted = [
|
||||
tc["function"]["arguments"]
|
||||
for m in out
|
||||
if m.get("role") == "assistant"
|
||||
for tc in m.get("tool_calls", [])
|
||||
]
|
||||
assert emitted == ["{}"]
|
||||
assert json.loads(emitted[0]) == {}
|
||||
# Canonical input is untouched — legalization is wire-copy only.
|
||||
assert msgs[1]["tool_calls"][0]["function"]["arguments"] == '{"command": "cat /va'
|
||||
|
||||
@@ -2,7 +2,11 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from turnstone.core.output_guard import evaluate_output, merge_guard_display_payload
|
||||
from turnstone.core.output_guard import (
|
||||
evaluate_output,
|
||||
merge_guard_display_payload,
|
||||
redact_credentials,
|
||||
)
|
||||
|
||||
|
||||
class TestBenignOutput:
|
||||
@@ -205,6 +209,47 @@ class TestCredentialLeakage:
|
||||
)
|
||||
assert "credential_leak" not in r.flags
|
||||
|
||||
def test_single_quote_json_secret(self) -> None:
|
||||
# Python dict reprs / JS object literals emit single quotes; these must
|
||||
# be detected and redacted just like the double-quoted JSON form.
|
||||
r = evaluate_output("headers = {'Authorization': 'Bearer canstillseethis'}")
|
||||
assert "credential_leak" in r.flags
|
||||
assert "json_secret_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "canstillseethis" not in r.sanitized
|
||||
|
||||
def test_single_quote_password(self) -> None:
|
||||
r = evaluate_output("{'password': 'hunter2hunter2'}")
|
||||
assert "json_secret_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "hunter2hunter2" not in r.sanitized
|
||||
|
||||
def test_mongodb_srv_connection_string(self) -> None:
|
||||
r = evaluate_output("uri: mongodb+srv://admin:s3cretpw@cluster.mongodb.net/db")
|
||||
assert "connection_string_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "s3cretpw" not in r.sanitized
|
||||
|
||||
def test_rediss_connection_string(self) -> None:
|
||||
r = evaluate_output("rediss://user:s3cretpw@redis.host:6380/0")
|
||||
assert "connection_string_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "s3cretpw" not in r.sanitized
|
||||
|
||||
def test_bearer_scheme_case_insensitive(self) -> None:
|
||||
# RFC 7235 scheme name is case-insensitive.
|
||||
r = evaluate_output("authorization: bearer eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.sig12345")
|
||||
assert "credential_leak" in r.flags
|
||||
|
||||
def test_prefixed_key_assignment_redacts_whole_token(self) -> None:
|
||||
# api_key=/secret_key=/access_token= must redact the entire assignment,
|
||||
# not chew only the tail into a garbled "api_[REDACTED:api_key]".
|
||||
secret = "abcdefghijklmnopqrstuvwxyz"
|
||||
for prefix in ("api_key", "secret_key", "session_key", "access_token", "key", "token"):
|
||||
out = redact_credentials(f"{prefix}={secret}")
|
||||
assert secret not in out, (prefix, out)
|
||||
assert out == "[REDACTED:api_key]", (prefix, out)
|
||||
|
||||
|
||||
class TestEncodedPayloads:
|
||||
"""Detect encoded/obfuscated payloads."""
|
||||
|
||||
@@ -521,16 +521,21 @@ class TestSpawnPersona:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Guard 7 — task_agent has no persona parameter (sub-agents keep their own
|
||||
# identity; persona is a workstream-level concept).
|
||||
# Guard 7 — task_agent HAS a persona parameter: a sub-agent's identity comes
|
||||
# from a persona (default = the built-in task-agent identity), validated at
|
||||
# prep against the interactive kind. Revises the original "no persona for
|
||||
# task agents" stance now that personas are first-class on every path.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_task_agent_schema_has_no_persona_param() -> None:
|
||||
def test_task_agent_schema_has_persona_param() -> None:
|
||||
from turnstone.core.tools import TOOLS
|
||||
|
||||
task_agent = next(t for t in TOOLS if t["function"]["name"] == "task_agent")
|
||||
assert "persona" not in task_agent["function"]["parameters"]["properties"]
|
||||
props = task_agent["function"]["parameters"]["properties"]
|
||||
assert "persona" in props
|
||||
# skill= is capability now — the description frames it that way.
|
||||
assert "capability" in props["skill"]["description"].lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1324,3 +1329,180 @@ class TestCreateStampsPersona:
|
||||
assert ws is not None and ws.session is not None
|
||||
assert ws.session._persona_name == ""
|
||||
assert not ws.persona
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Guard 10 — discovery: the calling LLM is TOLD which personas exist. The
|
||||
# live enabled interactive-kind list rides the `persona` parameter
|
||||
# description of task_agent / spawn_workstream / spawn_batch, rebuilt from
|
||||
# the pristine TOOLS base on every render; storage-less sessions keep the
|
||||
# base text untouched. Resolution is forgiving (case, unique display name)
|
||||
# but everything downstream carries the canonical slug.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _persona_desc(session: ChatSession, tool_name: str) -> str:
|
||||
tool = next(t for t in session._tools if t.get("function", {}).get("name") == tool_name)
|
||||
prop = ChatSession._persona_property(tool["function"]["parameters"]["properties"])
|
||||
assert prop is not None, f"{tool_name} has no persona parameter"
|
||||
return prop["description"]
|
||||
|
||||
|
||||
def _pristine_persona_desc(tool_name: str) -> str:
|
||||
from turnstone.core.tools import TOOLS
|
||||
|
||||
tool = next(t for t in TOOLS if t["function"]["name"] == tool_name)
|
||||
prop = ChatSession._persona_property(tool["function"]["parameters"]["properties"])
|
||||
assert prop is not None, f"{tool_name} has no persona parameter"
|
||||
return prop["description"]
|
||||
|
||||
|
||||
class TestPersonaDiscovery:
|
||||
def _seed(self) -> None:
|
||||
get_storage().create_persona(
|
||||
{
|
||||
"persona_id": "p-eng",
|
||||
"name": "engineer",
|
||||
"display_name": "Engineer",
|
||||
"description": "Default engineering identity",
|
||||
"base_prompt": "E",
|
||||
"applies_to_kinds": ["interactive"],
|
||||
"is_default": True,
|
||||
}
|
||||
)
|
||||
get_storage().create_persona(
|
||||
{
|
||||
"persona_id": "p-wri",
|
||||
"name": "writer",
|
||||
"display_name": "Creative Writer",
|
||||
"description": "Prose-first writing partner",
|
||||
"base_prompt": "W",
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
)
|
||||
|
||||
def _coord_session(self, mock_openai_client: Any) -> ChatSession:
|
||||
return _session(
|
||||
mock_openai_client,
|
||||
kind=WorkstreamKind.COORDINATOR,
|
||||
user_id="u1",
|
||||
coord_client=MagicMock(),
|
||||
)
|
||||
|
||||
def test_task_agent_description_lists_personas(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = _session(mock_openai_client)
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
assert desc.startswith(_pristine_persona_desc("task_agent"))
|
||||
assert "Available personas:" in desc
|
||||
# Default first, then A→Z, each with its one-line description.
|
||||
assert desc.index("`engineer` (default)") < desc.index("`writer`")
|
||||
assert "Prose-first writing partner" in desc
|
||||
|
||||
def test_spawn_tools_list_personas_for_coordinators(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
for tool_name in ("spawn_workstream", "spawn_batch"):
|
||||
desc = _persona_desc(session, tool_name)
|
||||
assert desc.startswith(_pristine_persona_desc(tool_name))
|
||||
assert "Available personas:" in desc
|
||||
assert "`engineer` (default)" in desc
|
||||
|
||||
def test_coordinator_kind_personas_are_not_offered(self, tmp_db, mock_openai_client) -> None:
|
||||
# Children and sub-agents are always interactive-kind; a
|
||||
# coordinator-only persona in the list would be a guaranteed error.
|
||||
self._seed()
|
||||
get_storage().create_persona(
|
||||
{
|
||||
"persona_id": "p-exe",
|
||||
"name": "executive",
|
||||
"base_prompt": "X",
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
}
|
||||
)
|
||||
session = self._coord_session(mock_openai_client)
|
||||
assert "`executive`" not in _persona_desc(session, "spawn_workstream")
|
||||
|
||||
def test_storage_down_keeps_pristine_base(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
with patch("turnstone.core.storage.is_storage_initialized", return_value=False):
|
||||
session = _session(mock_openai_client)
|
||||
assert _persona_desc(session, "task_agent") == _pristine_persona_desc("task_agent")
|
||||
|
||||
def test_rerender_is_idempotent_and_tracks_archive(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = _session(mock_openai_client)
|
||||
session._render_agent_tool_descriptions()
|
||||
session._render_agent_tool_descriptions()
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
assert desc.count("Available personas:") == 1
|
||||
# Archive one persona; the next render must drop it, not append.
|
||||
storage = get_storage()
|
||||
writer = storage.get_persona_by_name("writer")
|
||||
assert writer is not None
|
||||
storage.update_persona(writer["persona_id"], enabled=False)
|
||||
session._render_agent_tool_descriptions()
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
assert "`writer`" not in desc
|
||||
assert desc.count("Available personas:") == 1
|
||||
|
||||
def test_large_shelf_drops_prose_keeps_every_name(self, tmp_db, mock_openai_client) -> None:
|
||||
storage = get_storage()
|
||||
for i in range(26):
|
||||
storage.create_persona(
|
||||
{
|
||||
"persona_id": f"p-{i:02d}",
|
||||
"name": f"persona-{i:02d}",
|
||||
"description": "UNIQUE-PROSE-MARKER",
|
||||
"base_prompt": "x",
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
)
|
||||
session = _session(mock_openai_client)
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
for i in range(26):
|
||||
assert f"`persona-{i:02d}`" in desc
|
||||
assert "UNIQUE-PROSE-MARKER" not in desc
|
||||
|
||||
def test_spawn_forgives_case_and_display_name_but_stamps_slug(
|
||||
self, tmp_db, mock_openai_client
|
||||
) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
for variant in ("WRITER", "Writer", "Creative Writer"):
|
||||
item = session._prepare_spawn_workstream("c1", {"persona": variant})
|
||||
assert not item.get("error"), item.get("error")
|
||||
assert item["persona"] == "writer"
|
||||
|
||||
def test_spawn_batch_rows_land_on_canonical_slug(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
item = session._prepare_spawn_batch(
|
||||
"c1",
|
||||
{
|
||||
"children": [
|
||||
{"initial_message": "a", "persona": "WRITER"},
|
||||
{"initial_message": "b", "persona": "Creative Writer"},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert not item.get("error"), item.get("error")
|
||||
personas = [c["persona"] for c in item["children"] if "_error" not in c]
|
||||
assert personas == ["writer", "writer"]
|
||||
|
||||
def test_task_agent_prep_canonicalizes_header_and_stamp(
|
||||
self, tmp_db, mock_openai_client
|
||||
) -> None:
|
||||
self._seed()
|
||||
session = _session(mock_openai_client)
|
||||
item = session._prepare_task("t1", {"prompt": "go", "persona": "Writer"})
|
||||
assert not item.get("error"), item.get("error")
|
||||
assert item["persona"] == "writer"
|
||||
assert "persona: writer" in item["header"]
|
||||
|
||||
def test_unknown_persona_error_enumerates_live_names(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
item = session._prepare_spawn_workstream("c1", {"persona": "nope"})
|
||||
assert item.get("error")
|
||||
assert "Available for interactive: engineer (default), writer" in item["error"]
|
||||
|
||||
@@ -125,3 +125,191 @@ class TestConfigParsing:
|
||||
cfg["persona_memory"] = "True"
|
||||
with pytest.raises(ValueError, match="persona_memory"):
|
||||
snapshot_from_config(cfg)
|
||||
|
||||
|
||||
class _FakeStorage:
|
||||
"""Minimal storage double for resolve tests — exact-name index + list."""
|
||||
|
||||
def __init__(self, rows: list[dict]) -> None:
|
||||
self._rows = rows
|
||||
|
||||
def get_persona_by_name(self, name: str) -> dict | None:
|
||||
return next((dict(r) for r in self._rows if r["name"] == name), None)
|
||||
|
||||
def list_personas(self, include_disabled: bool = False) -> list[dict]:
|
||||
return [dict(r) for r in self._rows if include_disabled or r.get("enabled")]
|
||||
|
||||
|
||||
def _rows() -> list[dict]:
|
||||
return [
|
||||
{
|
||||
"name": "engineer",
|
||||
"display_name": "Engineer",
|
||||
"enabled": True,
|
||||
"is_default": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
{
|
||||
"name": "writer",
|
||||
"display_name": "Creative Writer",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
{
|
||||
"name": "executive",
|
||||
"display_name": "Executive",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
},
|
||||
{
|
||||
"name": "retired",
|
||||
"display_name": "Retired Persona",
|
||||
"enabled": False,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
class TestForgivingResolution:
|
||||
"""resolve_persona_for_kind — one shared rule, forgiving on all surfaces.
|
||||
|
||||
Exact slug first, then the lowercased input, then a UNIQUE
|
||||
case-insensitive display-name match; every failure enumerates the
|
||||
kind's live names (the self-correction path for stale tool
|
||||
descriptions), and callers stamp the returned row's canonical slug.
|
||||
"""
|
||||
|
||||
def _resolve(self, name: str, kind: str = "interactive", rows: list[dict] | None = None):
|
||||
from turnstone.core.personas import resolve_persona_for_kind
|
||||
|
||||
return resolve_persona_for_kind(_FakeStorage(rows or _rows()), name, kind)
|
||||
|
||||
def test_exact_slug_resolves(self) -> None:
|
||||
row, err = self._resolve("writer")
|
||||
assert err == "" and row is not None and row["name"] == "writer"
|
||||
|
||||
def test_case_variants_resolve_to_canonical_row(self) -> None:
|
||||
for variant in ("Writer", "WRITER", " writer "):
|
||||
row, err = self._resolve(variant)
|
||||
assert err == "" and row is not None and row["name"] == "writer"
|
||||
|
||||
def test_unique_display_name_resolves_to_slug(self) -> None:
|
||||
for variant in ("Creative Writer", "creative writer"):
|
||||
row, err = self._resolve(variant)
|
||||
assert err == "" and row is not None and row["name"] == "writer"
|
||||
|
||||
def test_ambiguous_display_name_names_the_candidates(self) -> None:
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "novelist",
|
||||
"display_name": "creative writer",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
]
|
||||
row, err = self._resolve("Creative Writer", rows=rows)
|
||||
assert row is None
|
||||
assert "more than one display name" in err
|
||||
assert "novelist" in err and "writer" in err
|
||||
assert "use the exact name" in err
|
||||
|
||||
def test_same_display_name_across_kinds_resolves_per_kind(self) -> None:
|
||||
# The label the caller saw came from a kind-filtered surface, so a
|
||||
# same-label persona of the OTHER kind must neither block (spurious
|
||||
# ambiguity) nor win (cross-kind resolution).
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "helper-coord",
|
||||
"display_name": "Helper",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
},
|
||||
{
|
||||
"name": "helper-int",
|
||||
"display_name": "Helper",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
]
|
||||
row, err = self._resolve("Helper", rows=rows)
|
||||
assert err == "" and row is not None and row["name"] == "helper-int"
|
||||
row, err = self._resolve("Helper", kind="coordinator", rows=rows)
|
||||
assert err == "" and row is not None and row["name"] == "helper-coord"
|
||||
|
||||
def test_wrong_kind_display_match_is_not_found_with_choices(self) -> None:
|
||||
# Display names are labels, not identifiers: a label that only exists
|
||||
# on another kind's persona reads as unknown for THIS kind (with the
|
||||
# kind's live choices attached) — never as a cross-kind resolution.
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "chief",
|
||||
"display_name": "The Chief",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
}
|
||||
]
|
||||
row, err = self._resolve("The Chief", rows=rows)
|
||||
assert row is None
|
||||
assert "not found or disabled" in err
|
||||
assert "Available for interactive: engineer (default), writer" in err
|
||||
|
||||
def test_whitespace_input_never_matches_blank_display_names(self) -> None:
|
||||
# display_name defaults to "" — a whitespace-only input (reachable via
|
||||
# CLI `--persona " "`) must read as unknown, never resolve to a
|
||||
# blank-labelled persona or report a bogus ambiguity.
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "unlabelled",
|
||||
"display_name": "",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
{
|
||||
"name": "unlabelled-too",
|
||||
"display_name": " ",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
]
|
||||
for raw in ("", " ", " "):
|
||||
row, err = self._resolve(raw, rows=rows)
|
||||
assert row is None
|
||||
assert "not found or disabled" in err
|
||||
assert "more than one display name" not in err
|
||||
|
||||
def test_unknown_error_lists_kind_names_default_first(self) -> None:
|
||||
row, err = self._resolve("nope")
|
||||
assert row is None
|
||||
assert "Persona not found or disabled: 'nope'" in err
|
||||
assert "Available for interactive: engineer (default), writer" in err
|
||||
assert "executive" not in err # wrong kind
|
||||
assert "retired" not in err # disabled
|
||||
|
||||
def test_kind_mismatch_reports_canonical_slug_and_choices(self) -> None:
|
||||
row, err = self._resolve("Executive") # case-forgiven, then kind-refused
|
||||
assert row is None
|
||||
assert "'executive' does not apply to kind 'interactive'" in err
|
||||
assert "Available for interactive: engineer (default), writer" in err
|
||||
|
||||
def test_disabled_persona_is_not_resolvable_by_any_route(self) -> None:
|
||||
for variant in ("retired", "RETIRED", "Retired Persona"):
|
||||
row, err = self._resolve(variant)
|
||||
assert row is None
|
||||
assert "not found or disabled" in err
|
||||
|
||||
def test_storage_none_is_a_distinct_error(self) -> None:
|
||||
from turnstone.core.personas import resolve_persona_for_kind
|
||||
|
||||
row, err = resolve_persona_for_kind(None, "writer", "interactive")
|
||||
assert row is None and err == "persona storage unavailable"
|
||||
|
||||
def test_listing_failure_degrades_to_plain_error(self) -> None:
|
||||
class _Broken(_FakeStorage):
|
||||
def list_personas(self, include_disabled: bool = False) -> list[dict]:
|
||||
raise RuntimeError("db gone")
|
||||
|
||||
from turnstone.core.personas import resolve_persona_for_kind
|
||||
|
||||
row, err = resolve_persona_for_kind(_Broken(_rows()), "nope", "interactive")
|
||||
assert row is None
|
||||
assert "Persona not found or disabled: 'nope'" in err
|
||||
|
||||
@@ -179,10 +179,10 @@ def test_bulk_live_admin_bypass_returns_live(storage):
|
||||
|
||||
|
||||
def test_bulk_live_cluster_wide_visibility(storage):
|
||||
"""Trusted-team visibility: any ``admin.cluster.inspect`` caller
|
||||
sees every row in ``results``. ``denied`` is reserved for ids
|
||||
that don't correspond to a persisted workstream (no existence
|
||||
oracle for unknown ids)."""
|
||||
"""A project-less workstream has no tenancy to enforce, so any
|
||||
``admin.cluster.inspect`` caller sees it in ``results``. ``denied``
|
||||
is reserved for ids that don't correspond to a persisted workstream
|
||||
(no existence oracle for unknown ids)."""
|
||||
ws_id = "b" * 32
|
||||
_seed_workstream(storage, ws_id=ws_id, node_id="node-a", user_id="stranger")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage))
|
||||
@@ -196,6 +196,48 @@ def test_bulk_live_cluster_wide_visibility(storage):
|
||||
assert body["denied"] == []
|
||||
|
||||
|
||||
def test_bulk_live_private_project_row_routes_to_denied(storage):
|
||||
"""A workstream in a private project the caller isn't a member of
|
||||
routes to ``denied``, not ``results`` — a cluster admin gets no
|
||||
private-project oracle from the bulk surface either."""
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
ws_id = "c" * 32
|
||||
storage.register_workstream(ws_id, node_id="node-a", user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage))
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/live?ids={ws_id}",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["results"] == {}
|
||||
assert body["denied"] == [ws_id]
|
||||
|
||||
|
||||
def test_bulk_live_private_project_row_visible_to_member(storage):
|
||||
"""A project member sees the row (routes to ``results``); the live
|
||||
block is null only because the coordinator row isn't loaded."""
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
ws_id = "c" * 32
|
||||
storage.register_workstream(
|
||||
ws_id,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage))
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/live?ids={ws_id}",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert ws_id in body["results"]
|
||||
assert body["denied"] == []
|
||||
|
||||
|
||||
def test_bulk_live_unknown_ids_route_to_denied(storage):
|
||||
"""Unknown ids (not in storage) land in ``denied`` so the endpoint
|
||||
can't be used as an existence oracle."""
|
||||
@@ -228,34 +270,42 @@ def test_bulk_live_coordinator_row_uses_manager_snapshot(storage):
|
||||
live = body["results"][ws.id]
|
||||
assert live is not None
|
||||
assert "pending_approval" in live
|
||||
# New field always present on the wire — None when no approval
|
||||
# is pending so the JS can `key in row` without surprise.
|
||||
assert "pending_approval_detail" in live
|
||||
assert live["pending_approval_detail"] is None
|
||||
# The details list is always present on the wire — empty when no
|
||||
# approval is pending so the JS can `key in row` without surprise.
|
||||
# Replaces 1.6's singular ``pending_approval_detail`` null
|
||||
# (breaking, 1.7).
|
||||
assert "pending_approval_details" in live
|
||||
assert live["pending_approval_details"] == []
|
||||
|
||||
|
||||
def test_bulk_live_coordinator_row_includes_pending_approval_detail(storage):
|
||||
"""When _pending_approval is set on a coord UI, the live block
|
||||
surfaces the merged items + judge_verdict payload through the
|
||||
coord-pseudo-node path. End-to-end equivalent of the dashboard
|
||||
test in test_server_authz, but for the console live-bulk
|
||||
endpoint that the coord tree UI actually consumes."""
|
||||
def test_bulk_live_coordinator_row_includes_pending_approval_details(storage):
|
||||
"""When an approval cycle is live on a coord UI, the live block
|
||||
surfaces one detail entry per cycle with merged items +
|
||||
judge_verdict through the coord-pseudo-node path. End-to-end
|
||||
equivalent of the dashboard test in test_server_authz, but for
|
||||
the console live-bulk endpoint that the coord tree UI actually
|
||||
consumes."""
|
||||
from turnstone.core.session_ui_base import ApprovalCycle
|
||||
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
ws.ui._pending_approval = {
|
||||
items = [
|
||||
{
|
||||
"call_id": "c-99",
|
||||
"header": "spawn_workstream",
|
||||
"preview": "{...}",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
]
|
||||
card = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-99",
|
||||
"header": "spawn_workstream",
|
||||
"preview": "{...}",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
],
|
||||
"cycle_id": "cyc-99",
|
||||
"items": ws.ui._serialize_approval_items(items),
|
||||
"judge_pending": False,
|
||||
}
|
||||
ws.ui._register_approval_cycle(ApprovalCycle(items, card, None))
|
||||
ws.ui._llm_verdicts["c-99"] = {
|
||||
"recommendation": "approve",
|
||||
"risk_level": "low",
|
||||
@@ -269,8 +319,10 @@ def test_bulk_live_coordinator_row_includes_pending_approval_detail(storage):
|
||||
assert resp.status_code == 200
|
||||
live = resp.json()["results"][ws.id]
|
||||
assert live["pending_approval"] is True # boolean derived flag
|
||||
detail = live["pending_approval_detail"]
|
||||
assert detail is not None
|
||||
details = live["pending_approval_details"]
|
||||
assert len(details) == 1
|
||||
detail = details[0]
|
||||
assert detail["cycle_id"] == "cyc-99"
|
||||
assert detail["call_id"] == "c-99"
|
||||
assert detail["items"][0]["func_name"] == "spawn_workstream"
|
||||
assert detail["items"][0]["judge_verdict"]["recommendation"] == "approve"
|
||||
|
||||
@@ -127,10 +127,14 @@ class TestWsVisiblePredicate:
|
||||
assert storage.get_project.call_count == 1
|
||||
|
||||
def test_for_request_bypass_rules(self) -> None:
|
||||
# Only service scope bypasses (node→console machine plumbing,
|
||||
# re-filtered per-user at the console edge).
|
||||
assert WorkstreamProjectVisibility.for_request(
|
||||
_request_for("bob", scopes=("service",))
|
||||
)._bypass
|
||||
assert WorkstreamProjectVisibility.for_request(
|
||||
# admin.cluster.inspect gates the inspect *surfaces* but does NOT
|
||||
# bypass private-project tenancy — the admin filters as themselves.
|
||||
assert not WorkstreamProjectVisibility.for_request(
|
||||
_request_for("bob", permissions=("admin.cluster.inspect",))
|
||||
)._bypass
|
||||
assert not WorkstreamProjectVisibility.for_request(_request_for("bob"))._bypass
|
||||
@@ -214,15 +218,17 @@ class TestResolveWorkstreamOwnerProjectGate:
|
||||
assert err is None
|
||||
assert owner == "bob"
|
||||
|
||||
def test_admin_inspect_bypasses(self, tmp_db: str) -> None:
|
||||
def test_admin_inspect_does_not_bypass(self, tmp_db: str) -> None:
|
||||
# A permitted admin (admin.cluster.inspect) who isn't the owner /
|
||||
# creator / member of a private project is still 403'd at the row
|
||||
# gate — the permission gates the inspect surface, not the tenancy.
|
||||
from turnstone.core.web_helpers import resolve_workstream_owner
|
||||
|
||||
self._seed(member=False)
|
||||
owner, err = resolve_workstream_owner(
|
||||
_request_for("bob", permissions=("admin.cluster.inspect",)), "ws-priv"
|
||||
)
|
||||
assert err is None
|
||||
assert owner == "alice"
|
||||
assert err is not None and err.status_code == 403
|
||||
|
||||
def test_missing_ws_still_404s(self, tmp_db: str) -> None:
|
||||
from turnstone.core.web_helpers import resolve_workstream_owner
|
||||
|
||||
@@ -88,10 +88,9 @@ def _make_session(**kwargs):
|
||||
|
||||
|
||||
def _sys_content(session: ChatSession) -> str:
|
||||
"""Extract the system message content."""
|
||||
msgs = [m for m in session.system_messages if m["role"] == "system"]
|
||||
assert msgs
|
||||
return msgs[0]["content"]
|
||||
"""Full prompt prefix: identity system message + any skill context message."""
|
||||
assert session.system_messages
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
def _create_template(db, template_id, name, content, is_default=False, **kwargs):
|
||||
@@ -187,17 +186,23 @@ class TestDefaultTemplates:
|
||||
content = _sys_content(session)
|
||||
assert "Not default." not in content
|
||||
|
||||
def test_templates_before_instructions(self, tmp_db):
|
||||
def test_default_template_stays_in_identity_system_message(self, tmp_db):
|
||||
"""Default (always-on) templates are the standing baseline and never
|
||||
change mid-session, so they stay in the identity system message (with
|
||||
user instructions) — only a NAMED applied skill moves to a separate
|
||||
context message. Template guidance still precedes user instructions."""
|
||||
from turnstone.core.storage import get_storage
|
||||
|
||||
db = get_storage()
|
||||
_create_template(db, "t1", "tpl", "TEMPLATE_CONTENT", is_default=True)
|
||||
|
||||
session = _make_session(instructions="USER_INSTRUCTIONS")
|
||||
content = _sys_content(session)
|
||||
tpl_pos = content.index("TEMPLATE_CONTENT")
|
||||
instr_pos = content.index("USER_INSTRUCTIONS")
|
||||
assert tpl_pos < instr_pos
|
||||
msgs = session.system_messages
|
||||
# Both the default template and instructions live in the system message,
|
||||
# template first — and no default triggers a user-role context message.
|
||||
assert all(m["role"] == "system" for m in msgs)
|
||||
content = msgs[0]["content"]
|
||||
assert content.index("TEMPLATE_CONTENT") < content.index("USER_INSTRUCTIONS")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -12,6 +12,7 @@ toggle.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
import os
|
||||
import sys
|
||||
from typing import Any
|
||||
@@ -21,6 +22,7 @@ import pytest
|
||||
|
||||
from tests._session_helpers import make_session as _make_session
|
||||
from turnstone.core.providers._anthropic import AnthropicProvider
|
||||
from turnstone.core.providers._protocol import ModelCapabilities
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
@@ -166,6 +168,178 @@ class TestCompatWireShape:
|
||||
assert kwargs["thinking"] == {"type": "enabled", "budget_tokens": 2048}
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# TestCompatReasoningControl
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
class TestCompatReasoningControl:
|
||||
"""Session effort knob → ``chat_template_kwargs`` on the compat lane.
|
||||
|
||||
vLLM's ``/v1/messages`` ignores the native ``thinking`` param — the
|
||||
reasoning levers live in the chat template.
|
||||
``merge_reasoning_template_kwargs`` maps the knob onto
|
||||
``caps.thinking_param`` (manual: knob "none" = off, mirroring
|
||||
``_reasoning_params``; adaptive: always on) and ``caps.effort_param``
|
||||
(graded value for gpt-oss-style templates). Verified live against
|
||||
qwen3.6 on vLLM 2026-07-03: ``{"enable_thinking": false}`` disables
|
||||
thinking, unknown chat_template_kwargs keys are silently ignored.
|
||||
"""
|
||||
|
||||
_MANUAL_CAPS = ModelCapabilities(
|
||||
token_param="max_tokens",
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
)
|
||||
|
||||
def setup_method(self) -> None:
|
||||
self.provider = AnthropicProvider(compat=True)
|
||||
|
||||
def _stream_kwargs(
|
||||
self,
|
||||
caps: ModelCapabilities | None,
|
||||
reasoning_effort: str,
|
||||
extra_params: dict[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
client = _capture_client()
|
||||
with patch("turnstone.core.providers._anthropic._ensure_anthropic"):
|
||||
list(
|
||||
self.provider.create_streaming(
|
||||
client=client,
|
||||
model="qwen3.6-27b",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
temperature=0.6,
|
||||
reasoning_effort=reasoning_effort,
|
||||
extra_params=extra_params,
|
||||
capabilities=caps,
|
||||
)
|
||||
)
|
||||
return client.messages.stream.call_args[1]
|
||||
|
||||
def test_manual_toggle_on(self) -> None:
|
||||
"""Any non-none effort turns the toggle on AND carries the graded
|
||||
value under the fallback key — the user's effort setting always
|
||||
reaches the wire; a template that doesn't reference the kwarg
|
||||
ignores it."""
|
||||
kwargs = self._stream_kwargs(self._MANUAL_CAPS, "medium")
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "medium"}
|
||||
}
|
||||
assert "thinking" not in kwargs
|
||||
assert kwargs["temperature"] == 0.6 # never forced to 1.0 on compat
|
||||
|
||||
@pytest.mark.parametrize("knob", ["none", ""])
|
||||
def test_manual_toggle_off(self, knob: str) -> None:
|
||||
"""Effort "none"/empty disables thinking — native manual-mode parity;
|
||||
no effort key rides when thinking is off."""
|
||||
kwargs = self._stream_kwargs(self._MANUAL_CAPS, knob)
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
assert "thinking" not in kwargs
|
||||
|
||||
def test_adaptive_always_on(self) -> None:
|
||||
"""Adaptive never knob-disables — native-adaptive contract, no native
|
||||
dict; the graded value rides for on-positions only."""
|
||||
caps = dataclasses.replace(self._MANUAL_CAPS, thinking_mode="adaptive")
|
||||
kwargs = self._stream_kwargs(caps, "high")
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "high"}
|
||||
}
|
||||
assert "thinking" not in kwargs
|
||||
assert kwargs["temperature"] == 0.6
|
||||
kwargs = self._stream_kwargs(caps, "none")
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": True}}
|
||||
assert "thinking" not in kwargs
|
||||
assert kwargs["temperature"] == 0.6
|
||||
|
||||
def test_default_caps_inject_nothing(self) -> None:
|
||||
"""Untouched compat defaults (thinking_mode=none) keep today's wire."""
|
||||
kwargs = self._stream_kwargs(None, "medium")
|
||||
assert "extra_body" not in kwargs
|
||||
assert "thinking" not in kwargs
|
||||
|
||||
def test_effort_param_validated_against_values(self) -> None:
|
||||
"""Off-list knob rounds up onto the declared values (ceiling-capped),
|
||||
never sent raw — and never snaps DOWN to the default."""
|
||||
caps = dataclasses.replace(
|
||||
self._MANUAL_CAPS,
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "xhigh")
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "high"}
|
||||
}
|
||||
|
||||
def test_effort_param_freeform_without_values(self) -> None:
|
||||
"""No declared values → knob forwarded as-is, template is authority."""
|
||||
caps = ModelCapabilities(
|
||||
token_param="max_tokens",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "xhigh")
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"reasoning_effort": "xhigh"}}
|
||||
|
||||
def test_effort_param_omitted_on_none(self) -> None:
|
||||
"""Knob "none" sends no effort key (and toggles thinking off)."""
|
||||
caps = dataclasses.replace(
|
||||
self._MANUAL_CAPS,
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "none")
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
def test_operator_override_wins(self) -> None:
|
||||
"""server_compat chat_template_kwargs entries beat the knob mapping."""
|
||||
kwargs = self._stream_kwargs(
|
||||
self._MANUAL_CAPS,
|
||||
"none",
|
||||
extra_params={"chat_template_kwargs": {"enable_thinking": True}, "foo": 1},
|
||||
)
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True},
|
||||
"foo": 1,
|
||||
}
|
||||
|
||||
def test_caller_extra_params_not_mutated(self) -> None:
|
||||
"""The session's extra_params dict must never be written through."""
|
||||
extra = {"chat_template_kwargs": {"foo": 1}}
|
||||
self._stream_kwargs(self._MANUAL_CAPS, "medium", extra_params=extra)
|
||||
assert extra == {"chat_template_kwargs": {"foo": 1}}
|
||||
|
||||
def test_no_output_config_on_compat(self) -> None:
|
||||
"""supports_effort must not leak Anthropic output_config to vLLM."""
|
||||
caps = dataclasses.replace(
|
||||
self._MANUAL_CAPS,
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "high")
|
||||
assert "output_config" not in kwargs
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "high"}
|
||||
}
|
||||
|
||||
def test_create_completion_same_injection(self) -> None:
|
||||
"""The non-streaming path shares _build_thinking_and_kwargs."""
|
||||
client = MagicMock()
|
||||
final = MagicMock(content=[], stop_reason="end_turn")
|
||||
stream = MagicMock()
|
||||
stream.get_final_message.return_value = final
|
||||
client.messages.stream.return_value.__enter__.return_value = stream
|
||||
with patch("turnstone.core.providers._anthropic._ensure_anthropic"):
|
||||
self.provider.create_completion(
|
||||
client=client,
|
||||
model="qwen3.6-27b",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
reasoning_effort="none",
|
||||
capabilities=self._MANUAL_CAPS,
|
||||
)
|
||||
kwargs = client.messages.stream.call_args[1]
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# TestCompatFactory
|
||||
# ===========================================================================
|
||||
|
||||
+200
-81
@@ -12,10 +12,12 @@ from turnstone.core.lowering import repair_wire_messages
|
||||
from turnstone.core.providers._openai import OpenAIProvider
|
||||
from turnstone.core.providers._openai_chat import OpenAIChatCompletionsProvider
|
||||
from turnstone.core.providers._openai_common import (
|
||||
OPENAI_COMPAT_DEFAULT,
|
||||
apply_cache_retention,
|
||||
apply_temperature_and_effort,
|
||||
apply_tool_search,
|
||||
format_citations,
|
||||
lookup_openai_capabilities,
|
||||
sanitize_messages,
|
||||
)
|
||||
from turnstone.core.providers._protocol import (
|
||||
@@ -153,51 +155,89 @@ class TestOpenAIProvider:
|
||||
def test_provider_name(self) -> None:
|
||||
assert self.provider.provider_name == "openai-compatible"
|
||||
|
||||
# -- _apply_thinking_mode -------------------------------------------------
|
||||
# -- reasoning template kwargs (_finalize_extra_body) ---------------------
|
||||
|
||||
def test_thinking_mode_none_does_nothing(self) -> None:
|
||||
"""No thinking params injected when thinking_mode is 'none'."""
|
||||
"""No toggle injected when thinking_mode is 'none'; operator keys pass."""
|
||||
caps = ModelCapabilities(thinking_mode="none")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert "enable_thinking" not in extra_body["chat_template_kwargs"]
|
||||
extra_params = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
eb = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert eb is not None
|
||||
assert "enable_thinking" not in eb["chat_template_kwargs"]
|
||||
assert eb["chat_template_kwargs"]["reasoning_effort"] == "medium"
|
||||
|
||||
def test_thinking_mode_manual_injects_param(self) -> None:
|
||||
"""Manual thinking mode injects enable_thinking into chat_template_kwargs."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is True
|
||||
assert extra_body["chat_template_kwargs"]["reasoning_effort"] == "medium"
|
||||
extra_params = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
eb = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert eb is not None
|
||||
assert eb["chat_template_kwargs"]["enable_thinking"] is True
|
||||
assert eb["chat_template_kwargs"]["reasoning_effort"] == "medium"
|
||||
|
||||
def test_thinking_mode_manual_knob_none_disables(self) -> None:
|
||||
"""Effort knob "none" turns the template toggle off, not just quiet."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
eb = self.provider._finalize_extra_body(None, caps, "none")
|
||||
assert eb == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
def test_thinking_mode_custom_param(self) -> None:
|
||||
"""Custom thinking_param (e.g. Granite's 'thinking') is used."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="thinking")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["thinking"] is True
|
||||
assert "enable_thinking" not in extra_body["chat_template_kwargs"]
|
||||
eb = self.provider._finalize_extra_body(None, caps, "medium")
|
||||
assert eb == {"chat_template_kwargs": {"thinking": True}}
|
||||
|
||||
def test_thinking_mode_does_not_override_explicit(self) -> None:
|
||||
"""If operator explicitly set the param to False, provider respects it."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is False
|
||||
extra_params = {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
eb = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert eb is not None
|
||||
assert eb["chat_template_kwargs"]["enable_thinking"] is False
|
||||
|
||||
def test_thinking_mode_creates_ctk_if_missing(self) -> None:
|
||||
"""Creates chat_template_kwargs dict if not present in extra_body."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_body: dict[str, Any] = {}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is True
|
||||
|
||||
def test_thinking_mode_adaptive(self) -> None:
|
||||
"""Adaptive thinking mode also injects the param."""
|
||||
def test_thinking_mode_adaptive_never_knob_disables(self) -> None:
|
||||
"""Adaptive = model self-regulates; knob "none" must not force false."""
|
||||
caps = ModelCapabilities(thinking_mode="adaptive")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is True
|
||||
for knob in ("high", "none", ""):
|
||||
eb = self.provider._finalize_extra_body(None, caps, knob)
|
||||
assert eb == {"chat_template_kwargs": {"enable_thinking": True}}
|
||||
|
||||
def test_effort_param_suppresses_flat_reasoning_effort(self) -> None:
|
||||
"""Declaring the ctk effort channel must not double-send the flat param."""
|
||||
from turnstone.core.providers._openai_common import apply_temperature_and_effort
|
||||
|
||||
caps = ModelCapabilities(
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
)
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, 0.5, "medium")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
# Without effort_param the flat param still flows (commercial path).
|
||||
flat_caps = ModelCapabilities(reasoning_effort_values=("low", "medium", "high"))
|
||||
kwargs = {}
|
||||
apply_temperature_and_effort(kwargs, flat_caps, 0.5, "medium")
|
||||
assert kwargs["reasoning_effort"] == "medium"
|
||||
|
||||
def test_effort_param_injects_knob_value(self) -> None:
|
||||
"""effort_param carries the knob into chat_template_kwargs (gpt-oss);
|
||||
a knob above the declared ceiling rides the ceiling, not the default."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eb = self.provider._finalize_extra_body(None, caps, "xhigh")
|
||||
assert eb == {"chat_template_kwargs": {"reasoning_effort": "high"}}
|
||||
assert self.provider._finalize_extra_body(None, caps, "none") is None
|
||||
|
||||
def test_caller_extra_params_not_mutated(self) -> None:
|
||||
"""The session dict and its ctk sub-dict survive injection untouched."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_params = {"chat_template_kwargs": {"foo": 1}}
|
||||
self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert extra_params == {"chat_template_kwargs": {"foo": 1}}
|
||||
|
||||
# -- _sanitize_messages ---------------------------------------------------
|
||||
|
||||
@@ -1633,6 +1673,25 @@ class TestProviderFactory:
|
||||
assert openai_prov.provider_name == "openai"
|
||||
assert compat.provider_name == "openai-compatible"
|
||||
|
||||
def test_openai_compatible_never_consults_commercial_registry(self) -> None:
|
||||
"""Local-lane model ids are operator-chosen strings — a prefix
|
||||
collision with a cloud model id must not inherit that model's
|
||||
sampling/effort contract, on either API surface. Cloud lookups
|
||||
are unaffected."""
|
||||
from turnstone.core.providers import create_provider
|
||||
|
||||
compat = create_provider("openai-compatible")
|
||||
compat_responses = create_provider("openai-compatible", api_surface="responses")
|
||||
for name in ("gpt-5.5-my-finetune", "o3-distill", "deepseek-v4-flash", ""):
|
||||
assert compat.get_capabilities(name) is OPENAI_COMPAT_DEFAULT
|
||||
assert compat_responses.get_capabilities(name) is OPENAI_COMPAT_DEFAULT
|
||||
# The commercial lane keeps resolving its registry rows — through
|
||||
# the factory AND through the non-compat class default.
|
||||
cloud = create_provider("openai").get_capabilities("gpt-5.5")
|
||||
assert cloud.default_reasoning_effort == "medium"
|
||||
assert "xhigh" in cloud.reasoning_effort_values
|
||||
assert create_provider("openai") is not compat_responses
|
||||
|
||||
def test_create_provider_returns_singleton(self) -> None:
|
||||
from turnstone.core.providers import create_provider
|
||||
|
||||
@@ -1765,6 +1824,39 @@ class TestProviderFactory:
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
class TestGoogleEffortKnob:
|
||||
"""The session effort knob reaches Gemini as a flat reasoning_effort."""
|
||||
|
||||
def _create_kwargs(self, reasoning_effort: str) -> dict[str, Any]:
|
||||
from turnstone.core.providers._google import GoogleProvider
|
||||
|
||||
prov = GoogleProvider()
|
||||
client = MagicMock()
|
||||
client.chat.completions.create.return_value = iter([])
|
||||
list(
|
||||
prov.create_streaming(
|
||||
client=client,
|
||||
model="gemini-3-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
reasoning_effort=reasoning_effort,
|
||||
)
|
||||
)
|
||||
return client.chat.completions.create.call_args[1]
|
||||
|
||||
def test_knob_values_forward_verbatim(self) -> None:
|
||||
for knob in ("minimal", "low", "medium", "high"):
|
||||
assert self._create_kwargs(knob)["reasoning_effort"] == knob
|
||||
|
||||
def test_off_list_knob_snaps_to_high(self) -> None:
|
||||
"""xhigh/max are not in Gemini's vocabulary — snap down to high."""
|
||||
for knob in ("xhigh", "max"):
|
||||
assert self._create_kwargs(knob)["reasoning_effort"] == "high"
|
||||
|
||||
def test_none_omits_the_param(self) -> None:
|
||||
"""Knob none never sends "none" — 2.5 Pro / 3.x reject disabling."""
|
||||
assert "reasoning_effort" not in self._create_kwargs("none")
|
||||
|
||||
|
||||
class TestGoogleProviderFidelity:
|
||||
"""Tests for thought_signature round-trip via provider_blocks."""
|
||||
|
||||
@@ -2003,49 +2095,64 @@ class TestOpenAIParameterGating:
|
||||
def setup_method(self) -> None:
|
||||
self.provider = OpenAIProvider()
|
||||
|
||||
def test_unknown_model_no_reasoning_effort(self) -> None:
|
||||
"""Unknown/local models should NOT receive top-level reasoning_effort."""
|
||||
def test_local_model_effort_forwarded_verbatim(self) -> None:
|
||||
"""Local-lane models receive the session knob verbatim on the flat
|
||||
param (effort_passthrough) — the user's effort setting always
|
||||
reaches the wire; "none" stays omitted (nothing to disable
|
||||
beyond the template toggle)."""
|
||||
caps = self.provider.get_capabilities("my-local-model")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "medium"
|
||||
assert kwargs["temperature"] == 0.7
|
||||
kwargs = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
|
||||
def test_gpt5_no_temperature_has_reasoning_effort(self) -> None:
|
||||
"""GPT-5 base: no temperature, reasoning_effort sent."""
|
||||
caps = self.provider.get_capabilities("gpt-5")
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="high")
|
||||
assert "temperature" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_gpt51_temperature_when_effort_none(self) -> None:
|
||||
"""GPT-5.1: temperature only when reasoning_effort='none'."""
|
||||
caps = self.provider.get_capabilities("gpt-5.1")
|
||||
"""GPT-5.1: temperature only when reasoning_effort='none'; the
|
||||
declared "none" level is forwarded explicitly (knob = off)."""
|
||||
caps = lookup_openai_capabilities("gpt-5.1")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert kwargs["temperature"] == 0.7
|
||||
assert "reasoning_effort" not in kwargs # "none" is skipped
|
||||
assert kwargs["reasoning_effort"] == "none"
|
||||
|
||||
def test_gpt51_no_temperature_when_reasoning_active(self) -> None:
|
||||
"""GPT-5.1: no temperature when reasoning is active."""
|
||||
caps = self.provider.get_capabilities("gpt-5.1")
|
||||
caps = lookup_openai_capabilities("gpt-5.1")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="high")
|
||||
assert "temperature" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_o_series_no_temperature_no_reasoning_effort(self) -> None:
|
||||
"""O-series: no temperature, no reasoning_effort."""
|
||||
caps = self.provider.get_capabilities("o3")
|
||||
def test_o_series_no_temperature_but_effort_forwarded(self) -> None:
|
||||
"""O-series: no temperature; low/medium/high ARE valid effort
|
||||
values (all o-series except o1-mini) and the knob reaches them."""
|
||||
caps = lookup_openai_capabilities("o3")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "temperature" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "medium"
|
||||
|
||||
def test_o1_mini_has_no_effort_control(self) -> None:
|
||||
"""o1-mini is the one o-series model without reasoning_effort."""
|
||||
caps = lookup_openai_capabilities("o1-mini")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
|
||||
def test_gpt5_pro_unsupported_effort_falls_back(self) -> None:
|
||||
"""GPT-5 pro only supports 'high'; unsupported values fall back to default."""
|
||||
caps = self.provider.get_capabilities("gpt-5-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5-pro")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "temperature" not in kwargs
|
||||
@@ -2053,19 +2160,19 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt5_pro_supported_effort_passes_through(self) -> None:
|
||||
"""GPT-5 pro accepts 'high' directly."""
|
||||
caps = self.provider.get_capabilities("gpt-5-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5-pro")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="high")
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_gpt54_1m_context_and_effort(self) -> None:
|
||||
"""GPT-5.4: 1M context, temperature when effort=none, xhigh supported."""
|
||||
caps = self.provider.get_capabilities("gpt-5.4")
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
assert caps.context_window == 1050000
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert kwargs["temperature"] == 0.7
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "none" # declared level, forwarded
|
||||
kwargs2: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs2, caps, temperature=0.7, reasoning_effort="xhigh")
|
||||
assert "temperature" not in kwargs2
|
||||
@@ -2073,7 +2180,7 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt54_pro_no_temperature_always_reasoning(self) -> None:
|
||||
"""GPT-5.4 pro: no temperature, medium/high/xhigh only."""
|
||||
caps = self.provider.get_capabilities("gpt-5.4-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5.4-pro")
|
||||
assert caps.context_window == 1050000
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="low")
|
||||
@@ -2082,14 +2189,14 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt55_1m_context_and_effort(self) -> None:
|
||||
"""GPT-5.5: 1M context, temperature when effort=none, xhigh supported."""
|
||||
caps = self.provider.get_capabilities("gpt-5.5")
|
||||
caps = lookup_openai_capabilities("gpt-5.5")
|
||||
assert caps.context_window == 1050000
|
||||
assert caps.supports_tool_search is True
|
||||
assert caps.supports_vision is True
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert kwargs["temperature"] == 0.7
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "none" # declared level, forwarded
|
||||
kwargs2: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs2, caps, temperature=0.7, reasoning_effort="xhigh")
|
||||
assert "temperature" not in kwargs2
|
||||
@@ -2097,7 +2204,7 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt55_pro_no_temperature_always_reasoning(self) -> None:
|
||||
"""GPT-5.5 pro: no temperature, medium/high/xhigh only."""
|
||||
caps = self.provider.get_capabilities("gpt-5.5-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5.5-pro")
|
||||
assert caps.context_window == 1050000
|
||||
assert caps.supports_tool_search is True
|
||||
kwargs: dict[str, Any] = {}
|
||||
@@ -2316,11 +2423,20 @@ class TestAnthropicReasoningNone:
|
||||
result = _map_reasoning_to_effort("xhigh", ("low", "medium", "high", "xhigh", "max"))
|
||||
assert result == "xhigh"
|
||||
|
||||
def test_map_xhigh_rejected_by_model_without_it(self) -> None:
|
||||
def test_map_xhigh_snaps_up_through_gap_to_max(self) -> None:
|
||||
"""Levels with a hole (no xhigh) round the knob UP to the next
|
||||
declared level rather than dropping output_config entirely."""
|
||||
from turnstone.core.providers._anthropic import _map_reasoning_to_effort
|
||||
|
||||
result = _map_reasoning_to_effort("xhigh", ("low", "medium", "high", "max"))
|
||||
assert result is None
|
||||
assert result == "max"
|
||||
|
||||
def test_map_above_ceiling_rides_ceiling(self) -> None:
|
||||
from turnstone.core.providers._anthropic import _map_reasoning_to_effort
|
||||
|
||||
assert _map_reasoning_to_effort("max", ("low", "medium", "high")) == "high"
|
||||
assert _map_reasoning_to_effort("minimal", ("low", "medium", "high")) == "low"
|
||||
assert _map_reasoning_to_effort("none", ("low", "medium", "high")) is None
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
@@ -2704,19 +2820,19 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_search_model_capability(self) -> None:
|
||||
"""Search models should have supports_web_search=True."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
assert caps.supports_web_search is True
|
||||
|
||||
def test_non_search_model_no_web_search(self) -> None:
|
||||
"""Regular models should not have supports_web_search."""
|
||||
caps = self.provider.get_capabilities("gpt-5")
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
assert caps.supports_web_search is False
|
||||
caps = self.provider.get_capabilities("gpt-5.2")
|
||||
caps = lookup_openai_capabilities("gpt-5.2")
|
||||
assert caps.supports_web_search is False
|
||||
|
||||
def test_apply_web_search_injects_options(self) -> None:
|
||||
"""For search models, web_search_options should be added to kwargs."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {"model": "gpt-5-search-api"}
|
||||
tools: list[dict[str, Any]] = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run bash"}},
|
||||
@@ -2733,7 +2849,7 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_apply_web_search_no_op_for_regular_models(self) -> None:
|
||||
"""For non-search models, no web_search_options, tools unchanged."""
|
||||
caps = self.provider.get_capabilities("gpt-5")
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
kwargs: dict[str, Any] = {"model": "gpt-5"}
|
||||
tools: list[dict[str, Any]] = [
|
||||
{"type": "function", "function": {"name": "web_search", "description": "Search"}},
|
||||
@@ -2744,7 +2860,7 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_apply_web_search_returns_none_when_only_web_search(self) -> None:
|
||||
"""If web_search was the only tool, return None after removing it."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {}
|
||||
tools: list[dict[str, Any]] = [
|
||||
{"type": "function", "function": {"name": "web_search", "description": "Search"}},
|
||||
@@ -2758,7 +2874,7 @@ class TestOpenAIWebSearch:
|
||||
toolset) must NOT gain native search — the option stays off and the
|
||||
tools pass through untouched. Contrast test_apply_web_search_with_
|
||||
no_tools, which covers the tool-less utility-call case."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
assert caps.supports_web_search is True
|
||||
kwargs: dict[str, Any] = {"model": "gpt-5-search-api"}
|
||||
tools: list[dict[str, Any]] = [
|
||||
@@ -2833,7 +2949,7 @@ class TestOpenAIWebSearch:
|
||||
visibility set, coordinator toolset, or a tool-less utility call —
|
||||
must not gain native search at the provider layer.
|
||||
"""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {}
|
||||
result = self.provider._apply_web_search(kwargs, caps, None)
|
||||
assert "web_search_options" not in kwargs
|
||||
@@ -2841,7 +2957,7 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_apply_web_search_replaces_client_def(self) -> None:
|
||||
"""With the client def present, it is filtered and the option set."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {}
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}]
|
||||
result = self.provider._apply_web_search(kwargs, caps, tools)
|
||||
@@ -2867,6 +2983,10 @@ class TestOpenAIWebSearch:
|
||||
"function": {"name": "web_search", "description": "Search"},
|
||||
},
|
||||
],
|
||||
# The local lane resolves no commercial rows — the search
|
||||
# model's capabilities ride in explicitly, as the session
|
||||
# layer would pass them.
|
||||
capabilities=lookup_openai_capabilities("gpt-5-search-api"),
|
||||
)
|
||||
)
|
||||
call_kwargs = client.chat.completions.create.call_args[1]
|
||||
@@ -3287,22 +3407,18 @@ class TestAnthropicToolSearch:
|
||||
|
||||
|
||||
class TestOpenAIToolSearch:
|
||||
"""Test OpenAI provider tool search injection."""
|
||||
"""Test OpenAI tool search injection (registry rows + shared helper)."""
|
||||
|
||||
@pytest.fixture()
|
||||
def provider(self):
|
||||
return OpenAIProvider()
|
||||
|
||||
def test_tool_search_capability_on_gpt54(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5.4")
|
||||
def test_tool_search_capability_on_gpt54(self):
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
assert caps.supports_tool_search is True
|
||||
|
||||
def test_tool_search_not_supported_on_gpt5(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5")
|
||||
def test_tool_search_not_supported_on_gpt5(self):
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
assert caps.supports_tool_search is False
|
||||
|
||||
def test_apply_tool_search_marks_deferred(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5.4")
|
||||
def test_apply_tool_search_marks_deferred(self):
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
tools = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run commands"}},
|
||||
{
|
||||
@@ -3318,16 +3434,16 @@ class TestOpenAIToolSearch:
|
||||
# slack tool deferred
|
||||
assert result[1]["defer_loading"] is True
|
||||
|
||||
def test_apply_tool_search_no_op_without_deferred(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5.4")
|
||||
def test_apply_tool_search_no_op_without_deferred(self):
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
tools = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run commands"}},
|
||||
]
|
||||
result = apply_tool_search(caps, tools, None)
|
||||
assert result == tools
|
||||
|
||||
def test_apply_tool_search_no_op_on_unsupported_model(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5")
|
||||
def test_apply_tool_search_no_op_on_unsupported_model(self):
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
tools = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run commands"}},
|
||||
]
|
||||
@@ -3395,13 +3511,12 @@ class TestVisionCapabilities:
|
||||
assert caps.supports_vision is False
|
||||
|
||||
def test_openai_commercial_supports_vision(self) -> None:
|
||||
provider = OpenAIProvider()
|
||||
for model in ("gpt-5", "gpt-5-mini", "gpt-5.4", "o3", "o4-mini"):
|
||||
caps = provider.get_capabilities(model)
|
||||
caps = lookup_openai_capabilities(model)
|
||||
assert caps.supports_vision is True, f"{model} should support vision"
|
||||
|
||||
def test_openai_default_no_vision(self) -> None:
|
||||
"""Unknown models (local servers) default to no vision."""
|
||||
"""Local-lane models (any name) default to no vision."""
|
||||
provider = OpenAIProvider()
|
||||
caps = provider.get_capabilities("some-local-model")
|
||||
assert caps.supports_vision is False
|
||||
@@ -3626,8 +3741,9 @@ class TestAnthropicPromptCaching:
|
||||
)
|
||||
assert kwargs["output_config"] == {"effort": "xhigh"}
|
||||
|
||||
def test_xhigh_effort_not_applied_to_opus_4_6(self) -> None:
|
||||
"""xhigh is not a valid effort level for Opus 4.6 — should be ignored."""
|
||||
def test_xhigh_effort_snaps_to_max_on_opus_4_6(self) -> None:
|
||||
"""Opus 4.6 declares (low, medium, high, max) — a knob of xhigh
|
||||
rounds up to max instead of silently dropping output_config."""
|
||||
caps = self.provider.get_capabilities("claude-opus-4-6")
|
||||
kwargs = self.provider._build_thinking_and_kwargs(
|
||||
caps=caps,
|
||||
@@ -3640,7 +3756,7 @@ class TestAnthropicPromptCaching:
|
||||
model="claude-opus-4-6",
|
||||
tools=None,
|
||||
)
|
||||
assert "output_config" not in kwargs
|
||||
assert kwargs["output_config"] == {"effort": "max"}
|
||||
|
||||
@patch("turnstone.core.providers._anthropic._ensure_anthropic")
|
||||
def test_streaming_message_start_cache_metrics(self, mock_ensure: MagicMock) -> None:
|
||||
@@ -4173,7 +4289,10 @@ class TestResponsesParamBuilding:
|
||||
assert kwargs["reasoning"] == {"effort": "high"}
|
||||
assert "reasoning_effort" not in kwargs
|
||||
|
||||
def test_no_reasoning_when_none_effort(self) -> None:
|
||||
def test_none_effort_sends_declared_none_level(self) -> None:
|
||||
"""gpt-5.4 declares an explicit "none" level — the knob's off
|
||||
position forwards it rather than omitting (omission would leave
|
||||
the server default in charge on models like gpt-5.5)."""
|
||||
kwargs = self.provider._build_kwargs(
|
||||
model="gpt-5.4",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
@@ -4183,7 +4302,7 @@ class TestResponsesParamBuilding:
|
||||
reasoning_effort="none",
|
||||
deferred_names=None,
|
||||
)
|
||||
assert "reasoning" not in kwargs
|
||||
assert kwargs["reasoning"] == {"effort": "none"}
|
||||
|
||||
def test_store_is_false(self) -> None:
|
||||
kwargs = self.provider._build_kwargs(
|
||||
|
||||
+67
-21
@@ -57,6 +57,39 @@ def _auth(
|
||||
return {"Authorization": f"Bearer {_make_jwt(user, scopes=scopes, permissions=permissions)}"}
|
||||
|
||||
|
||||
class TestAssignableScopes:
|
||||
"""``service`` scope is a cross-tenant bypass and must never be
|
||||
GRANTED via a user-facing token mint (admin API or CLI) — otherwise an
|
||||
``admin.users`` holder could self-mint it and see every private
|
||||
project's workstreams. Both mint paths route through
|
||||
:func:`reject_unassignable_scopes`."""
|
||||
|
||||
def test_service_scope_rejected(self) -> None:
|
||||
from turnstone.core.auth import reject_unassignable_scopes
|
||||
|
||||
assert reject_unassignable_scopes("service") is not None
|
||||
assert reject_unassignable_scopes("read,service") is not None
|
||||
assert reject_unassignable_scopes("read,write,approve,service") is not None
|
||||
|
||||
def test_service_not_in_assignable_set(self) -> None:
|
||||
from turnstone.core.auth import ASSIGNABLE_SCOPES, VALID_SCOPES
|
||||
|
||||
assert "service" in VALID_SCOPES # still a valid runtime scope
|
||||
assert "service" not in ASSIGNABLE_SCOPES # but not user-assignable
|
||||
|
||||
def test_ordinary_scopes_accepted(self) -> None:
|
||||
from turnstone.core.auth import reject_unassignable_scopes
|
||||
|
||||
assert reject_unassignable_scopes("read") is None
|
||||
assert reject_unassignable_scopes("read,write,approve") is None
|
||||
|
||||
def test_empty_and_unknown_rejected(self) -> None:
|
||||
from turnstone.core.auth import reject_unassignable_scopes
|
||||
|
||||
assert reject_unassignable_scopes("") is not None
|
||||
assert reject_unassignable_scopes("bogus") is not None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FakeUI / FakeSession doubles — match the shape the create handler expects
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -88,18 +121,19 @@ class _FakeUI:
|
||||
self._ws_turn_tool_calls = 0
|
||||
self._llm_verdicts: dict[str, dict[str, Any]] = {}
|
||||
|
||||
def serialize_pending_approval_detail(self) -> dict[str, Any] | None:
|
||||
# Mirrors SessionUIBase.serialize_pending_approval_detail —
|
||||
def serialize_pending_approval_details(self) -> list[dict[str, Any]]:
|
||||
# Mirrors SessionUIBase.serialize_pending_approval_details —
|
||||
# the fake is monkeypatched in for ``WebUI`` and the dashboard
|
||||
# handler reads this method during projection. Real subclasses
|
||||
# inherit from ``SessionUIBase``; the fake replicates the
|
||||
# shape directly to stay decoupled.
|
||||
# iterate their approval-cycle registry (one entry per live
|
||||
# cycle); the fake models a single slot, so the list carries
|
||||
# zero or one entries.
|
||||
pending = self._pending_approval
|
||||
if pending is None:
|
||||
return None
|
||||
return []
|
||||
items = pending.get("items") or []
|
||||
if not items:
|
||||
return None
|
||||
return []
|
||||
call_ids = [item.get("call_id", "") for item in items]
|
||||
# Match the real impl's pattern (session_ui_base.py): snapshot
|
||||
# references under the lock, copy after release. Writers only
|
||||
@@ -133,11 +167,14 @@ class _FakeUI:
|
||||
# fake here keeps test-vs-prod behavioural drift from
|
||||
# masking a real-shape regression.
|
||||
primary = next((cid for cid in call_ids if cid), "")
|
||||
return {
|
||||
"call_id": primary,
|
||||
"judge_pending": bool(pending.get("judge_pending", False)),
|
||||
"items": serialized,
|
||||
}
|
||||
return [
|
||||
{
|
||||
"cycle_id": pending.get("cycle_id", ""),
|
||||
"call_id": primary,
|
||||
"judge_pending": bool(pending.get("judge_pending", False)),
|
||||
"items": serialized,
|
||||
}
|
||||
]
|
||||
|
||||
def serialize_recent_auto_approvals(self) -> list[dict[str, Any]]:
|
||||
# Empty buffer for tests that don't exercise the auto-approve
|
||||
@@ -671,28 +708,35 @@ class TestDashboardTrustedTeamVisibility:
|
||||
owners = {w["user_id"] for w in data["workstreams"]}
|
||||
assert {"user-a", "user-b"}.issubset(owners)
|
||||
|
||||
def test_dashboard_pending_approval_detail_default_none(self, app_client):
|
||||
"""No pending approval → field is explicitly null on the wire so
|
||||
consumers can distinguish "not present" from "absent key"."""
|
||||
def test_dashboard_pending_approval_details_default_empty(self, app_client):
|
||||
"""No pending approval → the list field is explicitly empty on
|
||||
the wire so consumers can distinguish "nothing pending" from
|
||||
"absent key". Replaces 1.6's singular ``pending_approval_detail``
|
||||
null (breaking, 1.7)."""
|
||||
client, _mgr = app_client
|
||||
client.post("/v1/api/workstreams/new", json={"name": "a"}, headers=_auth("user-a"))
|
||||
resp = client.get("/v1/api/dashboard", headers=_auth("user-a"))
|
||||
assert resp.status_code == 200
|
||||
rows = resp.json()["workstreams"]
|
||||
assert len(rows) == 1
|
||||
assert "pending_approval_detail" in rows[0]
|
||||
assert rows[0]["pending_approval_detail"] is None
|
||||
assert "pending_approval_details" in rows[0]
|
||||
assert rows[0]["pending_approval_details"] == []
|
||||
# The 1.6 singular field is GONE, not null — a consumer still
|
||||
# reading it should break loudly, not read None forever.
|
||||
assert "pending_approval_detail" not in rows[0]
|
||||
|
||||
def test_dashboard_pending_approval_detail_merges_judge_verdict(self, app_client):
|
||||
def test_dashboard_pending_approval_details_merge_judge_verdict(self, app_client):
|
||||
"""When _pending_approval is set on a ws's UI, /dashboard
|
||||
embeds the merged items + judge_verdict so coord live-bulk
|
||||
callers can render inline approve/deny buttons."""
|
||||
embeds one detail entry per live cycle with merged items +
|
||||
judge_verdict so coord live-bulk callers can render inline
|
||||
approve/deny buttons."""
|
||||
client, mgr = app_client
|
||||
client.post("/v1/api/workstreams/new", json={"name": "a"}, headers=_auth("user-a"))
|
||||
ws_id = next(iter(mgr.list_all())).id
|
||||
ui = mgr.get(ws_id).ui
|
||||
ui._pending_approval = {
|
||||
"type": "approve_request",
|
||||
"cycle_id": "cyc-1",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
@@ -714,8 +758,10 @@ class TestDashboardTrustedTeamVisibility:
|
||||
resp = client.get("/v1/api/dashboard", headers=_auth("user-a"))
|
||||
assert resp.status_code == 200
|
||||
row = next(w for w in resp.json()["workstreams"] if w["ws_id"] == ws_id)
|
||||
detail = row["pending_approval_detail"]
|
||||
assert detail is not None
|
||||
details = row["pending_approval_details"]
|
||||
assert len(details) == 1
|
||||
detail = details[0]
|
||||
assert detail["cycle_id"] == "cyc-1"
|
||||
assert detail["call_id"] == "c-1"
|
||||
assert detail["judge_pending"] is False
|
||||
item = detail["items"][0]
|
||||
|
||||
+41
-17
@@ -208,7 +208,9 @@ class TestMergeServerCompat:
|
||||
|
||||
|
||||
class TestEndToEndRequestShaping:
|
||||
"""Compose both layers — session builds extra_params, provider applies thinking."""
|
||||
"""Compose both layers — session builds extra_params, provider applies reasoning."""
|
||||
|
||||
provider = OpenAIChatCompletionsProvider()
|
||||
|
||||
def test_vllm_gemma_full_flow(self) -> None:
|
||||
"""Gemma now needs only the thinking param — no server workaround."""
|
||||
@@ -216,18 +218,27 @@ class TestEndToEndRequestShaping:
|
||||
server_compat = {"server_type": "vllm"}
|
||||
# Step 1: session forwards (no auto-injection of reasoning_effort).
|
||||
extra_params = merge_server_compat(None, server_compat)
|
||||
# Step 2: provider injects thinking param into chat_template_kwargs.
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
# Step 2: provider folds the effort knob into chat_template_kwargs.
|
||||
extra_body = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"enable_thinking": True}}
|
||||
|
||||
def test_knob_none_disables_toggle(self) -> None:
|
||||
"""Session effort "none" turns the template toggle off dynamically."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, {"server_type": "vllm"}), caps, "none"
|
||||
)
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
def test_server_workaround_composes_with_thinking(self) -> None:
|
||||
"""A top-level server workaround forwards alongside the injected thinking param."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
compat = {"server_type": "llama.cpp", "extra_body": {"reasoning_format": "auto"}}
|
||||
extra_body = dict(merge_server_compat(None, compat))
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, compat), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {
|
||||
"chat_template_kwargs": {"enable_thinking": True},
|
||||
@@ -237,20 +248,20 @@ class TestEndToEndRequestShaping:
|
||||
def test_granite_thinking_key(self) -> None:
|
||||
"""Granite uses 'thinking' instead of 'enable_thinking'."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="thinking")
|
||||
extra_params = merge_server_compat(None, {})
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, {}), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"thinking": True}}
|
||||
|
||||
def test_non_thinking_model_no_injection(self) -> None:
|
||||
"""Non-thinking model gets no chat_template_kwargs at all."""
|
||||
"""Non-thinking model gets no extra_body at all."""
|
||||
caps = ModelCapabilities() # thinking_mode="none"
|
||||
extra_params = merge_server_compat(None, {})
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, {}), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {}
|
||||
assert extra_body is None
|
||||
|
||||
def test_operator_reasoning_effort_passthrough(self) -> None:
|
||||
"""Operator-supplied reasoning_effort under chat_template_kwargs is preserved."""
|
||||
@@ -259,9 +270,9 @@ class TestEndToEndRequestShaping:
|
||||
"server_type": "vllm",
|
||||
"extra_body": {"chat_template_kwargs": {"reasoning_effort": "high"}},
|
||||
}
|
||||
extra_params = merge_server_compat(None, compat)
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, compat), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {
|
||||
"chat_template_kwargs": {
|
||||
@@ -270,6 +281,19 @@ class TestEndToEndRequestShaping:
|
||||
},
|
||||
}
|
||||
|
||||
def test_operator_pin_beats_effort_param(self) -> None:
|
||||
"""A pinned effort key wins over the knob mapping (setdefault)."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
compat = {"extra_body": {"chat_template_kwargs": {"reasoning_effort": "low"}}}
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, compat), caps, "high"
|
||||
)
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"reasoning_effort": "low"}}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Probe integration: suggest_profile called from _detect_openai_compat
|
||||
|
||||
+395
-51
@@ -272,18 +272,19 @@ class TestChatSessionConstruction:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — _exec_task (optional skill substitutes the hardcoded identity)
|
||||
# Tests — _exec_task (identity from persona/default; skill = capability turn)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestTaskExec:
|
||||
"""Tests for _exec_task: optional skill= replaces the default persona,
|
||||
but operating guidance (one-shot, tool-use over narration, no follow-ups)
|
||||
is always preserved."""
|
||||
"""Tests for _exec_task: identity comes from ``persona=`` (or the default
|
||||
task-agent identity), NEVER the skill; a ``skill=`` rides a distinct
|
||||
capability turn. Operating guidance (one-shot, tool-use over narration,
|
||||
no follow-ups) always layers on top."""
|
||||
|
||||
@staticmethod
|
||||
def _capture_exec_messages(session, item):
|
||||
"""Run _exec_task with _run_agent patched; return system message text."""
|
||||
def _capture_exec_turns(session, item):
|
||||
"""Run _exec_task with _run_agent patched; return the turns list."""
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
@@ -292,27 +293,22 @@ class TestTaskExec:
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
return captured["messages"][0].text
|
||||
return captured["messages"]
|
||||
|
||||
def test_known_skill_renders_into_system_message(self, tmp_db) -> None:
|
||||
"""Validated skill content (with template vars resolved) replaces
|
||||
the default '# Task Agent' persona, but the operating guidance
|
||||
(the numbered list) is preserved — those are sub-agent semantics
|
||||
that a persona should layer on top of, not replace.
|
||||
|
||||
Covers the full prepare→exec round-trip so a future regression
|
||||
in either half (skill not stored on the item, or exec ignoring it)
|
||||
is caught."""
|
||||
def test_skill_delivered_as_capability_turn_not_identity(self, tmp_db) -> None:
|
||||
"""A skill= is CAPABILITY, not identity: its body (template vars
|
||||
resolved) rides a distinct turn AFTER the system message, while the
|
||||
default '# Task Agent' identity + operating guidance stay in the
|
||||
system message. Covers the full prepare→exec round-trip."""
|
||||
session = _make_session()
|
||||
skill = {
|
||||
"name": "research",
|
||||
"content": "# Research Agent\nws={{ws_id}} model={{model}} node={{node_id}}",
|
||||
"content": "# Research Skill\nws={{ws_id}} model={{model}} node={{node_id}}",
|
||||
}
|
||||
with patch("turnstone.core.session.get_skill_by_name", return_value=skill):
|
||||
item = session._prepare_task("c1", {"prompt": "investigate X", "skill": "research"})
|
||||
|
||||
# Item carries the minimized projection — name/content/risk_level
|
||||
# only — not the raw prompt_templates row.
|
||||
# Item carries the minimized projection — name/content/risk_level only.
|
||||
assert item["skill"] == {
|
||||
"name": "research",
|
||||
"content": skill["content"],
|
||||
@@ -321,35 +317,302 @@ class TestTaskExec:
|
||||
assert item.get("needs_approval") is True
|
||||
assert "skill: research" in item["header"]
|
||||
|
||||
sys_msg = self._capture_exec_messages(session, item)
|
||||
# Skill persona rendered with template vars resolved
|
||||
assert "# Research Agent" in sys_msg
|
||||
assert f"ws={session._ws_id}" in sys_msg
|
||||
assert f"model={session.model}" in sys_msg
|
||||
# Default persona is gone — skill substitutes for it.
|
||||
assert "# Task Agent" not in sys_msg
|
||||
assert "autonomous task agent with full tool access" not in sys_msg
|
||||
# Operating guidance survives regardless of skill.
|
||||
turns = self._capture_exec_turns(session, item)
|
||||
sys_msg = turns[0].text
|
||||
# Identity stays the DEFAULT — the skill does NOT become identity.
|
||||
assert ChatSession._TASK_DEFAULT_IDENTITY in sys_msg
|
||||
assert "# Task Agent" in sys_msg
|
||||
assert ChatSession._TASK_OPERATING_GUIDANCE in sys_msg
|
||||
# Skill body is NOT fused into the identity system message.
|
||||
assert "# Research Skill" not in sys_msg
|
||||
# It rides a distinct capability turn, template vars resolved.
|
||||
capability = turns[1].text
|
||||
assert ChatSession._TASK_SKILL_CAPABILITY_PREAMBLE in capability
|
||||
assert "# Research Skill" in capability
|
||||
assert f"ws={session._ws_id}" in capability
|
||||
assert f"model={session.model}" in capability
|
||||
# Task prompt is the final turn.
|
||||
assert turns[-1].text == "investigate X"
|
||||
|
||||
def test_omitted_skill_uses_hardcoded_identity(self, tmp_db) -> None:
|
||||
"""Regression guard: without skill=, the default '# Task Agent'
|
||||
persona AND the operating guidance both appear verbatim.
|
||||
|
||||
Pins the no-skill path so the substitution branch can't
|
||||
accidentally swallow the default case."""
|
||||
def test_omitted_skill_uses_default_identity(self, tmp_db) -> None:
|
||||
"""Without skill= or persona=, the default '# Task Agent' identity +
|
||||
operating guidance appear in the system message, and there is NO
|
||||
capability turn — just system + prompt."""
|
||||
session = _make_session()
|
||||
item = session._prepare_task("c1", {"prompt": "do x"})
|
||||
|
||||
assert item["skill"] is None
|
||||
assert item["persona"] == ""
|
||||
assert "skill:" not in item["header"]
|
||||
assert "persona:" not in item["header"]
|
||||
|
||||
sys_msg = self._capture_exec_messages(session, item)
|
||||
turns = self._capture_exec_turns(session, item)
|
||||
sys_msg = turns[0].text
|
||||
assert ChatSession._TASK_DEFAULT_IDENTITY in sys_msg
|
||||
assert ChatSession._TASK_OPERATING_GUIDANCE in sys_msg
|
||||
# Default-persona literals also present (sanity check on the constant).
|
||||
assert "# Task Agent" in sys_msg
|
||||
assert "autonomous task agent with full tool access" in sys_msg
|
||||
# No skill → no capability turn: just system + prompt.
|
||||
assert len(turns) == 2
|
||||
assert turns[-1].text == "do x"
|
||||
|
||||
def test_persona_sets_identity_skill_stays_capability(self, tmp_db) -> None:
|
||||
"""persona= sets the sub-agent identity (base prompt) in place of the
|
||||
default; a skill passed alongside stays a capability turn."""
|
||||
session = _make_session()
|
||||
persona_row = {
|
||||
"name": "engineer",
|
||||
"base_prompt": "# Engineer\nYou are an engineer.",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
skill = {"name": "research", "content": "# Research Skill"}
|
||||
with (
|
||||
patch("turnstone.core.session.get_skill_by_name", return_value=skill),
|
||||
patch("turnstone.core.session.get_storage") as gs,
|
||||
):
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task(
|
||||
"c1", {"prompt": "do x", "skill": "research", "persona": "engineer"}
|
||||
)
|
||||
|
||||
assert item.get("needs_approval") is True
|
||||
assert item["persona"] == "engineer"
|
||||
assert "persona: engineer" in item["header"]
|
||||
assert "skill: research" in item["header"]
|
||||
|
||||
turns = self._capture_exec_turns(session, item)
|
||||
sys_msg = turns[0].text
|
||||
# Identity = persona, not the default and not the skill.
|
||||
assert "# Engineer" in sys_msg
|
||||
assert ChatSession._TASK_DEFAULT_IDENTITY not in sys_msg
|
||||
assert "# Research Skill" not in sys_msg
|
||||
# Operating guidance still layers on the persona identity.
|
||||
assert ChatSession._TASK_OPERATING_GUIDANCE in sys_msg
|
||||
# Skill remains a capability turn.
|
||||
assert "# Research Skill" in turns[1].text
|
||||
|
||||
def test_unknown_persona_returns_error(self, tmp_db) -> None:
|
||||
"""Unknown persona name → clean error item, no approval."""
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = None
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "ghost"})
|
||||
assert item.get("needs_approval") is False
|
||||
assert "ghost" in item["error"]
|
||||
assert "Omit `persona`" in item["error"]
|
||||
|
||||
def test_persona_wrong_kind_returns_error(self, tmp_db) -> None:
|
||||
"""A coordinator-only persona can't serve as a task-agent identity."""
|
||||
session = _make_session()
|
||||
coord_row = {
|
||||
"name": "orchestrator",
|
||||
"base_prompt": "# Orchestrator",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = coord_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "orchestrator"})
|
||||
assert item.get("needs_approval") is False
|
||||
assert "interactive" in item["error"]
|
||||
|
||||
def test_persona_tool_allowlist_restricts_sub_agent_tools(self, tmp_db) -> None:
|
||||
"""A restrictive persona caps the sub-agent's TOOLS (Principle 7 /
|
||||
review fix), not just its identity text — stated identity must match
|
||||
granted authority."""
|
||||
session = _make_session()
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "write_file"}},
|
||||
{"function": {"name": "bash"}},
|
||||
]
|
||||
persona_row = {
|
||||
"name": "readonly",
|
||||
"base_prompt": "# Readonly reviewer",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": ["read_file", "search"], # excludes write_file/bash
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "edit auth", "persona": "readonly"})
|
||||
assert item["persona_tools"] == frozenset({"read_file", "search"})
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
tool_names = {t["function"]["name"] for t in captured["tools"]}
|
||||
# write_file + bash dropped by the persona; read_file kept (search was
|
||||
# never in the task tool set to begin with).
|
||||
assert tool_names == {"read_file"}
|
||||
|
||||
def test_no_persona_keeps_full_task_tools(self, tmp_db) -> None:
|
||||
session = _make_session()
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "bash"}},
|
||||
]
|
||||
item = session._prepare_task("c1", {"prompt": "do x"})
|
||||
assert item["persona_tools"] is None
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file", "bash"}
|
||||
|
||||
def test_parent_persona_caps_sub_agent_tools(self, tmp_db) -> None:
|
||||
"""A restricted PARENT session must not escalate authority by spawning:
|
||||
the sub-agent's tools are capped by the parent's own persona grant even
|
||||
with NO child persona (Principle 7 — delegation narrows, never widens;
|
||||
whole-PR review fix)."""
|
||||
session = _make_session()
|
||||
session._tool_search = None
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "write_file"}},
|
||||
{"function": {"name": "bash"}},
|
||||
]
|
||||
# Parent runs under a read-only persona.
|
||||
session._persona_tools = frozenset({"read_file", "search"})
|
||||
|
||||
item = session._prepare_task("c1", {"prompt": "edit auth"})
|
||||
assert item["persona_tools"] is None # no CHILD persona
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
# Parent's read-only grant caps the sub-agent: write_file + bash dropped.
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file"}
|
||||
|
||||
def test_child_persona_mcp_off_drops_mcp_tools(self, tmp_db) -> None:
|
||||
"""A child persona with mcp_enabled=False hides MCP tools (``mcp__*`` and
|
||||
the MCP-access read_resource / use_prompt) from the sub-agent, even when
|
||||
tool_allowlist is null (unrestricted native tools) — the mcp lever must
|
||||
not silently no-op on the task_agent path (whole-PR review fix)."""
|
||||
session = _make_session()
|
||||
session._tool_search = None
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "read_resource"}},
|
||||
{"function": {"name": "use_prompt"}},
|
||||
{"function": {"name": "mcp__github__search"}},
|
||||
]
|
||||
persona_row = {
|
||||
"name": "sandboxed",
|
||||
"base_prompt": "# Sandboxed",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None, # null = unrestricted native tools
|
||||
"mcp_enabled": False, # but MCP is OFF
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "sandboxed"})
|
||||
assert item["persona_mcp"] is False
|
||||
assert item["persona_tools"] is None
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
# MCP tools shed; native read_file kept.
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file"}
|
||||
|
||||
def test_child_persona_memory_off_drops_memory_tool(self, tmp_db) -> None:
|
||||
"""A child persona with memory_enabled=False drops the memory tool from
|
||||
the sub-agent's hands (lever 4), matching a main session under the same
|
||||
persona (whole-PR review fix)."""
|
||||
session = _make_session()
|
||||
session._tool_search = None
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "memory"}},
|
||||
]
|
||||
persona_row = {
|
||||
"name": "nomem",
|
||||
"base_prompt": "# No memory",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": False,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "nomem"})
|
||||
assert item["persona_memory"] is False
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file"}
|
||||
|
||||
def test_evaluate_intent_projects_persona_for_task_agent(self, tmp_db, monkeypatch) -> None:
|
||||
"""Judge/audit projection includes the persona name (review fix): a
|
||||
persona-driven identity shift must be visible to policy + audit, like
|
||||
spawn_workstream."""
|
||||
session = _make_session()
|
||||
fake_verdict = MagicMock()
|
||||
fake_verdict.to_dict.return_value = {"verdict_id": "v0", "tier": "heuristic"}
|
||||
fake_judge = MagicMock()
|
||||
fake_judge.evaluate.side_effect = lambda items, *_a, **_kw: [fake_verdict] * len(items)
|
||||
fake_judge.arg_budget_chars.return_value = 200_000
|
||||
monkeypatch.setattr(session, "_ensure_judge", lambda: fake_judge)
|
||||
|
||||
persona_row = {
|
||||
"name": "engineer",
|
||||
"base_prompt": "# Engineer",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "engineer"})
|
||||
session._evaluate_intent([item])
|
||||
assert item["func_args"]["persona"] == "engineer"
|
||||
|
||||
@pytest.mark.parametrize("skill_value", ["", " ", "\t\n"])
|
||||
def test_prepare_task_empty_or_whitespace_skill_treated_as_omitted(
|
||||
@@ -397,12 +660,11 @@ class TestTaskExec:
|
||||
# tell them apart at recovery time.
|
||||
assert "unknown skill" not in item["error"]
|
||||
|
||||
def test_prepare_task_high_risk_skill_surfaces_in_header(self, tmp_db, caplog) -> None:
|
||||
"""High/critical risk skills surface the tier in the approval header
|
||||
and emit a structured warning, mirroring the signal ``_load_skills``
|
||||
emits for session-level skills (session.py:1336)."""
|
||||
import logging
|
||||
|
||||
def test_prepare_task_denies_high_risk_skill(self, tmp_db) -> None:
|
||||
"""High/critical-risk skills are PRINCIPAL-load-only: task_agent(skill=…)
|
||||
DENIES them — the same gate skills(load) / spawn_* enforce, so a model
|
||||
cannot route around it by delegating activation to a sub-agent
|
||||
(whole-PR review fix — task_agent was the un-gated surface)."""
|
||||
session = _make_session()
|
||||
risky_skill = {
|
||||
"name": "danger",
|
||||
@@ -410,16 +672,15 @@ class TestTaskExec:
|
||||
"enabled": True,
|
||||
"risk_level": "critical",
|
||||
}
|
||||
with (
|
||||
caplog.at_level(logging.WARNING, logger="turnstone.core.session"),
|
||||
patch("turnstone.core.session.get_skill_by_name", return_value=risky_skill),
|
||||
):
|
||||
with patch("turnstone.core.session.get_skill_by_name", return_value=risky_skill):
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "skill": "danger"})
|
||||
assert item.get("needs_approval") is True
|
||||
assert "skill: danger" in item["header"]
|
||||
assert "risk: critical" in item["header"]
|
||||
warning_seen = any("high_risk_skill" in r.getMessage() for r in caplog.records)
|
||||
assert warning_seen, "expected task_agent.high_risk_skill warning"
|
||||
assert item.get("needs_approval") is False
|
||||
assert "principal-load-only" in item["header"]
|
||||
assert "/skill danger" in item["error"]
|
||||
# Distinct from the unknown/disabled errors so the model's recovery
|
||||
# path can tell them apart.
|
||||
assert "unknown skill" not in item["error"]
|
||||
assert "disabled" not in item["error"]
|
||||
|
||||
def test_prepare_task_normal_risk_skill_omits_tier_from_header(self, tmp_db) -> None:
|
||||
"""Header only surfaces high/critical — low/medium/safe skills don't
|
||||
@@ -537,6 +798,89 @@ class TestTaskExec:
|
||||
captured[0](fake_verdict) # must not raise
|
||||
session.ui.on_intent_verdict.assert_not_called()
|
||||
|
||||
def test_evaluate_intent_agent_gate_owns_generation_off_the_main_slot(
|
||||
self, tmp_db, monkeypatch
|
||||
) -> None:
|
||||
"""Sub-agent gates run the SAME judge pipeline as the main loop
|
||||
but as their OWN generation (release blocker #1: task_agent
|
||||
calls used to reach the gate judge-blind). The main-loop
|
||||
supersede slot stays untouched — with parallel task agents,
|
||||
publishing into it would make every sibling's verdicts look
|
||||
stale to the previous sibling's callback — while the generation
|
||||
is stamped on the items for the UI's origin checks, registered
|
||||
for ``close()``'s sweep, delivered alongside the verdict, and
|
||||
grounded on the SUB-AGENT's trajectory (its task prompt is the
|
||||
delegation contract), not the parent conversation."""
|
||||
import threading
|
||||
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.trajectory import turns_from_dicts
|
||||
|
||||
class _GateUI(SessionUIBase):
|
||||
pass
|
||||
|
||||
session = _make_session()
|
||||
ui = _GateUI(ws_id="ws-gate", user_id="u1")
|
||||
ui.on_intent_verdict = MagicMock() # shadow: capture delivery kwargs
|
||||
session.ui = ui
|
||||
|
||||
captured: dict[str, Any] = {}
|
||||
fake_verdict = MagicMock()
|
||||
fake_verdict.to_dict.return_value = {"verdict_id": "v0", "call_id": "c1", "tier": "llm"}
|
||||
fake_judge = MagicMock()
|
||||
|
||||
def _eval(items, convo, **kw):
|
||||
captured["convo"] = convo
|
||||
captured["callback"] = kw.get("callback")
|
||||
captured["cancel_event"] = kw.get("cancel_event")
|
||||
captured["done"] = kw.get("done_callback")
|
||||
return [fake_verdict] * len(items)
|
||||
|
||||
fake_judge.evaluate.side_effect = _eval
|
||||
monkeypatch.setattr(session, "_ensure_judge", lambda: fake_judge)
|
||||
|
||||
main_slot = threading.Event()
|
||||
session._judge_cancel_event = main_slot
|
||||
agent_turns = turns_from_dicts([{"role": "user", "content": "Task: reindex the docs tree"}])
|
||||
item = {"call_id": "c1", "func_name": "bash", "needs_approval": True, "command": "ls"}
|
||||
|
||||
ev = session._evaluate_intent([item], conversation=agent_turns, agent_gate=True)
|
||||
|
||||
assert ev is not None and ev is not main_slot
|
||||
# Main-loop slot untouched by the sub-agent spawn.
|
||||
assert session._judge_cancel_event is main_slot
|
||||
# Generation stamped for the UI's origin checks + close() sweep,
|
||||
# and handed to the daemon as its cancel event.
|
||||
assert item["_judge_event"] is ev
|
||||
assert ev in session._judge_cancel_events
|
||||
assert captured["cancel_event"] is ev
|
||||
# Judge grounded on the sub-agent trajectory, not session.messages.
|
||||
assert any("reindex the docs tree" in str(m) for m in captured["convo"])
|
||||
# Delivery rides the generation into the UI.
|
||||
captured["callback"](fake_verdict)
|
||||
assert ui.on_intent_verdict.call_args.kwargs.get("judge_event") is ev
|
||||
# Daemon completion keeps the close()-sweep set exact.
|
||||
captured["done"]()
|
||||
assert ev not in session._judge_cancel_events
|
||||
|
||||
def test_close_fires_agent_gate_judge_generations(self, tmp_db, monkeypatch) -> None:
|
||||
"""``close()`` aborts EVERY in-flight judge daemon — including
|
||||
sub-agent generations that never touched the main slot — so a
|
||||
torn-down session can't leave daemons running against a dead
|
||||
UI."""
|
||||
session = _make_session()
|
||||
fake_verdict = MagicMock()
|
||||
fake_verdict.to_dict.return_value = {"verdict_id": "v0", "call_id": "c1", "tier": "llm"}
|
||||
fake_judge = MagicMock()
|
||||
fake_judge.evaluate.side_effect = lambda items, *_a, **_kw: [fake_verdict] * len(items)
|
||||
monkeypatch.setattr(session, "_ensure_judge", lambda: fake_judge)
|
||||
|
||||
item = {"call_id": "c1", "func_name": "bash", "needs_approval": True, "command": "ls"}
|
||||
ev = session._evaluate_intent([item], conversation=[], agent_gate=True)
|
||||
assert ev is not None and not ev.is_set()
|
||||
session.close()
|
||||
assert ev.is_set()
|
||||
|
||||
def _drive_gate(self, session, monkeypatch, *, cancel_on_approval: bool):
|
||||
"""Run one needs_approval bash item through ``_execute_tools`` with a
|
||||
stubbed judge + approval gate; return the cancel event the judge
|
||||
|
||||
+732
-180
File diff suppressed because it is too large
Load Diff
@@ -49,6 +49,7 @@ _ESM_BUNDLES = [
|
||||
_SHARED / "composer_queue.js",
|
||||
_SHARED / "interactive.js",
|
||||
_SHARED / "conversation.js",
|
||||
_SHARED / "redact_credentials.js",
|
||||
]
|
||||
|
||||
# Sink scan: everything except renderer.js — the one sanctioned HTML-string
|
||||
@@ -68,6 +69,7 @@ _ESM_NO_VAR_BUNDLES = [
|
||||
_SHARED / "auth.js",
|
||||
_SHARED / "interactive.js",
|
||||
_SHARED / "conversation.js",
|
||||
_SHARED / "redact_credentials.js",
|
||||
]
|
||||
|
||||
# The same unsafe DOM-write / dynamic-code sink set that ``test_app_js.py``
|
||||
@@ -444,7 +446,8 @@ def test_shell_bridges_setrowbadge_for_classic_subsystems() -> None:
|
||||
badge the same way the gear deletion did)."""
|
||||
body = _SHELL_JS.read_text(encoding="utf-8")
|
||||
assert 'setRowBadge } from "./rail.js"' in body, "shell must import setRowBadge from rail.js"
|
||||
assert "notifySessionClosed, setRowBadge }" in body, (
|
||||
ts_shell = body[body.index("window.TS_SHELL = {") :][:200]
|
||||
assert "setRowBadge" in ts_shell, (
|
||||
"TS_SHELL must expose setRowBadge for classic subsystems (the consent-badge bridge)"
|
||||
)
|
||||
|
||||
@@ -1011,7 +1014,8 @@ def test_shell_closes_pane_on_ws_closed() -> None:
|
||||
assert 'pm.getPane("interactive", wsId)' in shell
|
||||
assert "if (p) pm.close(p.id)" in shell, "ws_closed closes the pane, not mark-dead"
|
||||
assert "showDeadBanner" in shell, "the banner lane must survive for non-closed deaths"
|
||||
assert "window.TS_SHELL = { panes: pm, caps, notifySessionClosed, setRowBadge }" in shell, (
|
||||
ts_shell = shell[shell.index("window.TS_SHELL = {") :][:200]
|
||||
assert "panes: pm" in ts_shell and "notifySessionClosed" in ts_shell, (
|
||||
"the seam must be exported on TS_SHELL for the console's Tier-1 handler"
|
||||
)
|
||||
app = _CONSOLE_APP.read_text(encoding="utf-8")
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
"""Tests for turnstone.eval skill-adherence measurement mode.
|
||||
|
||||
Two levels, neither requires a live model:
|
||||
|
||||
* ``TestSkillComposition`` is the load-bearing plumbing proof — it seeds a
|
||||
named skill, builds ``HeadlessSession`` under natural composition, and
|
||||
asserts the skill body folds into ``system_messages`` for the treatment
|
||||
arm and is absent for the control arm. This is what makes the two arms
|
||||
measure different things.
|
||||
* ``TestAdherenceLift`` unit-tests ``run_skill_adherence``'s lift math with
|
||||
the per-arm runner stubbed out.
|
||||
"""
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
from collections.abc import Iterator
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from openai import OpenAI
|
||||
|
||||
from turnstone.core.storage import get_storage, init_storage, reset_storage
|
||||
from turnstone.eval import core
|
||||
from turnstone.eval.core import HeadlessSession, run_skill_adherence
|
||||
|
||||
_SKILL = {
|
||||
"name": "search-first",
|
||||
"content": (
|
||||
"# Search First\n\nBefore answering ANY question about where something "
|
||||
"lives in the codebase, you MUST call the `search` tool first. "
|
||||
"SENTINEL_SKILL_BODY_MARKER."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_storage() -> Iterator[None]:
|
||||
"""Fresh sqlite storage in a temp dir, torn down after the test."""
|
||||
workdir = tempfile.mkdtemp(prefix="turnstone_skill_test_")
|
||||
reset_storage()
|
||||
init_storage("sqlite", path=os.path.join(workdir, ".eval.db"), run_migrations=False)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
reset_storage()
|
||||
import shutil
|
||||
|
||||
shutil.rmtree(workdir, ignore_errors=True)
|
||||
|
||||
|
||||
def _seed_skill(skill: dict[str, str]) -> None:
|
||||
"""Seed a named skill exactly as the runner does."""
|
||||
get_storage().create_prompt_template(
|
||||
template_id="eval-skill",
|
||||
name=skill["name"],
|
||||
category="eval",
|
||||
content=skill["content"],
|
||||
variables="[]",
|
||||
is_default=False,
|
||||
org_id="",
|
||||
created_by="eval",
|
||||
activation="named",
|
||||
enabled=True,
|
||||
)
|
||||
|
||||
|
||||
def _system_text(session: HeadlessSession) -> str:
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
class TestSkillComposition:
|
||||
"""Prove the treatment/control arms compose different system messages."""
|
||||
|
||||
def test_treatment_folds_skill_into_system(self, temp_storage: None) -> None:
|
||||
_seed_skill(_SKILL)
|
||||
client = OpenAI(base_url="http://localhost:9/v1", api_key="dummy")
|
||||
session = HeadlessSession(client=client, model="test-model")
|
||||
try:
|
||||
# Treatment arm activates the seeded skill via the real path.
|
||||
session.set_skill(_SKILL["name"])
|
||||
assert "SENTINEL_SKILL_BODY_MARKER" in _system_text(session)
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_control_omits_skill(self, temp_storage: None) -> None:
|
||||
# Control arm: no skill seeded, no set_skill — natural default only.
|
||||
client = OpenAI(base_url="http://localhost:9/v1", api_key="dummy")
|
||||
session = HeadlessSession(client=client, model="test-model")
|
||||
try:
|
||||
assert "SENTINEL_SKILL_BODY_MARKER" not in _system_text(session)
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_no_system_prompt_override_in_skill_mode(self, temp_storage: None) -> None:
|
||||
# skill_mode must NOT override the base identity — a real base prompt
|
||||
# (persona / composed developer message) must survive, or we'd be
|
||||
# measuring an empty prompt instead of the identity under test.
|
||||
client = OpenAI(base_url="http://localhost:9/v1", api_key="dummy")
|
||||
session = HeadlessSession(client=client, model="test-model")
|
||||
try:
|
||||
assert _system_text(session).strip(), "expected a composed base prompt"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestAdherenceLift:
|
||||
"""Unit-test the lift math with the per-arm runner stubbed."""
|
||||
|
||||
def test_lift_treatment_over_control(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# Stub _run_iteration: treatment (skill != None) passes 3/3, control
|
||||
# (skill is None) passes 1/3. run_skill_adherence must report the
|
||||
# difference as the lift.
|
||||
def fake_run_iteration(**kwargs: Any) -> dict[str, Any]:
|
||||
rate = 1.0 if kwargs.get("skill") is not None else 1.0 / 3.0
|
||||
return {"aggregate": {"overall_pass_rate": rate}}
|
||||
|
||||
monkeypatch.setattr(core, "_run_iteration", fake_run_iteration)
|
||||
|
||||
cases = [
|
||||
{
|
||||
"id": "search-first",
|
||||
"skill": _SKILL,
|
||||
"user_prompt": "where is X?",
|
||||
"expected_actions": [{"tool": "search"}],
|
||||
}
|
||||
]
|
||||
result = run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=3,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
|
||||
assert len(result["cases"]) == 1
|
||||
row = result["cases"][0]
|
||||
assert row["case_id"] == "search-first"
|
||||
assert row["skill"] == "search-first"
|
||||
assert row["treatment_rate"] == pytest.approx(1.0)
|
||||
assert row["control_rate"] == pytest.approx(1.0 / 3.0)
|
||||
assert row["lift"] == pytest.approx(2.0 / 3.0)
|
||||
assert row["n_runs"] == 3
|
||||
assert result["mean_lift"] == pytest.approx(2.0 / 3.0)
|
||||
|
||||
def test_rejects_malformed_skill(self) -> None:
|
||||
# A skill missing 'content' (or 'name') fails fast with a clear error,
|
||||
# not a KeyError mid-run (Copilot review). Validation raises before any
|
||||
# arm runs, so no _run_iteration stub is needed.
|
||||
cases = [
|
||||
{
|
||||
"id": "bad-skill",
|
||||
"skill": {"name": "x"}, # missing 'content'
|
||||
"user_prompt": "do x",
|
||||
"expected_actions": [{"tool": "search"}],
|
||||
}
|
||||
]
|
||||
with pytest.raises(ValueError, match="non-empty 'name' and 'content'"):
|
||||
run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=1,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
|
||||
def test_skipped_when_no_skill(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# A case with no skill is not measurable — it must be skipped, not
|
||||
# crash, and must not contribute to the mean.
|
||||
def fake_run_iteration(**kwargs: Any) -> dict[str, Any]:
|
||||
return {"aggregate": {"overall_pass_rate": 1.0}}
|
||||
|
||||
monkeypatch.setattr(core, "_run_iteration", fake_run_iteration)
|
||||
|
||||
cases = [{"id": "no-skill", "user_prompt": "hi", "expected_actions": []}]
|
||||
result = run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=3,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
assert result["cases"] == []
|
||||
assert result["mean_lift"] == 0.0
|
||||
|
||||
def test_mean_lift_averages_multiple_cases(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# Two skill cases with different lifts average into mean_lift.
|
||||
rates = iter([1.0, 0.0, 1.0, 0.5]) # t1, c1, t2, c2 -> lifts 1.0, 0.5
|
||||
|
||||
def fake_run_iteration(**kwargs: Any) -> dict[str, Any]:
|
||||
return {"aggregate": {"overall_pass_rate": next(rates)}}
|
||||
|
||||
monkeypatch.setattr(core, "_run_iteration", fake_run_iteration)
|
||||
|
||||
cases = [
|
||||
{"id": "a", "skill": _SKILL, "user_prompt": "q", "expected_actions": []},
|
||||
{"id": "b", "skill": _SKILL, "user_prompt": "q", "expected_actions": []},
|
||||
]
|
||||
result = run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=2,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
assert [c["lift"] for c in result["cases"]] == pytest.approx([1.0, 0.5])
|
||||
assert result["mean_lift"] == pytest.approx(0.75)
|
||||
@@ -135,9 +135,8 @@ def _create_skill(db: Any, skill_id: str, name: str, content: str, **kw: Any) ->
|
||||
|
||||
|
||||
def _sys_content(session: ChatSession) -> str:
|
||||
msgs = [m for m in session.system_messages if m["role"] == "system"]
|
||||
assert msgs
|
||||
return msgs[0]["content"]
|
||||
assert session.system_messages
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,295 @@
|
||||
"""Skill substitution unification (SKILL.md subsystem refactor, step 1).
|
||||
|
||||
Pins the invariant that skill-body placeholder substitution is IDENTICAL
|
||||
across every invocation context. Interactive load, default skills, and
|
||||
``task_agent`` sub-agents all route through
|
||||
``ChatSession._render_skill_body`` — so a skill reading ``$ARGUMENTS`` or
|
||||
``${TURNSTONE_EFFORT}`` resolves the same everywhere, rather than
|
||||
rendering literally on the ``task_agent`` path (which previously ran
|
||||
``_render_template`` alone).
|
||||
|
||||
Also covers the two behaviours the unified path newly guarantees:
|
||||
|
||||
* ``${TURNSTONE_*}`` env vars (canonical) and their ``${CLAUDE_*}``
|
||||
back-compat aliases both resolve.
|
||||
* ``${TURNSTONE_SKILL_DIR}`` resolves to the concrete materialized bundle
|
||||
path in the rendered body, because resources are materialized BEFORE
|
||||
substitution (the ordering fix).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from tests._session_helpers import make_session
|
||||
from turnstone.core.storage._registry import get_storage
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import pytest
|
||||
|
||||
|
||||
def _create_skill(db: Any, skill_id: str, name: str, content: str, **kw: Any) -> None:
|
||||
db.create_prompt_template(
|
||||
template_id=skill_id,
|
||||
name=name,
|
||||
category=kw.get("category", "general"),
|
||||
content=content,
|
||||
variables="[]",
|
||||
is_default=kw.get("is_default", False),
|
||||
org_id="",
|
||||
created_by="test",
|
||||
origin="manual",
|
||||
mcp_server="",
|
||||
readonly=False,
|
||||
description="",
|
||||
tags="[]",
|
||||
source_url="",
|
||||
version="1.0.0",
|
||||
author="",
|
||||
activation=kw.get("activation", "named"),
|
||||
token_estimate=0,
|
||||
model="",
|
||||
auto_approve=False,
|
||||
temperature=None,
|
||||
reasoning_effort="",
|
||||
max_tokens=None,
|
||||
token_budget=0,
|
||||
agent_max_turns=None,
|
||||
notify_on_complete="{}",
|
||||
enabled=True,
|
||||
allowed_tools="[]",
|
||||
priority=0,
|
||||
)
|
||||
|
||||
|
||||
class TestRenderSkillBodySharedPath:
|
||||
"""``_render_skill_body`` is the single substitution path — the one
|
||||
``task_agent`` now calls. With ``substitute_args=True`` (arg-capable
|
||||
invocations: interactive /skill, skills(load)) the spec arg forms
|
||||
resolve; with ``substitute_args=False`` (capability contexts: defaults,
|
||||
task_agent) literal ``$N``/``$ARGUMENTS`` are left untouched. Env vars
|
||||
resolve either way."""
|
||||
|
||||
def test_env_vars_resolve(self, tmp_db: str) -> None:
|
||||
session = make_session(reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body(
|
||||
"id ${TURNSTONE_SESSION_ID} effort ${TURNSTONE_EFFORT}"
|
||||
)
|
||||
assert out == f"id {session._ws_id} effort high"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_claude_aliases_resolve(self, tmp_db: str) -> None:
|
||||
session = make_session(reasoning_effort="low")
|
||||
try:
|
||||
out = session._render_skill_body("id ${CLAUDE_SESSION_ID} effort ${CLAUDE_EFFORT}")
|
||||
assert out == f"id {session._ws_id} effort low"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_arg_capable_path_clears_bare_arguments_when_no_args(self, tmp_db: str) -> None:
|
||||
# Arg-capable invocation (interactive /skill, skills(load)) with no
|
||||
# args → bare $ARGUMENTS clears to empty, per the SKILL.md spec.
|
||||
session = make_session()
|
||||
try:
|
||||
assert session._render_skill_body("before $ARGUMENTS after") == "before after"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_curly_and_spec_passes_both_apply(self, tmp_db: str) -> None:
|
||||
# Legacy ``{{model}}`` AND spec ``${TURNSTONE_EFFORT}`` in one body —
|
||||
# both passes run through the shared path.
|
||||
session = make_session(model="my-model", reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body("model {{model}} effort ${TURNSTONE_EFFORT}")
|
||||
assert out == "model my-model effort high"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_arg_capable_path_clears_positional_tokens_when_no_args(self, tmp_db: str) -> None:
|
||||
# Arg-capable path with no args → positional forms clear to empty (spec).
|
||||
session = make_session()
|
||||
try:
|
||||
out = session._render_skill_body("step $1 / $0 / $ARGUMENTS[2] done")
|
||||
assert out == "step / / done"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_capability_context_preserves_literal_arg_tokens(self, tmp_db: str) -> None:
|
||||
# Capability contexts (task_agent, defaults) never receive invocation
|
||||
# args, so substitute_args=False leaves literal $ARGUMENTS/$N/$name
|
||||
# untouched (they are prose/shell text) while env vars still resolve.
|
||||
# Pins the review fix that stopped blanking such tokens for sub-agents.
|
||||
session = make_session(reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body(
|
||||
"run ./deploy.sh $1 $2 at ${TURNSTONE_EFFORT}; process $ARGUMENTS",
|
||||
substitute_args=False,
|
||||
)
|
||||
assert out == "run ./deploy.sh $1 $2 at high; process $ARGUMENTS"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_skill_dir_literal_on_sub_agent_path(self, tmp_db: str) -> None:
|
||||
# task_agent calls _render_skill_body with no skill_dir (sub-agent
|
||||
# bundles aren't materialized yet), so ${TURNSTONE_SKILL_DIR} stays
|
||||
# literal on this path — unchanged from before, resolved in a later
|
||||
# step. The env vars that DO have values still resolve.
|
||||
session = make_session(reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body(
|
||||
"dir ${TURNSTONE_SKILL_DIR} effort ${TURNSTONE_EFFORT}"
|
||||
)
|
||||
assert out == "dir ${TURNSTONE_SKILL_DIR} effort high"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestSkillDirResolvesInBody:
|
||||
"""Materialize-before-substitute: ``${TURNSTONE_SKILL_DIR}`` in a skill
|
||||
body resolves to the concrete on-disk bundle path after a full load."""
|
||||
|
||||
def test_turnstone_skill_dir_in_body(self, tmp_db: str) -> None:
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "dir-skill", "Scripts under ${TURNSTONE_SKILL_DIR}/scripts.")
|
||||
db.create_skill_resource("r1", "s1", "scripts/go.py", "print('x')")
|
||||
|
||||
session = make_session(skill="dir-skill")
|
||||
try:
|
||||
base = session._skill_resources_dir
|
||||
assert base is not None
|
||||
assert session._skill_content == f"Scripts under {base}/scripts."
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_claude_skill_dir_not_aliased_in_body(self, tmp_db: str) -> None:
|
||||
# CLAUDE_SKILL_DIR is NOT a turnstone-owned alias: it stays a literal
|
||||
# placeholder even with a materialized bundle (that name belongs to the
|
||||
# host in bash; turnstone claims neither surface). The canonical
|
||||
# TURNSTONE_SKILL_DIR does resolve.
|
||||
db = get_storage()
|
||||
_create_skill(
|
||||
db,
|
||||
"s1",
|
||||
"dir-alias-skill",
|
||||
"Bundle at ${CLAUDE_SKILL_DIR} vs ${TURNSTONE_SKILL_DIR}",
|
||||
)
|
||||
db.create_skill_resource("r1", "s1", "references/a.md", "# a")
|
||||
|
||||
session = make_session(skill="dir-alias-skill")
|
||||
try:
|
||||
base = session._skill_resources_dir
|
||||
assert base is not None
|
||||
assert session._skill_content == f"Bundle at ${{CLAUDE_SKILL_DIR}} vs {base}"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_skill_dir_literal_without_resources(self, tmp_db: str) -> None:
|
||||
# No bundled resources → no dir → placeholder stays literal
|
||||
# (graceful degradation), not an empty path.
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "no-res-skill", "Path ${TURNSTONE_SKILL_DIR} here.")
|
||||
|
||||
session = make_session(skill="no-res-skill")
|
||||
try:
|
||||
assert session._skill_resources_dir is None
|
||||
assert session._skill_content == "Path ${TURNSTONE_SKILL_DIR} here."
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestSkillResourceEnvAliases:
|
||||
"""Bash env exposes the materialized bundle dir under turnstone-owned
|
||||
names (``TURNSTONE_SKILL_DIR`` / ``SKILL_RESOURCES_DIR``) unconditionally,
|
||||
and never under the foreign ``CLAUDE_SKILL_DIR`` — that name is the host's,
|
||||
so turnstone leaves it untouched whether or not the host has set it."""
|
||||
|
||||
def test_turnstone_owned_names_present_claude_absent(
|
||||
self, tmp_db: str, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.delenv("CLAUDE_SKILL_DIR", raising=False)
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "env-alias-skill", "content")
|
||||
db.create_skill_resource("r1", "s1", "scripts/t.py", "code")
|
||||
|
||||
session = make_session(skill="env-alias-skill")
|
||||
try:
|
||||
env = session._skill_resource_env()
|
||||
d = session._skill_resources_dir
|
||||
assert env["SKILL_RESOURCES_DIR"] == d
|
||||
assert env["TURNSTONE_SKILL_DIR"] == d
|
||||
# turnstone never supplies CLAUDE_SKILL_DIR (the host's namespace),
|
||||
# even when the host hasn't set it.
|
||||
assert "CLAUDE_SKILL_DIR" not in env
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_host_claude_skill_dir_untouched(
|
||||
self, tmp_db: str, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# turnstone as a node inside Claude Code: the host's CLAUDE_SKILL_DIR
|
||||
# must survive. turnstone never injects CLAUDE_SKILL_DIR into the
|
||||
# extra-env, so scrubbed_env's passthrough keeps the host value.
|
||||
monkeypatch.setenv("CLAUDE_SKILL_DIR", "/host/claude/skill")
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "env-host-skill", "content")
|
||||
db.create_skill_resource("r1", "s1", "scripts/t.py", "code")
|
||||
|
||||
session = make_session(skill="env-host-skill")
|
||||
try:
|
||||
env = session._skill_resource_env()
|
||||
d = session._skill_resources_dir
|
||||
assert env["TURNSTONE_SKILL_DIR"] == d
|
||||
assert env["SKILL_RESOURCES_DIR"] == d
|
||||
assert "CLAUDE_SKILL_DIR" not in env
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestSkillContextPlacement:
|
||||
"""Step 3: an applied skill's body rides its own capability context
|
||||
message (user role), separate from the identity system message — so it
|
||||
never occupies the cached identity prefix or reads as identity, and it
|
||||
does not leak into the task_agent base."""
|
||||
|
||||
def test_skill_body_in_context_message_not_identity(self, tmp_db: str) -> None:
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "place-skill", "PLACEMENT_MARKER body text")
|
||||
|
||||
session = make_session(skill="place-skill")
|
||||
try:
|
||||
msgs = session.system_messages
|
||||
# Identity system message is first, role=system, and skill-free.
|
||||
assert msgs[0]["role"] == "system"
|
||||
assert "PLACEMENT_MARKER" not in msgs[0]["content"]
|
||||
# Skill rides exactly one separate user-role capability message.
|
||||
skill_msgs = [m for m in msgs if m["role"] == "user"]
|
||||
assert len(skill_msgs) == 1
|
||||
assert "PLACEMENT_MARKER" in skill_msgs[0]["content"]
|
||||
# The intro names the active skill so the model knows what it is.
|
||||
assert "place-skill" in skill_msgs[0]["content"]
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_no_skill_no_context_message(self, tmp_db: str) -> None:
|
||||
# No applied skill and no defaults → only the identity system message.
|
||||
session = make_session()
|
||||
try:
|
||||
assert all(m["role"] == "system" for m in session.system_messages)
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_agent_prefix_excludes_skill_context(self, tmp_db: str) -> None:
|
||||
# task_agent base = the identity system block only; the parent's
|
||||
# applied skill does NOT leak into the sub-agent prefix.
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "leak-skill", "SHOULD_NOT_LEAK body")
|
||||
|
||||
session = make_session(skill="leak-skill")
|
||||
try:
|
||||
assert len(session._agent_system_messages) == 1
|
||||
assert session._agent_system_messages[0]["role"] == "system"
|
||||
assert "SHOULD_NOT_LEAK" not in session._agent_system_messages[0]["content"]
|
||||
finally:
|
||||
session.close()
|
||||
@@ -111,10 +111,9 @@ def _make_session(**kwargs):
|
||||
|
||||
|
||||
def _sys_content(session: ChatSession) -> str:
|
||||
"""Extract the system message content."""
|
||||
msgs = [m for m in session.system_messages if m["role"] == "system"]
|
||||
assert msgs
|
||||
return msgs[0]["content"]
|
||||
"""Full prompt prefix: identity system message + any skill context message."""
|
||||
assert session.system_messages
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
def _create_template(db, template_id, name, content, **kwargs):
|
||||
|
||||
@@ -1444,3 +1444,98 @@ class TestSkillCatalogDisclosure:
|
||||
# human-facing path); the tool-facing path is the new
|
||||
# ``skills(action='find')`` flow.
|
||||
assert "/skill" in content
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — model-initiated load risk gate (design §5.5 / Principle 7)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestSkillsLoadRiskGate:
|
||||
"""A high/critical-risk skill is PRINCIPAL-load-only: the model cannot
|
||||
activate it through skills(action='load') — that path is denied outright,
|
||||
forcing an explicit operator /skill. Lower-risk skills load as before.
|
||||
The scanner-computed risk_level is the gate signal (it already escalates
|
||||
for the auto_approve + allowed_tools authority the create path warns about),
|
||||
so no new schema is needed. The principal path (set_skill via /skill or
|
||||
cli --skill) is not routed through _prepare_skills_load and is not gated."""
|
||||
|
||||
@staticmethod
|
||||
def _load_item(session, name: str, risk: str):
|
||||
row = {"name": name, "risk_level": risk, "enabled": True, "content": "x"}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = row
|
||||
return session._prepare_skills("c1", {"action": "load", "name": name})
|
||||
|
||||
def test_high_risk_load_denied_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "danger", "high")
|
||||
assert item.get("needs_approval") is False
|
||||
assert "/skill danger" in item["error"]
|
||||
assert "high" in item["error"]
|
||||
|
||||
def test_critical_risk_load_denied_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "nuke", "critical")
|
||||
assert item.get("needs_approval") is False
|
||||
assert "critical" in item["error"]
|
||||
|
||||
def test_low_risk_load_allowed_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "safe", "low")
|
||||
assert item.get("needs_approval") is True
|
||||
assert item["action"] == "load"
|
||||
assert item["name"] == "safe"
|
||||
|
||||
def test_no_risk_load_allowed_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "plain", "")
|
||||
assert item.get("needs_approval") is True
|
||||
|
||||
def test_missing_skill_falls_through_to_exec_not_found(self) -> None:
|
||||
# Unknown name → the gate finds no row and passes through; the
|
||||
# not-found error is the exec path's job, so prep still asks for
|
||||
# approval rather than erroring on the gate.
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = None
|
||||
item = session._prepare_skills("c1", {"action": "load", "name": "ghost"})
|
||||
assert item.get("needs_approval") is True
|
||||
|
||||
def test_shared_helper_denies_high_and_critical_only(self) -> None:
|
||||
# The one shared gate used by skills(load) AND the spawn paths, so a
|
||||
# child spawn cannot route around it.
|
||||
session = _make_session()
|
||||
for tier in ("high", "critical"):
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = {
|
||||
"name": "x",
|
||||
"risk_level": tier,
|
||||
}
|
||||
assert "/skill x" in session._high_risk_skill_denied("x")
|
||||
for row in (
|
||||
{"name": "y", "risk_level": "low"},
|
||||
{"name": "y", "risk_level": ""},
|
||||
None,
|
||||
):
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = row
|
||||
assert session._high_risk_skill_denied("y") == ""
|
||||
|
||||
def test_risk_gate_fails_closed_on_storage_error(self) -> None:
|
||||
# A risk gate that can't read the row must DENY, never wave the skill
|
||||
# through (fail closed). Returning a denial (not "") also keeps
|
||||
# spawn_batch's per-row partial-success intact under a storage blip.
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.side_effect = RuntimeError("db down")
|
||||
denial = session._high_risk_skill_denied("z")
|
||||
assert denial != ""
|
||||
assert "z" in denial
|
||||
assert "/skill z" in denial
|
||||
|
||||
def test_risk_gate_fails_closed_on_no_storage(self) -> None:
|
||||
# Storage-unavailable (get_storage() is None) is treated the same as a
|
||||
# lookup fault — DENY, not allow. A gate that can't verify the risk
|
||||
# tier must not wave the skill through (Copilot review nit).
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage", return_value=None):
|
||||
denial = session._high_risk_skill_denied("z")
|
||||
assert denial != ""
|
||||
assert "/skill z" in denial
|
||||
|
||||
@@ -1,10 +1,14 @@
|
||||
"""Unit tests for ``_substitute_skill_args`` — SKILL.md spec placeholder
|
||||
substitution applied to skill bodies at load time.
|
||||
|
||||
Covers every placeholder form Turnstone implements (``${CLAUDE_SKILL_DIR}``
|
||||
is deferred — see #572) plus the spec's "append ARGUMENTS at end if no
|
||||
placeholder" rule and the single-pass guarantee against re-expansion of
|
||||
user-supplied values that happen to contain placeholder syntax.
|
||||
Covers every placeholder form Turnstone implements — including the
|
||||
``${TURNSTONE_*}`` canonical env vars and the ``${CLAUDE_SESSION_ID}`` /
|
||||
``${CLAUDE_EFFORT}`` back-compat aliases (there is deliberately NO
|
||||
``CLAUDE_SKILL_DIR`` alias), ``${TURNSTONE_SKILL_DIR}`` resolution when the
|
||||
caller supplies a materialized bundle path, plus the spec's "append
|
||||
ARGUMENTS at end if no placeholder" rule and the single-pass guarantee
|
||||
against re-expansion of user-supplied values that happen to contain
|
||||
placeholder syntax.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -136,6 +140,64 @@ class TestEnvironment:
|
||||
assert _sub("${CLAUDE_UNKNOWN_FOO}") == "${CLAUDE_UNKNOWN_FOO}"
|
||||
|
||||
|
||||
class TestEnvironmentAliases:
|
||||
"""``TURNSTONE_*`` is the canonical, vendor-neutral spelling;
|
||||
``CLAUDE_*`` is a permanent back-compat alias so skills imported from
|
||||
Claude Code / skills.sh keep resolving. Both map to one value."""
|
||||
|
||||
def test_turnstone_session_id(self) -> None:
|
||||
assert _sub("session ${TURNSTONE_SESSION_ID}") == "session ws-abc"
|
||||
|
||||
def test_turnstone_effort(self) -> None:
|
||||
assert _sub("effort ${TURNSTONE_EFFORT}") == "effort high"
|
||||
|
||||
def test_canonical_and_alias_agree(self) -> None:
|
||||
assert _sub("${CLAUDE_SESSION_ID}") == _sub("${TURNSTONE_SESSION_ID}") == "ws-abc"
|
||||
assert _sub("${CLAUDE_EFFORT}") == _sub("${TURNSTONE_EFFORT}") == "high"
|
||||
|
||||
def test_unknown_turnstone_var_left_as_literal(self) -> None:
|
||||
assert _sub("${TURNSTONE_UNKNOWN_FOO}") == "${TURNSTONE_UNKNOWN_FOO}"
|
||||
|
||||
|
||||
class TestSkillDir:
|
||||
"""``${TURNSTONE_SKILL_DIR}`` resolves to the materialized bundle path when
|
||||
the caller supplies one (it materializes resources BEFORE substituting) and
|
||||
degrades to a literal placeholder when the skill bundles no resources.
|
||||
``${CLAUDE_SKILL_DIR}`` is NOT a turnstone alias — it always stays literal
|
||||
(that name is the host's in bash; turnstone claims neither surface)."""
|
||||
|
||||
def test_turnstone_skill_dir_resolves(self) -> None:
|
||||
out = _substitute_skill_args(
|
||||
"cd ${TURNSTONE_SKILL_DIR}/scripts",
|
||||
arguments_str="",
|
||||
arg_names=[],
|
||||
ws_id="ws-abc",
|
||||
effort="high",
|
||||
skill_dir="/tmp/skill-xyz",
|
||||
)
|
||||
assert out == "cd /tmp/skill-xyz/scripts"
|
||||
|
||||
def test_claude_skill_dir_not_aliased(self) -> None:
|
||||
# Even with a materialized bundle, ${CLAUDE_SKILL_DIR} is left literal:
|
||||
# turnstone owns TURNSTONE_SKILL_DIR only, so the two never diverge from
|
||||
# the bash env (which likewise never sets CLAUDE_SKILL_DIR).
|
||||
out = _substitute_skill_args(
|
||||
"cd ${CLAUDE_SKILL_DIR}",
|
||||
arguments_str="",
|
||||
arg_names=[],
|
||||
ws_id="ws-abc",
|
||||
effort="high",
|
||||
skill_dir="/tmp/skill-xyz",
|
||||
)
|
||||
assert out == "cd ${CLAUDE_SKILL_DIR}"
|
||||
|
||||
def test_skill_dir_left_literal_when_unset(self) -> None:
|
||||
# Default skill_dir="" → placeholder stays literal (graceful),
|
||||
# not an empty path.
|
||||
assert _sub("${TURNSTONE_SKILL_DIR}") == "${TURNSTONE_SKILL_DIR}"
|
||||
assert _sub("${CLAUDE_SKILL_DIR}") == "${CLAUDE_SKILL_DIR}"
|
||||
|
||||
|
||||
class TestSinglePassGuarantee:
|
||||
"""A placeholder VALUE containing another placeholder must not be
|
||||
re-expanded — matches spec's "Substitution runs once" rule."""
|
||||
@@ -177,3 +239,45 @@ class TestIntegration:
|
||||
"Named: 123 resolved on main.\n"
|
||||
"Full: 123 main"
|
||||
)
|
||||
|
||||
|
||||
class TestSubstituteArgsToggle:
|
||||
"""``substitute_args=False`` (capability contexts: defaults, task_agent)
|
||||
leaves every invocation-arg form LITERAL while still resolving env vars,
|
||||
so literal ``$1`` / ``$ARGUMENTS`` prose or shell text isn't blanked."""
|
||||
|
||||
def test_arg_forms_left_literal(self) -> None:
|
||||
out = _substitute_skill_args(
|
||||
"run $0 $1 $ARGUMENTS $ARGUMENTS[2] $named",
|
||||
arguments_str="a b c", # present, but ignored under substitute_args=False
|
||||
arg_names=["named"],
|
||||
ws_id="ws-abc",
|
||||
effort="high",
|
||||
substitute_args=False,
|
||||
)
|
||||
assert out == "run $0 $1 $ARGUMENTS $ARGUMENTS[2] $named"
|
||||
|
||||
def test_env_still_resolves(self) -> None:
|
||||
out = _substitute_skill_args(
|
||||
"id ${TURNSTONE_SESSION_ID} at ${TURNSTONE_EFFORT} in ${TURNSTONE_SKILL_DIR}",
|
||||
arguments_str="",
|
||||
arg_names=[],
|
||||
ws_id="ws-abc",
|
||||
effort="high",
|
||||
skill_dir="/tmp/skill-x",
|
||||
substitute_args=False,
|
||||
)
|
||||
assert out == "id ws-abc at high in /tmp/skill-x"
|
||||
|
||||
def test_no_append_when_args_disabled(self) -> None:
|
||||
# The append-ARGUMENTS-at-end rule must not fire when arg substitution
|
||||
# is off, even if arguments_str is non-empty.
|
||||
out = _substitute_skill_args(
|
||||
"body with no placeholder",
|
||||
arguments_str="x y",
|
||||
arg_names=[],
|
||||
ws_id="ws-abc",
|
||||
effort="high",
|
||||
substitute_args=False,
|
||||
)
|
||||
assert out == "body with no placeholder"
|
||||
|
||||
@@ -21,12 +21,12 @@ exactly the case the visibility fix is meant to surface.
|
||||
from __future__ import annotations
|
||||
|
||||
import queue
|
||||
import threading
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.conftest import resolve_when_pending
|
||||
from turnstone.server import WebUI
|
||||
|
||||
|
||||
@@ -107,11 +107,11 @@ def test_policy_partial_allow_then_prompt_records_allowed_items() -> None:
|
||||
ui = WebUI(ws_id="ws-test")
|
||||
items = _make_items(("c1", "read_file"), ("c2", "bash"))
|
||||
|
||||
# ``approve_tools`` blocks on ``_approval_event.wait`` for the
|
||||
# prompt path. Schedule a deny-by-operator on a tiny timer so
|
||||
# the wait returns promptly; this test asserts on ring-buffer
|
||||
# state, not the verdict outcome, so a deny is fine.
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
# ``approve_tools`` blocks on ``_approval_event.wait`` for the prompt
|
||||
# path. Resolve (deny) once the approval is registered so the wait
|
||||
# returns promptly without the lost-wakeup race a fixed-delay timer has;
|
||||
# this test asserts on ring-buffer state, not the verdict, so a deny is fine.
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
|
||||
storage = MagicMock()
|
||||
|
||||
@@ -30,86 +30,37 @@ from typing import TYPE_CHECKING, Any, cast
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._wire_capture import RecordingClient
|
||||
from turnstone.core.lowering import repair_wire_messages
|
||||
from turnstone.core.providers._anthropic import AnthropicProvider
|
||||
from turnstone.core.providers._google import GoogleProvider
|
||||
from turnstone.core.providers._openai_chat import OpenAIChatCompletionsProvider
|
||||
from turnstone.core.providers._openai_responses import OpenAIResponsesProvider
|
||||
from turnstone.core.providers._protocol import ModelCapabilities
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
from turnstone.core.providers._protocol import LLMProvider
|
||||
|
||||
GOLDEN_DIR = Path(__file__).parent / "data" / "wire_payloads"
|
||||
_UPDATE = os.environ.get("UPDATE_WIRE_GOLDENS") == "1"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Recording fake client — captures the kwargs at each provider's SDK seam.
|
||||
# --------------------------------------------------------------------------- #
|
||||
class _EmptyStream:
|
||||
"""Stand-in for an SDK stream / stream-manager: empty iterable AND no-op CM."""
|
||||
|
||||
def __iter__(self) -> Iterator[Any]:
|
||||
return iter(())
|
||||
|
||||
def __enter__(self) -> _EmptyStream:
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc: object) -> None:
|
||||
return None
|
||||
|
||||
|
||||
class _Seam:
|
||||
"""Records the kwargs of a single SDK call, returns an empty stream stub."""
|
||||
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self._sink = sink
|
||||
|
||||
def __call__(self, **kwargs: Any) -> _EmptyStream:
|
||||
# Last write wins; only one seam is exercised per provider call.
|
||||
self._sink["payload"] = kwargs
|
||||
return _EmptyStream()
|
||||
|
||||
|
||||
class _Completions:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
|
||||
|
||||
class _Chat:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.completions = _Completions(sink)
|
||||
|
||||
|
||||
class _Messages:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class _Responses:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class RecordingClient:
|
||||
"""Fake SDK client exposing every provider's call seam, recording kwargs."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.captured: dict[str, Any] = {}
|
||||
self.messages = _Messages(self.captured)
|
||||
self.chat = _Chat(self.captured)
|
||||
self.responses = _Responses(self.captured)
|
||||
|
||||
|
||||
def _capture(
|
||||
provider: LLMProvider, *, model: str, messages: list[dict[str, Any]], **opts: Any
|
||||
provider: LLMProvider,
|
||||
*,
|
||||
model: str,
|
||||
messages: list[dict[str, Any]],
|
||||
caps: ModelCapabilities | None = None,
|
||||
**opts: Any,
|
||||
) -> dict[str, Any]:
|
||||
"""Drive ``create_streaming`` against a recording client; return the SDK kwargs."""
|
||||
"""Drive ``create_streaming`` against a recording client; return the SDK kwargs.
|
||||
|
||||
*caps* overrides the provider's own capability lookup — required for the
|
||||
anthropic-compatible lane, which has no static table (an operator-run model
|
||||
definition supplies its capabilities).
|
||||
"""
|
||||
client = RecordingClient()
|
||||
caps = provider.get_capabilities(model)
|
||||
caps = caps or provider.get_capabilities(model)
|
||||
# Mirror the session's wire prep: orphan repair runs once on the canonical
|
||||
# Turns (``ChatSession._prepare_wire_messages``), then the result is lowered
|
||||
# to the dict projection the translator consumes. Fixtures arrive
|
||||
@@ -145,7 +96,10 @@ def _assert_golden(name: str, payload: dict[str, Any]) -> None:
|
||||
norm = _normalize(payload)
|
||||
if _UPDATE:
|
||||
GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(norm, indent=2, sort_keys=True) + "\n")
|
||||
# ensure_ascii=False keeps non-ASCII (em dashes in repair
|
||||
# messages) literal, matching the existing baselines — a regen
|
||||
# must not churn unrelated lines into \uXXXX escapes.
|
||||
path.write_text(json.dumps(norm, indent=2, sort_keys=True, ensure_ascii=False) + "\n")
|
||||
return
|
||||
assert path.exists(), f"missing golden {name!r}; run UPDATE_WIRE_GOLDENS=1 to baseline"
|
||||
assert norm == json.loads(path.read_text()), f"wire-payload drift for {name!r}"
|
||||
@@ -321,3 +275,36 @@ def test_wire_payload(
|
||||
provider = factory()
|
||||
payload = _capture(provider, model=model, messages=[dict(m) for m in messages], **opts)
|
||||
_assert_golden(f"{provider_id}__{fixture_id}", payload)
|
||||
|
||||
|
||||
# The anthropic-compatible lane (vLLM /v1/messages) has no static capability
|
||||
# table and carries reasoning control in ``extra_body.chat_template_kwargs``
|
||||
# rather than the native ``thinking`` param — a wire shape the matrix above
|
||||
# never exercises (both AnthropicProvider rows are the native lane). Freeze it
|
||||
# with the capabilities a manual-mode model definition supplies and a real
|
||||
# effort level, so the graded ``reasoning_effort`` key is pinned in the golden
|
||||
# (never the native ``thinking`` param, and no forced ``temperature=1.0``).
|
||||
_COMPAT_CAPS = ModelCapabilities(
|
||||
context_window=262144,
|
||||
max_output_tokens=64000,
|
||||
token_param="max_tokens",
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
supports_reasoning_replay=True,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("fixture_id", sorted(_FIXTURES))
|
||||
def test_wire_payload_anthropic_compat(fixture_id: str) -> None:
|
||||
messages, opts = _FIXTURES[fixture_id]
|
||||
provider = AnthropicProvider(compat=True)
|
||||
payload = _capture(
|
||||
provider,
|
||||
model="qwen3.6-27b",
|
||||
messages=[dict(m) for m in messages],
|
||||
caps=_COMPAT_CAPS,
|
||||
reasoning_effort="high",
|
||||
**opts,
|
||||
)
|
||||
assert "thinking" not in payload, "compat lane must never send the native thinking param"
|
||||
_assert_golden(f"anthropic_compat__{fixture_id}", payload)
|
||||
|
||||
@@ -1598,7 +1598,7 @@ class TestDetailInteractive:
|
||||
"user_id": "test-user",
|
||||
"kind": "interactive",
|
||||
"pending_approval": False,
|
||||
"pending_approval_detail": None,
|
||||
"pending_approval_details": [],
|
||||
}
|
||||
|
||||
def test_pending_approval_fields_propagate_from_ui(self):
|
||||
@@ -1631,23 +1631,26 @@ class TestDetailInteractive:
|
||||
],
|
||||
"judge_pending": True,
|
||||
}
|
||||
loaded_ws.ui.serialize_pending_approval_detail = MagicMock(
|
||||
return_value={
|
||||
"call_id": "c-1",
|
||||
"judge_pending": True,
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
"func_name": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
"heuristic_verdict": {
|
||||
"recommendation": "approve",
|
||||
"risk_level": "low",
|
||||
"confidence": 0.9,
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
loaded_ws.ui.serialize_pending_approval_details = MagicMock(
|
||||
return_value=[
|
||||
{
|
||||
"cycle_id": "cyc-1",
|
||||
"call_id": "c-1",
|
||||
"judge_pending": True,
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
"func_name": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
"heuristic_verdict": {
|
||||
"recommendation": "approve",
|
||||
"risk_level": "low",
|
||||
"confidence": 0.9,
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
)
|
||||
mock_mgr = MagicMock()
|
||||
mock_mgr.get.return_value = loaded_ws
|
||||
@@ -1657,9 +1660,12 @@ class TestDetailInteractive:
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body["pending_approval"] is True
|
||||
assert body["pending_approval_detail"]["call_id"] == "c-1"
|
||||
assert body["pending_approval_detail"]["judge_pending"] is True
|
||||
items = body["pending_approval_detail"]["items"]
|
||||
details = body["pending_approval_details"]
|
||||
assert len(details) == 1
|
||||
assert details[0]["cycle_id"] == "cyc-1"
|
||||
assert details[0]["call_id"] == "c-1"
|
||||
assert details[0]["judge_pending"] is True
|
||||
items = details[0]["items"]
|
||||
assert len(items) == 1
|
||||
assert items[0]["func_name"] == "spawn_workstream"
|
||||
assert items[0]["needs_approval"] is True
|
||||
@@ -1680,7 +1686,7 @@ class TestDetailInteractive:
|
||||
loaded_ws.user_id = "test-user"
|
||||
loaded_ws.kind = "coordinator"
|
||||
loaded_ws.ui._pending_approval = {"items": []}
|
||||
loaded_ws.ui.serialize_pending_approval_detail = MagicMock(
|
||||
loaded_ws.ui.serialize_pending_approval_details = MagicMock(
|
||||
side_effect=RuntimeError("verdict object is malformed"),
|
||||
)
|
||||
mock_mgr = MagicMock()
|
||||
@@ -1691,7 +1697,7 @@ class TestDetailInteractive:
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body["pending_approval"] is True
|
||||
assert body["pending_approval_detail"] is None
|
||||
assert body["pending_approval_details"] == []
|
||||
|
||||
def test_lazy_rehydrates_on_miss(self):
|
||||
"""``mgr.get`` miss → ``mgr.open`` rehydrate. Same flow as coord;
|
||||
@@ -1801,7 +1807,7 @@ class TestTenantCheckOnReadEndpoints:
|
||||
# Sensitive fields the PR added must not surface for a
|
||||
# non-owning caller.
|
||||
assert "name" not in body
|
||||
assert "pending_approval_detail" not in body
|
||||
assert "pending_approval_details" not in body
|
||||
assert "user_id" not in body
|
||||
# And mgr.get was NEVER consulted — the gate fires first.
|
||||
mock_mgr.get.assert_not_called()
|
||||
@@ -1832,7 +1838,7 @@ class TestTenantCheckOnReadEndpoints:
|
||||
body = r.json()
|
||||
assert body["ws_id"] == ws_id
|
||||
assert body["pending_approval"] is False
|
||||
assert body["pending_approval_detail"] is None
|
||||
assert body["pending_approval_details"] == []
|
||||
|
||||
def test_history_404s_when_tenant_check_rejects(self, _inject_storage):
|
||||
"""A non-owning interactive caller reading another user's ws_id
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
"""turnstone - Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."""
|
||||
|
||||
__version__ = "1.7.0a6"
|
||||
__version__ = "1.7.0rc1"
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user