mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-13 07:22:24 -06:00
Compare commits
87 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4638d22bd0 | |||
| ee3bd1dcf2 | |||
| ae3a83ccce | |||
| 0f17433e1f | |||
| 043554bb2f | |||
| 8389808add | |||
| 6cbef4f633 | |||
| 2b6dde4f7e | |||
| fbe31b9885 | |||
| ef13f40cf5 | |||
| bfa1b104cf | |||
| 324a1d1a35 | |||
| 95ab88ff6f | |||
| d29840f985 | |||
| 44c0b9c340 | |||
| cdbdf3dc2b | |||
| 8aabb061c2 | |||
| 8bd638569f | |||
| 251dc44a46 | |||
| efd0a1d000 | |||
| 2f93c39fd3 | |||
| f27ce104c6 | |||
| 20a61b692b | |||
| 1966107efe | |||
| d5b2fe6e45 | |||
| 1569819750 | |||
| 4107a30148 | |||
| d06d88b83f | |||
| c411aac939 | |||
| 104715b650 | |||
| 59a9899149 | |||
| 4da7c3b91c | |||
| 012f4e3e16 | |||
| 3636724848 | |||
| eeda5ac312 | |||
| be872b840f | |||
| ee94ae8ba1 | |||
| 357d00400e | |||
| 3615f98c19 | |||
| 801d5dfb59 | |||
| a352b20786 | |||
| 6f8efaa44e | |||
| 9c90fe2722 | |||
| 1035fe05eb | |||
| 530958e06b | |||
| e136237b63 | |||
| 41e9907803 | |||
| 7c34d859b4 | |||
| 16ee4e12ef | |||
| 976c07d047 | |||
| 8da5dc3f5a | |||
| 7f0e0406b3 | |||
| 6c94514106 | |||
| 0d0fe8dd71 | |||
| 18c3301428 | |||
| 185dcc2960 | |||
| 0fe8e4106f | |||
| fd65a490dc | |||
| 3607517814 | |||
| a0e04a8588 | |||
| ffe8214cfe | |||
| 1f63f622c9 | |||
| f4701bf0f9 | |||
| 06cc184227 | |||
| 59a527f2f2 | |||
| d7941c88be | |||
| c64dc16319 | |||
| d564cee43d | |||
| 2cf23b6fe2 | |||
| 68b22adfa3 | |||
| 9289693730 | |||
| deff44bcea | |||
| 217d3a3a9b | |||
| bcf509a440 | |||
| 10f726f83d | |||
| acc262c405 | |||
| 62034378c6 | |||
| bcb8c5ab88 | |||
| fec5067fcd | |||
| b0ed67aa60 | |||
| c023272b16 | |||
| 3568a6db50 | |||
| 9bf8d5699b | |||
| c0ff00a1ff | |||
| 45010f5890 | |||
| 845df69031 | |||
| d47d528d9a |
@@ -0,0 +1,5 @@
|
||||
# Funding platforms for the GitHub "Sponsor" button.
|
||||
# https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/displaying-a-sponsor-button-in-your-repository
|
||||
|
||||
github: [eous]
|
||||
custom: ["https://paypal.me/eousphoros"]
|
||||
@@ -152,7 +152,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- run: uv lock --check
|
||||
@@ -161,7 +161,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@6c0083bb7289c31716797a039b6367b3079cc46e # v1
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
allowed_bots: 'renovate[bot]' # let Renovate PRs get reviewed
|
||||
|
||||
@@ -45,7 +45,7 @@ jobs:
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@6c0083bb7289c31716797a039b6367b3079cc46e # v1
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
|
||||
|
||||
@@ -54,7 +54,7 @@ jobs:
|
||||
|
||||
- name: Log in to GHCR
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v4
|
||||
uses: docker/login-action@af1e73f918a031802d376d3c8bbc3fe56130a9b0 # v4
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
@@ -78,7 +78,7 @@ jobs:
|
||||
fi
|
||||
echo "tags=${TAGS}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4
|
||||
- uses: docker/setup-buildx-action@bb05f3f5519dd87d3ba754cc423b652a5edd6d2c # v4
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
|
||||
- name: Build and push
|
||||
|
||||
@@ -28,3 +28,4 @@ tools/skill_audit_analysis/data/
|
||||
tools/skill_audit_analysis/output/
|
||||
design_ideas/
|
||||
.claude/
|
||||
docs/design/
|
||||
|
||||
+194
-8
@@ -6,14 +6,94 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||
and this project adheres to [PEP 440](https://peps.python.org/pep-0440/) for
|
||||
version numbers (`X.Y.Z`, with `X.Y.ZaN` / `bN` / `rcN` for pre-releases).
|
||||
|
||||
Three release tracks are maintained — the current stable, one prior
|
||||
stable, and the experimental line:
|
||||
Two active release tracks are maintained — the current stable and the
|
||||
experimental line:
|
||||
|
||||
- **`stable/1.5`** — patch-only (`v1.5.x`)
|
||||
- **`stable/1.6`** — patch-only (`v1.6.x`)
|
||||
- **`stable/1.7`** — patch-only (`v1.7.x`)
|
||||
- **`main`** — experimental (next major)
|
||||
|
||||
## [Unreleased]
|
||||
Earlier stable lines (`stable/1.6`, `stable/1.5`) are frozen.
|
||||
|
||||
## [1.7.1]
|
||||
|
||||
A maintenance and hardening patch for the 1.7 line. No schema migrations;
|
||||
the credential-redaction work below is additive and needs no configuration
|
||||
change. The one new operator-facing knob is the opt-in `[oidc]
|
||||
allow_private_network` flag (default off).
|
||||
|
||||
### Security
|
||||
|
||||
- **Credential redaction hardened across the tool-call surface** — the
|
||||
redactor that scrubs secrets from tool arguments and log previews was
|
||||
reworked on both the backend and the browser to close several leak paths
|
||||
and to fix false-positive and performance issues. Malformed tool-call
|
||||
arguments are now legalised before they reach the wire; the tool-args log
|
||||
preview scrubs credentials and control characters; and the coordinator's
|
||||
tool-call cards gain a matching client-side redaction pass so the JS and
|
||||
backend redactors stay at parity. Pattern coverage now includes
|
||||
`secret_access_key` / `aws_secret_access_key` multi-segment keys, bare
|
||||
`token=` / `key=` forms (guarded by a negative lookbehind to avoid
|
||||
false positives), and SQLAlchemy `+driver`-qualified connection-string
|
||||
schemes matched case-insensitively.
|
||||
- **OIDC SSRF guard: `[oidc] allow_private_network` opt-in** — self-hosted
|
||||
identity providers on private networks can now be reached by setting
|
||||
`allow_private_network = true` under `[oidc]` (default off; the MCP OAuth
|
||||
path stays strict). Rejections of discovered endpoints carry the opt-in
|
||||
hint so the misconfiguration is self-explanatory. See `docs/oidc.md`.
|
||||
|
||||
### Added
|
||||
|
||||
- **Persona discoverability + forgiving name resolution** — personas are
|
||||
now discoverable by agents, and persona-name resolution tolerates
|
||||
case/whitespace variation; a not-found resolution reports the offending
|
||||
input verbatim instead of a bare error.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **MCP transport lifecycles routed through per-entry owner tasks**
|
||||
(#787/#788) — static and pooled MCP transport lifecycles are now driven
|
||||
by per-server / per-entry owner tasks, with a hardened disarm-sweep loop
|
||||
guard and targeted exception handling in place of a broad `BaseException`
|
||||
arm, so a dying transport can no longer spin the CPU or strand delivery.
|
||||
- **Client-construction failures surface as misconfiguration, not raw
|
||||
500s** — a model whose client cannot be constructed now reports a factory
|
||||
misconfiguration, and the raw exception text is kept out of the resulting
|
||||
503 response.
|
||||
- **Postgres history search survives oversized rows** — a conversation row
|
||||
exceeding Postgres' full-text limits no longer aborts history search.
|
||||
- **Agent-tool render is idempotent** — tool rendering no longer deep-copies
|
||||
a tool definition until a description actually changes, so no-persona
|
||||
sessions share the tool constant (correctness plus a hot-path allocation
|
||||
win).
|
||||
- **Private-project workstream visibility scoped to members** — workstreams
|
||||
in a private project are visible to project members only, not to every
|
||||
admin; coordinator tenancy checks now use request-scoped storage.
|
||||
- **Pane hotkeys work off macOS and match across surfaces** — the pane
|
||||
keyboard shortcuts no longer collide with browser accelerators on
|
||||
non-macOS platforms and behave consistently across surfaces.
|
||||
|
||||
## [1.7.0]
|
||||
|
||||
The headline of the 1.7 line is **Personas** — operator-authored control
|
||||
over how each workstream composes its system message and capability
|
||||
envelope. The rest of the release hardens the pieces a persona leans on:
|
||||
concurrent approvals, cross-provider reasoning-effort control, cooperative
|
||||
compaction, multi-user session safety, and MCP resilience for unattended
|
||||
work.
|
||||
|
||||
> **⚠️ Before upgrading:** 1.7.0 adds Alembic migrations `062`–`065`,
|
||||
> applied automatically on first start (projects, personas, and two
|
||||
> smaller schema tidy-ups). Migration `063` creates the `personas` table
|
||||
> with its six seed personas and converts existing `creative_mode`
|
||||
> workstreams to the `writer` persona in place. The changes are additive
|
||||
> to your conversation data, but — as always — back up your storage before
|
||||
> upgrading (`pg_dump` for PostgreSQL; copy the database file for SQLite).
|
||||
|
||||
**Breaking changes at a glance** (details in the sections below): the
|
||||
`/creative` REPL toggle removed (replaced by the `writer` persona), the
|
||||
`turnstone-bootstrap` entry point renamed to `turnstone-doctor`, and the
|
||||
approval-status API/SDK field `pending_approval_details` changed from a
|
||||
single object to a list (one entry per concurrent approval cycle).
|
||||
|
||||
### Added
|
||||
|
||||
@@ -31,12 +111,106 @@ stable, and the experimental line:
|
||||
`turnstone --persona <name>`); authored in the console's new
|
||||
Governance → Personas tab (`persona.{create,read,write}` perms,
|
||||
archive-only lifecycle). See `docs/personas.md`.
|
||||
- **Projects — governed resource containers** (#724) — group workstreams
|
||||
and their resources under a project (migration `062`), with
|
||||
project-scoped memory, a per-project resources view, a project column on
|
||||
the saved list, and server-enforced private-project workstream
|
||||
visibility.
|
||||
- **Task-agent sub-harness** (#732) — a spawned task agent now runs on its
|
||||
own Turn-IR sub-harness with parent-tagged step events: its sub-tool
|
||||
steps nest inside an expandable card in the parent trajectory, its
|
||||
sub-trajectory is recallable, and each agent gets read isolation from
|
||||
its siblings.
|
||||
- **MCP static-server autonomous reconnect** (#768) — statically
|
||||
configured MCP servers are now kept live by a health loop
|
||||
(capped-jittered backoff, ping-based liveness) instead of silently
|
||||
staying dead after the first transport drop.
|
||||
- **Attachments — capability-gated client-side fallback** — when the
|
||||
active model can't natively handle an attachment, the client degrades
|
||||
gracefully (PDF → extracted text, audio → transcript) instead of
|
||||
failing the turn.
|
||||
- **Eval measurement / optimizer split** (#763, #765) — `turnstone-eval`
|
||||
is now a measure-only substrate with the prompt optimizer factored out,
|
||||
plus a new skill-adherence measurement mode.
|
||||
- **Deployment examples** — a vLLM + LiteLLM unified-memory inference
|
||||
example showing a 3-model co-resident stack with an HF loader (#686,
|
||||
#688), and an Altair + `vl-convert-python` visualization stack (#685).
|
||||
- **Concurrent approvals and a long-session frontend overhaul** (#754,
|
||||
#755, #773, #775) — the live-session frontend was reworked for long
|
||||
runs (the pipeline is wedge-proofed and its hot paths de-O(N)'d), and on
|
||||
top of it a workstream can now hold more than one tool call awaiting
|
||||
approval at a time. Each parallel batch gets its own approval cycle,
|
||||
with one card per pending call in the interactive and coordinator UIs,
|
||||
cycle-keyed tracking in Slack and Discord, and cycle-routed resolution
|
||||
across the server/console/SDK APIs; sub-agent tool gates run the
|
||||
intent-judge pipeline as their own generation. The send button no longer
|
||||
sticks disabled after a batch resolves — orphaned approval cycles are
|
||||
pruned and the app is the sole owner of the button state.
|
||||
*(BREAKING: the `pending_approval_details` field is now a list, oldest
|
||||
first.)*
|
||||
- **Reasoning-effort control on every provider lane** (#771, #774) — the
|
||||
session effort knob now reaches local backends too: it drives
|
||||
`chat_template_kwargs` on the anthropic-compatible and openai-compatible
|
||||
lanes and threads through to Gemini and xAI, alongside the commercial
|
||||
providers that handle effort natively. The console surfaces each model's
|
||||
effective effort ladder in plain words and adds an always-on
|
||||
thinking-mode option to the model form. Effort snapping is ordinal —
|
||||
it rounds up and caps at the model's ceiling rather than silently
|
||||
dropping.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Skills are capability-context, not identity** (#762) — a task agent's
|
||||
identity now comes from its persona; an applied skill's body is demoted
|
||||
to capability context and moved out of the identity system message.
|
||||
Skill-body substitution is unified across every invocation context so
|
||||
the same skill renders identically whether loaded interactively, by the
|
||||
model, or inside a sub-agent.
|
||||
- **`turnstone-doctor` replaces `turnstone-bootstrap`** (#718)
|
||||
*(BREAKING)* — the setup/diagnostics entry point is renamed; update any
|
||||
scripts or service units that invoke `turnstone-bootstrap`.
|
||||
- **Honest cancellation dispositions** — cancelled or timed-out
|
||||
side-effecting tools now report an `UNKNOWN` disposition rather than a
|
||||
flat failure, tool dispositions are typed (not just prose), and a
|
||||
coordinator cancel propagates down the sub-tree.
|
||||
- **Multi-user shared-workstream context** (#750) — in a shared
|
||||
workstream, send is gated to the acting participant while a turn is in
|
||||
flight (both the interactive and coordinator surfaces), cross-user
|
||||
mid-turn interjections are blocked, and shared-workstream state plus
|
||||
fork sender attribution are now durable.
|
||||
- **Cooperative compaction** (#730) — the context budget is anchored to
|
||||
the provider's true capacity, the summary call is chunked so it can't
|
||||
overflow, and the active plan and the outstanding ask are carried across
|
||||
compaction verbatim. The `recall` tool is scoped to the compacted-away
|
||||
past.
|
||||
- **Intent judge sees the full tool arguments** (#760) — the judge's
|
||||
argument projection is no longer narrowed, so it stops issuing confident
|
||||
false denials on a partial view. The output-guard judge sources its real
|
||||
context window, and `context_window = 0` in `config.toml` now means
|
||||
auto-detect.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Compaction resume hardening** (#731) — checkpoint markers are
|
||||
persisted so resume rehydration is bounded, context-overflow on resume
|
||||
is recovered across providers, and a recognized rate-limit is no longer
|
||||
misclassified as context overflow.
|
||||
- **MCP unattended-work resilience** (#706, #742, #767) — dead-transport
|
||||
handling is completed, consented OAuth (OBO) tokens are refreshed
|
||||
proactively so autonomous runs don't strand on an expired grant, the
|
||||
Entra ID on-behalf-of impersonation flow blockers are closed (migration
|
||||
`065` adds the OIDC `oid`), and OAuth refresh failures are classified so
|
||||
a transient blip never revokes consent nor a dead grant strands the
|
||||
user.
|
||||
- **Memory writes** (#735) — save/update is a single atomic upsert, and
|
||||
writing a memory no longer recomposes the system prefix mid-session.
|
||||
|
||||
### Removed
|
||||
|
||||
- **`/creative` removed** *(BREAKING)* — the REPL toggle (and its tab
|
||||
completion) is gone; the `writer` seed persona replaces it — start a
|
||||
session with `turnstone --persona writer` or pick *Writer* in the web
|
||||
- **`/creative` removed** *(BREAKING)* — subsumed by the Personas feature
|
||||
above: the REPL toggle (and its tab completion) is gone, and the
|
||||
`writer` seed persona replaces it — start a session with
|
||||
`turnstone --persona writer` or pick *Writer* in the web
|
||||
pickers. Unlike the old fork, the writer persona composes the full
|
||||
system message, so session context and mandatory prompt policies now
|
||||
apply to prose-only sessions too. The `creative_mode` key in
|
||||
@@ -45,6 +219,18 @@ stable, and the experimental line:
|
||||
automatically, so they resume as writing sessions rather than as
|
||||
legacy defaults.
|
||||
|
||||
### Security
|
||||
|
||||
- **High-risk skill activation is gated** (#762) — a model-initiated load
|
||||
of a `high`- or `critical`-risk skill is gated and fails closed when the
|
||||
backing storage is unavailable, so an untrusted turn can't silently
|
||||
pull in a dangerous capability.
|
||||
- **Dependency security floors** — `cryptography` and `starlette` are
|
||||
pinned to security-fixed minimums.
|
||||
- **CI publish hardening** — the vendored-JS dispatch path refuses fork
|
||||
PRs, and `workflow_run` publishing is gated to same-repo tag pushes, so
|
||||
a fork can't trigger a release build.
|
||||
|
||||
## [1.6.0]
|
||||
|
||||
The first stable release of the 1.6 line — and the first under Apache 2.0.
|
||||
|
||||
+1
-1
@@ -8,7 +8,7 @@ FROM python:3.14-slim
|
||||
LABEL org.opencontainers.image.title="turnstone" \
|
||||
org.opencontainers.image.description="Multi-node AI orchestration platform"
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.26 /uv /usr/local/bin/uv
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.27 /uv /usr/local/bin/uv
|
||||
|
||||
# Remove the slim image's man page exclusion so man-db has actual content
|
||||
RUN rm -f /etc/dpkg/dpkg.cfg.d/docker
|
||||
|
||||
+4
-2
@@ -137,9 +137,9 @@ With that caveat, **the interlingua and the certificate are one object seen twic
|
||||
|
||||
Borrowed theorems are real; the framings are not — keep them separate. Some framings are nonetheless *corroborated* — independently reached from another field — a third grade, weaker than proof and noted last.
|
||||
|
||||
**Proven (citable).** Foster–Lyapunov drift ⇒ positive recurrence + $\mathbb{E}[\tau]\le V(s_0)/\varepsilon$ (Foster 1953; Meyn & Tweedie, *Markov Chains and Stochastic Stability*, 1993) — positive recurrence needs the usual irreducibility/petite-set hypotheses, while the absorbing-halt case used here needs only the weaker supermartingale optional-stopping hitting-time bound. The minimal $V$ is the expected hitting time, by first-step analysis + optional stopping (Norris, *Markov Chains*, 1997). For an absorbing chain that expected hitting time is the row sum of the fundamental matrix $N=\sum_{n\ge0}Q_{\mathrm{tr}}^{\,n}$ (Kemeny & Snell, *Finite Markov Chains*, 1960), with the general-state analogue the potential (Green) operator (Revuz, *Markov Chains*, 1984). Koopman's linear-operator view of nonlinear dynamics is classical (Koopman 1931), and Lyapunov functions can be assembled from its eigenfunctions when the spectrum is suitable (Mauroy & Mezić, 2016). You certify a candidate $\hat V$ by a *proven* drift inequality rather than by deriving $V^\star$, and estimate it empirically only where a proof is out of reach — the empirical drift checks, it does not certify (neural-Lyapunov: Chang, Roohi & Gao, *Neural Lyapunov Control*, NeurIPS 2019, arXiv:2005.00611). A classical monotone data-flow analysis gets its $V$ for free because a finite-height lattice is a well-founded descent (Kildall, POPL 1973). The gate-a-plant architecture itself is classical: supervisory control theory synthesizes a deterministic supervisor that disables controllable events of a plant it does not author, with the supremal controllable sublanguage as the largest admissible behavior (Ramadge & Wonham, SIAM J. Control and Optimization, 1987) — $\gamma$ is that supervisor, with a learned stochastic plant on general state spaces. The successor representation is Dayan (*Improving Generalization for Temporal Difference Learning: The Successor Representation*, Neural Computation 1993). Dialect-stack architecture: MLIR (Lattner et al., CGO 2021, arXiv:2002.11054); learned pass-ordering: MLGO (Trofin et al., arXiv:2101.04808). Single-pass low-depth expressivity: log-precision transformers are simulable by constant-depth logspace-uniform threshold circuits ($\mathsf{TC}^0$) (Merrill & Sabharwal, *The Parallelism Tradeoff: Limitations of Log-Precision Transformers*, TACL 2023) — fixed/constant precision is a stronger restriction, added autoregressive steps escape it (Merrill & Sabharwal, *The Expressive Power of Transformers with Chain of Thought*, ICLR 2024), and growing precision changes the picture, so the bound is suggestive for deployed models, not literal.
|
||||
**Proven (citable).** Foster–Lyapunov drift ⇒ positive recurrence + $\mathbb{E}[\tau]\le V(s_0)/\varepsilon$ (Foster 1953; Meyn & Tweedie, *Markov Chains and Stochastic Stability*, 1993) — positive recurrence needs the usual irreducibility/petite-set hypotheses, while the absorbing-halt case used here needs only the weaker supermartingale optional-stopping hitting-time bound. The minimal $V$ is the expected hitting time, by first-step analysis + optional stopping (Norris, *Markov Chains*, 1997). For an absorbing chain that expected hitting time is the row sum of the fundamental matrix $N=\sum_{n\ge0}Q_{\mathrm{tr}}^{\,n}$ (Kemeny & Snell, *Finite Markov Chains*, 1960), with the general-state analogue the potential (Green) operator (Revuz, *Markov Chains*, 1984). Koopman's linear-operator view of nonlinear dynamics is classical (Koopman 1931), and Lyapunov functions can be assembled from its eigenfunctions when the spectrum is suitable (Mauroy & Mezić, 2016). You certify a candidate $\hat V$ by a *proven* drift inequality rather than by deriving $V^\star$, and estimate it empirically only where a proof is out of reach — the empirical drift checks, it does not certify (neural-Lyapunov: Chang, Roohi & Gao, *Neural Lyapunov Control*, NeurIPS 2019, arXiv:2005.00611). A classical monotone data-flow analysis gets its $V$ for free because a finite-height lattice is a well-founded descent (Kildall, POPL 1973). The gate-a-plant architecture itself is classical: supervisory control theory synthesizes a deterministic supervisor that disables controllable events of a plant it does not author, with the supremal controllable sublanguage as the largest admissible behavior (Ramadge & Wonham, SIAM J. Control and Optimization, 1987) — $\gamma$ is that supervisor, with a learned stochastic plant on general state spaces; the same theory's controllability condition (specifications must be closed under *uncontrollable* events) and its nonblocking requirement are the proven ancestors of gate-early-on-irreversibles and of the always-enabled escalation the appendix requires behind any learned veto. Covert-channel discipline — identify the channel, measure its bandwidth in bits, audit what cannot be closed — is the TCSEC lineage (*A Guide to Understanding Covert Channel Analysis of Trusted Systems*, NCSC-TG-030, 1993). The successor representation is Dayan (*Improving Generalization for Temporal Difference Learning: The Successor Representation*, Neural Computation 1993). Dialect-stack architecture: MLIR (Lattner et al., CGO 2021, arXiv:2002.11054); learned pass-ordering: MLGO (Trofin et al., arXiv:2101.04808). Single-pass low-depth expressivity: log-precision transformers are simulable by constant-depth logspace-uniform threshold circuits ($\mathsf{TC}^0$) (Merrill & Sabharwal, *The Parallelism Tradeoff: Limitations of Log-Precision Transformers*, TACL 2023) — fixed/constant precision is a stronger restriction, added autoregressive steps escape it (Merrill & Sabharwal, *The Expressive Power of Transformers with Chain of Thought*, ICLR 2024), and growing precision changes the picture, so the bound is suggestive for deployed models, not literal.
|
||||
|
||||
**Asserted (ours — not theorems).** That the harness is best modeled as nested stopped chains; that $V^\star$ is incompressible (no compression theorem); that "no lattice for $f(\cdot\,;W)$" means none is *known*, not that none exists; and everything under *Where this points* — including the Koopman/certificate co-determination, which is well-posed only under the spectral assumptions noted there, and the interlingua/certificate identification; and the design rules read off the objects rather than proven from them — the single-trusted-writer completion of the provenance partition, the narrow-only rule for learned checks, the composition law of the appendix. These organize the design; they are not results.
|
||||
**Asserted (ours — not theorems).** That the harness is best modeled as nested stopped chains; that $V^\star$ is incompressible (no compression theorem); that "no lattice for $f(\cdot\,;W)$" means none is *known*, not that none exists; and everything under *Where this points* — including the Koopman/certificate co-determination, which is well-posed only under the spectral assumptions noted there, and the interlingua/certificate identification; and the design rules read off the objects rather than proven from them — the single-trusted-writer completion of the provenance partition, the narrow-only rule for learned checks and its influence-side twin (verdict payloads to the plant selected, never generated), the composition law of the appendix. These organize the design; they are not results.
|
||||
|
||||
**Converged-upon (independently arrived at, from other framings).** The *Asserted* claims above are ours but not ours alone; several are reached independently, from starting points unconnected to this framing — which is the corroboration a definition earns: not a chorus of agreement (the systems below often disagree on method and goal), but that work approaching from capabilities, reinforcement learning, control theory, software architecture, and language-modeling theory each lands on a piece of the same object. That the **deterministic controller, not the model, carries the guarantee** is reached from four directions — capability and information-flow control (CaMeL: Debenedetti et al., *Defeating Prompt Injections by Design*, arXiv:2503.18813, securing the agent even when the underlying model is susceptible); reinforcement learning (shielding: Alshiekh et al., *Safe Reinforcement Learning via Shielding*, AAAI 2018, arXiv:1708.08611 — a deterministic reactive shield filtering a learned policy's actions against a temporal-logic specification); control theory (*Stable Agentic Control*, arXiv:2605.03034, enforcing finite action catalogs at the tool-output interface under a Lyapunov input-to-state-stability certificate against adversarial disturbance); and software architecture (the plan-then-execute / control-flow-integrity line, e.g. Beurer-Kellner et al., *Design Patterns for Securing LLM Agents against Prompt Injections*, arXiv:2506.08837). The **certified-vs-measured split** is reached from the construction side (CaMeL's provable security) and, independently, from the destruction side (guardrail-evasion results — *Bypassing Prompt Injection and Jailbreak Detection in LLM Guardrails*, arXiv:2504.11168, the v1 title — later versions retitle it; *No Free Lunch with Guardrails*, arXiv:2504.00441), with verification-oriented work stating it as the motivating gap (*Towards Verifiably Safe Tool Use for LLM Agents*, arXiv:2601.08012; VeriGuard, arXiv:2510.05156): a learned safeguard raises the odds of detection but cannot guarantee safety against a persistent attacker. The **inner readout as a composition of Markov kernels** is independently formalized in language-modeling theory — the autoregressive step as kernel composition in the category $\mathsf{Stoch}$ (*A Markov Categorical Framework for Language Modeling*, arXiv:2507.19247), and the broader "LLMs as Markov chains" line — though that work models the inner kernel alone and never closes it into an agentic loop, which is exactly the seam this definition adds. That **provenance shrinks the admissible adversary** is reached by datamarking / spotlighting (Hines et al., arXiv:2403.14720, 2024) and by CaMeL's data/control-flow separation; and a systematization of prompt injection against agentic coding assistants reaches the same verdict from the attack side — mitigation must be *architectural*, not model-level (*Prompt Injection Attacks on Agentic Coding Assistants*, arXiv:2601.17548); the sharper open problem this object is built to answer — formally specify the trust boundaries, then verify implementations respect them — is our phrasing of where that verdict points, not the paper's. Two convergences are weaker, and flagged. The **reach-avoid hitting-time certificate** is the independently developed reach-avoid supermartingale (RASM, arXiv:2210.05308, AAAI 2023) and stochastic Lyapunov–barrier apparatus, and its *hardness* is corroborated — expected-stopping-time problems for Markov chains are inter-reducible with the Positivity problem, a relative of the Skolem problem (Chatterjee & Doyen, *Stochastic Processes with Expected Stopping Time*, arXiv:2104.07278) — but this supports generic hardness only, not the specific incompressibility-at-$|W|$ conjecture, which remains ours and unproven. And **injection as an adversarial policy** is corroborated as a minimax game in the *detection* setting (DataSentinel: Liu et al., *A Game-Theoretic Detection of Prompt Injection Attacks*, arXiv:2504.11358) and as adversarial-disturbance robustness (*Stable Agentic Control*, above) — but no prior work assembles it as reach-avoid over the tool-output kernel with the gate as the irreversibility margin; here the relation is adjacency, not convergence.
|
||||
|
||||
@@ -175,6 +175,8 @@ The same pressure lands on tooling from a second direction. The $\mathsf{action\
|
||||
|
||||
That last kind marks the seam where the gate stops being able to stay pure, and it is the same seam the rest of this document is built around. The *structural* slice of intent — does the action cohere with the plan in $s$ — is a deterministic predicate over $s$ and $y$, effect-free, and belongs in $\gamma$ without reservation. But whether an action matches what the user *actually meant*, in the full semantic sense, is exactly the thing the definition says cannot be checked: natural language is all undefined behavior, with no source-language standard to validate against. So a semantic intent check is a *learned* check, and an LLM judging "is this what they wanted" is a **stochastic kernel** — putting it inside $\gamma$ breaks the property the gate exists to hold, by the same move flagged for the fold-back verifier: a learned judge is a kernel, and belongs in $M_W$, not in a deterministic map. Semantic intent therefore does not live *in* the gate; it is a plant call — a separate authorize-the-proposal pass through $M_W$ whose output $\gamma$ then deterministically gates — or it is drift you measure, never a guarantee you hold. That nested call is not a new kind of thing: it is a mini-harness inside the gate's decision — a judge $M_W$, its own syntactic readout, its own deterministic gate — so its failure case answers itself, the inner gate fail-closing on an unparseable or low-confidence judgment exactly as the outer one does, because it *is* one. The object is **closed under this construction**: semantic gating is added by recursion, not by a new primitive. One constraint on the recursion is load-bearing enough to be a rule, because it is where this entry meets the provenance partition of the body: the judge's verdict is derived, through a learned kernel, from the very content an adversary may have bent, so folding it into authorization is exactly the fold the partition forbids — *unless the verdict can only cost capability*. **A learned check may narrow the deterministic admissible set; it must never widen it.** Judge-as-veto is safe by construction: attacker influence over the judge can at worst manufacture a denial, a liveness cost the certificate already prices. Judge-as-approver — a verdict granting what the deterministic checks alone would refuse, or standing in for the trusted principal's confirmation — lowers the certified floor to those deterministic checks alone; if avoiding $B$ depended on the deny the judge now withholds on the adversary's behalf, the certificate is gone. Only the trusted principal widens authorization; learned kernels only narrow it. (The recursion already obeys this: the mini-harness's inner gate fail-closes to $\bot$ — a deny — which is why the construction was safe to add at all.) The cost is real and worth stating — a judge pass is another full model call, with its latency and tokens — so it is a decision about *which* actions warrant it, not a free wrapper for all of them. The gate widens to every deterministic predicate over $s$ and $y$; it does not widen to the one predicate the document says is not deterministically checkable.
|
||||
|
||||
One more caveat keeps the veto's pricing honest, because a denial is free only in the *authority* lattice. In the dynamics it is an input like any other — folded into $s$, lowered into the next context, conditioning the plant's next proposal — so adversarial influence over a judge is influence over the *trajectory*: a selection channel (deny all but the path toward $B$, and the admissible set the plant experiences is a maze the adversary curated), and a targeted-liveness channel against load-bearing actions — the unstated dual of judge-as-approver: if avoiding $B$ depends on the action the judge now denies on the adversary's behalf, fail-closed's safe landing is an obligation the design earns per-state, not an axiom it inherits. The supervisory ancestry supplies the discipline: a learned veto requires a **nonblocking escape it cannot disable** — an always-enabled route to the trusted principal behind a bounded retry budget — or manufactured denials strand the run, or steer it. And whatever a verdict carries *back to the plant* is a second channel, wearing the judge's authority framing. Free prose there is *generative* influence — injected context, priced by the minimax descent, never by the veto's zero-widening — so the narrow-only rule has an influence-side twin: **a learned verdict's payload to the plant is selected, never generated** — controller-authored symbols, typed citations validated like any effect record, template text with no interpolated model prose — its per-verdict capacity a designed constant rather than a measured hope, and the residual selection pattern audited as the covert channel it is. The alphabet's bound is not a count but two thresholds: symbols become tokens when their semantics stop being controller-authored — the registry the trusted writer can actually audit is the real constant, and borrowed alphabets with upstream owners (a linter's rule registry) spend that budget well — and tokens become language when composition turns productive, arrangement carrying meaning the controller never wrote. Below both thresholds the alphabet may be as large as the audit budget affords. The strongest form dissolves the learned verdict into *scheduling*: the learned component chooses which deterministic checks to run — pass-ordering over verification passes — and the only verdicts that flow anywhere are what the oracles actually said, leaving attention misallocation, a liveness cost, as the entire attack surface.
|
||||
|
||||
But "before any invocation" has to be read as *before any effect*, which is sharper than it sounds — and the reason is the irreversibility point above: you validate before execution because execution is what you cannot take back, so the real invariant is **no effect crosses $\gamma$ unvalidated**. That catches three cases the naive reading misses. *Reads are not free*: a read-only call is still an injection vector (it pulls attacker-controlled content into context) or an exfiltration vector (a request whose URL is the payload), so the gate authorizes the *call* regardless of whether it mutates. *Validation must not act*: a "validator" that resolves a call by hitting an API, expanding a template that fires a webhook, or evaluating an argument that runs code has collapsed validation into invocation, and the effect has already happened *inside* $\gamma$ — so $\gamma$ itself must be **effect-free**, pure and total over the proposal and the current $s$, with no network and no execution; if deciding validity *requires* a side effect, that side effect is itself an action and must go through the gate, recursively. *The output is an action too*: the user-visible response and any logging are effects — for model-authored text, emitted either as an authorized action through $\gamma$ or only after an accepted halt (shell-templated status on any halt is the controller speaking, not the model) — streaming raw tokens to a sink before $\gamma$ has cleared them is the same bug from the other end.
|
||||
|
||||
So the property, tightest: $\gamma$ is a **pure, effect-free authorization that every model-proposed action — tool call, read, write, or final output — must pass before any effect occurs**, with "before" enforced structurally by the gate being the only route from model text to $Q_E$. The two failure modes to design against are a path from model output to a sink that bypasses the gate, and a $\gamma$ that is not effect-free, so that "validating" a call already rang the bell. And the boundary, so the property does not overpromise: $\gamma$ guarantees *no unauthorized effect* — pure code ordering, fully in your control — but not that an *authorized* effect is safe or correct; that is the plant's problem, and the reason $\rho$ and the reach-avoid certificate exist. Fail-closed is the floor — nothing executes that did not pass the gate — not the ceiling.
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
[](https://pypi.org/project/turnstone/)
|
||||
[](LICENSE)
|
||||
[](https://discord.gg/Nh3bWMacaq)
|
||||
[](https://github.com/sponsors/eous)
|
||||
|
||||
Self-hosted, local-first orchestration for tool-using AI agents. Give LLMs real tools — shell, files, search, web — and run them across your own cluster with direct HTTP routing and interactive interfaces. Your code, your models, your data stay on hardware you control: no telemetry, no phone-home.
|
||||
|
||||
@@ -171,6 +172,14 @@ UML diagrams in [`docs/diagrams/`](docs/diagrams/):
|
||||
- Optional: Discord / Slack channel integrations (`pip install turnstone[discord,slack]`)
|
||||
- [Git LFS](https://git-lfs.com/) for cloning (diagram PNGs)
|
||||
|
||||
## Support
|
||||
|
||||
Turnstone is free, Apache-2.0, and self-hosted — no paid tier, no telemetry, no upsell. If it saves you time or you'd like to help keep development moving, you can sponsor the project:
|
||||
|
||||
**[❤ Sponsor Turnstone →](https://github.com/sponsors/eous)** · one-off via **[PayPal](https://paypal.me/eousphoros)**
|
||||
|
||||
Sponsorship is entirely optional and funds maintenance, new features, and infrastructure. Prefer to contribute in other ways? Filing issues, improving docs, and [pull requests](CONTRIBUTING.md) help just as much.
|
||||
|
||||
## Community
|
||||
|
||||
Questions, ideas, or want to show what you're building? Join us on Discord:
|
||||
|
||||
+108
-9
@@ -634,8 +634,17 @@ function tool (the model always searches). Citations from `url_citation`
|
||||
annotations are formatted as footnotes. Extended prompt cache retention
|
||||
(`prompt_cache_retention: "24h"`) is enabled for GPT-5.x models at no
|
||||
additional cost. Cached token counts are extracted from
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models (local servers) get
|
||||
permissive defaults with `supports_vision=False` and use SearxNG for web search.
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models get permissive
|
||||
defaults with `supports_vision=False` and use SearxNG for web search. The
|
||||
`openai-compatible` lane never consults this table at all — on either API
|
||||
surface (the responses pin is served by a compat-mode
|
||||
`OpenAIResponsesProvider`, mirroring `AnthropicProvider(compat=True)`): a
|
||||
local server serves whatever the operator named it (vLLM
|
||||
`--served-model-name` is a free string), so a prefix collision with a cloud
|
||||
model id must not inherit that model's sampling/effort contract — every
|
||||
local model gets the plain defaults, and anything beyond them is declared on
|
||||
the model definition (capabilities JSON + `server_compat`), matching the
|
||||
`anthropic-compatible` lane.
|
||||
|
||||
**AnthropicProvider** (`_anthropic.py`): converts OpenAI-format messages to
|
||||
Anthropic content blocks, maps `system`/`developer` roles to the `system`
|
||||
@@ -798,15 +807,105 @@ model = "deepseek-ai/DeepSeek-V4-Flash"
|
||||
supports_vision = true # multimodal checkpoints only
|
||||
supports_mid_conversation_system = true # template-dependent
|
||||
context_window = 131072
|
||||
thinking_mode = "manual" # session effort knob drives the template toggle
|
||||
thinking_param = "enable_thinking" # Qwen/Gemma key; "thinking" for Granite/DeepSeek
|
||||
```
|
||||
|
||||
The reasoning toggle does NOT use Anthropic's `thinking` request param.
|
||||
Toggle it through the chat template instead: set `{"chat_template_kwargs":
|
||||
{"thinking": false}}` as extra body params in the admin Models
|
||||
server-compat section (for this provider the section shows only the
|
||||
extra-body field — server type, API surface, and thinking mode are
|
||||
openai-compatible-only knobs); the provider forwards it via the SDK's
|
||||
`extra_body`.
|
||||
Reasoning control does NOT use Anthropic's `thinking` request param —
|
||||
the levers live in the chat template, reached through
|
||||
`chat_template_kwargs` in the request body. Two channels, dynamic first:
|
||||
|
||||
* **Session effort knob (dynamic).** Set the model's thinking mode to
|
||||
"Effort-knob controlled" in the admin Models form (or
|
||||
`thinking_mode = "manual"` + `thinking_param` under
|
||||
`[models.*.capabilities]`) and the provider maps the session's
|
||||
reasoning-effort knob onto the template toggle per-request: effort
|
||||
`none` sends `{<thinking_param>: false}`, any other level sends
|
||||
`true` — the same contract as the real lane's manual mode. ("Always
|
||||
on" / `thinking_mode = "adaptive"` instead always sends `true`: the
|
||||
model self-regulates, so the knob never force-disables — mirroring
|
||||
the native adaptive branch.) The graded effort value always rides
|
||||
alongside the toggle: under `effort_param` when the operator names
|
||||
the template's key, else under the conventional fallback key
|
||||
(`reasoning_effort`) on the anthropic-compatible lane — the user's
|
||||
effort setting always reaches the wire, and a template that doesn't
|
||||
reference the kwarg ignores it. On the openai-compatible lane the
|
||||
undeclared-key case rides the flat top-level `reasoning_effort`
|
||||
param instead (the documented compat field), forwarded verbatim.
|
||||
Optional `reasoning_effort_values` / `default_reasoning_effort`
|
||||
validate the knob before it reaches the server; without declared
|
||||
values the knob is forwarded as-is. The knob is ordinal, and validation
|
||||
respects that: an off-list knob value rounds UP onto the declared
|
||||
list and a value above the ceiling rides the ceiling
|
||||
(`snap_reasoning_effort`) — asking for more effort than the model
|
||||
declares never falls back to a lower default tier. The knob's
|
||||
`none` position is forwarded verbatim when the model declares an
|
||||
explicit `none` level (gpt-5.1+, grok-4.3) — omitting it there would
|
||||
leave a reasoning-on server default (e.g. gpt-5.5's `medium`) in
|
||||
charge of a knob that promises off — and omitted otherwise; `none`
|
||||
is never a snap target for other positions.
|
||||
`default_reasoning_effort` only catches values the ordinal snap
|
||||
cannot rank (custom strings). Declare values that match the
|
||||
template's documented vocabulary: for DeepSeek-V4, which officially
|
||||
accepts `high`/`max` (Think High is the default thinking tier;
|
||||
`low`/`medium` alias to `high`, `xhigh` to `max`), a
|
||||
`("high", "max")` values list reproduces the official aliasing
|
||||
exactly — `low`/`medium` round up to `high`, `xhigh` to `max` —
|
||||
and freeform passthrough matches it too. To map an undocumented
|
||||
template, probe with per-request `chat_template_kwargs` and compare
|
||||
`input_tokens`. Setting `effort_param` also suppresses the
|
||||
flat top-level `reasoning_effort` request param on the
|
||||
openai-compatible lane — the template channel replaces it, never
|
||||
doubles it. With the default `thinking_mode = "none"` nothing is
|
||||
injected and the server's template default decides.
|
||||
|
||||
Upgrade note: before 1.7.0a7 the openai-compatible lane sent the
|
||||
toggle unconditionally `true` whenever thinking mode was enabled. A
|
||||
stored per-model `reasoning_effort = "none"` now disables thinking
|
||||
on such models — pick any real level (or clear the override) to keep
|
||||
it on. Also since 1.7.0a7 the effort level itself always reaches the
|
||||
wire on the local lanes (previously dropped unless
|
||||
`reasoning_effort_values` was declared): flat `reasoning_effort` on
|
||||
openai-compatible, the `effort_param`-or-fallback template key on
|
||||
anthropic-compatible when reasoning control is engaged.
|
||||
* **Operator pin (static).** Entries under `{"chat_template_kwargs":
|
||||
...}` in the admin Models extra-body field ride the SDK's
|
||||
`extra_body` unconditionally and win over the knob mapping on key
|
||||
collision — e.g. pin `{"enable_thinking": true}` to keep thinking on
|
||||
regardless of the session knob. (Server type and API surface remain
|
||||
openai-compatible-only knobs and stay hidden for this provider.)
|
||||
|
||||
The same knob mapping drives the `openai-compatible` lane's Chat
|
||||
Completions requests — `merge_reasoning_template_kwargs` is shared by
|
||||
both local-server lanes, so `thinking_mode`/`thinking_param`/
|
||||
`effort_param` mean the same thing whichever endpoint serves the model.
|
||||
Only the Responses API surface (native reasoning) ignores it.
|
||||
|
||||
The console surfaces this projection as an *effective effort ladder*:
|
||||
the admin model form's per-model effort select and the skill
|
||||
launch-config effort select annotate each position with what the
|
||||
request will carry, in plain words — a position whose delivered level
|
||||
matches its name stays plain ("Max"), a snapped position says so
|
||||
("Low — sends high"), the adaptive lanes' none position warns
|
||||
"thinking stays on", and budget detail lives in the tooltip. A
|
||||
position is never labeled after a sibling that shares its wire (that
|
||||
rendered "Max (= minimal)", implying a downgrade the wire doesn't
|
||||
contain). Computed server-side by `providers/effort_ladder.py` from
|
||||
the same mapping functions the providers use at request time and
|
||||
shipped on `/v1/api/models` rows (every row carries `effort_ladder`,
|
||||
empty when the capabilities column fails to parse) and
|
||||
`POST /v1/api/admin/models/effort-ladder`. The ladder describes what
|
||||
Turnstone sends — a server-side template may alias further (DeepSeek-V4
|
||||
folds `low`/`medium` into its default `high` tier).
|
||||
|
||||
The `anthropic-compatible` lane never sends Anthropic's native
|
||||
`thinking`/`output_config` params — they are not in vLLM's request
|
||||
schema. The real `anthropic` provider is unaffected: official Claude
|
||||
models keep native thinking, budget mapping, and `output_config`
|
||||
effort. A gateway fronting *real* Claude on a Messages-shaped URL
|
||||
(e.g. a LiteLLM `anthropic/` route to the Claude API) should use
|
||||
`provider = "anthropic"` with a custom `base_url`, which keeps the
|
||||
native thinking params.
|
||||
|
||||
Verified quirks of vLLM's Anthropic endpoint:
|
||||
|
||||
|
||||
@@ -41,6 +41,7 @@ are set.
|
||||
| `TURNSTONE_OIDC_PASSWORD_ENABLED` | No | `true` | Set to `false` to hide the password form and block all username/password logins (including admin). API tokens continue to work. |
|
||||
| `TURNSTONE_OIDC_REDIRECT_BASE` | Yes | — | Externally-reachable origin for the OIDC redirect URI (e.g. `https://app.example.com`). Without this, OIDC will refuse to start. The previous Host-header fallback was unsafe under permissive reverse proxies. |
|
||||
| `TURNSTONE_OIDC_TRUSTED_ENDPOINT_HOSTS` | No | — | Comma-separated list of additional hostnames whose endpoints the IdP discovery document is allowed to reference. See [Cross-host endpoints](#cross-host-endpoints). |
|
||||
| `TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK` | No | `false` | Allow the issuer (and its discovered endpoints) to resolve to private/internal addresses — needed for a self-hosted IdP on an internal network. See [Self-hosted and internal IdPs](#self-hosted-and-internal-idps). |
|
||||
|
||||
All four required fields — issuer, client ID, client secret, and
|
||||
`TURNSTONE_OIDC_REDIRECT_BASE` — must be set. If any are missing OIDC
|
||||
@@ -99,6 +100,40 @@ The same scheme / no-userinfo / SSRF rules apply to allow-listed hosts —
|
||||
this knob only relaxes the same-origin check, not the security gates.
|
||||
Each entry is a hostname (no scheme, no path).
|
||||
|
||||
### Self-hosted and internal IdPs
|
||||
|
||||
By default Turnstone refuses an issuer whose hostname resolves to a
|
||||
private or internal address:
|
||||
|
||||
```
|
||||
OIDCError: endpoint URL resolves to non-public address (10.0.0.5): https://auth.example.site
|
||||
```
|
||||
|
||||
This is SSRF hardening, not a licensing or product restriction: the OIDC
|
||||
flow makes server-side HTTP requests (discovery, JWKS, token exchange),
|
||||
and refusing non-public destinations keeps a mistyped or maliciously
|
||||
steered issuer from aiming those fetches at internal services. For a
|
||||
self-hosted IdP (Keycloak, Authentik, Dex, …) on a private network,
|
||||
opt in explicitly in `config.toml`:
|
||||
|
||||
```toml
|
||||
[oidc]
|
||||
allow_private_network = true
|
||||
```
|
||||
|
||||
or via `TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK=true` (the env var wins
|
||||
when both are set).
|
||||
|
||||
The opt-in admits private-range (RFC 1918), unique-local, CGNAT
|
||||
(100.64/10 — tailnets), and loopback addresses. Link-local, multicast,
|
||||
and reserved ranges stay refused even with the opt-in — cloud metadata
|
||||
services (169.254.169.254) live there, and no legitimate IdP does. The
|
||||
HTTPS requirement and the same-origin endpoint checks are unaffected.
|
||||
|
||||
This knob only affects the login-flow IdP configured here. OAuth
|
||||
endpoints advertised by remote MCP servers are untrusted input and are
|
||||
always held to the strict public-address rule.
|
||||
|
||||
### config.toml alternative
|
||||
|
||||
```toml
|
||||
@@ -111,6 +146,8 @@ provider_name = "Google"
|
||||
role_claim = "groups"
|
||||
password_enabled = true
|
||||
redirect_base = "https://app.example.com"
|
||||
# Self-hosted IdP on an internal network (see "Self-hosted and internal IdPs")
|
||||
allow_private_network = false
|
||||
|
||||
[oidc.role_map]
|
||||
admin = "builtin-admin"
|
||||
|
||||
+37
-4
@@ -118,9 +118,37 @@ seeded):
|
||||
`persona` argument, validated when the coordinator prepares the spawn
|
||||
and re-checked by the node that creates the child (children are always
|
||||
interactive-kind). Omitted means the interactive **default** — a child
|
||||
never inherits its parent coordinator's persona. Sub-agents spawned via
|
||||
`task_agent` have no persona parameter at all; they keep their own
|
||||
identity and envelope.
|
||||
never inherits its parent coordinator's persona.
|
||||
- **Sub-agents**: `task_agent` takes a `persona` argument setting the
|
||||
sub-agent's identity and capability envelope (resolved against
|
||||
interactive-kind personas, frozen into the task at prep). Omitted keeps
|
||||
the default autonomous task-agent identity — never the parent's persona.
|
||||
|
||||
## How agents discover personas
|
||||
|
||||
Agents are told, not expected to guess: the live persona list (enabled,
|
||||
interactive-kind — children and sub-agents are always interactive) is
|
||||
injected into the `persona` parameter description of `task_agent`,
|
||||
`spawn_workstream`, and `spawn_batch` whenever the session's tool surface
|
||||
is rendered — session start, MCP catalog change, model-registry reload.
|
||||
Each entry carries the name, the default marker, and the persona's
|
||||
one-line description so the model can pick by purpose (descriptions drop
|
||||
out past 25 personas; the name list always enumerates completely).
|
||||
|
||||
A persona created after that render is still reachable — pass its name.
|
||||
Every resolve failure enumerates the names currently valid for the kind,
|
||||
so a stale list (or a typo) self-corrects on the next attempt.
|
||||
|
||||
Resolution is forgiving on all surfaces (they share one rule):
|
||||
|
||||
- names match case-insensitively (`Writer` resolves `writer`);
|
||||
- an input that uniquely matches a persona's **display name**
|
||||
(case-insensitive, among the kind's enabled personas — display names are
|
||||
not unique, and a same-label persona of another kind neither blocks nor
|
||||
wins) resolves to that persona; an ambiguous match errors, listing the
|
||||
candidate slugs;
|
||||
- whatever variant matched, the stamped identity, approval chrome, and
|
||||
wire always carry the canonical `name` slug.
|
||||
|
||||
## Authoring (console)
|
||||
|
||||
@@ -128,7 +156,12 @@ Personas are managed in the console's **Manage → Governance → Personas**
|
||||
tab. The admin shelf exposes exactly the four levers plus the kind
|
||||
list, the default marker, and archive. Rules:
|
||||
|
||||
- `name` is an immutable lowercase slug; edit `display_name` instead.
|
||||
- `name` is an immutable lowercase slug — and the identifier agents and
|
||||
the CLI launch the persona by (`persona=` on the spawn tools,
|
||||
`--persona` on the CLI); the create shelf says so under **Name**.
|
||||
`display_name` is a list label, editable any time, and deliberately
|
||||
not an identifier (a unique display name happens to resolve, as a
|
||||
forgiveness fallback — don't design workflows around it).
|
||||
- Exactly one default per kind, storage-enforced: flipping the flag on a
|
||||
successor demotes the incumbent atomically, defaults are single-kind,
|
||||
and a default cannot be archived.
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
{
|
||||
"defaults": {
|
||||
"n_runs": 3
|
||||
},
|
||||
"cases": [
|
||||
{
|
||||
"id": "search-first",
|
||||
"skill": {
|
||||
"name": "search-first",
|
||||
"content": "# Search First\n\nBefore answering ANY question about where something lives in the codebase, you MUST call the `search` tool first. Never answer from memory."
|
||||
},
|
||||
"user_prompt": "Where is JWT token validation implemented in this project?",
|
||||
"expected_actions": [{ "tool": "search" }],
|
||||
"match_mode": "ordered_subset",
|
||||
"max_turns": 4
|
||||
},
|
||||
{
|
||||
"id": "test-after-edit",
|
||||
"skill": {
|
||||
"name": "test-after-edit",
|
||||
"content": "# Test After Edit\n\nAfter editing or writing ANY file, you MUST run the test suite with `python -m pytest` via bash before you finish. Do not report done until tests have run."
|
||||
},
|
||||
"user_prompt": "Add a function `clamp(x, lo, hi)` that clamps x to [lo, hi] in utils.py.",
|
||||
"setup": {
|
||||
"files": {
|
||||
"utils.py": ""
|
||||
}
|
||||
},
|
||||
"expected_actions": [
|
||||
{ "tool": "write_file" },
|
||||
{ "tool": "bash", "args_pattern": { "command": "pytest" } }
|
||||
],
|
||||
"match_mode": "ordered_subset",
|
||||
"max_turns": 8
|
||||
},
|
||||
{
|
||||
"id": "changelog-update",
|
||||
"skill": {
|
||||
"name": "changelog-update",
|
||||
"content": "# Changelog Discipline\n\nWhenever you modify a file, you MUST also append a one-line entry to CHANGELOG.md describing the change in the same task."
|
||||
},
|
||||
"user_prompt": "Fix the off-by-one so pager.py shows the last page. Edit pager.py.",
|
||||
"setup": {
|
||||
"files": {
|
||||
"pager.py": "def last_page(total_items, per_page):\n # off-by-one: drops the final partial page\n return total_items // per_page\n",
|
||||
"CHANGELOG.md": "# Changelog\n"
|
||||
}
|
||||
},
|
||||
"expected_actions": [
|
||||
{ "tool": "edit_file", "args_pattern": { "path": "CHANGELOG.md" } }
|
||||
],
|
||||
"match_mode": "subset",
|
||||
"max_turns": 8
|
||||
}
|
||||
]
|
||||
}
|
||||
+1
-1
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "turnstone"
|
||||
version = "1.7.0a6"
|
||||
version = "1.7.1"
|
||||
description = "Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
|
||||
@@ -11201,6 +11201,7 @@
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline BASE override \u2014 required. Every persona must name a prompt source; built-in file-backed personas are seeded by migration, not created here, so an operator-created persona must supply base_prompt.",
|
||||
"title": "Base Prompt"
|
||||
},
|
||||
"tool_allowlist": {
|
||||
@@ -11259,7 +11260,7 @@
|
||||
"type": "object"
|
||||
},
|
||||
"UpdatePersonaRequest": {
|
||||
"description": "PATCH body \u2014 absent fields are left unchanged.\n\nExplicit ``null`` is meaningful only on the two resettable fields:\n``base_prompt: null`` clears the override back to the kind's stock\nBASE, and ``tool_allowlist: null`` resets to unrestricted. ``null``\non the boolean flags or ``applies_to_kinds`` is ignored (treated as\nabsent), so a client serializing unset optionals as null cannot\narchive a persona or flip levers by accident.\n\nArchive = ``{\"enabled\": false}``; default flip = ``{\"is_default\": true}``\non the successor (storage demotes the incumbent atomically). ``name``\nis immutable; existing workstreams are never affected by edits.",
|
||||
"description": "PATCH body \u2014 absent fields are left unchanged.\n\nExplicit ``null`` resets ``tool_allowlist`` to unrestricted, and \u2014 on a\nBUILT-IN persona only \u2014 clears ``base_prompt`` (the operator override),\nreverting to that persona's file-backed prompt. An OPERATOR persona has no\nfallback source, so ``base_prompt: null`` on one is rejected: every persona\nmust name a prompt source. ``null`` on the boolean flags or\n``applies_to_kinds`` is ignored (treated as absent), so a client serializing\nunset optionals as null cannot archive a persona or flip levers by accident.\n\nArchive = ``{\"enabled\": false}``; default flip = ``{\"is_default\": true}``\non the successor (storage demotes the incumbent atomically). ``name``\nis immutable; existing workstreams are never affected by edits.",
|
||||
"properties": {
|
||||
"display_name": {
|
||||
"anyOf": [
|
||||
@@ -13305,21 +13306,17 @@
|
||||
},
|
||||
"pending_approval": {
|
||||
"default": false,
|
||||
"description": "True when the workstream is parked on ``_approval_event`` awaiting an operator approve/deny. Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"description": "True when at least one approval cycle is live (a gate thread parked awaiting an operator approve/deny). Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"title": "Pending Approval",
|
||||
"type": "boolean"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload \u2014 same shape as ``DashboardWorkstream.pending_approval_detail``. ``None`` when no approval is pending. Lets a reload paint the action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payloads, one per live cycle, oldest first \u2014 same shape as ``DashboardWorkstream.pending_approval_details``. Empty when no approval is pending. Lets a reload paint every action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -13332,8 +13329,14 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalDetail": {
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nSet when a workstream's ``approve_tools`` is parked on\n``_approval_event``; ``None`` (omitted) otherwise. Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nOne entry per live approval CYCLE \u2014 a gate thread parked in\n``approve_tools`` awaiting the operator. Parallel task agents run\nconcurrent gates, so a workstream can have several of these at\nonce (``pending_approval_details``, oldest first). Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"properties": {
|
||||
"cycle_id": {
|
||||
"default": "",
|
||||
"description": "Identity of this approval cycle. Echo it back on ``POST /v1/api/workstreams/{ws_id}/approve`` to resolve exactly this round \u2014 required for correctness when several cycles are live (parallel task agents).",
|
||||
"title": "Cycle Id",
|
||||
"type": "string"
|
||||
},
|
||||
"call_id": {
|
||||
"default": "",
|
||||
"description": "Primary call_id \u2014 first non-empty call_id in items list order. Matches the 409 ``current_call_id`` response from ``POST /v1/api/workstreams/{ws_id}/approve`` so the UI can render the same identifier the server reports as current.",
|
||||
|
||||
@@ -2692,21 +2692,17 @@
|
||||
},
|
||||
"pending_approval": {
|
||||
"default": false,
|
||||
"description": "True when the workstream is parked on ``_approval_event`` awaiting an operator approve/deny. Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"description": "True when at least one approval cycle is live (a gate thread parked awaiting an operator approve/deny). Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"title": "Pending Approval",
|
||||
"type": "boolean"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload \u2014 same shape as ``DashboardWorkstream.pending_approval_detail``. ``None`` when no approval is pending. Lets a reload paint the action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payloads, one per live cycle, oldest first \u2014 same shape as ``DashboardWorkstream.pending_approval_details``. Empty when no approval is pending. Lets a reload paint every action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -2719,8 +2715,14 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalDetail": {
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nSet when a workstream's ``approve_tools`` is parked on\n``_approval_event``; ``None`` (omitted) otherwise. Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nOne entry per live approval CYCLE \u2014 a gate thread parked in\n``approve_tools`` awaiting the operator. Parallel task agents run\nconcurrent gates, so a workstream can have several of these at\nonce (``pending_approval_details``, oldest first). Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"properties": {
|
||||
"cycle_id": {
|
||||
"default": "",
|
||||
"description": "Identity of this approval cycle. Echo it back on ``POST /v1/api/workstreams/{ws_id}/approve`` to resolve exactly this round \u2014 required for correctness when several cycles are live (parallel task agents).",
|
||||
"title": "Cycle Id",
|
||||
"type": "string"
|
||||
},
|
||||
"call_id": {
|
||||
"default": "",
|
||||
"description": "Primary call_id \u2014 first non-empty call_id in items list order. Matches the 409 ``current_call_id`` response from ``POST /v1/api/workstreams/{ws_id}/approve`` so the UI can render the same identifier the server reports as current.",
|
||||
@@ -3004,17 +3006,13 @@
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload for the coordinator children-tree UI. Carries the merged ``_pending_approval`` items list + per-call_id LLM verdict cache so a coord can render approve/deny buttons + judge pill without a separate per-child round-trip. ``None`` when no approval is pending. Also surfaced (verbatim) on ``GET /v1/api/cluster/ws/live`` via the ``_CLUSTER_WS_LIVE_KEYS`` projection."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payload for the coordinator children-tree UI: EVERY live approval cycle, oldest first \u2014 parallel task agents gate concurrently, so a workstream can hold several prompts at once. Each entry carries the cycle's items + per-call_id LLM verdict cache so a coord can render approve/deny buttons + judge pill without a separate per-child round-trip; resolve each with its ``cycle_id``. Empty when no approval is pending. Also surfaced (verbatim) on ``GET /v1/api/cluster/ws/live`` via the ``_CLUSTER_WS_LIVE_KEYS`` projection. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
},
|
||||
"recent_auto_approvals": {
|
||||
"description": "Per-ws ring buffer (cap 10) of recent tool calls that bypassed the operator approval gate. Surfaces ``WebUI._recent_auto_approvals`` so the coord-tree row can render an 'auto-approved by ...' pill when the child's skill / blanket / admin-policy rules silently let a tool through. Also projected onto ``GET /v1/api/cluster/ws/live`` via ``_CLUSTER_WS_LIVE_KEYS``.",
|
||||
|
||||
Generated
+50
-50
@@ -409,16 +409,16 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@vitest/expect": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.9.tgz",
|
||||
"integrity": "sha512-vl/rYsUKcBr3SnQn166+XR5ZQcgMx3DQhFWdfli/cWpLnLUmbxZvyrJZotLFUryib+LtArYMSTJ5RbQ57ZqrlA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.10.tgz",
|
||||
"integrity": "sha512-YsCn+qAk1GWjQOWFEsEcL2gNQ0zmVmQu3T03qP6UyjhtmdtwtbuI+DASn/7iQB3HGTXkdBwGddzxPlmiql5vlA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@standard-schema/spec": "^1.1.0",
|
||||
"@types/chai": "^5.2.2",
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"chai": "^6.2.2",
|
||||
"tinyrainbow": "^3.1.0"
|
||||
},
|
||||
@@ -427,13 +427,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/mocker": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.9.tgz",
|
||||
"integrity": "sha512-EVkXzBjrPGM+cK8/ANWgBrkUCfJfb38/EfTSO8h7pWvKkyPkpWxvR7BkD2MyItMF62C97zAEoqdpUixwR/e+Rw==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.10.tgz",
|
||||
"integrity": "sha512-v0xaezt+DKEmKfaxg133ldzADrwLGd7Ze1MfQQTYfvs8OqZIwbxyxaYURivwV7sWy5fqn3rH5uOrSp07bp44Ow==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"estree-walker": "^3.0.3",
|
||||
"magic-string": "^0.30.21"
|
||||
},
|
||||
@@ -454,9 +454,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/pretty-format": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.9.tgz",
|
||||
"integrity": "sha512-s0iufns3iIFitdgm+YR7g1whCAaGtXz459VS9/PqyKDEEFgYIhsHOQmXgIgDuYCt7DeQmiZT0Qe2OA2p4ZPu5A==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.10.tgz",
|
||||
"integrity": "sha512-W1HsjSH4MXQ9YfmmhLAoIYf1HRfekQCGngeIgcei6MP5QQGWUe0gkopdZQaVCFO+JDJMrAJGwa5pRpNpvy4P8Q==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -467,13 +467,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/runner": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.9.tgz",
|
||||
"integrity": "sha512-KXLMDtc7oe70+3mJfGrPUWPesswH+3sTxAMAMl8DG7I8IUQT4XW718dY5ID3vPUcmlu27CcKfY4P3h3I29SLJg==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.10.tgz",
|
||||
"integrity": "sha512-IKI6kpIH+LmpROplyLwBBaCfMgOZOMsygVa6BARD6ahA04VRuJSa6OaVG7kRvSEMD870Vd91rSSw0eegtWyLGg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"pathe": "^2.0.3"
|
||||
},
|
||||
"funding": {
|
||||
@@ -481,14 +481,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/snapshot": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.9.tgz",
|
||||
"integrity": "sha512-Jc7RKGNBo8Z28WYIm0Niej4xdSPByRf6mU58VpHQkd6Zh05rlnA+twjbK5HyeIGHxrzsc3mJgS43uM0CZKzaIA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.10.tgz",
|
||||
"integrity": "sha512-xRkfOT1qpTAi/Ti4Y1LtfRc3kEuqxGw59eN2jN9pRWMtS/XDevekhcFSqvQqjUNGksfjMJu3Y+oJ+4Ypn2OaJw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"magic-string": "^0.30.21",
|
||||
"pathe": "^2.0.3"
|
||||
},
|
||||
@@ -497,9 +497,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/spy": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.9.tgz",
|
||||
"integrity": "sha512-fHpsS6mIi+PiEW+vcRVOMkX1oSaPKne3VOclSFICPcGOmfKgXPU5iAah+wcNcj2xPrCCmfq99IDGf+EojhhvhA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.10.tgz",
|
||||
"integrity": "sha512-PLf/Ugvoq5wO/b4rwYCR1h2PSIdXz7wnkQFMiUpLdtM7l6pqVFcQIBEHyT1+l+cj7mNwAfZHzqXqDyjvOuwbDw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
@@ -507,13 +507,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/utils": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.9.tgz",
|
||||
"integrity": "sha512-A51o8ymO5PpqlWNnBP9ZHPXDIpuMtTLlGSjN7la4US+LJzoUMyhwjA5QXlm39JexgwHKW4Xjs8Z2d3dLCXOeuA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.10.tgz",
|
||||
"integrity": "sha512-fy9am/HWxbaGt/Sawrp90vt6Y6jQwf1RX77cz3uwoJwJVMli/e1IEwRPnMNJ7vKfPTwo0diXifkpPvwH9v7nGA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"convert-source-map": "^2.0.0",
|
||||
"tinyrainbow": "^3.1.0"
|
||||
},
|
||||
@@ -949,9 +949,9 @@
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/picomatch": {
|
||||
"version": "4.0.4",
|
||||
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz",
|
||||
"integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==",
|
||||
"version": "4.0.5",
|
||||
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.5.tgz",
|
||||
"integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -1122,9 +1122,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vite": {
|
||||
"version": "8.1.2",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.2.tgz",
|
||||
"integrity": "sha512-6YYPbRXTxx6bRXmOn7XdnQAy5DQNHhDgtjhDHI13oe4pY93kkcdGJWxpGwOm++/Wh0QpQhDrpIoVMrmrsI5AGQ==",
|
||||
"version": "8.1.3",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.3.tgz",
|
||||
"integrity": "sha512-Ds+gBRbj0lwRO2Y5hwnUBdxSwlAve9LeRyU4sNnAr0ewW0gWF0n5bgXgUzbgZ49MV9BVUAQUFYVcDUcilUExMA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -1200,19 +1200,19 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vitest": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.9.tgz",
|
||||
"integrity": "sha512-nE3/LEyc0z87uHYLZebqCUOaJr2hdtuPp7BQ4BosVFnfltxgAvMG08NyrSGlPpOUWvR27c5flSmYFTNr78L9GQ==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.10.tgz",
|
||||
"integrity": "sha512-R9jUTe5S4Qb0HCd4TNqpC7oGcrMssMRGXLW80ubjWsW9VH5GF8y1Y0SFLY9AbqSk6nt0PnOx4H4WNJYZ13GUPw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/expect": "4.1.9",
|
||||
"@vitest/mocker": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/runner": "4.1.9",
|
||||
"@vitest/snapshot": "4.1.9",
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/expect": "4.1.10",
|
||||
"@vitest/mocker": "4.1.10",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"@vitest/runner": "4.1.10",
|
||||
"@vitest/snapshot": "4.1.10",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"es-module-lexer": "^2.0.0",
|
||||
"expect-type": "^1.3.0",
|
||||
"magic-string": "^0.30.21",
|
||||
@@ -1240,12 +1240,12 @@
|
||||
"@edge-runtime/vm": "*",
|
||||
"@opentelemetry/api": "^1.9.0",
|
||||
"@types/node": "^20.0.0 || ^22.0.0 || >=24.0.0",
|
||||
"@vitest/browser-playwright": "4.1.9",
|
||||
"@vitest/browser-preview": "4.1.9",
|
||||
"@vitest/browser-webdriverio": "4.1.9",
|
||||
"@vitest/coverage-istanbul": "4.1.9",
|
||||
"@vitest/coverage-v8": "4.1.9",
|
||||
"@vitest/ui": "4.1.9",
|
||||
"@vitest/browser-playwright": "4.1.10",
|
||||
"@vitest/browser-preview": "4.1.10",
|
||||
"@vitest/browser-webdriverio": "4.1.10",
|
||||
"@vitest/coverage-istanbul": "4.1.10",
|
||||
"@vitest/coverage-v8": "4.1.10",
|
||||
"@vitest/ui": "4.1.10",
|
||||
"happy-dom": "*",
|
||||
"jsdom": "*",
|
||||
"vite": "^6.0.0 || ^7.0.0 || ^8.0.0"
|
||||
|
||||
@@ -75,15 +75,35 @@ export interface ToolInfoEvent {
|
||||
items: Array<Record<string, unknown>>;
|
||||
}
|
||||
|
||||
/** One approval CYCLE awaiting the operator. Several can be outstanding
|
||||
* at once (parallel task agents each gate their own tool calls) — key
|
||||
* prompt UI by `cycle_id` and echo it back on the approve POST.
|
||||
*
|
||||
* `cycle_id` is optional because it was added in 1.7: a pre-1.7 server
|
||||
* omits it on the wire, so a current SDK talking to an older node sees
|
||||
* `undefined`. Resolve those the legacy way (no selector → oldest
|
||||
* cycle). A current server always sends it. */
|
||||
export interface ApproveRequestEvent {
|
||||
type: "approve_request";
|
||||
cycle_id?: string;
|
||||
items: Array<Record<string, unknown>>;
|
||||
judge_pending?: boolean;
|
||||
}
|
||||
|
||||
/** A specific approval cycle resolved; `cycle_id`/`call_ids` identify
|
||||
* which prompt to dismiss.
|
||||
*
|
||||
* Both are optional for the same reason as `ApproveRequestEvent.cycle_id`
|
||||
* — a pre-1.7 server emits neither, so a bare "something resolved"
|
||||
* dismisses the sole tracked prompt (the legacy fallback the UI and
|
||||
* channel adapters keep). A current server always sends both. */
|
||||
export interface ApprovalResolvedEvent {
|
||||
type: "approval_resolved";
|
||||
approved: boolean;
|
||||
feedback: string;
|
||||
always?: boolean;
|
||||
cycle_id?: string;
|
||||
call_ids?: string[];
|
||||
}
|
||||
|
||||
export interface ToolResultEvent {
|
||||
|
||||
@@ -166,6 +166,13 @@ export class TurnstoneServer extends BaseClient {
|
||||
approved?: boolean;
|
||||
feedback?: string | null;
|
||||
always?: boolean;
|
||||
/** Resolve exactly this approval cycle (from ApproveRequestEvent.cycle_id).
|
||||
* Omitting it resolves the OLDEST live cycle — ambiguous when parallel
|
||||
* task agents have several prompts outstanding, so pass it whenever the
|
||||
* triggering event is known. */
|
||||
cycleId?: string;
|
||||
/** Alternative selector: any call_id inside the target cycle. */
|
||||
callId?: string;
|
||||
}): Promise<StatusResponse> {
|
||||
return this.request(
|
||||
"POST",
|
||||
@@ -175,6 +182,8 @@ export class TurnstoneServer extends BaseClient {
|
||||
approved: opts.approved ?? true,
|
||||
feedback: opts.feedback,
|
||||
always: opts.always,
|
||||
cycle_id: opts.cycleId,
|
||||
call_id: opts.callId,
|
||||
},
|
||||
},
|
||||
);
|
||||
|
||||
@@ -104,7 +104,6 @@ describe("TurnstoneServer attachments", () => {
|
||||
const [, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(JSON.parse(init.body)).toEqual({
|
||||
message: "hi",
|
||||
ws_id: "ws-X",
|
||||
attachment_ids: ["a1", "a2"],
|
||||
});
|
||||
});
|
||||
@@ -117,7 +116,7 @@ describe("TurnstoneServer attachments", () => {
|
||||
});
|
||||
await client.send("hi", "ws-X");
|
||||
const [, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "hi", ws_id: "ws-X" });
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "hi" });
|
||||
});
|
||||
|
||||
it("createWorkstream with attachments sends multipart and auto-generates ws_id", async () => {
|
||||
|
||||
@@ -74,8 +74,8 @@ describe("TurnstoneServer", () => {
|
||||
await client.send("Hello", "ws1");
|
||||
|
||||
const [url, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(url).toBe("http://test/v1/api/send");
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "Hello", ws_id: "ws1" });
|
||||
expect(url).toBe("http://test/v1/api/workstreams/ws1/send");
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "Hello" });
|
||||
});
|
||||
|
||||
it("injects auth header when token provided", async () => {
|
||||
|
||||
@@ -51,6 +51,12 @@ def make_replay_mocks(
|
||||
ui._ws_messages = 0
|
||||
for key, value in ui_overrides.items():
|
||||
setattr(ui, key, value)
|
||||
# Both replay paths read cycle cards via ``pending_approval_cards()``
|
||||
# (one card per concurrent approval cycle). Model it from the
|
||||
# single-slot ``_pending_approval`` override so tests keep seeding
|
||||
# the one field; a bare MagicMock here would iterate empty and
|
||||
# silently drop the approve_request from the replay.
|
||||
ui.pending_approval_cards = lambda: [ui._pending_approval] if ui._pending_approval else []
|
||||
ws = MagicMock()
|
||||
ws.session = session
|
||||
request = MagicMock()
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Recording fake SDK client — captures the kwargs at each provider's seam.
|
||||
|
||||
Every provider's ``create_streaming`` assembles its kwargs and calls the
|
||||
SDK *eagerly* before returning the stream iterator (Anthropic
|
||||
``client.messages.stream``, OpenAI ``client.chat.completions.create``,
|
||||
Responses ``client.responses.create/stream``), so driving a provider
|
||||
against a :class:`RecordingClient` captures the full composed request
|
||||
payload without a network round-trip.
|
||||
|
||||
Shared by the wire-payload golden harness (``test_wire_payload_golden``)
|
||||
and the effort-ladder parity harness (``test_effort_ladder_wire_parity``)
|
||||
so both assert against the same capture seam.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
|
||||
class _EmptyStream:
|
||||
"""Stand-in for an SDK stream / stream-manager: empty iterable AND no-op CM."""
|
||||
|
||||
def __iter__(self) -> Iterator[Any]:
|
||||
return iter(())
|
||||
|
||||
def __enter__(self) -> _EmptyStream:
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc: object) -> None:
|
||||
return None
|
||||
|
||||
|
||||
class _Seam:
|
||||
"""Records the kwargs of a single SDK call, returns an empty stream stub."""
|
||||
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self._sink = sink
|
||||
|
||||
def __call__(self, **kwargs: Any) -> _EmptyStream:
|
||||
# Last write wins; only one seam is exercised per provider call.
|
||||
self._sink["payload"] = kwargs
|
||||
return _EmptyStream()
|
||||
|
||||
|
||||
class _Completions:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
|
||||
|
||||
class _Chat:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.completions = _Completions(sink)
|
||||
|
||||
|
||||
class _Messages:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class _Responses:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class RecordingClient:
|
||||
"""Fake SDK client exposing every provider's call seam, recording kwargs."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.captured: dict[str, Any] = {}
|
||||
self.messages = _Messages(self.captured)
|
||||
self.chat = _Chat(self.captured)
|
||||
self.responses = _Responses(self.captured)
|
||||
+69
-1
@@ -52,8 +52,76 @@ def serve_until_exit(server: Any) -> None:
|
||||
loop.close()
|
||||
|
||||
|
||||
class _PendingResolver:
|
||||
"""Race-free drop-in for ``threading.Timer(delay, ui.resolve_approval)``.
|
||||
|
||||
``approve_tools`` runs ``_approval_event.clear()`` -> register
|
||||
``_pending_approval`` -> ``_approval_event.wait(_APPROVAL_WAIT_TIMEOUT)``
|
||||
(3600s). A *fixed-delay* timer can fire ``resolve_approval``
|
||||
(``_approval_event.set()``) BEFORE that ``.clear()`` on a slow/loaded
|
||||
runner, so the set is wiped by the clear and ``approve_tools`` blocks the
|
||||
full hour -- surfacing as a CI hang. This instead waits until the approval
|
||||
is actually registered (which happens *after* the clear), then resolves, so
|
||||
the wakeup can never be lost. ``start()`` / ``cancel()`` mirror
|
||||
``threading.Timer`` so it drops into existing scaffolding. ``cancel()``
|
||||
signals the worker to stop and joins it, so a test that errors *before* the
|
||||
approval registers can't leak the thread or resolve late into a finished
|
||||
test. ``before`` runs just before resolving -- e.g. to snapshot
|
||||
pending-state fields the test asserts on.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
ui: Any,
|
||||
*args: Any,
|
||||
before: Callable[[], None] | None = None,
|
||||
deadline: float = 10.0,
|
||||
**kwargs: Any,
|
||||
) -> None:
|
||||
self._ui = ui
|
||||
self._args = args
|
||||
self._kwargs = kwargs
|
||||
self._before = before
|
||||
self._deadline = deadline
|
||||
self._cancelled = threading.Event()
|
||||
self._started = False
|
||||
self._thread = threading.Thread(target=self._run, name="resolve-when-pending", daemon=True)
|
||||
|
||||
def _run(self) -> None:
|
||||
end = time.monotonic() + self._deadline
|
||||
while time.monotonic() < end:
|
||||
if self._cancelled.is_set():
|
||||
return
|
||||
# getattr (not a bare read) so a UI without _pending_approval can't
|
||||
# crash the worker into a silent death that leaves approve_tools
|
||||
# blocked for the full _APPROVAL_WAIT_TIMEOUT.
|
||||
if getattr(self._ui, "_pending_approval", None) is not None:
|
||||
if self._before is not None:
|
||||
self._before()
|
||||
self._ui.resolve_approval(*self._args, **self._kwargs)
|
||||
return
|
||||
time.sleep(0.001)
|
||||
# Deadline without registration: approve_tools isn't parked on the
|
||||
# approval event (returned early, or never reached it) -- don't resolve
|
||||
# into an unknown state; let the test's own assertions speak.
|
||||
|
||||
def start(self) -> None:
|
||||
self._started = True
|
||||
self._thread.start()
|
||||
|
||||
def cancel(self) -> None:
|
||||
self._cancelled.set()
|
||||
if self._started:
|
||||
self._thread.join(timeout=5)
|
||||
|
||||
|
||||
def resolve_when_pending(ui: Any, *args: Any, **kwargs: Any) -> _PendingResolver:
|
||||
"""Build a race-free approval resolver (see :class:`_PendingResolver`)."""
|
||||
return _PendingResolver(ui, *args, **kwargs)
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
from collections.abc import Callable, Iterator
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager, StaticServerState
|
||||
from turnstone.core.mcp_crypto import MCPTokenCipher
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris and London?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
},
|
||||
{
|
||||
"id": "call_2",
|
||||
"input": {
|
||||
"city": "London"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_2",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"text": "Actually, never mind London.",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "What's in this image?",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"source": {
|
||||
"data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==",
|
||||
"media_type": "image/png",
|
||||
"type": "base64"
|
||||
},
|
||||
"type": "image"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Think about the weather.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"signature": "sig-abc",
|
||||
"thinking": "The user wants weather.",
|
||||
"type": "thinking"
|
||||
},
|
||||
{
|
||||
"text": "Let me check.",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Think about the weather.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"signature": "sig-abc",
|
||||
"thinking": "The user wants weather.",
|
||||
"type": "thinking"
|
||||
},
|
||||
{
|
||||
"text": "Let me check.",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Run the deploy.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {},
|
||||
"name": "deploy",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "deployed",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"text": "Great, what's next?",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Hi there.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "Hello! How can I help?",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": "What's the weather in Paris?",
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "It's 18C and clear in Paris.",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -43,6 +43,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -43,6 +43,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
+196
-23
@@ -9,6 +9,8 @@ manual testing.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
@@ -17,6 +19,12 @@ import pytest
|
||||
|
||||
_APP_JS = Path(__file__).resolve().parent.parent / "turnstone/ui/static/app.js"
|
||||
_INTERACTIVE_JS = Path(__file__).resolve().parent.parent / "turnstone/shared_static/interactive.js"
|
||||
_SHELL_JS = Path(__file__).resolve().parent.parent / "turnstone/shared_static/shell.js"
|
||||
_REDACT_CREDENTIALS_JS = (
|
||||
Path(__file__).resolve().parent.parent / "turnstone/shared_static/redact_credentials.js"
|
||||
)
|
||||
_CONSOLE_APP_JS = Path(__file__).resolve().parent.parent / "turnstone/console/static/app.js"
|
||||
_CONSOLE_INDEX = Path(__file__).resolve().parent.parent / "turnstone/console/static/index.html"
|
||||
|
||||
|
||||
def _pane_method_offset(body: str, name: str) -> int:
|
||||
@@ -555,7 +563,6 @@ _CONSOLE_ADMIN_JS = Path(__file__).resolve().parent.parent / "turnstone/console/
|
||||
_CONSOLE_GOVERNANCE_JS = (
|
||||
Path(__file__).resolve().parent.parent / "turnstone/console/static/governance.js"
|
||||
)
|
||||
_CONSOLE_INTERACTIVE_JS = Path(__file__).resolve().parent.parent / "turnstone/console/static/app.js"
|
||||
|
||||
|
||||
_UNSAFE_CODE_SINK_LINT_TARGETS = [
|
||||
@@ -566,7 +573,7 @@ _UNSAFE_CODE_SINK_LINT_TARGETS = [
|
||||
("turnstone/console/static/coordinator/coordinator.js", _COORD_JS),
|
||||
("turnstone/console/static/admin.js", _CONSOLE_ADMIN_JS),
|
||||
("turnstone/console/static/governance.js", _CONSOLE_GOVERNANCE_JS),
|
||||
("turnstone/console/static/app.js", _CONSOLE_INTERACTIVE_JS),
|
||||
("turnstone/console/static/app.js", _CONSOLE_APP_JS),
|
||||
]
|
||||
|
||||
|
||||
@@ -955,6 +962,7 @@ _CONST_GUARD_BUNDLES = _SWEPT_BUNDLES + [
|
||||
_REPO_ROOT / "turnstone/shared_static/rail.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/interactive.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/conversation.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/redact_credentials.js",
|
||||
]
|
||||
|
||||
|
||||
@@ -1267,41 +1275,102 @@ def test_swept_bundle_has_no_const_reassign(bundle: Path) -> None:
|
||||
)
|
||||
|
||||
|
||||
def test_redact_api_keys_runtime_smoke() -> None:
|
||||
"""Runtime smoke for ``_redactApiKeys``. The function is pure — no
|
||||
DOM dependency — so it transplants cleanly into a standalone
|
||||
``node -e`` invocation. This is the bit that would have caught
|
||||
the original ``const redacted`` bug (which ``node --check`` and a
|
||||
pure-static keyword scan both miss; the ``TypeError`` only fires
|
||||
at call-time)."""
|
||||
body = _INTERACTIVE_JS.read_text(encoding="utf-8")
|
||||
m = re.search(
|
||||
r"function _redactApiKeys\(text\) \{.*?\n\}\n",
|
||||
body,
|
||||
re.DOTALL,
|
||||
)
|
||||
assert m is not None, "_redactApiKeys not found in app.js"
|
||||
fn = m.group(0)
|
||||
script = (
|
||||
fn
|
||||
+ "\nconst q = _redactApiKeys('https://x?api_key=abc&u=foo');\n"
|
||||
def test_redact_credentials_runtime_smoke() -> None:
|
||||
"""Runtime smoke for ``redactCredentials`` via a temp harness file.
|
||||
The function is pure (no DOM dependency). Tests the shared module
|
||||
directly via ESM import (replaces the legacy ``_redactApiKeys`` test
|
||||
which now delegates to this).
|
||||
|
||||
The tempfile is written with a ``.mjs`` extension so Node forces ESM
|
||||
parsing regardless of any ``package.json`` ``type`` field in parent
|
||||
directories. The ``redact_credentials.js`` source file is imported
|
||||
by absolute path so resolution is unambiguous.
|
||||
"""
|
||||
import tempfile
|
||||
|
||||
mod_path = _REDACT_CREDENTIALS_JS.resolve()
|
||||
harness = (
|
||||
"import { redactCredentials } from "
|
||||
+ json.dumps(str(mod_path))
|
||||
+ ";\n"
|
||||
+ "const q = redactCredentials('https://x?api_key=abc&u=foo');\n"
|
||||
+ 'if (q !== "https://x?api_key=***&u=foo") '
|
||||
+ "throw new Error('query-string redact failed: ' + q);\n"
|
||||
+ 'const j = _redactApiKeys(\'{"api_key":"abc"}\');\n'
|
||||
+ 'const j = redactCredentials(\'{"api_key":"abc"}\');\n'
|
||||
+ 'if (j !== \'{"api_key":"***"}\') '
|
||||
+ "throw new Error('json-style redact failed: ' + j);\n"
|
||||
+ "// Bearer token redaction (raw input)\n"
|
||||
+ "const b = redactCredentials('Authorization: Bearer eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjMifQ.test-token_here');\n"
|
||||
+ "if (!b.includes('[REDACTED:api_key]')) "
|
||||
+ "throw new Error('bearer redact failed: ' + b);\n"
|
||||
+ "// Connection string redaction (raw input)\n"
|
||||
+ "const c = redactCredentials('postgresql://user:supersecret@localhost/db');\n"
|
||||
+ "if (!c.includes('[REDACTED:password]')) "
|
||||
+ "throw new Error('conn-string redact failed: ' + c);\n"
|
||||
+ "// Authorization JSON key redaction (step 6 comprehensive)\n"
|
||||
+ 'const a = redactCredentials(\'{"Authorization": "Bearer canstillseethis"}\');\n'
|
||||
+ "if (!a.includes('[REDACTED:secret]')) "
|
||||
+ "throw new Error('authorization JSON redact failed: ' + a);\n"
|
||||
+ "// Single-quote JSON (Python dict repr / JS object literal)\n"
|
||||
+ "const sq = redactCredentials(\"{'Authorization': 'Bearer canstillseethis'}\");\n"
|
||||
+ "if (!sq.includes('[REDACTED:secret]')) "
|
||||
+ "throw new Error('single-quote authorization redact failed: ' + sq);\n"
|
||||
+ "// mongodb+srv connection string (Atlas SRV)\n"
|
||||
+ "const ms = redactCredentials('mongodb+srv://u:s3cretpw@cluster.mongodb.net/db');\n"
|
||||
+ "if (!ms.includes('[REDACTED:password]')) "
|
||||
+ "throw new Error('mongodb+srv redact failed: ' + ms);\n"
|
||||
+ "// lowercase bearer scheme (RFC 7235 case-insensitive)\n"
|
||||
+ "const lb = redactCredentials('authorization: bearer "
|
||||
+ "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.sig12345');\n"
|
||||
+ "if (!lb.includes('[REDACTED:api_key]')) "
|
||||
+ "throw new Error('lowercase bearer redact failed: ' + lb);\n"
|
||||
+ "// api_key= assignment redacts the whole token, not a garbled api_[REDACTED\n"
|
||||
+ "const ak = redactCredentials('api_key=abcdefghijklmnopqrstuvwxyz');\n"
|
||||
+ "if (ak !== '[REDACTED:api_key]') "
|
||||
+ "throw new Error('api_key= clean redact failed: ' + ak);\n"
|
||||
+ "// Prefilter fast path: plain text with no anchor substring is unchanged\n"
|
||||
+ "const fp = redactCredentials('build ok in 42s - 3 tests passed');\n"
|
||||
+ "if (fp !== 'build ok in 42s - 3 tests passed') "
|
||||
+ "throw new Error('prefilter fast-path no-op failed: ' + fp);\n"
|
||||
+ "// Bare credentials with no =, quote or @ anywhere must still redact\n"
|
||||
+ "// (these pin the prefilter as a superset of the pattern set)\n"
|
||||
+ "const bk = redactCredentials('loaded sk-abcdefghijklmnopqrstuvwx');\n"
|
||||
+ "if (bk !== 'loaded [REDACTED:api_key]') "
|
||||
+ "throw new Error('bare sk- redact failed: ' + bk);\n"
|
||||
+ "const aw = redactCredentials('using AKIAABCDEFGHIJKLMNOP now');\n"
|
||||
+ "if (aw !== 'using [REDACTED:api_key] now') "
|
||||
+ "throw new Error('bare AKIA redact failed: ' + aw);\n"
|
||||
+ "const bt = redactCredentials('Bearer abcdefghijklmnopqrstuvwxyz');\n"
|
||||
+ "if (bt !== '[REDACTED:api_key]') "
|
||||
+ "throw new Error('bare bearer redact failed: ' + bt);\n"
|
||||
+ "// SQLAlchemy dialect+driver connection URLs (psycopg2/asyncpg)\n"
|
||||
+ "const pg2 = redactCredentials('postgresql+psycopg2://user:s3cret@db:5432/app');\n"
|
||||
+ "if (pg2 !== 'postgresql+psycopg2://user:[REDACTED:password]@db:5432/app') "
|
||||
+ "throw new Error('psycopg2 conn redact failed: ' + pg2);\n"
|
||||
+ "const apg = redactCredentials('postgresql+asyncpg://user:s3cret@db/app');\n"
|
||||
+ "if (apg !== 'postgresql+asyncpg://user:[REDACTED:password]@db/app') "
|
||||
+ "throw new Error('asyncpg conn redact failed: ' + apg);\n"
|
||||
+ "// RFC 3986 schemes are case-insensitive - uppercase must not bypass\n"
|
||||
+ "const up = redactCredentials('POSTGRESQL+PSYCOPG2://user:s3cret@db/app');\n"
|
||||
+ "if (up !== 'POSTGRESQL+PSYCOPG2://user:[REDACTED:password]@db/app') "
|
||||
+ "throw new Error('uppercase scheme conn redact failed: ' + up);\n"
|
||||
)
|
||||
with tempfile.NamedTemporaryFile(mode="w", suffix=".mjs", delete=False) as f:
|
||||
f.write(harness)
|
||||
tmp = f.name
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["node", "-e", script],
|
||||
["node", tmp],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=15,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
pytest.skip("node binary not available on PATH")
|
||||
finally:
|
||||
os.unlink(tmp)
|
||||
assert proc.returncode == 0, (
|
||||
f"_redactApiKeys runtime smoke failed. stdout={proc.stdout!r} stderr={proc.stderr!r}"
|
||||
f"redactCredentials runtime smoke failed. stdout={proc.stdout!r} stderr={proc.stderr!r}"
|
||||
)
|
||||
|
||||
|
||||
@@ -1757,3 +1826,107 @@ def test_global_stream_recovery_floor_and_render_coalescing() -> None:
|
||||
assert "requestAnimationFrame(" in body[fire : fire + 700], (
|
||||
"fireRender must coalesce subscriber repaints to one per frame"
|
||||
)
|
||||
|
||||
|
||||
def test_server_global_accels_are_platform_aware_and_scoped() -> None:
|
||||
"""The standalone's keydown handler owns only the GLOBAL accels — new
|
||||
workstream, switch, dashboard. They pick the modifier per platform (Ctrl on
|
||||
macOS where the browser owns Cmd, Alt elsewhere) so Ctrl+T/1-9 aren't eaten
|
||||
by the browser off macOS. The per-pane verbs (edit/refresh/fork/delete/
|
||||
close) moved to shell.js, so the handler must not invoke them itself."""
|
||||
body = _APP_JS.read_text(encoding="utf-8")
|
||||
assert "const IS_MAC" in body and 'navigator.platform.indexOf("Mac")' in body, (
|
||||
"the accelerators need a platform check to choose Ctrl vs Alt"
|
||||
)
|
||||
handler = body[body.index('document.addEventListener("keydown"') :]
|
||||
assert "const paneMod" in handler, (
|
||||
"global accels must gate on the platform-aware paneMod, not raw ctrlKey"
|
||||
)
|
||||
assert 'e.ctrlKey && e.key === "t"' not in handler, (
|
||||
"Ctrl+T is browser-reserved off macOS — new workstream must bind via paneMod"
|
||||
)
|
||||
assert "newWorkstream()" in handler and "switchTab(" in handler, (
|
||||
"the standalone handler still owns new + switch"
|
||||
)
|
||||
# macOS Ctrl+T / Ctrl+D are the Cocoa transpose / delete-forward text
|
||||
# bindings; the creation/dashboard chords must yield while typing, through
|
||||
# the shared TS_SHELL.inEditable guard (not a per-file copy).
|
||||
assert "TS_SHELL.inEditable(" in handler, (
|
||||
"new + dashboard must yield to text editing (macOS Ctrl+T / Ctrl+D)"
|
||||
)
|
||||
# The per-pane verbs are shell.js's job now — the standalone handler must not
|
||||
# double-bind them (shell.js drives them off the active pane's menu).
|
||||
for verb in ("editWorkstreamTitle()", "forkWorkstream()", "confirmDeleteWorkstream()"):
|
||||
assert verb not in handler, (
|
||||
f"{verb} moved to shell.js — the app.js handler must not also bind it"
|
||||
)
|
||||
|
||||
|
||||
def test_shortcut_overlay_labels_match_the_platform_modifier() -> None:
|
||||
"""The '?' help overlay must advertise the same modifier the handler
|
||||
listens for — Ctrl on macOS, Alt on Windows/Linux — instead of a hardcoded
|
||||
Ctrl that is wrong (and non-functional) off macOS."""
|
||||
index = _INDEX_HTML.read_text(encoding="utf-8")
|
||||
assert "const PANE_MOD" in index and 'navigator.platform.indexOf("Mac")' in index, (
|
||||
"the overlay must compute its modifier label per platform"
|
||||
)
|
||||
assert "${PANE_MOD}+T" in index, "the New-workstream badge must render through PANE_MOD"
|
||||
assert '<span class="kb-key">Ctrl+T</span>' not in index, (
|
||||
"the New-workstream badge must not hardcode Ctrl (wrong off macOS)"
|
||||
)
|
||||
|
||||
|
||||
def test_pane_menu_accels_are_shared_and_platform_aware() -> None:
|
||||
"""shell.js is the single source of truth for the per-pane tab-menu
|
||||
shortcuts: the badge string and the keydown handler come from ONE registry,
|
||||
so a badge can't advertise a chord the handler ignores. Badges must be
|
||||
platform-aware (no hardcoded Ctrl), and the shared handler must drive the
|
||||
ACTIVE pane's own menu so each surface contributes only what it supports."""
|
||||
shell = _SHELL_JS.read_text(encoding="utf-8")
|
||||
assert "PANE_MENU_ACCELS" in shell and "function paneAccelBadge" in shell, (
|
||||
"shell.js must own the accel registry + badge builder"
|
||||
)
|
||||
assert "const PANE_MOD_LABEL" in shell and 'navigator.platform.indexOf("Mac")' in shell, (
|
||||
"the shared badge must be platform-aware (Ctrl on macOS, Alt elsewhere)"
|
||||
)
|
||||
# The tab-menu items carry a stable accel + a computed badge, NOT a hardcoded
|
||||
# Ctrl string that would lie on Windows/Linux.
|
||||
for accel in ("close-pane", "edit-title", "refresh-title", "delete"):
|
||||
assert f'accel: "{accel}"' in shell, f"tab menu must tag the {accel} item"
|
||||
assert 'key: "Ctrl+Shift+E"' not in shell and 'key: "Ctrl+W"' not in shell, (
|
||||
"tab-menu badges must go through paneAccelBadge, not hardcoded Ctrl"
|
||||
)
|
||||
# The shared handler resolves the active pane and runs its menu item by accel.
|
||||
assert "paneAccelFor(e)" in shell and "pane.tabMenu()" in shell, (
|
||||
"the shared keydown handler must drive the active pane's menu by accel"
|
||||
)
|
||||
# The typing guard is shared (TS_SHELL.inEditable), not copied per surface.
|
||||
assert "function inEditable(" in shell and "inEditable," in shell, (
|
||||
"shell.js must define + expose the shared inEditable guard on TS_SHELL"
|
||||
)
|
||||
ui = _APP_JS.read_text(encoding="utf-8")
|
||||
console = _CONSOLE_APP_JS.read_text(encoding="utf-8")
|
||||
assert "_inEditable" not in ui and "_consoleInEditable" not in console, (
|
||||
"surfaces must use TS_SHELL.inEditable, not a per-file copy of the guard"
|
||||
)
|
||||
|
||||
|
||||
def test_console_has_matching_pane_hotkeys() -> None:
|
||||
"""The console regained pane hotkeys to match the standalone: a keydown
|
||||
handler for switch (Mod+1-9) + dashboard (Ctrl+D), and a '?' overlay that
|
||||
advertises them platform-aware. New workstream and Fork are intentionally
|
||||
omitted (no console fork / blank-new surface)."""
|
||||
app = _CONSOLE_APP_JS.read_text(encoding="utf-8")
|
||||
assert (
|
||||
"_CONSOLE_IS_MAC" in app and "statefulTabs()" in app and 'openPane("dashboard")' in app
|
||||
), "the console must wire switch (statefulTabs) + dashboard hotkeys"
|
||||
index = _CONSOLE_INDEX.read_text(encoding="utf-8")
|
||||
assert "const PANE_MOD" in index and '"Panes"' in index, (
|
||||
"the console '?' overlay needs a platform-aware Panes section"
|
||||
)
|
||||
assert "${PANE_MOD}+W" in index and "${PANE_MOD}+Shift+E" in index, (
|
||||
"console badges must render through PANE_MOD"
|
||||
)
|
||||
assert '"Fork"' not in index and "New workstream" not in index, (
|
||||
"Fork + New are intentionally omitted on the console"
|
||||
)
|
||||
|
||||
@@ -41,6 +41,11 @@ def _bind_ws_event_handlers(bot, cls):
|
||||
attr = getattr(cls, name)
|
||||
if callable(attr):
|
||||
setattr(bot, name, attr.__get__(bot, cls))
|
||||
# ``_handle_stream_end`` delegates the all-cycles sweep to
|
||||
# ``_pop_ws_approvals``; bind the real method too so dispatcher
|
||||
# tests observe the pop instead of a spec'd AsyncMock no-op.
|
||||
if hasattr(cls, "_pop_ws_approvals"):
|
||||
bot._pop_ws_approvals = cls._pop_ws_approvals.__get__(bot, cls)
|
||||
|
||||
|
||||
def _make_message(*, bot=False, guild=True, content="hello", channel=None, reference=None):
|
||||
@@ -537,7 +542,7 @@ class TestApprovalVerdictDisplay:
|
||||
},
|
||||
}
|
||||
]
|
||||
event = ApproveRequestEvent(ws_id="ws-1", items=items)
|
||||
event = ApproveRequestEvent(ws_id="ws-1", cycle_id="cyc-1", items=items)
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
# thread.send was called with an embed containing a verdict field
|
||||
@@ -551,8 +556,8 @@ class TestApprovalVerdictDisplay:
|
||||
assert "HIGH" in field.value
|
||||
assert "85%" in field.value
|
||||
|
||||
# Pending approval message tracked
|
||||
assert "ws-1" in bot._pending_approval_msgs
|
||||
# Pending approval message tracked under (ws_id, cycle_id).
|
||||
assert ("ws-1", "cyc-1") in bot._pending_approval_msgs
|
||||
|
||||
def test_approval_without_verdict(self):
|
||||
"""ApproveRequestEvent items without verdict still work normally."""
|
||||
@@ -585,10 +590,11 @@ class TestApprovalVerdictDisplay:
|
||||
embed = MagicMock()
|
||||
msg.embeds = [embed]
|
||||
msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (msg, frozenset({"c-1"}))
|
||||
|
||||
event = IntentVerdictEvent(
|
||||
ws_id="ws-1",
|
||||
call_id="c-1",
|
||||
func_name="bash",
|
||||
risk_level="high",
|
||||
recommendation="deny",
|
||||
@@ -628,7 +634,10 @@ class TestApprovalVerdictDisplay:
|
||||
bot._streaming = {}
|
||||
bot._thinking_msgs = {}
|
||||
bot._tool_info_msgs = {}
|
||||
bot._pending_approval_msgs = {"ws-1": MagicMock()}
|
||||
bot._pending_approval_msgs = {
|
||||
("ws-1", "cyc-1"): (MagicMock(), frozenset()),
|
||||
("ws-1", "cyc-2"): (MagicMock(), frozenset()),
|
||||
}
|
||||
bot._notify_reply_channels = {}
|
||||
_bind_ws_event_handlers(bot, TurnstoneBot)
|
||||
|
||||
@@ -636,7 +645,8 @@ class TestApprovalVerdictDisplay:
|
||||
event = StreamEndEvent(ws_id="ws-1")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
assert "ws-1" not in bot._pending_approval_msgs
|
||||
# ALL of the ws's cycles are swept, not just one entry.
|
||||
assert not bot._pending_approval_msgs
|
||||
|
||||
|
||||
class TestStreamEndBehavior:
|
||||
@@ -1657,19 +1667,21 @@ class TestApprovalResolved:
|
||||
bot = self._make_bot()
|
||||
thread = AsyncMock()
|
||||
|
||||
# Set up a pending approval message with components.
|
||||
# Set up a pending approval message with components. The event
|
||||
# below carries no cycle_id (pre-multi-cycle server) — the
|
||||
# legacy fallback clears the ws's single tracked entry.
|
||||
approval_msg = MagicMock()
|
||||
approval_msg.embeds = [MagicMock()]
|
||||
approval_msg.components = []
|
||||
approval_msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = approval_msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (approval_msg, frozenset())
|
||||
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=False, feedback="timeout")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
approval_msg.edit.assert_awaited_once()
|
||||
# Pending approval message should be removed.
|
||||
assert "ws-1" not in bot._pending_approval_msgs
|
||||
assert not bot._pending_approval_msgs
|
||||
|
||||
def test_disables_buttons_on_approved(self):
|
||||
from turnstone.sdk.events import ApprovalResolvedEvent
|
||||
@@ -1681,9 +1693,11 @@ class TestApprovalResolved:
|
||||
approval_msg.embeds = [MagicMock()]
|
||||
approval_msg.components = []
|
||||
approval_msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = approval_msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (approval_msg, frozenset())
|
||||
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True)
|
||||
# Cycle-routed resolution: the event's cycle_id selects exactly
|
||||
# this tracked message.
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True, cycle_id="cyc-1")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
approval_msg.edit.assert_awaited_once()
|
||||
|
||||
@@ -87,7 +87,7 @@ class TestSendApproval:
|
||||
monkeypatch.setattr(router._server, "approve", mock_approve)
|
||||
await router.send_approval("ws-1", "corr-abc", approved=True, feedback="ok")
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=True, feedback="ok", always=False
|
||||
ws_id="ws-1", approved=True, feedback="ok", always=False, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
@pytest.mark.anyio
|
||||
@@ -99,7 +99,7 @@ class TestSendApproval:
|
||||
monkeypatch.setattr(router._server, "approve", mock_approve)
|
||||
await router.send_approval("ws-1", "corr-abc", approved=False)
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=False, feedback=None, always=False
|
||||
ws_id="ws-1", approved=False, feedback=None, always=False, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
@pytest.mark.anyio
|
||||
@@ -110,7 +110,9 @@ class TestSendApproval:
|
||||
mock_approve = AsyncMock()
|
||||
monkeypatch.setattr(console_router._console, "route_approve", mock_approve)
|
||||
await console_router.send_approval("ws-1", "corr-abc", approved=True, always=True)
|
||||
mock_approve.assert_awaited_once_with(ws_id="ws-1", approved=True, feedback="", always=True)
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=True, feedback="", always=True, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteRoute:
|
||||
|
||||
@@ -576,10 +576,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -598,10 +599,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -620,10 +622,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -776,7 +779,9 @@ class TestWsEventDispatch:
|
||||
bot, client = self._make_ws_bot()
|
||||
|
||||
event = ApproveRequestEvent(
|
||||
ws_id="ws-1", items=[{"func_name": "bash", "needs_approval": True}]
|
||||
ws_id="ws-1",
|
||||
cycle_id="cyc-1",
|
||||
items=[{"call_id": "c-1", "func_name": "bash", "needs_approval": True}],
|
||||
)
|
||||
route = SlackRoute(channel="C1", user_id="U12345", thread_ts="123.456")
|
||||
_run(bot._on_ws_event("ws-1", route, event)) # type: ignore[attr-defined]
|
||||
@@ -784,8 +789,12 @@ class TestWsEventDispatch:
|
||||
client.chat_postMessage.assert_awaited_once()
|
||||
call_kwargs = client.chat_postMessage.call_args[1]
|
||||
assert "blocks" in call_kwargs
|
||||
assert "ws-1" in bot._pending_approval # type: ignore[attr-defined]
|
||||
assert bot._pending_approval["ws-1"].owner_user_id == "U12345" # type: ignore[attr-defined]
|
||||
# Tracked under (ws_id, cycle_id) so concurrent cycles each get
|
||||
# their own Slack message.
|
||||
entry = bot._pending_approval[("ws-1", "cyc-1")] # type: ignore[attr-defined]
|
||||
assert entry.owner_user_id == "U12345"
|
||||
assert entry.cycle_id == "cyc-1"
|
||||
assert entry.call_ids == frozenset({"c-1"})
|
||||
|
||||
def test_intent_verdict_updates_approval_message(self) -> None:
|
||||
from turnstone.channels.slack.bot import PendingApproval
|
||||
@@ -797,14 +806,17 @@ class TestWsEventDispatch:
|
||||
return_value={"ok": True, "messages": [{"blocks": []}]}
|
||||
)
|
||||
|
||||
bot._pending_approval["ws-1"] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[("ws-1", "cyc-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C1",
|
||||
message_ts="999.000",
|
||||
owner_user_id="U12345",
|
||||
cycle_id="cyc-1",
|
||||
call_ids=frozenset({"c-1"}),
|
||||
)
|
||||
|
||||
event = IntentVerdictEvent(
|
||||
ws_id="ws-1",
|
||||
call_id="c-1",
|
||||
func_name="bash",
|
||||
risk_level="high",
|
||||
confidence=0.9,
|
||||
@@ -821,17 +833,20 @@ class TestWsEventDispatch:
|
||||
from turnstone.sdk.events import ApprovalResolvedEvent
|
||||
|
||||
bot, client = self._make_ws_bot()
|
||||
bot._pending_approval["ws-1"] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[("ws-1", "cyc-9")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C1",
|
||||
message_ts="999.000",
|
||||
owner_user_id="U12345",
|
||||
cycle_id="cyc-9",
|
||||
)
|
||||
|
||||
# Event WITHOUT a cycle_id (pre-multi-cycle server): the legacy
|
||||
# fallback clears the ws's single tracked entry, as before.
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True)
|
||||
route = SlackRoute(channel="C1", user_id="U12345", thread_ts="123.456")
|
||||
_run(bot._on_ws_event("ws-1", route, event)) # type: ignore[attr-defined]
|
||||
|
||||
assert "ws-1" not in bot._pending_approval # type: ignore[attr-defined]
|
||||
assert not bot._pending_approval # type: ignore[attr-defined]
|
||||
client.chat_update.assert_awaited_once()
|
||||
|
||||
def test_link_prefix_does_not_hijack_regular_prompt(self) -> None:
|
||||
|
||||
@@ -1600,6 +1600,82 @@ class TestConsoleProxy:
|
||||
# browser's interactive UI 403-loops on every retry.
|
||||
assert sse_mock.await_args.kwargs.get("use_service_auth") is True
|
||||
|
||||
def test_proxy_events_global_403_without_cluster_inspect(self, mock_collector):
|
||||
"""A plain authenticated user (no service scope, no
|
||||
admin.cluster.inspect) cannot reach the node's cross-tenant
|
||||
firehose through the proxy: elevating to the console's service
|
||||
identity would bypass per-user filtering, so the path is
|
||||
operator-gated. _proxy_sse must NOT be reached."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from starlette.responses import Response
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from turnstone.console.server import _load_static, create_app
|
||||
from turnstone.core.auth import JWT_AUD_CONSOLE, create_jwt
|
||||
|
||||
_load_static()
|
||||
app = create_app(collector=mock_collector, jwt_secret=_TEST_JWT_SECRET)
|
||||
user_jwt = create_jwt(
|
||||
user_id="plain-user",
|
||||
scopes=frozenset({"read"}),
|
||||
source="test",
|
||||
secret=_TEST_JWT_SECRET,
|
||||
audience=JWT_AUD_CONSOLE,
|
||||
permissions=frozenset(),
|
||||
)
|
||||
user_client = TestClient(
|
||||
app,
|
||||
raise_server_exceptions=False,
|
||||
headers={"Authorization": f"Bearer {user_jwt}"},
|
||||
)
|
||||
with patch(
|
||||
"turnstone.console.server._proxy_sse",
|
||||
new_callable=AsyncMock,
|
||||
return_value=Response("ok", status_code=200),
|
||||
) as sse_mock:
|
||||
resp = user_client.get("/node/node-a/v1/api/events/global")
|
||||
assert resp.status_code == 403
|
||||
assert sse_mock.await_count == 0
|
||||
user_client.close()
|
||||
|
||||
def test_proxy_events_global_allows_cluster_inspect(self, mock_collector):
|
||||
"""An operator holding admin.cluster.inspect passes the gate and
|
||||
reaches the SSE proxy with the service token."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from starlette.responses import Response
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from turnstone.console.server import _load_static, create_app
|
||||
from turnstone.core.auth import JWT_AUD_CONSOLE, create_jwt
|
||||
|
||||
_load_static()
|
||||
app = create_app(collector=mock_collector, jwt_secret=_TEST_JWT_SECRET)
|
||||
op_jwt = create_jwt(
|
||||
user_id="operator",
|
||||
scopes=frozenset({"read"}),
|
||||
source="test",
|
||||
secret=_TEST_JWT_SECRET,
|
||||
audience=JWT_AUD_CONSOLE,
|
||||
permissions=frozenset({"admin.cluster.inspect"}),
|
||||
)
|
||||
op_client = TestClient(
|
||||
app,
|
||||
raise_server_exceptions=False,
|
||||
headers={"Authorization": f"Bearer {op_jwt}"},
|
||||
)
|
||||
with patch(
|
||||
"turnstone.console.server._proxy_sse",
|
||||
new_callable=AsyncMock,
|
||||
return_value=Response("ok", status_code=200),
|
||||
) as sse_mock:
|
||||
resp = op_client.get("/node/node-a/v1/api/events/global")
|
||||
assert resp.status_code == 200
|
||||
assert sse_mock.await_count == 1
|
||||
assert sse_mock.await_args.kwargs.get("use_service_auth") is True
|
||||
op_client.close()
|
||||
|
||||
def test_proxy_api_per_ws_events_uses_user_auth_not_service(self, client, mock_collector):
|
||||
"""Per-ws events route uses the user's re-minted JWT, not the
|
||||
service token — the upstream per-ws SSE handler scopes by
|
||||
|
||||
@@ -336,10 +336,88 @@ def test_channel_default_alias_blanked_when_disabled(
|
||||
|
||||
|
||||
def test_models_payload_strips_secret_fields(storage: SQLiteBackend) -> None:
|
||||
"""Regression guard: only alias/model/provider land in the response,
|
||||
never api_key / base_url / context_window / capabilities."""
|
||||
"""Regression guard: only alias/model/provider (+ the derived
|
||||
effort_ladder) land in the response, never api_key / base_url /
|
||||
context_window / raw capabilities."""
|
||||
_seed_model(storage, definition_id="m1", alias="primary")
|
||||
body = _get_models(_make_client(storage))
|
||||
assert body["models"] == [
|
||||
{"alias": "primary", "model": "model-x", "provider": "openai-compatible"}
|
||||
]
|
||||
assert len(body["models"]) == 1
|
||||
entry = body["models"][0]
|
||||
assert set(entry) == {"alias", "model", "provider", "effort_ladder"}
|
||||
assert entry["alias"] == "primary"
|
||||
assert entry["model"] == "model-x"
|
||||
assert entry["provider"] == "openai-compatible"
|
||||
|
||||
|
||||
def test_effort_ladder_parses_string_capabilities(storage: SQLiteBackend) -> None:
|
||||
"""The capabilities column is a JSON STRING — the ladder must survive
|
||||
the parse (regression: .items() on the raw string threw and the
|
||||
guard silently dropped the field from every row)."""
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="qwen",
|
||||
model="qwen3.6-27b",
|
||||
provider="anthropic-compatible",
|
||||
base_url="http://localhost:8000",
|
||||
api_key="dummy",
|
||||
context_window=262144,
|
||||
capabilities='{"thinking_mode": "manual", "thinking_param": "enable_thinking"}',
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
ladder = {r["value"]: r["effective"] for r in body["models"][0]["effort_ladder"]}
|
||||
assert ladder["none"] == "off"
|
||||
assert ladder["medium"] == "on+medium"
|
||||
assert ladder["max"] == "on+max"
|
||||
|
||||
|
||||
def test_effort_ladder_key_survives_malformed_capabilities(
|
||||
storage: SQLiteBackend,
|
||||
) -> None:
|
||||
"""A capabilities column that fails to parse must not drop the key —
|
||||
every row carries ``effort_ladder`` (empty on failure) so clients can
|
||||
index it unconditionally instead of null-checking per row."""
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="broken",
|
||||
model="model-x",
|
||||
provider="openai-compatible",
|
||||
base_url="http://localhost:8000/v1",
|
||||
api_key="dummy",
|
||||
context_window=131072,
|
||||
capabilities="{not valid json",
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
entry = body["models"][0]
|
||||
assert set(entry) == {"alias", "model", "provider", "effort_ladder"}
|
||||
assert entry["effort_ladder"] == []
|
||||
|
||||
|
||||
def test_effort_ladder_honors_responses_api_surface(storage: SQLiteBackend) -> None:
|
||||
"""server_compat.api_surface (namespaced inside the capabilities JSON)
|
||||
switches the projection to the flat-param path — no template toggle."""
|
||||
caps = (
|
||||
'{"thinking_mode": "manual", "thinking_param": "enable_thinking",'
|
||||
' "reasoning_effort_values": ["low", "medium", "high"],'
|
||||
' "server_compat": {"api_surface": "responses"}}'
|
||||
)
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="mistral",
|
||||
model="mistral-medium",
|
||||
provider="openai-compatible",
|
||||
base_url="http://localhost:8000/v1",
|
||||
api_key="dummy",
|
||||
context_window=131072,
|
||||
capabilities=caps,
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
ladder = {r["value"]: r["effective"] for r in body["models"][0]["effort_ladder"]}
|
||||
# Responses surface: flat param only — no "on+"/"off" toggle tokens.
|
||||
assert ladder["medium"] == "medium"
|
||||
assert ladder["none"] == "default"
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
"""``POST /v1/api/admin/models/effort-ladder`` — live modal projection.
|
||||
|
||||
Pure computation over (provider, model, unsaved capability overrides,
|
||||
api_surface); every malformed input must land as a 400, never a 500 —
|
||||
the body is operator-typed form state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from starlette.applications import Starlette
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.routing import Route
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from tests._coord_test_helpers import _AuthMiddleware
|
||||
from turnstone.console.server import admin_effort_ladder
|
||||
|
||||
|
||||
def _make_client() -> TestClient:
|
||||
app = Starlette(
|
||||
routes=[Route("/v1/api/admin/models/effort-ladder", admin_effort_ladder, methods=["POST"])],
|
||||
middleware=[Middleware(_AuthMiddleware)],
|
||||
)
|
||||
client = TestClient(app)
|
||||
client.headers.update({"X-Test-User": "admin", "X-Test-Perms": "admin.models"})
|
||||
return client
|
||||
|
||||
|
||||
def _post(client: TestClient, body: Any) -> Any:
|
||||
return client.post("/v1/api/admin/models/effort-ladder", json=body)
|
||||
|
||||
|
||||
def test_valid_request_returns_ladder() -> None:
|
||||
resp = _post(
|
||||
_make_client(),
|
||||
{
|
||||
"provider": "anthropic-compatible",
|
||||
"model": "qwen3.6-27b",
|
||||
"capabilities": {"thinking_mode": "manual", "thinking_param": "enable_thinking"},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
ladder = {r["value"]: r["effective"] for r in resp.json()["ladder"]}
|
||||
assert ladder["none"] == "off"
|
||||
assert ladder["high"] == "on+high"
|
||||
|
||||
|
||||
def test_api_surface_switches_projection() -> None:
|
||||
body = {
|
||||
"provider": "openai-compatible",
|
||||
"model": "m",
|
||||
"capabilities": {
|
||||
"thinking_mode": "manual",
|
||||
"reasoning_effort_values": ["low", "medium", "high"],
|
||||
},
|
||||
}
|
||||
client = _make_client()
|
||||
chat = {r["value"]: r["effective"] for r in _post(client, body).json()["ladder"]}
|
||||
body["api_surface"] = "responses"
|
||||
responses = {r["value"]: r["effective"] for r in _post(client, body).json()["ladder"]}
|
||||
assert chat["medium"] == "on+medium" # toggle + flat on the chat surface
|
||||
assert responses["medium"] == "medium" # flat only on the responses surface
|
||||
|
||||
|
||||
def test_non_dict_json_body_is_400_not_500() -> None:
|
||||
client = _make_client()
|
||||
for body in (None, [], "x", 7):
|
||||
resp = _post(client, body)
|
||||
assert resp.status_code == 400, (body, resp.status_code, resp.text)
|
||||
|
||||
|
||||
def test_unknown_provider_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "nope", "model": "m"})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_missing_model_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "openai", "model": ""})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_non_dict_capabilities_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "openai", "model": "m", "capabilities": [1]})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_garbage_capability_value_types_are_400() -> None:
|
||||
"""Wrong-typed override values raise inside the resolver → clean 400."""
|
||||
resp = _post(
|
||||
_make_client(),
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-fable-5",
|
||||
"capabilities": {"supports_effort": True, "effort_levels": 5},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_requires_admin_models_permission() -> None:
|
||||
client = _make_client()
|
||||
client.headers.update({"X-Test-Perms": "read"})
|
||||
resp = _post(client, {"provider": "openai", "model": "m"})
|
||||
assert resp.status_code in (401, 403)
|
||||
@@ -16,10 +16,10 @@ to ``SessionUIBase`` automatically enables:
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from tests.conftest import resolve_when_pending
|
||||
from turnstone.console.coordinator_ui import ConsoleCoordinatorUI
|
||||
|
||||
|
||||
@@ -153,7 +153,7 @@ def test_coord_heuristic_verdict_persists_to_storage() -> None:
|
||||
items[0]["_heuristic_verdict"] = hv
|
||||
|
||||
storage = MagicMock()
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(storage):
|
||||
@@ -246,9 +246,8 @@ def test_coord_pending_approval_sets_activity_tag() -> None:
|
||||
def _capture_activity() -> None:
|
||||
captured["activity"] = ui._ws_current_activity
|
||||
captured["state"] = ui._ws_activity_state
|
||||
ui.resolve_approval(False)
|
||||
|
||||
timer = threading.Timer(0.05, _capture_activity)
|
||||
timer = resolve_when_pending(ui, False, before=_capture_activity)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -292,7 +291,7 @@ def test_coord_judge_pending_flag_dynamic_when_heuristic_present() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -338,7 +337,7 @@ def test_coord_judge_pending_false_when_no_heuristic_verdict() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -410,7 +409,7 @@ def test_coord_budget_override_prompts_even_under_blanket_auto_approve() -> None
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(True))
|
||||
timer = resolve_when_pending(ui, True)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -453,7 +452,7 @@ def test_coord_budget_override_survives_wildcard_allow_policy() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(True))
|
||||
timer = resolve_when_pending(ui, True)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()), _patch_policies({"__budget_override__": "allow"}):
|
||||
@@ -526,12 +525,16 @@ class TestBroadcastApprovalResolved:
|
||||
collector = MagicMock()
|
||||
ConsoleCoordinatorUI._collector = collector
|
||||
try:
|
||||
ui._broadcast_approval_resolved(True, "lgtm", always=True)
|
||||
ui._broadcast_approval_resolved(
|
||||
True, "lgtm", always=True, cycle_id="cyc-1", call_ids=("c-1", "c-2")
|
||||
)
|
||||
collector.emit_console_ws_approval_resolved.assert_called_once_with(
|
||||
"coord-a",
|
||||
approved=True,
|
||||
feedback="lgtm",
|
||||
always=True,
|
||||
cycle_id="cyc-1",
|
||||
call_ids=["c-1", "c-2"],
|
||||
)
|
||||
finally:
|
||||
ConsoleCoordinatorUI._collector = None
|
||||
@@ -547,6 +550,8 @@ class TestBroadcastApprovalResolved:
|
||||
approved=False,
|
||||
feedback="",
|
||||
always=False,
|
||||
cycle_id="",
|
||||
call_ids=[],
|
||||
)
|
||||
finally:
|
||||
ConsoleCoordinatorUI._collector = None
|
||||
|
||||
@@ -195,6 +195,21 @@ def test_emit_tolerates_collector_exception() -> None:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_cleanup_ui_sweeps_all_approval_cycles_on_registry_uis() -> None:
|
||||
"""The real ConsoleCoordinatorUI carries the approval-cycle
|
||||
registry: cleanup denies + wakes EVERY parked gate via
|
||||
``resolve_all_approvals`` (parallel task agents can hold several),
|
||||
not the pre-cycle single-slot kick."""
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
ws.ui.resolve_all_approvals = MagicMock(return_value=2) # type: ignore[attr-defined]
|
||||
adapter.cleanup_ui(ws)
|
||||
ws.ui.resolve_all_approvals.assert_called_once_with( # type: ignore[attr-defined]
|
||||
False, "Workstream closed"
|
||||
)
|
||||
assert ws.ui._fg_event.is_set() # type: ignore[attr-defined]
|
||||
|
||||
|
||||
def test_cleanup_ui_unblocks_events_and_broadcasts_to_listeners() -> None:
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
|
||||
@@ -42,6 +42,7 @@ from turnstone.console.server import (
|
||||
_coord_create_post_install,
|
||||
_coord_create_validate_request,
|
||||
_coord_saved_loaded_lookup,
|
||||
_coordinator_tenant_check,
|
||||
_require_admin_coordinator,
|
||||
_require_coord_mgr,
|
||||
cluster_ws_detail,
|
||||
@@ -83,15 +84,24 @@ def _coord_attach_owner(request, ws_id, mgr):
|
||||
|
||||
Kind-strict — coord attachments can only be accessed for
|
||||
workstreams currently held by ``coord_mgr``; no storage fallback
|
||||
so cross-kind ws_ids 404 instead of leaking through storage.
|
||||
so cross-kind ws_ids 404 instead of leaking through storage. Also
|
||||
project-tenancy-strict: mirrors ``_coord_attachment_owner`` so a
|
||||
private-project coordinator's attachments 404-mask non-members.
|
||||
"""
|
||||
from starlette.responses import JSONResponse
|
||||
|
||||
from turnstone.core.auth import WorkstreamProjectVisibility
|
||||
from turnstone.core.web_helpers import auth_user_id
|
||||
|
||||
ws = mgr.get(ws_id)
|
||||
if ws is None:
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
storage = getattr(request.app.state, "auth_storage", None)
|
||||
if storage is None:
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
visibility = WorkstreamProjectVisibility.for_request(request, storage=storage)
|
||||
if not visibility.ws_visible(getattr(ws, "project_id", "") or "", ws_owner=ws.user_id or ""):
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
return ws.user_id or auth_user_id(request), None
|
||||
|
||||
|
||||
@@ -101,7 +111,7 @@ def _coord_attach_owner(request, ws_id, mgr):
|
||||
_coord_endpoint_config = SessionEndpointConfig(
|
||||
permission_gate=_require_admin_coordinator,
|
||||
manager_lookup=_require_coord_mgr,
|
||||
tenant_check=None,
|
||||
tenant_check=_coordinator_tenant_check,
|
||||
not_found_label="coordinator not found",
|
||||
audit_action_prefix="coordinator",
|
||||
supports_attachments=True,
|
||||
@@ -1103,18 +1113,7 @@ def test_approve_resolves_ui_event(storage):
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
assert isinstance(ws.ui, ConsoleCoordinatorUI)
|
||||
ws.ui._pending_approval = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
],
|
||||
}
|
||||
ws.ui._approval_event.clear()
|
||||
cycle = _seed_pending(ws, "c-1")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1122,34 +1121,46 @@ def test_approve_resolves_ui_event(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert ws.ui._approval_result == (True, None)
|
||||
assert resp.json()["cycle_id"] == cycle.cycle_id
|
||||
assert cycle.event.is_set()
|
||||
assert cycle.result == (True, None)
|
||||
assert "spawn_workstream" in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
def _seed_pending(ws, *call_ids: str) -> None:
|
||||
ws.ui._pending_approval = {
|
||||
def _seed_pending(ws, *call_ids: str, func_name: str = "spawn_workstream"):
|
||||
"""Register a live ApprovalCycle on the coord UI the way its
|
||||
``approve_tools`` gate does, returning the cycle for direct
|
||||
event/result assertions (the pre-cycle singleton
|
||||
``_approval_event`` / ``_approval_result`` slots are gone)."""
|
||||
from turnstone.core.session_ui_base import ApprovalCycle
|
||||
|
||||
items = [
|
||||
{
|
||||
"call_id": cid,
|
||||
"func_name": func_name,
|
||||
"approval_label": func_name,
|
||||
"needs_approval": True,
|
||||
}
|
||||
for cid in call_ids
|
||||
]
|
||||
card = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": cid,
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
for cid in call_ids
|
||||
],
|
||||
"cycle_id": f"cyc-{'-'.join(call_ids)}",
|
||||
"items": ws.ui._serialize_approval_items(items),
|
||||
"judge_pending": False,
|
||||
}
|
||||
ws.ui._approval_event.clear()
|
||||
cycle = ApprovalCycle(items, card, None)
|
||||
ws.ui._register_approval_cycle(cycle)
|
||||
return cycle
|
||||
|
||||
|
||||
def test_approve_409_on_stale_call_id(storage):
|
||||
"""Body call_id doesn't match any pending item → 409 with the
|
||||
current primary call_id so the UI can re-render against the
|
||||
new round."""
|
||||
current primary call_id + cycle_id so the UI can re-render
|
||||
against the new round."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-current")
|
||||
cycle = _seed_pending(ws, "c-current")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1160,17 +1171,17 @@ def test_approve_409_on_stale_call_id(storage):
|
||||
body = resp.json()
|
||||
assert body["error"] == "stale call_id"
|
||||
assert body["current_call_id"] == "c-current"
|
||||
# Approval event must NOT be set — no resolve_approval ran.
|
||||
assert not ws.ui._approval_event.is_set()
|
||||
assert body["current_cycle_id"] == cycle.cycle_id
|
||||
# The live cycle must NOT have been resolved.
|
||||
assert not cycle.event.is_set()
|
||||
|
||||
|
||||
def test_approve_409_when_no_pending_and_call_id_sent(storage):
|
||||
"""Body sends a call_id but the UI has no pending approval —
|
||||
409 with current_call_id=None so the UI knows to clear the row."""
|
||||
"""Body sends a call_id but the UI has no live cycle — 409 with
|
||||
current_call_id=None so the UI knows to clear the row."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
# No _pending_approval seeded → ui._pending_approval is None.
|
||||
ws.ui._approval_event.clear()
|
||||
# No cycle registered.
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1179,18 +1190,18 @@ def test_approve_409_when_no_pending_and_call_id_sent(storage):
|
||||
)
|
||||
assert resp.status_code == 409
|
||||
body = resp.json()
|
||||
assert body["error"] == "no pending approval"
|
||||
assert body["error"] == "stale call_id"
|
||||
assert body["current_call_id"] is None
|
||||
assert not ws.ui._approval_event.is_set()
|
||||
assert body["current_cycle_id"] is None
|
||||
|
||||
|
||||
def test_approve_no_call_id_preserves_backward_compat(storage):
|
||||
"""Existing clients (CLI, channel adapters) that omit call_id
|
||||
must still resolve approvals — the guard only kicks in when
|
||||
call_id is present in the body."""
|
||||
must still resolve approvals — a selector-less body lands on the
|
||||
oldest live cycle."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-1")
|
||||
cycle = _seed_pending(ws, "c-1")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1198,18 +1209,18 @@ def test_approve_no_call_id_preserves_backward_compat(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert resp.json()["cycle_id"] == cycle.cycle_id
|
||||
assert cycle.event.is_set()
|
||||
|
||||
|
||||
def test_approve_no_call_id_no_pending_falls_through(storage):
|
||||
"""Legacy clients (no call_id) calling approve when pending is
|
||||
None hit the existing resolve_approval no-op path — the new
|
||||
guard must not change that behavior. Regression guard for the
|
||||
legacy code path that the call_id check intentionally bypasses."""
|
||||
def test_approve_no_call_id_no_pending_resolves_nothing(storage):
|
||||
"""Legacy clients (no call_id) calling approve with no live cycle:
|
||||
200 with ``cycle_id: null`` — the handler resolves NOTHING rather
|
||||
than racing a cycle that registers between its lookup and its
|
||||
resolve (the client can't have been looking at one)."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
ws.ui._approval_event.clear()
|
||||
# No _pending_approval seeded.
|
||||
# No cycle registered.
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1217,7 +1228,7 @@ def test_approve_no_call_id_no_pending_falls_through(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert resp.json()["cycle_id"] is None
|
||||
|
||||
|
||||
def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
@@ -1226,7 +1237,7 @@ def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
one-boolean semantics of resolve_approval."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-1", "c-2", "c-3")
|
||||
cycle = _seed_pending(ws, "c-1", "c-2", "c-3")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1234,7 +1245,61 @@ def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert cycle.event.is_set()
|
||||
|
||||
|
||||
def test_selectorless_always_whitelists_only_the_resolved_oldest_cycle(storage):
|
||||
"""sweep-3 regression: with several live cycles, a selector-less
|
||||
"Approve + Always" must whitelist the tools of the cycle it
|
||||
actually resolved (the oldest) — not a sibling's."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
oldest = _seed_pending(ws, "a-1", func_name="spawn_workstream")
|
||||
newer = _seed_pending(ws, "b-1", func_name="send_message")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
json={"approved": True, "always": True}, # no selector
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["cycle_id"] == oldest.cycle_id
|
||||
assert oldest.event.is_set()
|
||||
assert not newer.event.is_set()
|
||||
assert "spawn_workstream" in ws.ui.auto_approve_tools
|
||||
assert "send_message" not in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
def test_approve_always_skips_whitelist_when_pinned_cycle_lost_the_race(storage):
|
||||
"""sweep-3 regression: the handler collects always-names from the
|
||||
cycle its lookup pinned; if that cycle is resolved by someone else
|
||||
(gate timeout, peer tab) between lookup and resolve, the whitelist
|
||||
must NOT grow — approving a card that already resolved must not
|
||||
auto-approve anything."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "a-1", func_name="spawn_workstream")
|
||||
ui = ws.ui
|
||||
real_find = ui.find_approval_cycle
|
||||
|
||||
def racing_find(**kwargs):
|
||||
card = real_find(**kwargs)
|
||||
if card is not None:
|
||||
# A concurrent resolver wins the gap between the handler's
|
||||
# lookup and its (pinned) resolve.
|
||||
ui.resolve_approval(False, "raced", cycle_id=card["cycle_id"])
|
||||
return card
|
||||
|
||||
ui.find_approval_cycle = racing_find
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
json={"approved": True, "always": True},
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["cycle_id"] is None
|
||||
assert "spawn_workstream" not in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1353,6 +1418,110 @@ def test_history_any_admin_coordinator_caller_can_read(storage):
|
||||
assert resp.json()["ws_id"] == ws.id
|
||||
|
||||
|
||||
def test_history_private_project_hidden_from_non_member(storage):
|
||||
# admin.coordinator gates the surface, but a coordinator in a private
|
||||
# project the caller isn't a member of is 404-masked — the conversation
|
||||
# does not leak to a non-member operator.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/history",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_history_private_project_visible_to_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/history",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert any(m.get("content") == "secret plan" for m in resp.json()["messages"])
|
||||
|
||||
|
||||
def test_export_private_project_hidden_from_non_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/export",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_children_private_project_hidden_from_non_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/children",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_open_private_project_hidden_from_non_member(storage):
|
||||
# `open` rehydrates + returns the auto-titled name, so an ungated open is a
|
||||
# private-project existence/metadata oracle AND an unauthorized resurrection.
|
||||
# The tenant_check must fire before the already-loaded shortcut and mgr.open.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{'c' * 32}/open",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_coord_attachments_private_project_hidden_from_non_member(storage):
|
||||
# Attachment list/serve resolves the owner as the coord owner and only
|
||||
# enforced cross-kind before — a non-member operator could enumerate and
|
||||
# download the owner's staged blobs. Now 404-masked by project tenancy.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{ws.id}/attachments",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_coord_attachments_private_project_visible_to_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{ws.id}/attachments",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
|
||||
|
||||
def test_history_serves_storage_only_workstream(storage):
|
||||
"""Persisted-but-not-loaded coordinators (closed / evicted) are still
|
||||
readable via /history without rehydrating. Mirrors the pre-lift
|
||||
@@ -1521,15 +1690,19 @@ def test_export_404_when_kind_interactive(storage):
|
||||
|
||||
|
||||
def test_cancel_resolves_pending_approval(storage):
|
||||
"""Cancel addresses the workstream, not one batch — EVERY live
|
||||
cycle resolves (parallel task agents can hold several gates)."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
assert isinstance(ws.ui, ConsoleCoordinatorUI)
|
||||
ws.ui._pending_approval = {"type": "approve_request", "items": []}
|
||||
ws.ui._approval_event.clear()
|
||||
first = _seed_pending(ws, "c-1")
|
||||
second = _seed_pending(ws, "c-2")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(f"/v1/api/workstreams/{ws.id}/cancel", headers=_COORD_HEADERS)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert first.event.is_set()
|
||||
assert second.event.is_set()
|
||||
assert first.result == (False, "Cancelled by user")
|
||||
|
||||
|
||||
def test_cancel_response_always_includes_dropped_key(storage):
|
||||
@@ -2049,6 +2222,10 @@ def test_open_any_admin_coordinator_caller_succeeds_in_memory(storage):
|
||||
|
||||
def test_open_rehydrates_when_not_in_memory(storage, monkeypatch):
|
||||
mgr = _build_mgr(storage)
|
||||
# The tenancy gate resolves the row from storage before rehydrating, so a
|
||||
# legitimately-openable coordinator must exist there (it always does in
|
||||
# production — open rehydrates a persisted row).
|
||||
storage.register_workstream("coord-rehy", kind="coordinator", user_id="user-1")
|
||||
rehydrated = MagicMock()
|
||||
rehydrated.id = "coord-rehy"
|
||||
rehydrated.name = "rehydrated"
|
||||
@@ -2082,6 +2259,7 @@ def test_open_503_on_coord_mgr_unavailable(storage):
|
||||
|
||||
def test_open_correlation_id_on_factory_failure(storage, monkeypatch):
|
||||
mgr = _build_mgr(storage)
|
||||
storage.register_workstream("bad-ws", kind="coordinator", user_id="user-1")
|
||||
monkeypatch.setattr(mgr, "open", MagicMock(side_effect=RuntimeError("boom")))
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post("/v1/api/workstreams/bad-ws/open", headers=_COORD_HEADERS)
|
||||
@@ -2092,6 +2270,7 @@ def test_open_correlation_id_on_factory_failure(storage, monkeypatch):
|
||||
def test_open_503_when_open_raises_value_error(storage, monkeypatch):
|
||||
"""ValueError from the factory surfaces as 503 with the remediation text."""
|
||||
mgr = _build_mgr(storage)
|
||||
storage.register_workstream("bad-ws", kind="coordinator", user_id="user-1")
|
||||
monkeypatch.setattr(mgr, "open", MagicMock(side_effect=ValueError("coord registry missing")))
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post("/v1/api/workstreams/bad-ws/open", headers=_COORD_HEADERS)
|
||||
@@ -2257,7 +2436,8 @@ def test_cluster_inspect_invalid_ws_id_400(storage):
|
||||
|
||||
|
||||
def test_cluster_inspect_any_inspect_caller_sees_detail(storage):
|
||||
# Trusted-team visibility: admin.cluster.inspect sees every row.
|
||||
# A project-less workstream has no tenancy to enforce, so any
|
||||
# admin.cluster.inspect caller sees it (trusted-team default).
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="owner")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
@@ -2269,6 +2449,46 @@ def test_cluster_inspect_any_inspect_caller_sees_detail(storage):
|
||||
assert resp.json()["persisted"]["ws_id"] == ws.id
|
||||
|
||||
|
||||
def test_cluster_inspect_private_project_hidden_from_non_member(storage):
|
||||
# admin.cluster.inspect gates the surface, but a workstream in a
|
||||
# private project the caller isn't a member of is masked as 404 —
|
||||
# no private-project oracle even for a cluster admin.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/{'c' * 32}/detail",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_cluster_inspect_private_project_visible_to_member(storage):
|
||||
# A project member (even a non-owner) still sees the persisted row.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
storage.register_workstream(
|
||||
"c" * 32,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/{'c' * 32}/detail",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["persisted"]["ws_id"] == "c" * 32
|
||||
|
||||
|
||||
def test_cluster_inspect_coordinator_self_path(storage):
|
||||
"""A coordinator row returns live from the in-process manager."""
|
||||
mgr = _build_mgr(storage)
|
||||
@@ -2399,6 +2619,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
ws_id = "f0" * 16
|
||||
_seed_node_workstream(storage, ws_id=ws_id, node_id="node-a")
|
||||
detail = {
|
||||
"cycle_id": "cyc-bash",
|
||||
"call_id": "c-bash",
|
||||
"judge_pending": False,
|
||||
"items": [
|
||||
@@ -2427,7 +2648,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
"activity_state": "approval",
|
||||
"activity": "awaiting approval",
|
||||
"tokens": 100,
|
||||
"pending_approval_detail": detail,
|
||||
"pending_approval_details": [detail],
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -2438,7 +2659,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
assert resp.status_code == 200
|
||||
live = resp.json()["live"]
|
||||
assert live["pending_approval"] is True # derived bool, existing behavior
|
||||
assert live["pending_approval_detail"] == detail # full payload, new behavior
|
||||
assert live["pending_approval_details"] == [detail] # full payload passthrough
|
||||
|
||||
|
||||
def test_cluster_inspect_node_backed_pending_approval_synthesized(storage):
|
||||
|
||||
@@ -313,17 +313,17 @@ def test_coordinator_js_handle_child_state_no_longer_reads_sse_pending_approval_
|
||||
)
|
||||
|
||||
# The merge body must preserve BOTH pending_approval and
|
||||
# pending_approval_detail from prev — preserving only one would
|
||||
# pending_approval_details from prev — preserving only one would
|
||||
# render a row with a phantom badge but no buttons (or vice versa).
|
||||
merge_body = re.search(
|
||||
r"mergedLive\s*=\s*Object\.assign\(\s*\{\}\s*,\s*live\s*,\s*\{"
|
||||
r"[^}]*pending_approval:\s*prev\.live\.pending_approval[^}]*"
|
||||
r"pending_approval_detail:\s*prev\.live\.pending_approval_detail",
|
||||
r"pending_approval_details:\s*prev\.live\.pending_approval_details",
|
||||
body,
|
||||
)
|
||||
assert merge_body is not None, (
|
||||
"Merge body must preserve both pending_approval AND "
|
||||
"pending_approval_detail from prev.live — preserving only one "
|
||||
"pending_approval_details from prev.live — preserving only one "
|
||||
"creates a half-rendered approval row."
|
||||
)
|
||||
|
||||
|
||||
@@ -198,6 +198,23 @@ def test_spawn_prepare_needs_approval(coord_session):
|
||||
assert item["skill"] == "s"
|
||||
|
||||
|
||||
def test_spawn_prepare_denies_high_risk_skill(coord_session):
|
||||
"""Review fix: the high/critical-risk gate that blocks skills(load) also
|
||||
blocks spawn_workstream(skill=…), so a child spawn can't route around it."""
|
||||
sess, _coord, _ui = coord_session
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = {
|
||||
"name": "danger",
|
||||
"risk_level": "critical",
|
||||
}
|
||||
item = sess._prepare_tool(
|
||||
_tc("spawn_workstream", {"initial_message": "go", "skill": "danger"})
|
||||
)
|
||||
assert "error" in item
|
||||
assert "/skill danger" in item["error"]
|
||||
assert item.get("needs_approval") is not True
|
||||
|
||||
|
||||
def test_spawn_exec_calls_client_and_returns_summary(coord_session):
|
||||
sess, coord, _ui = coord_session
|
||||
coord.spawn.return_value = {
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
"""Tests for the effective effort-ladder projection.
|
||||
|
||||
The ladder must mirror the request-time mapping functions exactly —
|
||||
equal ``effective`` tokens promise byte-identical effort behavior on
|
||||
the wire, which is what the UI annotations lean on.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from turnstone.core.providers._protocol import ModelCapabilities
|
||||
from turnstone.core.providers.effort_ladder import (
|
||||
KNOB_VALUES,
|
||||
effort_ladder,
|
||||
effort_ladder_for_model,
|
||||
)
|
||||
|
||||
|
||||
def _as_map(ladder: list[dict[str, str]]) -> dict[str, str]:
|
||||
assert [r["value"] for r in ladder] == list(KNOB_VALUES)
|
||||
return {r["value"]: r["effective"] for r in ladder}
|
||||
|
||||
|
||||
class TestLocalLanes:
|
||||
def test_toggle_engaged_carries_graded_value_per_position(self) -> None:
|
||||
"""No declared effort key: the toggle rides the knob AND the graded
|
||||
value is forwarded under the fallback template key — the user's
|
||||
effort setting always reaches the wire (a template that doesn't
|
||||
reference the kwarg ignores it), so every position is distinct."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
eff = _as_map(effort_ladder("anthropic-compatible", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["minimal"] == "on+minimal"
|
||||
assert eff["max"] == "on+max"
|
||||
assert len({eff[k] for k in KNOB_VALUES}) == len(KNOB_VALUES)
|
||||
|
||||
def test_freeform_effort_param_forwards_each_value(self) -> None:
|
||||
"""deepseek-style config: toggle + verbatim effort per position."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic-compatible", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["low"] == "on+low"
|
||||
assert eff["max"] == "on+max"
|
||||
|
||||
def test_validated_effort_param_shows_snapping(self) -> None:
|
||||
"""Off-list positions round up onto the declared values; above the
|
||||
ceiling they ride the ceiling — never the (possibly lower) default."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["minimal"] == "on+low"
|
||||
assert eff["high"] == "on+high"
|
||||
assert eff["xhigh"] == "on+high"
|
||||
assert eff["max"] == "on+high"
|
||||
|
||||
def test_openai_compatible_flat_param_without_effort_param(self) -> None:
|
||||
caps = ModelCapabilities(
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == "high" # ceiling, not default
|
||||
|
||||
def test_adaptive_local_never_off(self) -> None:
|
||||
caps = ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking")
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["none"] == "on"
|
||||
assert eff["max"] == "on"
|
||||
|
||||
|
||||
class TestNativeAnthropicLane:
|
||||
def test_adaptive_with_effort_levels(self) -> None:
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="adaptive",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high", "xhigh", "max"),
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["none"] == "adaptive" # thinking on, model decides
|
||||
assert eff["minimal"] == "low" # rounds up onto the declared levels
|
||||
assert eff["low"] == "low"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_sonnet_5_registry_row(self) -> None:
|
||||
"""claude-sonnet-5: adaptive + full effort ladder incl. xhigh/max —
|
||||
every knob level above none is a distinct wire behavior."""
|
||||
eff = _as_map(effort_ladder_for_model("anthropic", "claude-sonnet-5", None))
|
||||
assert eff["none"] == "adaptive"
|
||||
assert eff["minimal"] == "low" # rounds up onto declared levels
|
||||
assert eff["low"] == "low"
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_sonnet_4_6_xhigh_rides_max(self) -> None:
|
||||
"""Sonnet 4.6 declares (low, medium, high, max) — no xhigh, so the
|
||||
knob's xhigh snaps up onto max rather than down onto high."""
|
||||
eff = _as_map(effort_ladder_for_model("anthropic", "claude-sonnet-4-6", None))
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == "max"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_manual_budget_ladder(self) -> None:
|
||||
"""Budgets are monotone over the whole knob domain."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["minimal"] == eff["low"] == "budget:1024" # 1024 = API floor
|
||||
assert eff["medium"] == "budget:4096"
|
||||
assert eff["high"] == "budget:16384"
|
||||
assert eff["xhigh"] == "budget:32768"
|
||||
assert eff["max"] == "budget:65536"
|
||||
|
||||
|
||||
class TestFlatParamLanes:
|
||||
def test_google_default_caps(self) -> None:
|
||||
eff = _as_map(effort_ladder_for_model("google", "gemini-3-flash", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "minimal"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_google_override_routes_through_chat_lane(self) -> None:
|
||||
"""GoogleProvider inherits _finalize_extra_body — a thinking_mode
|
||||
override changes real requests, and the ladder must mirror it."""
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"google",
|
||||
"gemini-3-flash",
|
||||
{"thinking_mode": "manual", "thinking_param": "enable_thinking"},
|
||||
)
|
||||
)
|
||||
assert eff["none"] == "off"
|
||||
assert eff["medium"] == "on+medium" # toggle + inherited flat param
|
||||
|
||||
def test_responses_surface_projects_flat_only(self) -> None:
|
||||
caps_overrides = {
|
||||
"thinking_mode": "manual",
|
||||
"reasoning_effort_values": ["low", "medium", "high"],
|
||||
}
|
||||
chat = _as_map(effort_ladder_for_model("openai-compatible", "m", caps_overrides))
|
||||
responses = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"openai-compatible", "m", caps_overrides, api_surface="responses"
|
||||
)
|
||||
)
|
||||
assert chat["medium"] == "on+medium"
|
||||
assert responses["medium"] == "medium"
|
||||
assert responses["none"] == "default"
|
||||
|
||||
def test_xai_projects_flat_only(self) -> None:
|
||||
"""grok-4.3 declares values (none/low/medium/high, default low);
|
||||
knob positions above the ceiling ride the ceiling (high). The
|
||||
declared "none" IS forwarded for the knob's off position (xAI
|
||||
documents it as disabling reasoning) but is never a snap target
|
||||
for other positions."""
|
||||
eff = _as_map(effort_ladder_for_model("xai", "grok-4.3", None))
|
||||
assert eff["none"] == "none" # explicit disable, declared by grok
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["low"] == "low"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_xai_ignores_template_overrides(self) -> None:
|
||||
"""XAIProvider subclasses OpenAIResponsesProvider, which drops
|
||||
extra_body — a thinking_mode/effort_param override cannot change
|
||||
an xai request, so it must not change the ladder either."""
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"xai",
|
||||
"grok-4.3",
|
||||
{
|
||||
"thinking_mode": "manual",
|
||||
"thinking_param": "enable_thinking",
|
||||
"effort_param": "reasoning_effort",
|
||||
},
|
||||
)
|
||||
)
|
||||
assert eff["none"] == "none" # flat channel, not an "off" toggle
|
||||
assert eff["medium"] == "medium"
|
||||
assert all("+" not in v and v not in ("on", "off") for v in eff.values())
|
||||
|
||||
def test_openai_gpt55_registry_row(self) -> None:
|
||||
"""gpt-5.5 declares none/low/medium/high/xhigh with default medium:
|
||||
knob none sends the explicit "none" level (server default is
|
||||
MEDIUM, so omission would not disable), max rides the xhigh
|
||||
ceiling, minimal rounds up to low."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.5", None))
|
||||
assert eff["none"] == "none"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_openai_o3_registry_row(self) -> None:
|
||||
"""o-series (except o1-mini) accept low/medium/high; no declared
|
||||
"none" level, so the knob's off position omits the param."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "o3", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["medium"] == "medium"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_openai_codex_max_has_xhigh(self) -> None:
|
||||
"""gpt-5.1-codex-max must not prefix-fall onto the gpt-5.1 row
|
||||
(which lacks xhigh) — xhigh reaches the wire verbatim."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.1-codex-max", None))
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_anthropic_effort_applies_even_with_thinking_mode_none(self) -> None:
|
||||
"""output_config gates on supports_effort alone at request time."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["high"] == "high"
|
||||
assert eff["none"] == "default"
|
||||
|
||||
def test_overrides_merge_and_unknown_keys_ignored(self) -> None:
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"google",
|
||||
"gemini-3-flash",
|
||||
{"reasoning_effort_values": [], "not_a_field": True},
|
||||
)
|
||||
)
|
||||
# Operator cleared the values → nothing effort-related is sent.
|
||||
assert set(eff.values()) == {"default"}
|
||||
@@ -0,0 +1,410 @@
|
||||
"""Ladder↔wire parity harness — the effort ladder must tell the truth.
|
||||
|
||||
``effort_ladder`` *projects* the session effort knob through the same
|
||||
mapping functions the providers use at request time. This suite proves
|
||||
that projection against the REAL request path: for every provider lane
|
||||
and capability shape, each knob position is driven through the actual
|
||||
provider ``create_streaming`` against a recording fake client (the same
|
||||
SDK-seam capture the wire-payload goldens use), the effort-relevant
|
||||
subset of the captured kwargs is extracted, and it must equal what the
|
||||
ladder token decodes to. Two invariants per shape:
|
||||
|
||||
1. **Semantics** — each ladder token decodes to an expected wire subset
|
||||
(``on``/``off`` ⇒ the chat-template toggle, ``budget:N`` ⇒ Anthropic
|
||||
thinking budget, a bare level ⇒ the lane's flat/effort channel) and
|
||||
the observed wire subset must match it exactly.
|
||||
2. **Grouping** — the ladder's core promise: two knob positions carry
|
||||
equal ``effective`` tokens if and only if they produce identical
|
||||
effort-relevant wire payloads.
|
||||
|
||||
A failure here means the UI annotates behavior the wire does not have —
|
||||
the bug class that shipped xai in the ladder's chat-lane set even though
|
||||
``XAIProvider`` rides the Responses surface, which drops ``extra_body``.
|
||||
|
||||
The harness goes through ``create_provider`` (not direct classes) so the
|
||||
provider ROUTING the ladder assumes — e.g. ``api_surface="responses"``
|
||||
selecting the Responses adapter — is itself under test.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import dataclasses
|
||||
import itertools
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._wire_capture import RecordingClient
|
||||
from turnstone.core.providers import create_provider
|
||||
from turnstone.core.providers._protocol import (
|
||||
EFFORT_TEMPLATE_FALLBACK_PARAM,
|
||||
ModelCapabilities,
|
||||
)
|
||||
from turnstone.core.providers.effort_ladder import KNOB_VALUES, effort_ladder
|
||||
|
||||
# Above the largest manual-mode thinking budget (max: 65536) so the
|
||||
# request path's budget<max_tokens clamp never fires — the ladder
|
||||
# documents budgets unclamped, so the capture must be too. (At small
|
||||
# per-request max_tokens the clamp can genuinely alias adjacent budget
|
||||
# tiers on the wire; that is the ladder's documented approximation, not
|
||||
# a parity break.)
|
||||
_MAX_TOKENS = 128_000
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
class Shape:
|
||||
"""One (provider lane, capability shape) point of the parity matrix."""
|
||||
|
||||
id: str
|
||||
provider: str
|
||||
caps: ModelCapabilities
|
||||
api_surface: str = ""
|
||||
model: str = "m"
|
||||
|
||||
|
||||
# Real registry rows for the lanes whose defaults carry effort values —
|
||||
# parity should cover what ships, not only synthetic shapes.
|
||||
_GEMINI_CAPS = create_provider("google").get_capabilities("gemini-3-flash")
|
||||
_GROK_CAPS = create_provider("xai").get_capabilities("grok-4.3")
|
||||
_GPT55_CAPS = create_provider("openai").get_capabilities("gpt-5.5")
|
||||
|
||||
SHAPES: tuple[Shape, ...] = (
|
||||
# -- anthropic-compatible (vLLM /v1/messages): template channel only --
|
||||
Shape(
|
||||
"compat-toggle-manual",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"compat-toggle-adaptive",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"compat-freeform-effort",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
# DeepSeek-V4 official contract: toggle + effort in {high, max}.
|
||||
"compat-validated-effort",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("high", "max"),
|
||||
default_reasoning_effort="high",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"compat-inert",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="none"),
|
||||
),
|
||||
# -- openai-compatible on the Chat Completions surface: both channels --
|
||||
Shape(
|
||||
"oc-toggle-only",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"oc-toggle-plus-flat",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-effort-param-suppresses-flat",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-flat-only",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-adaptive",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking"),
|
||||
),
|
||||
# -- openai-compatible pinned to the Responses surface: template caps
|
||||
# become inert and only the native flat channel remains --
|
||||
Shape(
|
||||
"oc-responses-surface",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
api_surface="responses",
|
||||
),
|
||||
# -- commercial flat lanes --
|
||||
Shape(
|
||||
# Real registry row: none/low/medium/high/xhigh, default medium.
|
||||
# Knob none must send the EXPLICIT "none" level (omission would
|
||||
# leave the server default medium reasoning on); knob max rides
|
||||
# the xhigh ceiling.
|
||||
"openai-gpt-5.5",
|
||||
"openai",
|
||||
_GPT55_CAPS,
|
||||
model="gpt-5.5",
|
||||
),
|
||||
Shape("google-default", "google", _GEMINI_CAPS, model="gemini-3-flash"),
|
||||
Shape(
|
||||
# GoogleProvider subclasses the chat provider, so a template
|
||||
# override DOES change real requests — hybrid toggle + flat.
|
||||
"google-manual-override",
|
||||
"google",
|
||||
dataclasses.replace(_GEMINI_CAPS, thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
model="gemini-3-flash",
|
||||
),
|
||||
Shape("xai-default", "xai", _GROK_CAPS, model="grok-4.3"),
|
||||
Shape(
|
||||
# XAIProvider rides the Responses surface: template overrides are
|
||||
# inert on the wire, and the ladder must not pretend otherwise.
|
||||
"xai-template-override-inert",
|
||||
"xai",
|
||||
dataclasses.replace(
|
||||
_GROK_CAPS,
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
),
|
||||
model="grok-4.3",
|
||||
),
|
||||
# -- native Anthropic --
|
||||
Shape(
|
||||
"anthropic-adaptive-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="adaptive",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high", "xhigh", "max"),
|
||||
),
|
||||
model="claude-fable-5",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-adaptive-plain",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="adaptive"),
|
||||
model="claude-fable-5",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-manual-budgets",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="manual"),
|
||||
model="claude-3-7-sonnet-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-manual-plus-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
),
|
||||
model="claude-3-7-sonnet-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-none-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
),
|
||||
model="claude-3-5-haiku-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-inert",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="none"),
|
||||
model="claude-3-5-haiku-latest",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Wire capture + effort-subset extraction
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _wire_payload(shape: Shape, knob: str) -> dict[str, Any]:
|
||||
"""Drive the real provider request path; return the captured SDK kwargs."""
|
||||
provider = create_provider(shape.provider, api_surface=shape.api_surface or None)
|
||||
client = RecordingClient()
|
||||
gen = provider.create_streaming(
|
||||
client=client,
|
||||
model=shape.model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=_MAX_TOKENS,
|
||||
reasoning_effort=knob,
|
||||
capabilities=shape.caps,
|
||||
)
|
||||
# kwargs are recorded eagerly during the call above; close the
|
||||
# unconsumed iterator so stream-manager cleanup runs on the stub.
|
||||
close = getattr(gen, "close", None)
|
||||
if callable(close):
|
||||
with contextlib.suppress(Exception):
|
||||
close()
|
||||
assert "payload" in client.captured, f"{shape.id}: provider made no SDK call"
|
||||
return dict(client.captured["payload"])
|
||||
|
||||
|
||||
def _effort_wire_subset(payload: dict[str, Any], shape: Shape) -> dict[str, Any]:
|
||||
"""Every effort-related lever in *payload*, normalized across lanes.
|
||||
|
||||
Keys: ``thinking`` (native Anthropic param), ``output_effort``
|
||||
(Anthropic ``output_config.effort``), ``flat`` (Chat Completions
|
||||
``reasoning_effort`` / Responses ``reasoning.effort``), ``toggle``
|
||||
and ``template_effort`` (``extra_body.chat_template_kwargs`` — the
|
||||
graded key is ``caps.effort_param``, else the fallback template key
|
||||
on the anthropic-compatible lane, whose only effort channel is the
|
||||
template).
|
||||
"""
|
||||
caps = shape.caps
|
||||
effort_key = caps.effort_param or (
|
||||
EFFORT_TEMPLATE_FALLBACK_PARAM if shape.provider == "anthropic-compatible" else ""
|
||||
)
|
||||
subset: dict[str, Any] = {}
|
||||
if "thinking" in payload:
|
||||
subset["thinking"] = payload["thinking"]
|
||||
output_config = payload.get("output_config")
|
||||
if isinstance(output_config, dict) and "effort" in output_config:
|
||||
subset["output_effort"] = output_config["effort"]
|
||||
if "reasoning_effort" in payload:
|
||||
subset["flat"] = payload["reasoning_effort"]
|
||||
reasoning = payload.get("reasoning")
|
||||
if isinstance(reasoning, dict) and "effort" in reasoning:
|
||||
subset["flat"] = reasoning["effort"]
|
||||
extra_body = payload.get("extra_body")
|
||||
ctk = extra_body.get("chat_template_kwargs") if isinstance(extra_body, dict) else None
|
||||
if isinstance(ctk, dict):
|
||||
known = {caps.thinking_param, effort_key} - {""}
|
||||
unexpected = set(ctk) - known
|
||||
assert not unexpected, f"unexpected chat_template_kwargs keys: {unexpected}"
|
||||
if caps.thinking_param in ctk:
|
||||
subset["toggle"] = ctk[caps.thinking_param]
|
||||
if effort_key and effort_key in ctk:
|
||||
subset["template_effort"] = ctk[effort_key]
|
||||
return subset
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Ladder-token decoding — the token grammar, made executable
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _decode_token(shape: Shape, token: str) -> dict[str, Any]:
|
||||
"""Expected effort wire subset for a ladder ``effective`` token."""
|
||||
caps = shape.caps
|
||||
if shape.provider == "anthropic":
|
||||
return _decode_native(caps, token)
|
||||
if shape.provider in ("openai", "xai") or shape.api_surface == "responses":
|
||||
return {} if token == "default" else {"flat": token}
|
||||
return _decode_template(shape.provider, caps, token)
|
||||
|
||||
|
||||
def _decode_native(caps: ModelCapabilities, token: str) -> dict[str, Any]:
|
||||
if caps.thinking_mode == "adaptive":
|
||||
# Thinking is unconditionally adaptive; a non-"adaptive" token is
|
||||
# the output_config effort level riding on top.
|
||||
expected: dict[str, Any] = {"thinking": {"type": "adaptive"}}
|
||||
if token != "adaptive":
|
||||
expected["output_effort"] = token
|
||||
return expected
|
||||
if token in ("default", "off"):
|
||||
return {}
|
||||
effort, sep, budget = token.partition("·budget:")
|
||||
if sep:
|
||||
return {
|
||||
"output_effort": effort,
|
||||
"thinking": {"type": "enabled", "budget_tokens": int(budget)},
|
||||
}
|
||||
if token.startswith("budget:"):
|
||||
budget_tokens = int(token.removeprefix("budget:"))
|
||||
return {"thinking": {"type": "enabled", "budget_tokens": budget_tokens}}
|
||||
return {"output_effort": token}
|
||||
|
||||
|
||||
def _decode_template(provider: str, caps: ModelCapabilities, token: str) -> dict[str, Any]:
|
||||
if token == "default":
|
||||
return {}
|
||||
parts = token.split("+")
|
||||
expected: dict[str, Any] = {}
|
||||
if parts[0] in ("on", "off"):
|
||||
expected["toggle"] = parts[0] == "on"
|
||||
parts = parts[1:]
|
||||
if parts:
|
||||
assert len(parts) == 1, f"unparseable ladder token: {token!r}"
|
||||
if caps.effort_param or provider == "anthropic-compatible":
|
||||
# Declared graded key, or the anthropic-compatible fallback
|
||||
# template key — that lane has no flat channel, so a graded
|
||||
# part there is always template-borne.
|
||||
expected["template_effort"] = parts[0]
|
||||
else:
|
||||
expected["flat"] = parts[0]
|
||||
return expected
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# The parity tests
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", SHAPES, ids=lambda s: s.id)
|
||||
def test_ladder_tokens_match_wire(shape: Shape) -> None:
|
||||
"""Invariant 1: each token's decoded meaning equals the captured wire."""
|
||||
ladder = effort_ladder(shape.provider, shape.caps, shape.api_surface)
|
||||
assert [row["value"] for row in ladder] == list(KNOB_VALUES)
|
||||
for row in ladder:
|
||||
knob, token = row["value"], row["effective"]
|
||||
observed = _effort_wire_subset(_wire_payload(shape, knob), shape)
|
||||
expected = _decode_token(shape, token)
|
||||
assert observed == expected, (
|
||||
f"{shape.id}/knob={knob}: ladder says {token!r} which decodes to "
|
||||
f"{expected}, but the wire carries {observed}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", SHAPES, ids=lambda s: s.id)
|
||||
def test_equal_tokens_iff_equal_wire(shape: Shape) -> None:
|
||||
"""Invariant 2: token equality ⇔ effort-wire equality, per shape."""
|
||||
tokens = {
|
||||
row["value"]: row["effective"]
|
||||
for row in effort_ladder(shape.provider, shape.caps, shape.api_surface)
|
||||
}
|
||||
subsets = {knob: _effort_wire_subset(_wire_payload(shape, knob), shape) for knob in KNOB_VALUES}
|
||||
for a, b in itertools.combinations(KNOB_VALUES, 2):
|
||||
same_token = tokens[a] == tokens[b]
|
||||
same_wire = subsets[a] == subsets[b]
|
||||
assert same_token == same_wire, (
|
||||
f"{shape.id}: knobs {a!r}/{b!r} have "
|
||||
f"{'equal' if same_token else 'distinct'} tokens "
|
||||
f"({tokens[a]!r} vs {tokens[b]!r}) but "
|
||||
f"{'identical' if same_wire else 'different'} wire subsets "
|
||||
f"({subsets[a]} vs {subsets[b]})"
|
||||
)
|
||||
@@ -409,3 +409,23 @@ def test_pane_handles_cross_user_409() -> None:
|
||||
assert "r.status === 409" in body
|
||||
assert 'status: "cross_user_interjection"' in body
|
||||
assert 'data.status === "cross_user_interjection"' in body
|
||||
|
||||
|
||||
def test_sync_approval_state_prunes_orphan_cycles() -> None:
|
||||
"""``_syncApprovalState`` prunes cycles whose block elements are no longer
|
||||
in the living DOM (``.isConnected === false``). This covers the rare case
|
||||
where an ``approve_request`` event is processed between a DOM wipe
|
||||
(``clear_ui`` / ``replay_truncated`` / ``replaceChildren``) and the
|
||||
refetch-restore — the cycle card lives in a detached subtree, the matching
|
||||
``approval_resolved`` never arrives, and the send button stays disabled
|
||||
forever without this guard. The pin guards against a future refactor that
|
||||
drops the orphan prune but doesn't otherwise break ``_syncApprovalState``."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
fn_start = body.index("_syncApprovalState() {")
|
||||
assert "entry.blockEls && !entry.blockEls.some((el) => el.isConnected)" in body, (
|
||||
"orphan pruning must check .isConnected on block elements"
|
||||
)
|
||||
tail = body[fn_start : body.index("_oldestCycleId()", fn_start)]
|
||||
assert "this.approvalCycles.delete(cid);" in tail, (
|
||||
"orphan pruning must delete the cycle from the Map"
|
||||
)
|
||||
|
||||
@@ -11,12 +11,16 @@ this pins the behaviour the old ``_anthropic`` ``pc_tool_ids`` /
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from turnstone.core.lowering import (
|
||||
CANCELLED_TOOL_RESULT,
|
||||
_find_orphaned_tool_calls,
|
||||
repair_wire_messages,
|
||||
sanitize_tool_call_arguments,
|
||||
tool_args_preview,
|
||||
wire_valid_arguments,
|
||||
)
|
||||
|
||||
|
||||
@@ -180,3 +184,159 @@ def test_repair_does_not_mutate_input() -> None:
|
||||
repair_wire_messages(msgs)
|
||||
assert len(msgs) == original_len # caller's list untouched
|
||||
assert "tool_calls" in msgs[0]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# wire_valid_arguments — the shared "is this renderable" predicate
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_wire_valid_arguments_accepts_json_objects() -> None:
|
||||
assert wire_valid_arguments("{}") is True
|
||||
assert wire_valid_arguments('{"command": "ls -la"}') is True
|
||||
assert wire_valid_arguments(' { "a": 1 }\n') is True # surrounding whitespace ok
|
||||
|
||||
|
||||
def test_wire_valid_arguments_rejects_unrenderable() -> None:
|
||||
assert wire_valid_arguments('{"command": "cat /va') is False # unterminated (the incident)
|
||||
assert wire_valid_arguments("") is False # empty (no-arg call) — json.loads raises
|
||||
assert wire_valid_arguments("[]") is False # array, not object
|
||||
assert wire_valid_arguments("5") is False # bare scalar
|
||||
assert wire_valid_arguments('"hi"') is False # bare string
|
||||
assert wire_valid_arguments(None) is False # missing
|
||||
assert wire_valid_arguments({"a": 1}) is False # raw dict — not a string on the wire
|
||||
|
||||
|
||||
def test_wire_valid_arguments_totals_on_deeply_nested_json() -> None:
|
||||
# Deeply-nested JSON makes json.loads raise RecursionError (not a ValueError);
|
||||
# the predicate must return False, not propagate and crash the send.
|
||||
deep = "[" * 5000 + "]" * 5000
|
||||
assert wire_valid_arguments(deep) is False
|
||||
|
||||
|
||||
def test_tool_args_preview_stringifies_and_caps() -> None:
|
||||
assert tool_args_preview("x" * 500) == "x" * 120
|
||||
assert tool_args_preview(None) == "None"
|
||||
assert tool_args_preview({"a": 1}) == "{'a': 1}"
|
||||
|
||||
|
||||
def test_tool_args_preview_redacts_credentials() -> None:
|
||||
# Secrets in tool args (bash commands, tokens) must not reach logs — the preview
|
||||
# runs output_guard.redact_credentials over the full value first (PR #778 review).
|
||||
out = tool_args_preview('{"command": "aws configure set key AKIAIOSFODNN7EXAMPLE"}')
|
||||
assert "AKIAIOSFODNN7EXAMPLE" not in out
|
||||
assert "[REDACTED:api_key]" in out
|
||||
|
||||
|
||||
def test_tool_args_preview_is_single_line() -> None:
|
||||
# Control chars (LF/CR/TAB) collapse to spaces so the preview stays one log line.
|
||||
raw = "line1" + chr(10) + "line2" + chr(13) + "end" + chr(9) + "z"
|
||||
out = tool_args_preview(raw)
|
||||
assert chr(10) not in out and chr(13) not in out and chr(9) not in out
|
||||
assert "line1" in out and "end" in out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# sanitize_tool_call_arguments — the legalize pass
|
||||
# --------------------------------------------------------------------------- #
|
||||
def _call(call_id: str, arguments: Any, name: str = "bash") -> dict[str, Any]:
|
||||
return {"id": call_id, "type": "function", "function": {"name": name, "arguments": arguments}}
|
||||
|
||||
|
||||
def _assistant_calls(*calls: dict[str, Any]) -> dict[str, Any]:
|
||||
return {"role": "assistant", "content": "", "tool_calls": list(calls)}
|
||||
|
||||
|
||||
def test_sanitize_identity_when_all_valid() -> None:
|
||||
msgs = [_assistant_calls(_call("c1", "{}"), _call("c2", '{"a": 1}')), _tool("c1"), _tool("c2")]
|
||||
# Every arguments already a JSON object → same object returned (allocation-free).
|
||||
assert sanitize_tool_call_arguments(msgs) is msgs
|
||||
|
||||
|
||||
def test_sanitize_identity_when_no_tool_calls() -> None:
|
||||
msgs = [{"role": "user", "content": "hi"}, {"role": "assistant", "content": "yo"}]
|
||||
assert sanitize_tool_call_arguments(msgs) is msgs
|
||||
|
||||
|
||||
def test_sanitize_legalizes_unterminated_arguments() -> None:
|
||||
# The production incident: deepseek-v4-flash emitted an unterminated args string
|
||||
# with a non-``length`` finish reason, so it was committed and replayed verbatim.
|
||||
msgs = [_assistant_calls(_call("c1", '{"command": "cat /va')), _tool("c1", "retry")]
|
||||
out = sanitize_tool_call_arguments(msgs)
|
||||
assert out is not msgs # copied on repair
|
||||
assert out[0]["tool_calls"][0]["function"]["arguments"] == "{}"
|
||||
assert json.loads(out[0]["tool_calls"][0]["function"]["arguments"]) == {}
|
||||
|
||||
|
||||
def test_sanitize_legalizes_empty_arguments() -> None:
|
||||
# A no-arg tool call sends ``""``; json.loads("") raises, so deepseek_v4 would 400.
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", ""))])
|
||||
assert out[0]["tool_calls"][0]["function"]["arguments"] == "{}"
|
||||
|
||||
|
||||
def test_sanitize_legalizes_non_object_json() -> None:
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", "[]"), _call("c2", "5"))])
|
||||
assert [tc["function"]["arguments"] for tc in out[0]["tool_calls"]] == ["{}", "{}"]
|
||||
|
||||
|
||||
def test_sanitize_serializes_raw_dict_arguments() -> None:
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", {"command": "ls"}))])
|
||||
got = out[0]["tool_calls"][0]["function"]["arguments"]
|
||||
assert isinstance(got, str) and json.loads(got) == {"command": "ls"}
|
||||
|
||||
|
||||
def test_sanitize_falls_back_when_dict_not_serializable() -> None:
|
||||
# Defensive branch: a dict arguments carrying a non-JSON-encodable value
|
||||
# (a set) makes json.dumps raise TypeError — it collapses to "{}", not a crash.
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(_call("c1", {"x": {1, 2, 3}}))])
|
||||
assert out[0]["tool_calls"][0]["function"]["arguments"] == "{}"
|
||||
|
||||
|
||||
def test_sanitize_touches_only_the_offending_call() -> None:
|
||||
good = _call("c1", '{"a": 1}')
|
||||
bad = _call("c2", "{oops")
|
||||
out = sanitize_tool_call_arguments([_assistant_calls(good, bad)])
|
||||
# Valid sibling preserved by identity; only the bad call is rebuilt.
|
||||
assert out[0]["tool_calls"][0] is good
|
||||
assert out[0]["tool_calls"][1]["function"]["arguments"] == "{}"
|
||||
|
||||
|
||||
def test_sanitize_does_not_mutate_input() -> None:
|
||||
raw = '{"command": "cat /va'
|
||||
bad = _call("c1", raw)
|
||||
msgs = [_assistant_calls(bad)]
|
||||
sanitize_tool_call_arguments(msgs)
|
||||
assert bad["function"]["arguments"] == raw # caller's dict untouched
|
||||
assert msgs[0]["tool_calls"][0] is bad
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# legalize ∘ repair — the two send-time validity passes compose
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_legalize_then_repair_answered_call() -> None:
|
||||
# Malformed-but-answered (the poison-pill shape): args legalized, no orphan added.
|
||||
msgs = [_assistant_calls(_call("c1", "{bad")), _tool("c1", "retry with valid JSON")]
|
||||
out = repair_wire_messages(sanitize_tool_call_arguments(msgs))
|
||||
assert [m["role"] for m in out] == ["assistant", "tool"]
|
||||
assert json.loads(out[0]["tool_calls"][0]["function"]["arguments"]) == {}
|
||||
|
||||
|
||||
def test_legalize_then_repair_orphaned_call() -> None:
|
||||
# Malformed AND unanswered: legalized args + a synthesized cancellation result.
|
||||
msgs = [_assistant_calls(_call("c1", "{bad"))]
|
||||
out = repair_wire_messages(sanitize_tool_call_arguments(msgs))
|
||||
assert [m["role"] for m in out] == ["assistant", "tool"]
|
||||
assert json.loads(out[0]["tool_calls"][0]["function"]["arguments"]) == {}
|
||||
assert out[1]["content"] == CANCELLED_TOOL_RESULT
|
||||
|
||||
|
||||
def test_pipeline_every_emitted_arguments_is_a_json_object() -> None:
|
||||
# The end-state invariant a strict renderer relies on.
|
||||
msgs = [
|
||||
_assistant_calls(_call("c1", ""), _call("c2", "{oops"), _call("c3", '{"ok": true}')),
|
||||
_tool("c1"),
|
||||
_tool("c2"),
|
||||
_tool("c3"),
|
||||
]
|
||||
out = repair_wire_messages(sanitize_tool_call_arguments(msgs))
|
||||
for m in out:
|
||||
for tc in m.get("tool_calls", []):
|
||||
assert isinstance(json.loads(tc["function"]["arguments"]), dict)
|
||||
|
||||
+1145
-73
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,207 @@
|
||||
"""Live flaky-server smoke test: SIGKILL-flap a real MCP server, no CPU spin.
|
||||
|
||||
End-to-end regression for the flaky-server 100%-CPU incident: a real
|
||||
streamable-http MCP server (FastMCP, subprocess) is SIGKILLed and restarted
|
||||
several times underneath a real ``MCPClientManager`` with the health loop
|
||||
running on compressed timings. The production failure signature was armed
|
||||
anyio ``CancelScope``s — each one re-delivers cancellation via ``call_soon``
|
||||
every event-loop iteration, forever (~10^5+ callbacks/s), one more per flap
|
||||
cycle — so the pass criterion is structural: after the flaps settle, ZERO
|
||||
armed scopes exist on the mcp-loop, exactly one transport owner is alive, the
|
||||
health loop still runs, and a real tool call round-trips.
|
||||
|
||||
Self-contained (spawns its own server; no LLM backend, no network beyond
|
||||
127.0.0.1) — deliberately NOT marked ``live``. Wall clock ~10-15s.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import gc
|
||||
import signal
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import textwrap
|
||||
import time
|
||||
from typing import TYPE_CHECKING
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
SERVER_SRC = textwrap.dedent(
|
||||
'''
|
||||
"""Healthy streamable-http MCP server; the test SIGKILLs it to flap."""
|
||||
import sys
|
||||
|
||||
from mcp.server.fastmcp import FastMCP
|
||||
|
||||
port = int(sys.argv[1])
|
||||
mcp = FastMCP("flaky-victim", host="127.0.0.1", port=port)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def ping_me(x: int) -> int:
|
||||
"""Return x + 1."""
|
||||
return x + 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
mcp.run(transport="streamable-http")
|
||||
'''
|
||||
).lstrip()
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return int(s.getsockname()[1])
|
||||
|
||||
|
||||
def _wait_tcp_ready(port: int, timeout: float) -> bool:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
def _wait_session_live(mgr: MCPClientManager, name: str, timeout: float) -> bool:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
state = mgr._static_servers.get(name)
|
||||
if state is not None and state.session is not None:
|
||||
return True
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
async def _armed_scope_count() -> int:
|
||||
"""Armed scopes hosted on THIS (the mcp) loop — mirrors the production
|
||||
disarm sweep's scoping, and keeps an unrelated scope on another loop that
|
||||
is momentarily mid-cancellation from flaking the assertion."""
|
||||
import asyncio as _asyncio
|
||||
|
||||
from anyio._backends._asyncio import CancelScope
|
||||
|
||||
this_loop = _asyncio.get_running_loop()
|
||||
armed = 0
|
||||
for obj in gc.get_objects():
|
||||
if not isinstance(obj, CancelScope):
|
||||
continue
|
||||
if getattr(obj, "_cancel_handle", None) is None:
|
||||
continue
|
||||
host = getattr(obj, "_host_task", None)
|
||||
if host is not None and host.get_loop() is not this_loop:
|
||||
continue
|
||||
armed += 1
|
||||
return armed
|
||||
|
||||
|
||||
async def _live_owner_count() -> int:
|
||||
return sum(
|
||||
1
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-transport-owner:") and not t.done()
|
||||
)
|
||||
|
||||
|
||||
class TestFlakyServerNoSpin:
|
||||
def test_sigkill_flap_cycle_no_armed_scopes_and_recovers(
|
||||
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# The subprocess runs sys.executable, so importability HERE is a
|
||||
# faithful proxy for the server side. Environment gaps skip, not fail.
|
||||
pytest.importorskip("mcp.server.fastmcp")
|
||||
script = tmp_path / "flaky_srv.py"
|
||||
script.write_text(SERVER_SRC)
|
||||
port = _free_port()
|
||||
|
||||
# Compress recovery timings so 3 flap cycles fit a unit-test budget.
|
||||
monkeypatch.setattr(MCPClientManager, "_CONNECT_TIMEOUT", 3)
|
||||
monkeypatch.setattr(MCPClientManager, "_TCP_PROBE_TIMEOUT", 1)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_ATTEMPT_TIMEOUT_S", 5.0)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_CALLER_TIMEOUT_S", 6.0)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_BASE_S", 0.2)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_RECONNECT_MAX_S", 0.8)
|
||||
monkeypatch.setattr(MCPClientManager, "_STATIC_HEALTH_PING_TIMEOUT_S", 1.5)
|
||||
|
||||
def _spawn_server(*, initial: bool = False) -> subprocess.Popen[bytes]:
|
||||
proc = subprocess.Popen(
|
||||
[sys.executable, str(script), str(port)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
if not _wait_tcp_ready(port, 10.0):
|
||||
proc.kill()
|
||||
proc.wait(timeout=5)
|
||||
if initial:
|
||||
# Environment gap (loaded CI runner, sandboxed sockets) —
|
||||
# not a regression signal. Mid-test respawns DO fail: the
|
||||
# server already bound once, so a vanishing rebind is real.
|
||||
pytest.skip("flaky-server subprocess did not come up")
|
||||
raise AssertionError("flaky server did not come back up mid-test")
|
||||
return proc
|
||||
|
||||
proc: subprocess.Popen[bytes] | None = None
|
||||
mgr: MCPClientManager | None = None
|
||||
try:
|
||||
proc = _spawn_server(initial=True)
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.load_config",
|
||||
return_value={"static_health_check_seconds": 0.4},
|
||||
):
|
||||
mgr = MCPClientManager(
|
||||
{"flaky": {"type": "http", "url": f"http://127.0.0.1:{port}/mcp"}}
|
||||
)
|
||||
mgr.start()
|
||||
assert _wait_session_live(mgr, "flaky", 8.0), "initial connect failed"
|
||||
|
||||
for _cycle in range(3):
|
||||
proc.send_signal(signal.SIGKILL)
|
||||
proc.wait()
|
||||
time.sleep(0.6) # dead window: health loop sees the corpse
|
||||
proc = _spawn_server()
|
||||
assert _wait_session_live(mgr, "flaky", 10.0), (
|
||||
f"no reconnect after flap cycle {_cycle}"
|
||||
)
|
||||
|
||||
# Let in-flight teardown/backoff machinery fully settle.
|
||||
time.sleep(1.5)
|
||||
|
||||
assert mgr._loop is not None
|
||||
armed = asyncio.run_coroutine_threadsafe(_armed_scope_count(), mgr._loop).result(
|
||||
timeout=10
|
||||
)
|
||||
owners = asyncio.run_coroutine_threadsafe(_live_owner_count(), mgr._loop).result(
|
||||
timeout=10
|
||||
)
|
||||
health = mgr._static_health_task
|
||||
|
||||
# The production failure signature: one armed scope per flap cycle.
|
||||
assert armed == 0, f"{armed} armed cancel scope(s) — the CPU-spin signature"
|
||||
# Exactly the current session's owner is alive; the flapped ones
|
||||
# all unwound instead of leaking.
|
||||
assert owners == 1
|
||||
# The recovery machinery itself survived every flap.
|
||||
assert health is not None and not health.done()
|
||||
# The structural fix did the work — the disarm backstop never ran.
|
||||
assert mgr._last_scope_disarm == 0.0
|
||||
|
||||
# And the recovered session actually dispatches.
|
||||
out = mgr.call_tool_sync("mcp__flaky__ping_me", {"x": 41}, timeout=10)
|
||||
assert "42" in out
|
||||
finally:
|
||||
if mgr is not None:
|
||||
mgr.shutdown()
|
||||
if proc is not None:
|
||||
proc.send_signal(signal.SIGKILL)
|
||||
proc.wait(timeout=5)
|
||||
@@ -630,8 +630,6 @@ class TestCallback:
|
||||
server_name="srv-oauth",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id="ws-1",
|
||||
last_tool_call_id="tool-1",
|
||||
now_iso="2026-05-11T12:00:00",
|
||||
)
|
||||
storage.upsert_mcp_pending_consent(
|
||||
@@ -639,8 +637,6 @@ class TestCallback:
|
||||
server_name="srv-oauth",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T12:00:00",
|
||||
)
|
||||
token_store = _make_token_store(storage)
|
||||
|
||||
@@ -408,6 +408,97 @@ class TestRefreshFailureClassification:
|
||||
assert ("user-1", "srv-oauth") not in state.mcp_oauth_refresh_locks
|
||||
|
||||
|
||||
class TestObserveOnlyLookup:
|
||||
"""``revoke_on_failure=False`` (the background token-freshness sweep): still
|
||||
refresh a healthy token, but on failure NEVER delete a token or mutate the
|
||||
shared streak — a timer must not destroy consent or move a foreground user's
|
||||
revoke threshold. A permanent rejection surfaces as ``refresh_failed`` with
|
||||
the row INTACT; an ambiguous one as transient with the streak untouched."""
|
||||
|
||||
def _lookup(self, state: SimpleNamespace) -> Any:
|
||||
from turnstone.core.mcp_oauth import get_user_access_token_classified
|
||||
|
||||
async def _run() -> Any:
|
||||
with _public_addr_patch():
|
||||
return await get_user_access_token_classified(
|
||||
app_state=state,
|
||||
user_id="user-1",
|
||||
server_name="srv-oauth",
|
||||
force_refresh=True,
|
||||
revoke_on_failure=False,
|
||||
)
|
||||
|
||||
return asyncio.run(_run())
|
||||
|
||||
def test_permanent_invalid_grant_does_not_revoke(self, storage: SQLiteBackend) -> None:
|
||||
"""The exact contrast to ``test_permanent_invalid_grant_revokes``: same
|
||||
dead-grant signal, but observe-only leaves the row for the lazy path."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
client.post = AsyncMock(return_value=_mk_response(400, {"error": "invalid_grant"}))
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000)
|
||||
|
||||
result = self._lookup(state)
|
||||
|
||||
assert result.kind == "refresh_failed"
|
||||
assert state.mcp_token_store.get_user_token("user-1", "srv-oauth") is not None
|
||||
|
||||
def test_ambiguous_does_not_touch_shared_streak(self, storage: SQLiteBackend) -> None:
|
||||
"""Repeated observe-mode ambiguous failures never bump the shared
|
||||
ambiguous_streak, so a later foreground dispatch is not pushed over the
|
||||
escalation edge by background activity (the finding this guards)."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
client.post = AsyncMock(return_value=_mk_response(400, None))
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000)
|
||||
|
||||
with patch("turnstone.core.mcp_oauth._AMBIGUOUS_ESCALATION_THRESHOLD", 2):
|
||||
for _ in range(5):
|
||||
assert self._lookup(state).kind == "refresh_failed_transient"
|
||||
|
||||
backoff = getattr(state, "mcp_oauth_refresh_backoff", {})
|
||||
entry = backoff.get(("user-1", "srv-oauth"))
|
||||
assert entry is None or entry.ambiguous_streak == 0
|
||||
assert state.mcp_token_store.get_user_token("user-1", "srv-oauth") is not None
|
||||
|
||||
def test_expired_no_refresh_does_not_revoke(self, storage: SQLiteBackend) -> None:
|
||||
"""An expired token with no refresh token surfaces as a dead grant but is
|
||||
NOT deleted on the observe path."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000, refresh=None)
|
||||
|
||||
result = self._lookup(state)
|
||||
|
||||
assert result.kind == "refresh_failed"
|
||||
assert state.mcp_token_store.get_user_token("user-1", "srv-oauth") is not None
|
||||
|
||||
def test_healthy_token_still_refreshes(self, storage: SQLiteBackend) -> None:
|
||||
"""Observe mode is not read-only: a near-expiry token is still refreshed
|
||||
(only the destructive failure paths change)."""
|
||||
_seed_server(storage)
|
||||
client = MagicMock(spec=httpx.AsyncClient)
|
||||
client.get = AsyncMock(return_value=_mk_response(200, _good_as_metadata_doc()))
|
||||
client.post = AsyncMock(
|
||||
return_value=_mk_response(
|
||||
200, {"access_token": "fresh-bbb", "expires_in": 3600, "token_type": "Bearer"}
|
||||
)
|
||||
)
|
||||
state = _make_app_state(storage, http_client=client)
|
||||
_seed_token(state, expires_in_seconds=-1000)
|
||||
|
||||
result = self._lookup(state)
|
||||
|
||||
assert result.kind == "token"
|
||||
assert result.token == "fresh-bbb"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Happy paths
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -108,8 +108,6 @@ def _seed_pending(
|
||||
server_name=server_name,
|
||||
error_code=error_code,
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=now_iso,
|
||||
)
|
||||
|
||||
|
||||
@@ -24,8 +24,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required="read write",
|
||||
last_ws_id="ws-1",
|
||||
last_tool_call_id="tool-1",
|
||||
now_iso=_iso(),
|
||||
)
|
||||
rows = backend.list_mcp_pending_consent_by_user("user-a")
|
||||
@@ -35,8 +33,6 @@ class TestUpsertAndList:
|
||||
assert r["server_name"] == "srv-x"
|
||||
assert r["error_code"] == "mcp_consent_required"
|
||||
assert r["scopes_required"] == "read write"
|
||||
assert r["last_ws_id"] == "ws-1"
|
||||
assert r["last_tool_call_id"] == "tool-1"
|
||||
assert r["occurrence_count"] == 1
|
||||
assert r["first_seen_at"] == r["last_seen_at"]
|
||||
|
||||
@@ -46,8 +42,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T12:00:00",
|
||||
)
|
||||
backend.upsert_mcp_pending_consent(
|
||||
@@ -55,8 +49,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_insufficient_scope",
|
||||
scopes_required="read",
|
||||
last_ws_id="ws-2",
|
||||
last_tool_call_id="tool-2",
|
||||
now_iso="2026-05-11T13:00:00",
|
||||
)
|
||||
rows = backend.list_mcp_pending_consent_by_user("user-a")
|
||||
@@ -66,8 +58,6 @@ class TestUpsertAndList:
|
||||
assert r["occurrence_count"] == 2
|
||||
assert r["error_code"] == "mcp_insufficient_scope"
|
||||
assert r["scopes_required"] == "read"
|
||||
assert r["last_ws_id"] == "ws-2"
|
||||
assert r["last_tool_call_id"] == "tool-2"
|
||||
assert r["last_seen_at"] == "2026-05-11T13:00:00"
|
||||
# first_seen_at preserved — that's the load-bearing audit value.
|
||||
assert r["first_seen_at"] == "2026-05-11T12:00:00"
|
||||
@@ -78,8 +68,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-old",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T10:00:00",
|
||||
)
|
||||
backend.upsert_mcp_pending_consent(
|
||||
@@ -87,8 +75,6 @@ class TestUpsertAndList:
|
||||
server_name="srv-new",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso="2026-05-11T11:00:00",
|
||||
)
|
||||
rows = backend.list_mcp_pending_consent_by_user("user-a")
|
||||
@@ -100,8 +86,6 @@ class TestUpsertAndList:
|
||||
server_name="srv",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
assert backend.list_mcp_pending_consent_by_user("user-b") == []
|
||||
@@ -114,8 +98,6 @@ class TestDelete:
|
||||
server_name="srv-x",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
assert backend.delete_mcp_pending_consent("user-a", "srv-x") is True
|
||||
@@ -133,8 +115,6 @@ class TestDelete:
|
||||
server_name=name,
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
# Cross-user row that must NOT be touched.
|
||||
@@ -143,8 +123,6 @@ class TestDelete:
|
||||
server_name="srv-z",
|
||||
error_code="mcp_consent_required",
|
||||
scopes_required=None,
|
||||
last_ws_id=None,
|
||||
last_tool_call_id=None,
|
||||
now_iso=_iso(),
|
||||
)
|
||||
assert backend.delete_all_mcp_pending_consent_by_user("user-a") == 3
|
||||
|
||||
@@ -1033,20 +1033,22 @@ class TestStaticPathUnchanged:
|
||||
|
||||
from turnstone.core import mcp_client
|
||||
|
||||
source = inspect.getsource(mcp_client.MCPClientManager._connect_one)
|
||||
# The static path's streamablehttp_client call site lives in the
|
||||
# transport owner task (``_static_transport_owner``); ``_connect_one``
|
||||
# is a per-name-lock wrapper and ``_connect_one_locked`` only waits on
|
||||
# the owner's readiness.
|
||||
source = inspect.getsource(mcp_client.MCPClientManager._static_transport_owner)
|
||||
|
||||
# The static path's streamablehttp_client invocation should NOT
|
||||
# mention ``httpx_client_factory``. Pool path keeps it.
|
||||
# Find the streamablehttp_client(...) call inside _connect_one.
|
||||
assert "streamablehttp_client" in source
|
||||
# The call site in _connect_one is bare — no factory keyword.
|
||||
# We grep by line: the factory keyword must not appear in the
|
||||
# static-path source.
|
||||
# The call site in the owner is bare — no factory keyword. We grep by
|
||||
# line: the factory keyword must not appear in the static-path source.
|
||||
for line in source.splitlines():
|
||||
if "httpx_client_factory" in line:
|
||||
pytest.fail(
|
||||
"_connect_one (static path) passes httpx_client_factory to "
|
||||
"streamablehttp_client; hard invariant 1 violated."
|
||||
"_static_transport_owner (static path) passes httpx_client_factory "
|
||||
"to streamablehttp_client; hard invariant 1 violated."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,478 @@
|
||||
"""Pool transport owner-task lifecycle + anyio cancel-scope regressions.
|
||||
|
||||
The pool (auth_type=oauth_user) sibling of ``test_mcp_transport_owner.py``.
|
||||
Each ``(user, server)`` pool entry's transport + ``ClientSession`` cms are now
|
||||
entered, parked, and exited by ONE long-lived owner task
|
||||
(``_pool_transport_owner``) with a one-cancel close protocol, so a cancel scope
|
||||
whose host task has finished can never be left re-delivering cancellation in a
|
||||
``call_soon`` loop (the SDK #2147 100%-CPU spin). These fast mock-transport
|
||||
tests pin that protocol for the pool path; the real-server integration coverage
|
||||
lives in ``test_mcp_pool_auth_integration.py``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import threading
|
||||
import time
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager, PoolEntryState, _AuthCapture
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def running_loop_mgr():
|
||||
"""Background-loop fixture matching the pool-path test convention.
|
||||
|
||||
Teardown drains the eviction / sweep / health tasks AND any parked pool
|
||||
transport owner a successful connect left installed — the conftest fails
|
||||
leaked threads and an undrained owner is destroyed pending at GC.
|
||||
"""
|
||||
cfg: dict[str, Any] = {}
|
||||
mgr = MCPClientManager(cfg)
|
||||
loop = asyncio.new_event_loop()
|
||||
thread = threading.Thread(target=loop.run_forever, daemon=True, name="mcp-pool-owner-test-loop")
|
||||
thread.start()
|
||||
mgr._loop = loop
|
||||
try:
|
||||
yield mgr, loop, thread
|
||||
finally:
|
||||
|
||||
async def _drain(m: MCPClientManager) -> None:
|
||||
for attr in (
|
||||
"_user_pool_eviction_task",
|
||||
"_user_token_sweep_task",
|
||||
"_static_health_task",
|
||||
):
|
||||
task = getattr(m, attr)
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
await asyncio.gather(task, return_exceptions=True)
|
||||
setattr(m, attr, None)
|
||||
for entry in list(m._user_pool_entries.values()):
|
||||
owner = entry.owner_task
|
||||
if owner is not None and not owner.done():
|
||||
if entry.close_requested is not None:
|
||||
entry.close_requested.set()
|
||||
owner.cancel()
|
||||
await asyncio.gather(owner, return_exceptions=True)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
asyncio.run_coroutine_threadsafe(_drain(mgr), loop).result(timeout=5)
|
||||
loop.call_soon_threadsafe(loop.stop)
|
||||
thread.join(timeout=5)
|
||||
if not thread.is_alive():
|
||||
loop.close()
|
||||
|
||||
|
||||
def _run(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 5.0) -> Any:
|
||||
return asyncio.run_coroutine_threadsafe(coro, loop).result(timeout=timeout)
|
||||
|
||||
|
||||
def _http_cfg() -> dict[str, Any]:
|
||||
return {"type": "streamable-http", "url": "https://mcp.example.com/mcp", "headers": {}}
|
||||
|
||||
|
||||
def _make_pool_session_mock() -> AsyncMock:
|
||||
"""A ClientSession-shaped mock good enough for pool connect + discovery."""
|
||||
session = AsyncMock()
|
||||
session.initialize = AsyncMock()
|
||||
# None caps → resources/prompts discovery is skipped; only list_tools runs.
|
||||
session.get_server_capabilities = MagicMock(return_value=None)
|
||||
session.list_tools = AsyncMock(return_value=MagicMock(tools=[]))
|
||||
return session
|
||||
|
||||
|
||||
def _fake_transport_and_session(patches: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Build fake streamable-http transport + ClientSession cms.
|
||||
|
||||
Records enter/exit events and captures the kwargs that reach
|
||||
``streamablehttp_client`` (so the bearer-header / factory contract is
|
||||
observable).
|
||||
"""
|
||||
events: list[str] = []
|
||||
captured_kwargs: dict[str, Any] = {}
|
||||
session = _make_pool_session_mock()
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_streamablehttp_client(**kwargs: Any):
|
||||
captured_kwargs.clear()
|
||||
captured_kwargs.update(kwargs)
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
yield (AsyncMock(), AsyncMock(), lambda: None)
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_client_session_cm():
|
||||
events.append("session_enter")
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
events.append("session_exit")
|
||||
|
||||
def fake_client_session(_read: Any, _write: Any, message_handler: Any = None):
|
||||
return fake_client_session_cm()
|
||||
|
||||
patches["streamablehttp_client"] = fake_streamablehttp_client
|
||||
patches["ClientSession"] = fake_client_session
|
||||
return {"events": events, "session": session, "kwargs": captured_kwargs}
|
||||
|
||||
|
||||
async def _connect_under_lock(
|
||||
mgr: MCPClientManager, key: tuple[str, str], cfg: dict[str, Any], **kw: Any
|
||||
) -> PoolEntryState:
|
||||
"""Drive ``_connect_one_pool`` the way production does — under open_lock."""
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
async with entry.open_lock:
|
||||
return await mgr._connect_one_pool(key, cfg, "tok-aaa", **kw)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Owner lifecycle
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestPoolTransportOwnerLifecycle:
|
||||
def test_connect_installs_owner_and_teardown_closes_gracefully(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client", patches["streamablehttp_client"]
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
entry = _run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
assert entry.session is fake["session"]
|
||||
owner = entry.owner_task
|
||||
assert owner is not None and not owner.done()
|
||||
assert entry.close_requested is not None
|
||||
assert fake["events"] == ["transport_enter", "session_enter"]
|
||||
|
||||
_run(loop, mgr._teardown_pool_entry(key))
|
||||
|
||||
# Graceful close: the parked owner exits via the event — no cancel —
|
||||
# and unwinds BOTH cms in-task, inner-out (session before transport).
|
||||
assert owner.done() and not owner.cancelled()
|
||||
assert fake["events"] == [
|
||||
"transport_enter",
|
||||
"session_enter",
|
||||
"session_exit",
|
||||
"transport_exit",
|
||||
]
|
||||
assert entry.session is None
|
||||
assert entry.owner_task is None
|
||||
assert entry.close_requested is None
|
||||
# The entry itself is NOT popped — teardown leaves map/catalog cleanup
|
||||
# to callers.
|
||||
assert key in mgr._user_pool_entries
|
||||
|
||||
def test_owner_death_during_discovery_fails_fast(self, running_loop_mgr) -> None:
|
||||
"""The owner-died branch of ``_await_owner_discovery`` — the reason the
|
||||
helper exists: discovery runs in the caller while the transport is
|
||||
hosted by the owner, so a transport collapse mid-discovery cancels the
|
||||
OWNER and a bare await on the response stream would hang until the 30s
|
||||
phase timeout. The race must convert that into a PROMPT
|
||||
``ConnectionError``, reap the parked discovery future, and leave the
|
||||
entry torn down."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
discovery_parked = asyncio.Event()
|
||||
|
||||
async def _parked_list_tools() -> Any:
|
||||
discovery_parked.set()
|
||||
await asyncio.sleep(3600) # the transport never answers
|
||||
|
||||
fake["session"].list_tools = AsyncMock(side_effect=_parked_list_tools)
|
||||
|
||||
async def _drive() -> tuple[float, BaseException | None]:
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
|
||||
async def _collapse_owner_when_parked() -> None:
|
||||
await discovery_parked.wait()
|
||||
owner = entry.owner_task # installed before discovery begins
|
||||
assert owner is not None
|
||||
# The transport task group collapsing under live discovery
|
||||
# (e.g. an upstream 401) surfaces as the owner being cancelled.
|
||||
owner.cancel()
|
||||
|
||||
collapser = asyncio.create_task(_collapse_owner_when_parked())
|
||||
t0 = asyncio.get_running_loop().time()
|
||||
exc: BaseException | None = None
|
||||
try:
|
||||
async with entry.open_lock:
|
||||
await mgr._connect_one_pool(key, _http_cfg(), "tok-aaa")
|
||||
except Exception as e:
|
||||
# The expected ConnectionError; anything else (a cancel leak,
|
||||
# an interpreter exit) propagates and fails the test loudly.
|
||||
exc = e
|
||||
_ = await collapser # synchronization point; failures propagate
|
||||
return asyncio.get_running_loop().time() - t0, exc
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client", patches["streamablehttp_client"]
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
elapsed, exc = _run(loop, _drive(), timeout=15)
|
||||
|
||||
assert isinstance(exc, ConnectionError)
|
||||
assert "died during discovery" in str(exc)
|
||||
assert elapsed < 5.0 # prompt fail — not the 30s phase timeout
|
||||
entry = mgr._user_pool_entries[key]
|
||||
assert entry.session is None # discovery-failure teardown ran
|
||||
assert entry.owner_task is None
|
||||
|
||||
def test_cancelled_discovery_future_converts_to_connection_error(
|
||||
self, running_loop_mgr
|
||||
) -> None:
|
||||
"""A discovery future that completes CANCELLED without this race's own
|
||||
reap (an SDK-internal cancellation shape) is the transport-failure
|
||||
class, not the caller's cancellation — ``_await_owner_discovery`` must
|
||||
surface it as ``ConnectionError``, never a bare ``CancelledError`` the
|
||||
caller would misread as its own cancel."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
async def _drive() -> BaseException | None:
|
||||
parked = asyncio.Event()
|
||||
|
||||
async def _parked_owner() -> None:
|
||||
await parked.wait()
|
||||
|
||||
owner = asyncio.create_task(_parked_owner())
|
||||
await asyncio.sleep(0)
|
||||
|
||||
async def _self_cancelling_discovery() -> Any:
|
||||
# A coroutine raising CancelledError makes its wrapping task
|
||||
# complete CANCELLED — the shape of an SDK-internal cancel.
|
||||
raise asyncio.CancelledError
|
||||
|
||||
exc: BaseException | None = None
|
||||
try:
|
||||
await mgr._await_owner_discovery(owner, _self_cancelling_discovery())
|
||||
except (Exception, asyncio.CancelledError) as e:
|
||||
# Exception covers the expected ConnectionError; CancelledError
|
||||
# covers the exact regression this test guards (the bare cancel
|
||||
# leaking through instead of being converted).
|
||||
exc = e
|
||||
parked.set()
|
||||
_ = await owner # synchronization point; failures propagate
|
||||
return exc
|
||||
|
||||
exc = _run(loop, _drive())
|
||||
assert isinstance(exc, ConnectionError)
|
||||
assert "cancelled by transport failure" in str(exc)
|
||||
|
||||
def test_teardown_single_cancel_escalation(self, running_loop_mgr) -> None:
|
||||
"""A parked owner whose in-task unwind stalls past the graceful window
|
||||
gets EXACTLY ONE cancel — never a second (a second abandons an anyio
|
||||
scope exit mid-flight and mints the zombie the protocol prevents)."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._OWNER_CLOSE_GRACE_S = 0.1
|
||||
mgr._OWNER_CANCEL_GRACE_S = 1.0
|
||||
|
||||
events: list[str] = []
|
||||
cancels = {"n": 0}
|
||||
session = _make_pool_session_mock()
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_streamablehttp_client(**_kwargs: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
yield (AsyncMock(), AsyncMock(), lambda: None)
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_session_cm():
|
||||
events.append("session_enter")
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
# Stall the graceful unwind so teardown must escalate; count
|
||||
# each cancellation that reaches this in-task exit.
|
||||
try:
|
||||
await asyncio.sleep(3600)
|
||||
except asyncio.CancelledError:
|
||||
cancels["n"] += 1
|
||||
raise
|
||||
finally:
|
||||
events.append("session_exit")
|
||||
|
||||
def fake_session(_read: Any, _write: Any, message_handler: Any = None):
|
||||
return fake_session_cm()
|
||||
|
||||
key = ("user-1", "pool-srv")
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.streamablehttp_client", fake_streamablehttp_client),
|
||||
patch("turnstone.core.mcp_client.ClientSession", fake_session),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
entry = _run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
owner = entry.owner_task
|
||||
assert owner is not None
|
||||
_run(loop, mgr._teardown_pool_entry(key), timeout=10)
|
||||
|
||||
assert owner.done() and owner.cancelled()
|
||||
assert cancels["n"] == 1
|
||||
assert events[-1] == "transport_exit"
|
||||
assert entry.session is None and entry.owner_task is None
|
||||
|
||||
def test_owner_death_evicts_session_keeps_entry_and_catalog(self, running_loop_mgr) -> None:
|
||||
"""The transport collapsing under a live session (owner dies with no
|
||||
requested close) evicts the session via the done-callback but leaves the
|
||||
entry AND its discovered catalog in place for the next dispatch."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client", patches["streamablehttp_client"]
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
entry = _run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
owner = entry.owner_task
|
||||
assert owner is not None and entry.session is fake["session"]
|
||||
# Seed a catalog so we can prove the death-callback leaves it alone.
|
||||
entry.tools = [{"name": "mcp__pool-srv__ping", "server": "pool-srv"}]
|
||||
|
||||
# Simulate the transport task group collapsing: the owner gets a
|
||||
# stray cancellation (exactly what anyio's scope delivery does).
|
||||
loop.call_soon_threadsafe(owner.cancel)
|
||||
deadline = time.monotonic() + 5
|
||||
while time.monotonic() < deadline and entry.owner_task is not None:
|
||||
time.sleep(0.02)
|
||||
|
||||
assert owner.done()
|
||||
assert entry.session is None # evicted by the done-callback
|
||||
assert entry.owner_task is None
|
||||
assert key in mgr._user_pool_entries # entry kept
|
||||
assert entry.tools == [
|
||||
{"name": "mcp__pool-srv__ping", "server": "pool-srv"}
|
||||
] # catalog kept
|
||||
# The cms were still unwound in-task despite the stray cancel.
|
||||
assert fake["events"][-2:] == ["session_exit", "transport_exit"]
|
||||
|
||||
def test_caller_cancel_mid_connect_does_not_abandon_cms(self, running_loop_mgr) -> None:
|
||||
"""Cancelling the CONNECTING caller (an eviction giving up, shutdown, a
|
||||
sync boundary timing out) must close the owner via the one-cancel
|
||||
protocol — the transport cm still exits, in-task."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
events: list[str] = []
|
||||
entered = asyncio.Event()
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
@asynccontextmanager
|
||||
async def hanging_streamablehttp_client(**_kwargs: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
entered.set()
|
||||
await asyncio.sleep(3600) # server accepted, then stalled
|
||||
yield (AsyncMock(), AsyncMock(), lambda: None)
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
async def _drive() -> None:
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
|
||||
async def _connect() -> None:
|
||||
async with entry.open_lock:
|
||||
await mgr._connect_one_pool(key, _http_cfg(), "tok-aaa")
|
||||
|
||||
connect = asyncio.create_task(_connect())
|
||||
await asyncio.wait_for(entered.wait(), timeout=5)
|
||||
connect.cancel() # the attempt-timeout / shutdown shape
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
_ = await connect # only the expected cancel is absorbed
|
||||
# The owner must be closed (one cancel) and fully unwound.
|
||||
deadline = asyncio.get_running_loop().time() + 5
|
||||
while asyncio.get_running_loop().time() < deadline:
|
||||
owners = [
|
||||
t
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-pool-owner:") and not t.done()
|
||||
]
|
||||
if not owners:
|
||||
return
|
||||
await asyncio.sleep(0.02)
|
||||
raise AssertionError("owner task still alive after caller cancel")
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.streamablehttp_client", hanging_streamablehttp_client),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
_run(loop, _drive(), timeout=15)
|
||||
|
||||
assert events == ["transport_enter", "transport_exit"]
|
||||
assert mgr._user_pool_entries[key].session is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Client-kwargs contract (bearer header + auth-capture factory)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestPoolOwnerClientKwargs:
|
||||
def test_client_factory_present_iff_auth_capture(self, running_loop_mgr) -> None:
|
||||
"""The caller builds ``client_kwargs`` and the owner passes them to
|
||||
``streamablehttp_client`` verbatim: the auth-capture
|
||||
``httpx_client_factory`` is present exactly when a carrier is supplied,
|
||||
and the per-user bearer always reaches the wire."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
key = ("user-1", "pool-srv")
|
||||
|
||||
# With auth_capture → factory present.
|
||||
patches_a: dict[str, Any] = {}
|
||||
fake_a = _fake_transport_and_session(patches_a)
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client",
|
||||
patches_a["streamablehttp_client"],
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches_a["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
_run(loop, _connect_under_lock(mgr, key, _http_cfg(), auth_capture=_AuthCapture()))
|
||||
assert "httpx_client_factory" in fake_a["kwargs"]
|
||||
assert fake_a["kwargs"]["headers"]["Authorization"] == "Bearer tok-aaa"
|
||||
_run(loop, mgr._teardown_pool_entry(key))
|
||||
|
||||
# Without auth_capture → factory absent (but bearer still present).
|
||||
patches_b: dict[str, Any] = {}
|
||||
fake_b = _fake_transport_and_session(patches_b)
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.streamablehttp_client",
|
||||
patches_b["streamablehttp_client"],
|
||||
),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches_b["ClientSession"]),
|
||||
patch.object(mgr, "_tcp_probe", new=AsyncMock()),
|
||||
):
|
||||
_run(loop, _connect_under_lock(mgr, key, _http_cfg()))
|
||||
assert "httpx_client_factory" not in fake_b["kwargs"]
|
||||
assert fake_b["kwargs"]["headers"]["Authorization"] == "Bearer tok-aaa"
|
||||
_run(loop, mgr._teardown_pool_entry(key))
|
||||
@@ -0,0 +1,488 @@
|
||||
"""Transport owner-task lifecycle + anyio cancel-scope zombie regressions.
|
||||
|
||||
Covers the two bugs behind the flaky-MCP-server 100%-CPU incident:
|
||||
|
||||
* Bug 1 — an anyio cancel scope whose host task has finished can never be
|
||||
exited; once cancelled (SDK task-group child death, or a teardown racing a
|
||||
connect) anyio re-delivers cancellation to it via ``call_soon`` every loop
|
||||
iteration, forever. The fix routes every transport cm through a long-lived
|
||||
per-server OWNER task (enter, park, exit — all in one task) with a
|
||||
one-cancel close protocol; these tests pin the protocol's behavior.
|
||||
* Bug 2 — ``BaseExceptionGroup`` (BaseException-derived) escaping
|
||||
``except Exception`` killed ``_connect_all`` before the health/sweep loops
|
||||
were created, silently disabling all autonomous recovery.
|
||||
|
||||
The live end-to-end flap test (real server, SIGKILL cycle) lives in
|
||||
``test_mcp_live_flaky_server.py``; these are fast mock-transport unit tests.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import threading
|
||||
import time
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def running_loop_mgr():
|
||||
"""Background-loop fixture matching the static-path test convention."""
|
||||
cfg: dict[str, Any] = {"srv": {"type": "stdio", "command": "fake-cmd"}}
|
||||
mgr = MCPClientManager(cfg)
|
||||
loop = asyncio.new_event_loop()
|
||||
thread = threading.Thread(target=loop.run_forever, daemon=True, name="mcp-owner-test-loop")
|
||||
thread.start()
|
||||
mgr._loop = loop
|
||||
try:
|
||||
yield mgr, loop, thread
|
||||
finally:
|
||||
|
||||
async def _drain(m: MCPClientManager) -> None:
|
||||
for attr in (
|
||||
"_user_pool_eviction_task",
|
||||
"_user_token_sweep_task",
|
||||
"_static_health_task",
|
||||
):
|
||||
task = getattr(m, attr)
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
await asyncio.gather(task, return_exceptions=True)
|
||||
setattr(m, attr, None)
|
||||
for state in m._static_servers.values():
|
||||
owner = state.owner_task
|
||||
if owner is not None and not owner.done():
|
||||
if state.close_requested is not None:
|
||||
state.close_requested.set()
|
||||
owner.cancel()
|
||||
await asyncio.gather(owner, return_exceptions=True)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
asyncio.run_coroutine_threadsafe(_drain(mgr), loop).result(timeout=5)
|
||||
loop.call_soon_threadsafe(loop.stop)
|
||||
thread.join(timeout=5)
|
||||
if not thread.is_alive():
|
||||
loop.close()
|
||||
|
||||
|
||||
def _run(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 5.0) -> Any:
|
||||
return asyncio.run_coroutine_threadsafe(coro, loop).result(timeout=timeout)
|
||||
|
||||
|
||||
def _make_session_mock() -> AsyncMock:
|
||||
"""A ClientSession-shaped mock good enough for connect + discovery."""
|
||||
session = AsyncMock()
|
||||
session.initialize = AsyncMock()
|
||||
session.get_server_capabilities = MagicMock(return_value=None)
|
||||
session.list_tools = AsyncMock(return_value=MagicMock(tools=[]))
|
||||
return session
|
||||
|
||||
|
||||
def _fake_transport_and_session(mgr_module_patches: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Build fake stdio transport + ClientSession cms, recording enter/exit."""
|
||||
events: list[str] = []
|
||||
session = _make_session_mock()
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_stdio_client(_params: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
yield (AsyncMock(), AsyncMock())
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
@asynccontextmanager
|
||||
async def fake_client_session_cm():
|
||||
events.append("session_enter")
|
||||
try:
|
||||
yield session
|
||||
finally:
|
||||
events.append("session_exit")
|
||||
|
||||
def fake_client_session(_read: Any, _write: Any, message_handler: Any = None):
|
||||
return fake_client_session_cm()
|
||||
|
||||
mgr_module_patches["stdio_client"] = fake_stdio_client
|
||||
mgr_module_patches["ClientSession"] = fake_client_session
|
||||
return {"events": events, "session": session}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Owner lifecycle
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestTransportOwnerLifecycle:
|
||||
def test_connect_installs_owner_and_teardown_closes_gracefully(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", patches["stdio_client"]),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
):
|
||||
_run(loop, mgr._connect_one_locked("srv", mgr._server_configs["srv"]))
|
||||
state = mgr._static_servers["srv"]
|
||||
assert state.session is fake["session"]
|
||||
owner = state.owner_task
|
||||
assert owner is not None and not owner.done()
|
||||
assert state.close_requested is not None
|
||||
assert fake["events"] == ["transport_enter", "session_enter"]
|
||||
|
||||
_run(loop, mgr._teardown_static_session("srv"))
|
||||
|
||||
# Graceful close: the parked owner exits via the event — no cancel —
|
||||
# and unwinds BOTH cms in-task, inner-out.
|
||||
assert owner.done() and not owner.cancelled()
|
||||
assert fake["events"] == [
|
||||
"transport_enter",
|
||||
"session_enter",
|
||||
"session_exit",
|
||||
"transport_exit",
|
||||
]
|
||||
assert state.session is None
|
||||
assert state.owner_task is None
|
||||
assert state.close_requested is None
|
||||
|
||||
def test_owner_death_evicts_session(self, running_loop_mgr) -> None:
|
||||
"""Trigger-A observer: the transport collapsing under a live session
|
||||
(owner task dies without a requested close) evicts the session so the
|
||||
health loop / next dispatch reconnects instead of probing a corpse."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", patches["stdio_client"]),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
):
|
||||
_run(loop, mgr._connect_one_locked("srv", mgr._server_configs["srv"]))
|
||||
state = mgr._static_servers["srv"]
|
||||
owner = state.owner_task
|
||||
assert owner is not None and state.session is fake["session"]
|
||||
|
||||
# Simulate the transport task group collapsing: the owner gets a
|
||||
# stray cancellation (exactly what anyio's scope delivery does).
|
||||
loop.call_soon_threadsafe(owner.cancel)
|
||||
deadline = time.monotonic() + 5
|
||||
while time.monotonic() < deadline and state.owner_task is not None:
|
||||
time.sleep(0.02)
|
||||
|
||||
assert owner.done()
|
||||
assert state.session is None # evicted by the done-callback
|
||||
assert state.owner_task is None
|
||||
# The cms were still unwound in-task despite the stray cancel.
|
||||
assert fake["events"][-2:] == ["session_exit", "transport_exit"]
|
||||
|
||||
def test_owner_death_during_discovery_fails_fast(self, running_loop_mgr) -> None:
|
||||
"""The static sibling of the pool's owner-death discovery race:
|
||||
discovery runs in the connecting caller while the transport is hosted
|
||||
by the owner, so a transport collapse mid-discovery cancels the OWNER
|
||||
and a bare await on the response stream would hang to the caller-side
|
||||
attempt timeout (~45s). ``_await_owner_discovery`` must convert it
|
||||
into a PROMPT ``ConnectionError`` and leave the state torn down."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
patches: dict[str, Any] = {}
|
||||
fake = _fake_transport_and_session(patches)
|
||||
|
||||
discovery_parked = asyncio.Event()
|
||||
|
||||
async def _parked_list_tools() -> Any:
|
||||
discovery_parked.set()
|
||||
await asyncio.sleep(3600) # the transport never answers
|
||||
|
||||
fake["session"].list_tools = AsyncMock(side_effect=_parked_list_tools)
|
||||
|
||||
async def _drive() -> tuple[float, BaseException | None]:
|
||||
async def _collapse_owner_when_parked() -> None:
|
||||
await discovery_parked.wait()
|
||||
owner = mgr._static_servers["srv"].owner_task
|
||||
assert owner is not None
|
||||
owner.cancel() # the transport task group collapsing
|
||||
|
||||
collapser = asyncio.create_task(_collapse_owner_when_parked())
|
||||
t0 = asyncio.get_running_loop().time()
|
||||
exc: BaseException | None = None
|
||||
try:
|
||||
await mgr._connect_one_locked("srv", mgr._server_configs["srv"])
|
||||
except Exception as e:
|
||||
# The expected ConnectionError; anything else (a cancel leak,
|
||||
# an interpreter exit) propagates and fails the test loudly.
|
||||
exc = e
|
||||
_ = await collapser # synchronization point; failures propagate
|
||||
return asyncio.get_running_loop().time() - t0, exc
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", patches["stdio_client"]),
|
||||
patch("turnstone.core.mcp_client.ClientSession", patches["ClientSession"]),
|
||||
):
|
||||
elapsed, exc = _run(loop, _drive(), timeout=15)
|
||||
|
||||
assert isinstance(exc, ConnectionError)
|
||||
assert "died during discovery" in str(exc)
|
||||
assert elapsed < 5.0 # prompt fail — not the attempt-timeout hang
|
||||
assert mgr._static_servers["srv"].session is None
|
||||
# The owner unwound its cms despite dying mid-discovery.
|
||||
assert fake["events"][-2:] == ["session_exit", "transport_exit"]
|
||||
|
||||
def test_base_exception_escape_resolves_waiter_and_propagates(self, running_loop_mgr) -> None:
|
||||
"""A BaseException-derived escape that is neither CancelledError nor
|
||||
Exception/group (a library control-flow escape; SystemExit and
|
||||
KeyboardInterrupt take the same path but additionally stop the loop —
|
||||
asyncio semantics, unobservable in-process) is NOT swallowed — it
|
||||
propagates from the owner task — but the waiter must still be resolved
|
||||
with a transport-failure error, or the connecting caller would block
|
||||
until its outer bound (and ``_connect_all``'s initial connect has
|
||||
none)."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
class _TransportLibraryEscape(BaseException):
|
||||
pass
|
||||
|
||||
@asynccontextmanager
|
||||
async def escaping_stdio_client(_params: Any):
|
||||
raise _TransportLibraryEscape("control-flow escape")
|
||||
yield # pragma: no cover
|
||||
|
||||
async def _drive() -> tuple[BaseException | None, BaseException | None]:
|
||||
ready: asyncio.Future[Any] = asyncio.get_running_loop().create_future()
|
||||
close_requested = asyncio.Event()
|
||||
owner = asyncio.create_task(
|
||||
mgr._static_transport_owner(
|
||||
"srv", mgr._server_configs["srv"], ready, close_requested
|
||||
)
|
||||
)
|
||||
waiter_exc: BaseException | None = None
|
||||
try:
|
||||
await ready
|
||||
except (Exception, _TransportLibraryEscape) as e:
|
||||
# Exception covers the expected ConnectionError; the escape
|
||||
# type covers the exact regression this test guards (the raw
|
||||
# escape leaking to the waiter instead of being converted).
|
||||
waiter_exc = e
|
||||
await asyncio.wait({owner}, timeout=5)
|
||||
owner_exc = owner.exception() if owner.done() and not owner.cancelled() else None
|
||||
return waiter_exc, owner_exc
|
||||
|
||||
with patch("turnstone.core.mcp_client.stdio_client", escaping_stdio_client):
|
||||
waiter_exc, owner_exc = _run(loop, _drive(), timeout=10)
|
||||
|
||||
assert isinstance(waiter_exc, ConnectionError) # waiter resolved, never hung
|
||||
assert isinstance(owner_exc, _TransportLibraryEscape) # propagated, unswallowed
|
||||
|
||||
def test_connect_failure_unwinds_owner_and_raises(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
@asynccontextmanager
|
||||
async def failing_stdio_client(_params: Any):
|
||||
raise ConnectionError("refused")
|
||||
yield # pragma: no cover
|
||||
|
||||
with (
|
||||
patch("turnstone.core.mcp_client.stdio_client", failing_stdio_client),
|
||||
pytest.raises(ConnectionError, match="refused"),
|
||||
):
|
||||
_run(loop, mgr._connect_one_locked("srv", mgr._server_configs["srv"]))
|
||||
|
||||
state = mgr._static_servers["srv"]
|
||||
assert state.session is None
|
||||
assert state.owner_task is None
|
||||
|
||||
async def _no_owner_tasks() -> int:
|
||||
return sum(
|
||||
1
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-transport-owner:") and not t.done()
|
||||
)
|
||||
|
||||
assert _run(loop, _no_owner_tasks()) == 0
|
||||
|
||||
def test_caller_cancel_mid_connect_does_not_abandon_cms(self, running_loop_mgr) -> None:
|
||||
"""Bug-1 core regression: cancelling the CONNECTING caller (attempt
|
||||
timeout, shutdown, sync boundary giving up) must close the owner via
|
||||
the one-cancel protocol — the transport cm still exits, in-task."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
events: list[str] = []
|
||||
entered = asyncio.Event()
|
||||
|
||||
@asynccontextmanager
|
||||
async def hanging_stdio_client(_params: Any):
|
||||
events.append("transport_enter")
|
||||
try:
|
||||
entered.set()
|
||||
await asyncio.sleep(3600) # server accepted, then stalled
|
||||
yield (AsyncMock(), AsyncMock())
|
||||
finally:
|
||||
events.append("transport_exit")
|
||||
|
||||
async def _drive() -> None:
|
||||
connect = asyncio.create_task(
|
||||
mgr._connect_one_locked("srv", mgr._server_configs["srv"])
|
||||
)
|
||||
await asyncio.wait_for(entered.wait(), timeout=5)
|
||||
connect.cancel() # the attempt-timeout / shutdown shape
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
_ = await connect # only the expected cancel is absorbed
|
||||
# The owner must be closed (one cancel) and fully unwound.
|
||||
deadline = asyncio.get_running_loop().time() + 5
|
||||
while asyncio.get_running_loop().time() < deadline:
|
||||
owners = [
|
||||
t
|
||||
for t in asyncio.all_tasks()
|
||||
if t.get_name().startswith("mcp-transport-owner:") and not t.done()
|
||||
]
|
||||
if not owners:
|
||||
return
|
||||
await asyncio.sleep(0.02)
|
||||
raise AssertionError("owner task still alive after caller cancel")
|
||||
|
||||
with patch("turnstone.core.mcp_client.stdio_client", hanging_stdio_client):
|
||||
_run(loop, _drive(), timeout=15)
|
||||
|
||||
assert events == ["transport_enter", "transport_exit"]
|
||||
assert mgr._static_servers["srv"].session is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bug 2: BaseExceptionGroup vs except Exception
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestBaseExceptionGroupHardening:
|
||||
def test_connect_all_survives_group_and_starts_loops(self, running_loop_mgr) -> None:
|
||||
"""A transport failure wrapped in BaseExceptionGroup (e.g. an
|
||||
accept-then-RST server collapsing the SDK task group with a stray
|
||||
CancelledError inside) must not kill ``_connect_all`` before the
|
||||
health/sweep loops are started — that silently disabled ALL
|
||||
autonomous recovery."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
# Pin the loop cadences: the assertions below require both loops to be
|
||||
# ENABLED, independent of whatever mcp config the environment carries.
|
||||
mgr._user_token_sweep_s = 240.0
|
||||
mgr._static_health_check_s = 30.0
|
||||
|
||||
async def _exploding_connect(name: str, _cfg: dict[str, Any]) -> None:
|
||||
raise BaseExceptionGroup("transport collapsed", [asyncio.CancelledError()])
|
||||
|
||||
with patch.object(mgr, "_connect_one", side_effect=_exploding_connect):
|
||||
_run(loop, mgr._connect_all())
|
||||
|
||||
assert mgr._connected.is_set()
|
||||
assert "srv" in mgr._last_error
|
||||
health = mgr._static_health_task
|
||||
sweep = mgr._user_token_sweep_task
|
||||
assert health is not None and not health.done()
|
||||
assert sweep is not None and not sweep.done()
|
||||
|
||||
def test_health_loop_survives_group(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
ticks: list[int] = []
|
||||
|
||||
async def _tick_then_group() -> float:
|
||||
ticks.append(1)
|
||||
if len(ticks) == 1:
|
||||
raise BaseExceptionGroup("boom", [asyncio.CancelledError()])
|
||||
return 3600.0
|
||||
|
||||
mgr._static_health_check_s = 0.05 # quick recovery sleep after the group
|
||||
with patch.object(mgr, "_static_health_tick", side_effect=_tick_then_group):
|
||||
|
||||
async def _drive() -> asyncio.Task[None]:
|
||||
task = asyncio.create_task(mgr._static_health_loop())
|
||||
deadline = asyncio.get_running_loop().time() + 5
|
||||
while asyncio.get_running_loop().time() < deadline and len(ticks) < 2:
|
||||
await asyncio.sleep(0.02)
|
||||
assert len(ticks) >= 2, "loop died on BaseExceptionGroup"
|
||||
assert not task.done()
|
||||
task.cancel()
|
||||
with contextlib.suppress(asyncio.CancelledError):
|
||||
_ = await task # only the expected cancel is absorbed
|
||||
return task
|
||||
|
||||
_run(loop, _drive(), timeout=10)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Orphaned-scope disarm backstop
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestScopeDisarmBackstop:
|
||||
def test_disarms_exactly_the_all_done_scope_on_this_loop(self, running_loop_mgr) -> None:
|
||||
"""One sweep over three armed scopes must touch EXACTLY the true
|
||||
orphan: the all-done-tasks scope hosted on the mcp-loop. The
|
||||
live-task scope (its task may still drain the scope) and the
|
||||
hostless scope (loop unknown — not ours to reach into) stay armed.
|
||||
Asserting ``disarmed == 1`` discriminates both failure directions:
|
||||
a no-op sweep and an over-eager one."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
|
||||
async def _arm_and_sweep() -> dict[str, Any]:
|
||||
from anyio._backends._asyncio import CancelScope
|
||||
|
||||
this_loop = asyncio.get_running_loop()
|
||||
|
||||
async def _noop() -> None:
|
||||
return None
|
||||
|
||||
blocker = asyncio.Event()
|
||||
|
||||
async def _parked() -> None:
|
||||
await blocker.wait()
|
||||
|
||||
done_task = asyncio.create_task(_noop())
|
||||
_ = await done_task # synchronization point; failures propagate
|
||||
live_task = asyncio.create_task(_parked())
|
||||
await asyncio.sleep(0)
|
||||
|
||||
orphan = CancelScope()
|
||||
orphan._host_task = done_task
|
||||
orphan._tasks.add(done_task)
|
||||
orphan._cancel_handle = this_loop.call_soon(lambda: None)
|
||||
|
||||
live_scope = CancelScope()
|
||||
live_scope._host_task = live_task
|
||||
live_scope._tasks.add(live_task)
|
||||
live_scope._cancel_handle = this_loop.call_soon(lambda: None)
|
||||
|
||||
hostless = CancelScope()
|
||||
hostless._tasks.add(done_task)
|
||||
hostless._cancel_handle = this_loop.call_soon(lambda: None)
|
||||
|
||||
mgr._last_scope_disarm = 0.0
|
||||
disarmed = mgr._maybe_disarm_orphaned_scopes("unit test")
|
||||
results = {
|
||||
"disarmed": disarmed,
|
||||
"orphan_handle_cleared": orphan._cancel_handle is None,
|
||||
"orphan_tasks_cleared": len(orphan._tasks) == 0,
|
||||
"live_still_armed": live_scope._cancel_handle is not None,
|
||||
"live_task_kept": live_task in live_scope._tasks,
|
||||
"hostless_still_armed": hostless._cancel_handle is not None,
|
||||
"rate_limited_second": mgr._maybe_disarm_orphaned_scopes("again"),
|
||||
}
|
||||
for scope in (live_scope, hostless):
|
||||
if scope._cancel_handle is not None:
|
||||
scope._cancel_handle.cancel()
|
||||
scope._cancel_handle = None
|
||||
scope._tasks.clear()
|
||||
blocker.set()
|
||||
_ = await live_task # synchronization point; failures propagate
|
||||
return results
|
||||
|
||||
r = _run(loop, _arm_and_sweep())
|
||||
assert r["disarmed"] == 1
|
||||
assert r["orphan_handle_cleared"] and r["orphan_tasks_cleared"]
|
||||
assert r["live_still_armed"] and r["live_task_kept"]
|
||||
assert r["hostless_still_armed"]
|
||||
assert r["rate_limited_second"] == 0
|
||||
+448
-15
@@ -15,13 +15,13 @@ from __future__ import annotations
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from contextlib import AsyncExitStack
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -114,12 +114,31 @@ def running_loop_mgr():
|
||||
# handlers don't fire after pytest has torn its handlers down. Mirrors
|
||||
# the production ``shutdown()`` shape.
|
||||
async def _drain(m: MCPClientManager) -> None:
|
||||
task = m._user_pool_eviction_task
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
with contextlib.suppress(BaseException):
|
||||
await task
|
||||
m._user_pool_eviction_task = None
|
||||
# ``_static_health_task`` included: since the BaseExceptionGroup
|
||||
# hardening, ``_connect_all`` reliably starts (and keeps alive) the
|
||||
# health loop even when every configured connect fails — a test
|
||||
# that drives ``_connect_all`` must drain it like production
|
||||
# ``shutdown()`` does, or the task is destroyed pending at GC.
|
||||
for attr in (
|
||||
"_user_pool_eviction_task",
|
||||
"_user_token_sweep_task",
|
||||
"_static_health_task",
|
||||
):
|
||||
task = getattr(m, attr)
|
||||
if task is not None:
|
||||
task.cancel()
|
||||
await asyncio.gather(task, return_exceptions=True)
|
||||
setattr(m, attr, None)
|
||||
# Close any parked pool transport owners a successful
|
||||
# ``_connect_one_pool`` left installed, mirroring production
|
||||
# ``shutdown()`` — an undrained owner is destroyed pending at GC.
|
||||
for entry in list(m._user_pool_entries.values()):
|
||||
owner = entry.owner_task
|
||||
if owner is not None and not owner.done():
|
||||
if entry.close_requested is not None:
|
||||
entry.close_requested.set()
|
||||
owner.cancel()
|
||||
await asyncio.gather(owner, return_exceptions=True)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
asyncio.run_coroutine_threadsafe(_drain(mgr), loop).result(timeout=2)
|
||||
@@ -341,25 +360,42 @@ class TestEviction:
|
||||
assert ("u4", "pool-srv") in mgr._user_pool_entries
|
||||
assert ("u3", "pool-srv") in mgr._user_pool_entries
|
||||
|
||||
def test_eviction_resilient_to_close_errors(self, running_loop_mgr) -> None:
|
||||
def test_eviction_resilient_to_owner_unwind_errors(self, running_loop_mgr) -> None:
|
||||
"""Owner-model successor to the old ``resilient_to_close_errors`` test.
|
||||
|
||||
Teardown reaps the entry's owner through a bounded ``asyncio.wait`` that
|
||||
never re-raises, so even an owner whose in-task unwind raises cannot
|
||||
break eviction. The old failure mode this guarded — a cross-task
|
||||
``stack.aclose()`` raising ``RuntimeError('...different task...')`` — is
|
||||
structurally impossible now: the transport cms live in, and unwind in,
|
||||
the owner task, never the evictor.
|
||||
"""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._user_pool_idle_ttl_s = 0.0
|
||||
|
||||
broken_stack = MagicMock(spec=AsyncExitStack)
|
||||
broken_stack.aclose = AsyncMock(side_effect=RuntimeError("close failed"))
|
||||
|
||||
async def _seed() -> None:
|
||||
for i in range(2):
|
||||
entry = await mgr._ensure_pool_entry((f"u{i}", "pool-srv"))
|
||||
key = (f"u{i}", "pool-srv")
|
||||
entry = await mgr._ensure_pool_entry(key)
|
||||
event = asyncio.Event()
|
||||
|
||||
async def _owner(ev: asyncio.Event = event) -> None:
|
||||
await ev.wait()
|
||||
raise RuntimeError("unwind failed")
|
||||
|
||||
owner = asyncio.create_task(_owner(), name=f"mcp-pool-owner-test:{i}")
|
||||
# Retrieve the exception so the raising owner doesn't warn at GC.
|
||||
owner.add_done_callback(lambda t: None if t.cancelled() else t.exception())
|
||||
entry.session = MagicMock()
|
||||
entry.stack = broken_stack
|
||||
entry.owner_task = owner
|
||||
entry.close_requested = event
|
||||
|
||||
_run_on_loop(loop, _seed())
|
||||
|
||||
async def _evict() -> None:
|
||||
await mgr._evict_idle_pool_entries()
|
||||
|
||||
# Eviction must not raise even if close fails.
|
||||
# Eviction must not raise even if the owner's unwind raises.
|
||||
_run_on_loop(loop, _evict())
|
||||
# All entries removed from the dict regardless.
|
||||
assert mgr._user_pool_entries == {}
|
||||
@@ -969,3 +1005,400 @@ class TestUserIdThreadThrough:
|
||||
assert result == "static-output"
|
||||
# No pool entries were created.
|
||||
assert mgr._user_pool_entries == {}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Background token-freshness sweep (oauth_user keep-hot, no connection warming)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestUserTokenFreshnessSweep:
|
||||
"""The background sweep that keeps every consented ``oauth_user`` grant hot
|
||||
for unattended / autonomous work: refresh-on-expiry via the canonical path,
|
||||
proactive dead-grant badging, once-only surfacing, and — the load-bearing
|
||||
property — total invisibility to static / no-auth deployments."""
|
||||
|
||||
def _wire(self, mgr: MCPClientManager, storage: SQLiteBackend, cipher: Any) -> None:
|
||||
mgr.set_storage(storage)
|
||||
mgr.set_app_state(_make_app_state(storage, cipher=cipher))
|
||||
mgr._oauth_user_server_names = {"pool-srv"}
|
||||
|
||||
@staticmethod
|
||||
def _classified(kind: str, token: str | None = None):
|
||||
async def _fake(**kwargs: Any) -> Any:
|
||||
return SimpleNamespace(kind=kind, token=token)
|
||||
|
||||
return _fake
|
||||
|
||||
# -- no-auth / static safety: the sweep must be structurally invisible ----
|
||||
|
||||
def test_sweep_noop_without_oauth_servers(self, running_loop_mgr, storage) -> None:
|
||||
"""A static-only / no-auth deployment: the OBO gate returns before any
|
||||
DB scan or AS round-trip — the single most important property."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._oauth_user_server_names = set() # no oauth_user server configured
|
||||
storage.list_mcp_user_token_reconcile_targets = MagicMock(return_value=[]) # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=AsyncMock(),
|
||||
) as classified:
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
storage.list_mcp_user_token_reconcile_targets.assert_not_called() # no token-table scan
|
||||
classified.assert_not_awaited() # no AS round-trip
|
||||
|
||||
def test_sweep_noop_before_storage_wired(self, running_loop_mgr) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._oauth_user_server_names = {"pool-srv"} # oauth configured but app not wired yet
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=AsyncMock(),
|
||||
) as classified:
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
classified.assert_not_awaited()
|
||||
|
||||
def test_sweep_skips_server_not_in_oauth_set(self, running_loop_mgr, storage) -> None:
|
||||
"""A token row lingering for a since-demoted / renamed server is not
|
||||
reconciled — only pairs whose server is currently ``oauth_user``."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="ghost-srv")
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=AsyncMock(),
|
||||
) as classified:
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
classified.assert_not_awaited() # ghost-srv is not in _oauth_user_server_names
|
||||
|
||||
# -- classification branches --------------------------------------------
|
||||
|
||||
def test_healthy_token_no_badge(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("token", token="access-aaa"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
storage.upsert_mcp_pending_consent.assert_not_called()
|
||||
assert ("u1", "pool-srv") not in mgr._token_sweep_warned
|
||||
|
||||
def test_dead_grant_badges_once_and_dedups(self, running_loop_mgr, storage, caplog) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed"),
|
||||
),
|
||||
caplog.at_level(logging.WARNING, logger="turnstone.core.mcp_client"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness()) # second tick: no re-badge
|
||||
|
||||
# Badge raised exactly once, proactively, with the dashboard's code.
|
||||
storage.upsert_mcp_pending_consent.assert_called_once()
|
||||
assert (
|
||||
storage.upsert_mcp_pending_consent.call_args.kwargs["error_code"]
|
||||
== "mcp_consent_required"
|
||||
)
|
||||
assert ("u1", "pool-srv") in mgr._token_sweep_warned
|
||||
escalations = [r for r in caplog.records if "needs re-consent" in r.getMessage()]
|
||||
assert len(escalations) == 1 # logged loud-once, not every tick
|
||||
|
||||
def test_decrypt_failure_warns_but_does_not_badge(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("decrypt_failure"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
# Operator-actionable (key unknown) — surfaced in the warned set, but NOT
|
||||
# a user-consent badge (outside the dashboard's scope).
|
||||
storage.upsert_mcp_pending_consent.assert_not_called()
|
||||
assert ("u1", "pool-srv") in mgr._token_sweep_warned
|
||||
|
||||
def test_transient_failure_is_silent(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock() # type: ignore[method-assign]
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed_transient"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
storage.upsert_mcp_pending_consent.assert_not_called()
|
||||
assert ("u1", "pool-srv") not in mgr._token_sweep_warned # retryable, not surfaced
|
||||
|
||||
def test_recovery_rearms_and_clears_badge(self, running_loop_mgr, storage) -> None:
|
||||
"""A dead grant that later returns healthy clears its warned pin AND drops
|
||||
the stale badge — the self-heal for a spurious invalid_grant that has
|
||||
since recovered. Production-reachable now that the observe-only sweep no
|
||||
longer deletes the row on refresh_failed, so the pair keeps enumerating."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.delete_mcp_pending_consent = MagicMock(return_value=True) # type: ignore[method-assign]
|
||||
key = ("u1", "pool-srv")
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert key in mgr._token_sweep_warned
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("token", token="access-aaa"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert key not in mgr._token_sweep_warned # recovered → re-armed
|
||||
storage.delete_mcp_pending_consent.assert_called_once_with("u1", "pool-srv")
|
||||
|
||||
def test_dead_grant_not_pinned_when_badge_persist_fails(
|
||||
self, running_loop_mgr, storage
|
||||
) -> None:
|
||||
"""If the badge write fails, the pair is NOT pinned, so the next tick
|
||||
retries — a single failed persist must not permanently lose the only
|
||||
proactive signal for a sweep-detected dead grant."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
storage.upsert_mcp_pending_consent = MagicMock( # type: ignore[method-assign]
|
||||
side_effect=RuntimeError("db down")
|
||||
)
|
||||
key = ("u1", "pool-srv")
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("refresh_failed"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert key not in mgr._token_sweep_warned # not pinned — will retry
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
# Retried on the second tick rather than deduped away by a phantom pin.
|
||||
assert storage.upsert_mcp_pending_consent.call_count == 2
|
||||
|
||||
def test_sweep_uses_non_revoking_observe_mode(self, running_loop_mgr, storage) -> None:
|
||||
"""The background sweep MUST call the canonical lookup non-destructively:
|
||||
a timer may never delete a token or move a foreground user's revoke
|
||||
threshold."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
seen_kwargs: list[dict[str, Any]] = []
|
||||
|
||||
async def _spy(**kwargs: Any) -> Any:
|
||||
seen_kwargs.append(kwargs)
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_spy):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert seen_kwargs and seen_kwargs[0]["revoke_on_failure"] is False
|
||||
assert seen_kwargs[0]["revoke_ambiguous_escalation"] is False
|
||||
|
||||
# -- keepalive refresh (exercise the refresh token before it idles out) ---
|
||||
|
||||
def test_keepalive_refresh_due_logic(self) -> None:
|
||||
mgr = MCPClientManager({})
|
||||
mgr._user_token_refresh_keepalive_s = 3600.0
|
||||
old = (datetime.now(UTC) - timedelta(hours=2)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
recent = (datetime.now(UTC) - timedelta(minutes=1)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
assert mgr._keepalive_refresh_due(old) is True # past the window → force
|
||||
assert mgr._keepalive_refresh_due(recent) is False # still warm
|
||||
assert mgr._keepalive_refresh_due(None) is True # unknown → force once, safe
|
||||
assert mgr._keepalive_refresh_due("not-a-date") is True # unparseable → force
|
||||
mgr._user_token_refresh_keepalive_s = 0.0
|
||||
assert mgr._keepalive_refresh_due(old) is False # disabled → never force
|
||||
|
||||
def test_keepalive_due_forces_refresh(self, running_loop_mgr, storage) -> None:
|
||||
"""A grant whose refresh token has idled past the window is force-refreshed
|
||||
even though its access token may be fresh — the [6] fix: keep the refresh
|
||||
token alive so an unattended run never finds it aged out."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._user_token_refresh_keepalive_s = 1800.0
|
||||
stale = (datetime.now(UTC) - timedelta(hours=2)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
storage.list_mcp_user_token_reconcile_targets = MagicMock( # type: ignore[method-assign]
|
||||
return_value=[("u1", "pool-srv", stale)]
|
||||
)
|
||||
seen_kwargs: list[dict[str, Any]] = []
|
||||
|
||||
async def _spy(**kwargs: Any) -> Any:
|
||||
seen_kwargs.append(kwargs)
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_spy):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert seen_kwargs and seen_kwargs[0]["force_refresh"] is True
|
||||
|
||||
def test_keepalive_not_due_does_not_force(self, running_loop_mgr, storage) -> None:
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._user_token_refresh_keepalive_s = 1800.0
|
||||
recent = (datetime.now(UTC) - timedelta(minutes=1)).strftime("%Y-%m-%dT%H:%M:%S")
|
||||
storage.list_mcp_user_token_reconcile_targets = MagicMock( # type: ignore[method-assign]
|
||||
return_value=[("u1", "pool-srv", recent)]
|
||||
)
|
||||
seen_kwargs: list[dict[str, Any]] = []
|
||||
|
||||
async def _spy(**kwargs: Any) -> Any:
|
||||
seen_kwargs.append(kwargs)
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_spy):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert seen_kwargs and seen_kwargs[0]["force_refresh"] is False # still warm
|
||||
|
||||
def test_warned_set_pruned_to_consented_pairs(self, running_loop_mgr, storage) -> None:
|
||||
"""A warned pair that is no longer consented (row gone) is dropped from
|
||||
the dedup set so it can't grow unbounded across transient dead grants."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
_seed_user_token(storage, cipher, user_id="u1", server_name="pool-srv")
|
||||
mgr._token_sweep_warned = {("gone-user", "pool-srv"), ("u1", "pool-srv")}
|
||||
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.get_user_access_token_classified",
|
||||
new=self._classified("token", token="access-aaa"),
|
||||
):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert ("gone-user", "pool-srv") not in mgr._token_sweep_warned # pruned
|
||||
assert ("u1", "pool-srv") not in mgr._token_sweep_warned # healthy → cleared
|
||||
|
||||
def test_per_pair_failure_isolated(self, running_loop_mgr, storage) -> None:
|
||||
"""One pair raising must not starve the rest of the pass."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
cipher = make_mcp_token_cipher()
|
||||
self._wire(mgr, storage, cipher)
|
||||
mgr._oauth_user_server_names = {"pool-srv"}
|
||||
_seed_user_token(storage, cipher, user_id="u-bad", server_name="pool-srv")
|
||||
_seed_user_token(storage, cipher, user_id="u-ok", server_name="pool-srv")
|
||||
seen: list[str] = []
|
||||
|
||||
async def _flaky(**kwargs: Any) -> Any:
|
||||
uid = kwargs["user_id"]
|
||||
seen.append(uid)
|
||||
if uid == "u-bad":
|
||||
raise RuntimeError("boom")
|
||||
return SimpleNamespace(kind="token", token="access-aaa")
|
||||
|
||||
with patch("turnstone.core.mcp_client.get_user_access_token_classified", new=_flaky):
|
||||
_run_on_loop(loop, mgr._sweep_user_token_freshness())
|
||||
assert {"u-bad", "u-ok"} <= set(seen) # both attempted despite one raising
|
||||
|
||||
def test_sweep_loop_cancel_returns_cleanly(self, running_loop_mgr) -> None:
|
||||
"""The loop body exits on cancellation without raising (mirrors the
|
||||
eviction loop's teardown contract)."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._user_token_sweep_s = 999.0 # park in the sleep
|
||||
|
||||
async def _spawn() -> asyncio.Task[None]:
|
||||
return asyncio.ensure_future(mgr._user_token_sweep_loop())
|
||||
|
||||
task = _run_on_loop(loop, _spawn())
|
||||
|
||||
async def _cancel() -> None:
|
||||
task.cancel()
|
||||
with contextlib.suppress(BaseException):
|
||||
await task
|
||||
|
||||
_run_on_loop(loop, _cancel())
|
||||
assert task.cancelled() or task.done()
|
||||
|
||||
def test_connect_all_starts_the_sweep_task(self, running_loop_mgr) -> None:
|
||||
"""Wiring guard: ``_connect_all`` must start the sweep once, even with no
|
||||
servers configured — otherwise the whole keep-hot mechanism is dead code."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
assert mgr._user_token_sweep_task is None
|
||||
|
||||
_run_on_loop(loop, mgr._connect_all())
|
||||
try:
|
||||
task = mgr._user_token_sweep_task
|
||||
assert task is not None and not task.done() # live, single instance
|
||||
finally:
|
||||
|
||||
async def _drain() -> None:
|
||||
t = mgr._user_token_sweep_task
|
||||
if t is not None:
|
||||
t.cancel()
|
||||
with contextlib.suppress(BaseException):
|
||||
await t
|
||||
mgr._user_token_sweep_task = None
|
||||
|
||||
_run_on_loop(loop, _drain())
|
||||
|
||||
def test_disabled_sweep_not_started_by_connect_all(self, running_loop_mgr) -> None:
|
||||
"""Cadence <= 0 disables the sweep entirely — no task is spawned."""
|
||||
mgr, loop, _ = running_loop_mgr
|
||||
mgr._user_token_sweep_s = 0.0
|
||||
_run_on_loop(loop, mgr._connect_all())
|
||||
assert mgr._user_token_sweep_task is None
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("configured", "expected"),
|
||||
[
|
||||
(0, 0.0), # explicit disable
|
||||
(-5, 0.0), # negative disables (no busy-loop)
|
||||
(1, 30.0), # tiny positive floored to _MIN_USER_TOKEN_SWEEP_S
|
||||
(600, 600.0), # normal value passes through
|
||||
],
|
||||
)
|
||||
def test_cadence_clamped_or_disabled(self, configured, expected) -> None:
|
||||
"""The config cadence is floored (positive) or disabled (<= 0) so an
|
||||
``asyncio.sleep(0)`` busy-loop is unreachable."""
|
||||
with patch(
|
||||
"turnstone.core.mcp_client.load_config",
|
||||
return_value={"user_token_sweep_seconds": configured},
|
||||
):
|
||||
mgr = MCPClientManager({})
|
||||
assert mgr._user_token_sweep_s == expected
|
||||
|
||||
# -- storage enumerator --------------------------------------------------
|
||||
|
||||
def test_reconcile_targets_pairs_expiry_unfiltered_with_last_exercised(self, storage) -> None:
|
||||
cipher = make_mcp_token_cipher()
|
||||
# alice consents to two servers → two rows.
|
||||
_seed_user_token(storage, cipher, user_id="alice", server_name="srv-a")
|
||||
_seed_user_token(storage, cipher, user_id="alice", server_name="srv-b")
|
||||
# bob's access token is expired but the refresh token is live — still a
|
||||
# consented, reconcilable grant, so bob must be enumerated.
|
||||
_seed_user_token(
|
||||
storage, cipher, user_id="bob", server_name="srv-a", expires_in_seconds=-999
|
||||
)
|
||||
targets = storage.list_mcp_user_token_reconcile_targets()
|
||||
# (user, server) identity, all three grants present regardless of expiry.
|
||||
assert sorted((u, s) for u, s, _ in targets) == [
|
||||
("alice", "srv-a"),
|
||||
("alice", "srv-b"),
|
||||
("bob", "srv-a"),
|
||||
]
|
||||
# last_exercised = COALESCE(last_refreshed, created); never-refreshed rows
|
||||
# fall back to created, so it is always populated (drives the keepalive).
|
||||
assert all(last_exercised for _, _, last_exercised in targets)
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Tests for alembic migration 065 (capture Entra oid/tid on oidc_identities).
|
||||
|
||||
Drives ``command.upgrade``/``downgrade`` against an isolated SQLite database per
|
||||
test (the 060/062/063 harness pattern), then asserts:
|
||||
|
||||
* upgrade adds the ``oid``/``tid`` columns and the ``idx_oidc_identities_oid``
|
||||
index;
|
||||
* a pre-065 row migrates cleanly, gaining ``""`` for the new columns;
|
||||
* downgrade removes the columns + index, returning ``oidc_identities`` to its
|
||||
exact pre-065 shape — this pins the **clean-rollback** guarantee (the change
|
||||
can be backed out with no orphaned state if the upstream PR is rejected).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import command
|
||||
from alembic.config import Config
|
||||
|
||||
_MIGRATIONS_DIR = str(
|
||||
Path(__file__).resolve().parent.parent / "turnstone" / "core" / "storage" / "migrations"
|
||||
)
|
||||
|
||||
|
||||
def _alembic_cfg(db_path: Path) -> Config:
|
||||
cfg = Config()
|
||||
cfg.set_main_option("script_location", _MIGRATIONS_DIR)
|
||||
cfg.set_main_option("sqlalchemy.url", f"sqlite:///{db_path}")
|
||||
return cfg
|
||||
|
||||
|
||||
class TestMigration065:
|
||||
def test_upgrade_adds_oid_tid_and_index(self, tmp_path: Path) -> None:
|
||||
db_path = tmp_path / "065-up.db"
|
||||
command.upgrade(_alembic_cfg(db_path), "065")
|
||||
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
insp = sa.inspect(engine)
|
||||
cols = {c["name"] for c in insp.get_columns("oidc_identities")}
|
||||
assert {"oid", "tid"} <= cols
|
||||
idx = {i["name"] for i in insp.get_indexes("oidc_identities")}
|
||||
assert "idx_oidc_identities_oid" in idx
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
def test_preexisting_row_migrates_with_empty_default(self, tmp_path: Path) -> None:
|
||||
db_path = tmp_path / "065-default.db"
|
||||
cfg = _alembic_cfg(db_path)
|
||||
# Stop at 064, insert a pre-065 identity, THEN upgrade to 065.
|
||||
command.upgrade(cfg, "064")
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
with engine.begin() as conn:
|
||||
conn.execute(
|
||||
sa.text(
|
||||
"INSERT INTO oidc_identities "
|
||||
"(issuer, subject, user_id, email, created, last_login) "
|
||||
"VALUES ('iss', 'sub', 'u1', '', "
|
||||
"'2026-01-01T00:00:00', '2026-01-01T00:00:00')"
|
||||
)
|
||||
)
|
||||
command.upgrade(cfg, "065")
|
||||
with engine.connect() as conn:
|
||||
row = conn.execute(
|
||||
sa.text("SELECT oid, tid FROM oidc_identities WHERE subject = 'sub'")
|
||||
).fetchone()
|
||||
assert row is not None
|
||||
assert row[0] == "" and row[1] == ""
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
def test_downgrade_removes_oid_tid_and_index(self, tmp_path: Path) -> None:
|
||||
db_path = tmp_path / "065-down.db"
|
||||
cfg = _alembic_cfg(db_path)
|
||||
command.upgrade(cfg, "065")
|
||||
command.downgrade(cfg, "064")
|
||||
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
insp = sa.inspect(engine)
|
||||
cols = {c["name"] for c in insp.get_columns("oidc_identities")}
|
||||
assert "oid" not in cols and "tid" not in cols
|
||||
idx = {i["name"] for i in insp.get_indexes("oidc_identities")}
|
||||
assert "idx_oidc_identities_oid" not in idx
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
def test_downgrade_then_upgrade_round_trip(self, tmp_path: Path) -> None:
|
||||
"""up -> down -> up must land cleanly (no leftover column/index conflict)."""
|
||||
db_path = tmp_path / "065-roundtrip.db"
|
||||
cfg = _alembic_cfg(db_path)
|
||||
command.upgrade(cfg, "065")
|
||||
command.downgrade(cfg, "064")
|
||||
command.upgrade(cfg, "065")
|
||||
|
||||
engine = sa.create_engine(f"sqlite:///{db_path}")
|
||||
try:
|
||||
cols = {c["name"] for c in sa.inspect(engine).get_columns("oidc_identities")}
|
||||
assert {"oid", "tid"} <= cols
|
||||
finally:
|
||||
engine.dispose()
|
||||
@@ -167,6 +167,43 @@ class TestModelRegistry:
|
||||
with pytest.raises(ValueError, match="Unknown model alias"):
|
||||
reg.get_client("nonexistent")
|
||||
|
||||
def test_client_construction_failure_is_value_error(self) -> None:
|
||||
# Environment failures inside SDK construction (e.g. httpx raising
|
||||
# FileNotFoundError for a CA bundle deleted by a venv rebuild) must
|
||||
# surface as ValueError so routes answer 503-with-message instead
|
||||
# of an opaque 500.
|
||||
reg = self._make_registry()
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.model_registry.create_client",
|
||||
side_effect=FileNotFoundError(2, "No such file", "/gone/cacert.pem"),
|
||||
),
|
||||
pytest.raises(ValueError, match="'default'.*FileNotFoundError") as excinfo,
|
||||
):
|
||||
reg.get_client("default")
|
||||
assert isinstance(excinfo.value.__cause__, FileNotFoundError)
|
||||
# The message is echoed in 503 bodies: exception TYPE only — the
|
||||
# raw exception text can embed filesystem paths and must stay in
|
||||
# the server log.
|
||||
assert "/gone/cacert.pem" not in str(excinfo.value)
|
||||
assert "No such file" not in str(excinfo.value)
|
||||
# Nothing half-constructed may be cached — a later call with a
|
||||
# repaired environment must construct for real.
|
||||
assert "default" not in reg._clients
|
||||
|
||||
def test_client_construction_value_error_passes_through(self) -> None:
|
||||
# create_client's own misconfig ValueErrors already carry
|
||||
# remediation text and must not be double-wrapped.
|
||||
reg = self._make_registry()
|
||||
with (
|
||||
patch(
|
||||
"turnstone.core.model_registry.create_client",
|
||||
side_effect=ValueError("anthropic-compatible requires base_url"),
|
||||
),
|
||||
pytest.raises(ValueError, match="^anthropic-compatible requires base_url$"),
|
||||
):
|
||||
reg.get_client("default")
|
||||
|
||||
def test_shutdown(self) -> None:
|
||||
reg = self._make_registry()
|
||||
reg.get_client("default")
|
||||
|
||||
@@ -15,6 +15,7 @@ import pytest
|
||||
|
||||
from turnstone.core.oauth_ssrf import (
|
||||
OAuthSSRFError,
|
||||
OAuthSSRFPrivateAddressError,
|
||||
effective_port,
|
||||
is_localhost,
|
||||
validate_discovered_endpoint,
|
||||
@@ -86,6 +87,55 @@ class TestValidateUrlNoSSRF:
|
||||
):
|
||||
validate_url_no_ssrf("https://corp.example.com", allow_http=False)
|
||||
|
||||
def test_private_address_raises_distinct_subclass(self) -> None:
|
||||
# Callers with an operator opt-in (OIDC) catch the subclass to
|
||||
# append the remediation hint; plain OAuthSSRFError catches still work.
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=self._PRIVATE_ADDR),
|
||||
pytest.raises(OAuthSSRFPrivateAddressError),
|
||||
):
|
||||
validate_url_no_ssrf("https://corp.example.com", allow_http=False)
|
||||
|
||||
def test_allow_private_accepts_rfc1918(self) -> None:
|
||||
with patch("socket.getaddrinfo", return_value=self._PRIVATE_ADDR):
|
||||
parsed = validate_url_no_ssrf(
|
||||
"https://auth.corp.example.com", allow_http=False, allow_private=True
|
||||
)
|
||||
assert parsed.hostname == "auth.corp.example.com"
|
||||
|
||||
def test_allow_private_accepts_cgnat(self) -> None:
|
||||
# 100.64/10 (RFC 6598, shared address space) — e.g. a tailnet-hosted IdP.
|
||||
with patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("100.64.0.7", 0))]):
|
||||
validate_url_no_ssrf("https://idp.tail.example", allow_http=False, allow_private=True)
|
||||
|
||||
def test_allow_private_accepts_loopback_hostname(self) -> None:
|
||||
# A non-localhost hostname resolving to loopback (IdP behind a
|
||||
# local reverse proxy) is operator-trusted under the opt-in.
|
||||
with patch("socket.getaddrinfo", return_value=self._LOOPBACK_ADDR):
|
||||
validate_url_no_ssrf("https://auth.internal", allow_http=False, allow_private=True)
|
||||
|
||||
def test_allow_private_still_rejects_link_local(self) -> None:
|
||||
# Cloud metadata services live on link-local; no legitimate IdP does.
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("169.254.169.254", 0))]),
|
||||
pytest.raises(OAuthSSRFError, match="refused even with private"),
|
||||
):
|
||||
validate_url_no_ssrf("https://md.example.com", allow_http=False, allow_private=True)
|
||||
|
||||
def test_allow_private_still_rejects_unspecified(self) -> None:
|
||||
# The message names the class so 0.0.0.0/:: rejections are unambiguous.
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("0.0.0.0", 0))]),
|
||||
pytest.raises(OAuthSSRFError, match="unspecified"),
|
||||
):
|
||||
validate_url_no_ssrf("https://zero.example.com", allow_http=False, allow_private=True)
|
||||
|
||||
def test_allow_private_does_not_relax_https(self) -> None:
|
||||
with pytest.raises(OAuthSSRFError, match="must use HTTPS"):
|
||||
validate_url_no_ssrf(
|
||||
"http://auth.corp.example.com", allow_http=False, allow_private=True
|
||||
)
|
||||
|
||||
def test_rejects_unresolvable(self) -> None:
|
||||
import socket
|
||||
|
||||
@@ -122,6 +172,19 @@ class TestValidateDiscoveredEndpoint:
|
||||
trusted_endpoint_hosts=frozenset(),
|
||||
)
|
||||
|
||||
def test_allow_private_passes_through(self) -> None:
|
||||
# Same-origin endpoint on a private-resolving issuer host is accepted
|
||||
# when the operator opted in.
|
||||
issuer = urllib.parse.urlparse("https://auth.corp.example.com")
|
||||
with patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("10.0.0.5", 0))]):
|
||||
validate_discovered_endpoint(
|
||||
"https://auth.corp.example.com/token",
|
||||
issuer,
|
||||
allow_http=False,
|
||||
trusted_endpoint_hosts=frozenset(),
|
||||
allow_private=True,
|
||||
)
|
||||
|
||||
def test_trusted_endpoint_host_passes(self) -> None:
|
||||
issuer = urllib.parse.urlparse("https://idp.example.com")
|
||||
with patch("socket.getaddrinfo", return_value=self._PUBLIC_ADDR):
|
||||
|
||||
@@ -90,6 +90,42 @@ class TestLoadOIDCConfig:
|
||||
assert cfg.scopes == "openid"
|
||||
assert cfg.provider_name == "Okta"
|
||||
|
||||
def test_load_oidc_config_allow_private_network_env(self, monkeypatch):
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_ISSUER", "https://auth.internal.example")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_CLIENT_ID", "cid")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_CLIENT_SECRET", "csecret")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK", "true")
|
||||
|
||||
with patch("turnstone.core.config.load_config", return_value={}):
|
||||
cfg = load_oidc_config()
|
||||
|
||||
assert cfg.allow_private_network is True
|
||||
|
||||
def test_load_oidc_config_allow_private_network_toml(self, monkeypatch):
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_ISSUER", "https://auth.internal.example")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_CLIENT_ID", "cid")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_CLIENT_SECRET", "csecret")
|
||||
monkeypatch.delenv("TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK", raising=False)
|
||||
|
||||
with patch(
|
||||
"turnstone.core.config.load_config",
|
||||
return_value={"allow_private_network": True},
|
||||
):
|
||||
cfg = load_oidc_config()
|
||||
|
||||
assert cfg.allow_private_network is True
|
||||
|
||||
def test_load_oidc_config_allow_private_network_default_off(self, monkeypatch):
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_ISSUER", "https://auth.example.com")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_CLIENT_ID", "cid")
|
||||
monkeypatch.setenv("TURNSTONE_OIDC_CLIENT_SECRET", "csecret")
|
||||
monkeypatch.delenv("TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK", raising=False)
|
||||
|
||||
with patch("turnstone.core.config.load_config", return_value={}):
|
||||
cfg = load_oidc_config()
|
||||
|
||||
assert cfg.allow_private_network is False
|
||||
|
||||
def test_load_oidc_config_disabled_when_missing(self, monkeypatch):
|
||||
monkeypatch.delenv("TURNSTONE_OIDC_ISSUER", raising=False)
|
||||
monkeypatch.delenv("TURNSTONE_OIDC_CLIENT_ID", raising=False)
|
||||
@@ -345,6 +381,27 @@ class TestValidateIssuerURL:
|
||||
):
|
||||
validate_issuer_url("https://idp.example.com")
|
||||
|
||||
def test_private_address_hint_mentions_opt_in(self):
|
||||
"""The rejection message points the operator at allow_private_network."""
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("10.0.0.5", 0))]),
|
||||
pytest.raises(OIDCError, match="allow_private_network"),
|
||||
):
|
||||
validate_issuer_url("https://auth.internal.example")
|
||||
|
||||
def test_allow_private_accepts_private_issuer(self):
|
||||
"""The opt-in accepts an issuer resolving to RFC 1918 space."""
|
||||
with patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("10.0.0.5", 0))]):
|
||||
validate_issuer_url("https://auth.internal.example", allow_private=True)
|
||||
|
||||
def test_allow_private_still_rejects_link_local(self):
|
||||
"""Link-local (cloud metadata) is refused even with the opt-in."""
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("169.254.169.254", 0))]),
|
||||
pytest.raises(OIDCError, match="refused even with private"),
|
||||
):
|
||||
validate_issuer_url("https://md.internal.example", allow_private=True)
|
||||
|
||||
def test_rejects_http_non_localhost(self):
|
||||
"""HTTP is rejected for non-localhost hosts."""
|
||||
with pytest.raises(OIDCError, match="must use HTTPS"):
|
||||
@@ -508,6 +565,20 @@ class TestValidateDiscoveredEndpoint:
|
||||
trusted_endpoint_hosts=frozenset(),
|
||||
)
|
||||
|
||||
def test_private_endpoint_hint_mentions_opt_in(self):
|
||||
"""A discovered endpoint resolving private carries the opt-in hint
|
||||
just like the issuer does — the remediation is the same knob."""
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("10.0.0.5", 0))]),
|
||||
pytest.raises(OIDCError, match="allow_private_network"),
|
||||
):
|
||||
validate_discovered_endpoint(
|
||||
"https://idp.example.com/token",
|
||||
self._issuer(),
|
||||
allow_http=False,
|
||||
trusted_endpoint_hosts=frozenset(),
|
||||
)
|
||||
|
||||
def test_rejects_http_when_issuer_is_https(self):
|
||||
"""http:// discovered endpoint rejected when issuer is https://."""
|
||||
with (
|
||||
@@ -1643,6 +1714,65 @@ class TestProvisionOIDCUser:
|
||||
|
||||
storage.assign_role.assert_not_called()
|
||||
|
||||
def test_provision_oidc_user_null_oid_tid_collapse_to_empty(self):
|
||||
"""A present-but-null oid/tid claim must store "" — never the string "None".
|
||||
|
||||
`claims.get("oid", "")` returns None (not the "" default) when the key is
|
||||
present with a JSON null, and str(None) == "None" would slip past both the
|
||||
server_default and the truthy backfill guard, storing a bogus non-empty
|
||||
sentinel that collides across every null-emitting user. New-user path.
|
||||
"""
|
||||
config = _make_config()
|
||||
storage = _mock_storage()
|
||||
storage.get_user.return_value = {
|
||||
"user_id": "u-new",
|
||||
"username": "bob",
|
||||
"display_name": "Bob",
|
||||
"password_hash": "!oidc",
|
||||
}
|
||||
|
||||
claims = {"sub": "sub-null", "preferred_username": "bob", "oid": None, "tid": None}
|
||||
with patch("turnstone.core.oidc.uuid") as mock_uuid:
|
||||
mock_uuid.uuid4.return_value = MagicMock(hex="u-new-hex-00000000000000000000")
|
||||
provision_oidc_user(storage, config, claims)
|
||||
|
||||
kwargs = storage.create_oidc_user.call_args.kwargs
|
||||
assert kwargs["oid"] == ""
|
||||
assert kwargs["tid"] == ""
|
||||
|
||||
def test_provision_oidc_user_null_oid_tid_not_backfilled_existing(self):
|
||||
"""Existing-identity path: null oid/tid claims must not backfill "None".
|
||||
|
||||
The truthy guard in update_oidc_identity_login only protects against ""; a
|
||||
"None" produced by str(None) is truthy and would be written, clobbering a
|
||||
real value captured on an earlier login.
|
||||
"""
|
||||
config = _make_config()
|
||||
existing_user = {
|
||||
"user_id": "u1",
|
||||
"username": "alice",
|
||||
"display_name": "Alice",
|
||||
"password_hash": "!oidc",
|
||||
}
|
||||
existing_identity = {
|
||||
"issuer": "https://idp.example.com",
|
||||
"subject": "sub-123",
|
||||
"user_id": "u1",
|
||||
"email": "alice@example.com",
|
||||
"created": "2024-01-01T00:00:00",
|
||||
"last_login": "2024-01-01T00:00:00",
|
||||
"oid": "obj-real",
|
||||
"tid": "ten-real",
|
||||
}
|
||||
storage = _mock_storage(identity=existing_identity, user=existing_user)
|
||||
|
||||
claims = {"sub": "sub-123", "email": "alice@example.com", "oid": None, "tid": None}
|
||||
provision_oidc_user(storage, config, claims)
|
||||
|
||||
kwargs = storage.update_oidc_identity_login.call_args.kwargs
|
||||
assert kwargs["oid"] == ""
|
||||
assert kwargs["tid"] == ""
|
||||
|
||||
def test_existing_identity_self_heals_zero_roles(self):
|
||||
"""Existing identity user with zero roles -> safety-net assigns builtin-viewer.
|
||||
|
||||
@@ -2107,6 +2237,58 @@ class TestDiscoverOIDC:
|
||||
|
||||
asyncio.run(_run())
|
||||
|
||||
def test_discover_oidc_private_issuer_rejected_by_default(self):
|
||||
"""Without the opt-in, a private-resolving issuer disables OIDC."""
|
||||
config = _make_config(
|
||||
issuer="https://auth.internal.example",
|
||||
authorization_endpoint="",
|
||||
token_endpoint="",
|
||||
userinfo_endpoint="",
|
||||
jwks_uri="",
|
||||
)
|
||||
|
||||
async def _run():
|
||||
with patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("10.0.0.5", 0))]):
|
||||
result = await discover_oidc(config)
|
||||
assert result.enabled is False
|
||||
|
||||
asyncio.run(_run())
|
||||
|
||||
def test_discover_oidc_private_issuer_with_opt_in(self):
|
||||
"""allow_private_network=True lets a private-resolving IdP discover."""
|
||||
config = _make_config(
|
||||
issuer="https://auth.internal.example",
|
||||
allow_private_network=True,
|
||||
authorization_endpoint="",
|
||||
token_endpoint="",
|
||||
userinfo_endpoint="",
|
||||
jwks_uri="",
|
||||
)
|
||||
|
||||
discovery_doc = {
|
||||
"authorization_endpoint": "https://auth.internal.example/authorize",
|
||||
"token_endpoint": "https://auth.internal.example/token",
|
||||
"userinfo_endpoint": "https://auth.internal.example/userinfo",
|
||||
"jwks_uri": "https://auth.internal.example/jwks",
|
||||
}
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.json.return_value = discovery_doc
|
||||
mock_response.raise_for_status = MagicMock()
|
||||
|
||||
async def _run():
|
||||
client = _mock_async_client(lambda url: _async_return(mock_response))
|
||||
with (
|
||||
patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("10.0.0.5", 0))]),
|
||||
patch("httpx.AsyncClient", return_value=client),
|
||||
):
|
||||
result = await discover_oidc(config)
|
||||
|
||||
assert result.enabled is True
|
||||
assert result.token_endpoint == "https://auth.internal.example/token"
|
||||
|
||||
asyncio.run(_run())
|
||||
|
||||
def test_discover_oidc_failure(self):
|
||||
"""Mock httpx error -> enabled=False returned."""
|
||||
config = _make_config(
|
||||
|
||||
@@ -84,6 +84,40 @@ class TestCreateOIDCUser:
|
||||
assert identity is not None
|
||||
assert identity["user_id"] == "u-other"
|
||||
|
||||
def test_create_oidc_user_captures_oid_tid(self, db):
|
||||
"""Entra oid/tid are persisted and returned on the identity."""
|
||||
db.create_oidc_user(
|
||||
user_id="u-oid",
|
||||
username="carol",
|
||||
display_name="Carol",
|
||||
password_hash="!oidc",
|
||||
issuer="https://idp.example.com",
|
||||
subject="sub-oid",
|
||||
email="carol@example.com",
|
||||
oid="obj-123",
|
||||
tid="tenant-abc",
|
||||
)
|
||||
identity = db.get_oidc_identity("https://idp.example.com", "sub-oid")
|
||||
assert identity is not None
|
||||
assert identity["oid"] == "obj-123"
|
||||
assert identity["tid"] == "tenant-abc"
|
||||
|
||||
def test_create_oidc_user_oid_tid_default_empty(self, db):
|
||||
"""Omitting oid/tid (non-Entra IdP) stores "" — never NULL."""
|
||||
db.create_oidc_user(
|
||||
user_id="u-noid",
|
||||
username="dave",
|
||||
display_name="Dave",
|
||||
password_hash="!oidc",
|
||||
issuer="https://idp.example.com",
|
||||
subject="sub-noid",
|
||||
email="dave@example.com",
|
||||
)
|
||||
identity = db.get_oidc_identity("https://idp.example.com", "sub-noid")
|
||||
assert identity is not None
|
||||
assert identity["oid"] == ""
|
||||
assert identity["tid"] == ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# OIDC Identity CRUD
|
||||
@@ -137,6 +171,33 @@ class TestOIDCIdentityCRUD:
|
||||
result = db.update_oidc_identity_login("https://idp.example.com", "sub-999")
|
||||
assert result is False
|
||||
|
||||
def test_update_oidc_identity_login_backfills_oid_tid(self, db):
|
||||
"""A login carrying oid/tid backfills them onto a pre-existing row."""
|
||||
db.create_oidc_identity("https://idp.example.com", "sub-bf", "u1", "a@example.com")
|
||||
before = db.get_oidc_identity("https://idp.example.com", "sub-bf")
|
||||
assert before is not None and before["oid"] == ""
|
||||
|
||||
db.update_oidc_identity_login("https://idp.example.com", "sub-bf", oid="obj-9", tid="ten-9")
|
||||
|
||||
after = db.get_oidc_identity("https://idp.example.com", "sub-bf")
|
||||
assert after is not None
|
||||
assert after["oid"] == "obj-9"
|
||||
assert after["tid"] == "ten-9"
|
||||
|
||||
def test_update_oidc_identity_login_omitted_does_not_clobber_oid_tid(self, db):
|
||||
"""A later login WITHOUT oid/tid must not wipe previously-captured values."""
|
||||
db.create_oidc_identity("https://idp.example.com", "sub-keep", "u1", "a@example.com")
|
||||
db.update_oidc_identity_login(
|
||||
"https://idp.example.com", "sub-keep", oid="obj-keep", tid="ten-keep"
|
||||
)
|
||||
# Simulate a subsequent login where the token omitted oid/tid.
|
||||
db.update_oidc_identity_login("https://idp.example.com", "sub-keep")
|
||||
|
||||
identity = db.get_oidc_identity("https://idp.example.com", "sub-keep")
|
||||
assert identity is not None
|
||||
assert identity["oid"] == "obj-keep"
|
||||
assert identity["tid"] == "ten-keep"
|
||||
|
||||
def test_list_oidc_identities_for_user(self, db):
|
||||
"""Two identities for same user, list returns both."""
|
||||
db.create_oidc_identity("https://idp1.example.com", "sub-A", "u1", "alice@idp1.com")
|
||||
|
||||
@@ -8,6 +8,7 @@ capability-gated emission in ``ChatSession._init_system_messages``.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
@@ -330,3 +331,38 @@ class TestEmptyUserTurnDrop:
|
||||
assert len(user_turns) == 1
|
||||
assert f"[start system-reminder_{nonce}]" in user_turns[0]["content"]
|
||||
assert "child done" in user_turns[0]["content"]
|
||||
|
||||
|
||||
class TestToolArgumentLegalization:
|
||||
"""``_prepare_wire_messages`` legalizes malformed tool-call ``arguments`` so a
|
||||
strict renderer (vLLM ``deepseek_v4``) can ``json.loads`` every arguments string
|
||||
— the sibling send-time validity pass to orphan repair."""
|
||||
|
||||
def test_unterminated_arguments_legalized_on_the_wire(self) -> None:
|
||||
s = make_session()
|
||||
msgs = [
|
||||
{"role": "user", "content": "go"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "c1",
|
||||
"type": "function",
|
||||
"function": {"name": "bash", "arguments": '{"command": "cat /va'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "retry with valid JSON"},
|
||||
]
|
||||
out = s._prepare_wire_messages(msgs)
|
||||
emitted = [
|
||||
tc["function"]["arguments"]
|
||||
for m in out
|
||||
if m.get("role") == "assistant"
|
||||
for tc in m.get("tool_calls", [])
|
||||
]
|
||||
assert emitted == ["{}"]
|
||||
assert json.loads(emitted[0]) == {}
|
||||
# Canonical input is untouched — legalization is wire-copy only.
|
||||
assert msgs[1]["tool_calls"][0]["function"]["arguments"] == '{"command": "cat /va'
|
||||
|
||||
@@ -2,7 +2,11 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from turnstone.core.output_guard import evaluate_output, merge_guard_display_payload
|
||||
from turnstone.core.output_guard import (
|
||||
evaluate_output,
|
||||
merge_guard_display_payload,
|
||||
redact_credentials,
|
||||
)
|
||||
|
||||
|
||||
class TestBenignOutput:
|
||||
@@ -205,6 +209,73 @@ class TestCredentialLeakage:
|
||||
)
|
||||
assert "credential_leak" not in r.flags
|
||||
|
||||
def test_single_quote_json_secret(self) -> None:
|
||||
# Python dict reprs / JS object literals emit single quotes; these must
|
||||
# be detected and redacted just like the double-quoted JSON form.
|
||||
r = evaluate_output("headers = {'Authorization': 'Bearer canstillseethis'}")
|
||||
assert "credential_leak" in r.flags
|
||||
assert "json_secret_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "canstillseethis" not in r.sanitized
|
||||
|
||||
def test_single_quote_password(self) -> None:
|
||||
r = evaluate_output("{'password': 'hunter2hunter2'}")
|
||||
assert "json_secret_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "hunter2hunter2" not in r.sanitized
|
||||
|
||||
def test_mongodb_srv_connection_string(self) -> None:
|
||||
r = evaluate_output("uri: mongodb+srv://admin:s3cretpw@cluster.mongodb.net/db")
|
||||
assert "connection_string_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "s3cretpw" not in r.sanitized
|
||||
|
||||
def test_rediss_connection_string(self) -> None:
|
||||
r = evaluate_output("rediss://user:s3cretpw@redis.host:6380/0")
|
||||
assert "connection_string_leak" in r.flags
|
||||
assert r.sanitized is not None
|
||||
assert "s3cretpw" not in r.sanitized
|
||||
|
||||
def test_sqlalchemy_driver_connection_string(self) -> None:
|
||||
# SQLAlchemy dialect+driver URLs must match — the bare-dialect
|
||||
# list alone leaked these (only +psycopg was enumerated).
|
||||
for url in (
|
||||
"postgresql+psycopg2://admin:s3cret_pass@db.internal:5432/prod",
|
||||
"postgresql+asyncpg://admin:s3cret_pass@db.internal/prod",
|
||||
"mysql+pymysql://admin:s3cret_pass@db.internal/prod",
|
||||
):
|
||||
r = evaluate_output(url)
|
||||
assert "connection_string_leak" in r.flags, url
|
||||
assert r.sanitized is not None, url
|
||||
assert "s3cret_pass" not in r.sanitized, url
|
||||
assert ":[REDACTED:password]@" in r.sanitized, url
|
||||
|
||||
def test_uppercase_scheme_connection_string(self) -> None:
|
||||
# RFC 3986 schemes are case-insensitive; an uppercase scheme must
|
||||
# not bypass redaction.
|
||||
for url in (
|
||||
"POSTGRESQL+PSYCOPG2://admin:s3cret_pass@db.internal/prod",
|
||||
"HTTPS://admin:s3cret_pass@api.internal/x",
|
||||
):
|
||||
r = evaluate_output(url)
|
||||
assert "connection_string_leak" in r.flags, url
|
||||
assert r.sanitized is not None, url
|
||||
assert "s3cret_pass" not in r.sanitized, url
|
||||
|
||||
def test_bearer_scheme_case_insensitive(self) -> None:
|
||||
# RFC 7235 scheme name is case-insensitive.
|
||||
r = evaluate_output("authorization: bearer eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.sig12345")
|
||||
assert "credential_leak" in r.flags
|
||||
|
||||
def test_prefixed_key_assignment_redacts_whole_token(self) -> None:
|
||||
# api_key=/secret_key=/access_token= must redact the entire assignment,
|
||||
# not chew only the tail into a garbled "api_[REDACTED:api_key]".
|
||||
secret = "abcdefghijklmnopqrstuvwxyz"
|
||||
for prefix in ("api_key", "secret_key", "session_key", "access_token", "key", "token"):
|
||||
out = redact_credentials(f"{prefix}={secret}")
|
||||
assert secret not in out, (prefix, out)
|
||||
assert out == "[REDACTED:api_key]", (prefix, out)
|
||||
|
||||
|
||||
class TestEncodedPayloads:
|
||||
"""Detect encoded/obfuscated payloads."""
|
||||
|
||||
@@ -521,16 +521,21 @@ class TestSpawnPersona:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Guard 7 — task_agent has no persona parameter (sub-agents keep their own
|
||||
# identity; persona is a workstream-level concept).
|
||||
# Guard 7 — task_agent HAS a persona parameter: a sub-agent's identity comes
|
||||
# from a persona (default = the built-in task-agent identity), validated at
|
||||
# prep against the interactive kind. Revises the original "no persona for
|
||||
# task agents" stance now that personas are first-class on every path.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_task_agent_schema_has_no_persona_param() -> None:
|
||||
def test_task_agent_schema_has_persona_param() -> None:
|
||||
from turnstone.core.tools import TOOLS
|
||||
|
||||
task_agent = next(t for t in TOOLS if t["function"]["name"] == "task_agent")
|
||||
assert "persona" not in task_agent["function"]["parameters"]["properties"]
|
||||
props = task_agent["function"]["parameters"]["properties"]
|
||||
assert "persona" in props
|
||||
# skill= is capability now — the description frames it that way.
|
||||
assert "capability" in props["skill"]["description"].lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1324,3 +1329,180 @@ class TestCreateStampsPersona:
|
||||
assert ws is not None and ws.session is not None
|
||||
assert ws.session._persona_name == ""
|
||||
assert not ws.persona
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Guard 10 — discovery: the calling LLM is TOLD which personas exist. The
|
||||
# live enabled interactive-kind list rides the `persona` parameter
|
||||
# description of task_agent / spawn_workstream / spawn_batch, rebuilt from
|
||||
# the pristine TOOLS base on every render; storage-less sessions keep the
|
||||
# base text untouched. Resolution is forgiving (case, unique display name)
|
||||
# but everything downstream carries the canonical slug.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _persona_desc(session: ChatSession, tool_name: str) -> str:
|
||||
tool = next(t for t in session._tools if t.get("function", {}).get("name") == tool_name)
|
||||
prop = ChatSession._persona_property(tool["function"]["parameters"]["properties"])
|
||||
assert prop is not None, f"{tool_name} has no persona parameter"
|
||||
return prop["description"]
|
||||
|
||||
|
||||
def _pristine_persona_desc(tool_name: str) -> str:
|
||||
from turnstone.core.tools import TOOLS
|
||||
|
||||
tool = next(t for t in TOOLS if t["function"]["name"] == tool_name)
|
||||
prop = ChatSession._persona_property(tool["function"]["parameters"]["properties"])
|
||||
assert prop is not None, f"{tool_name} has no persona parameter"
|
||||
return prop["description"]
|
||||
|
||||
|
||||
class TestPersonaDiscovery:
|
||||
def _seed(self) -> None:
|
||||
get_storage().create_persona(
|
||||
{
|
||||
"persona_id": "p-eng",
|
||||
"name": "engineer",
|
||||
"display_name": "Engineer",
|
||||
"description": "Default engineering identity",
|
||||
"base_prompt": "E",
|
||||
"applies_to_kinds": ["interactive"],
|
||||
"is_default": True,
|
||||
}
|
||||
)
|
||||
get_storage().create_persona(
|
||||
{
|
||||
"persona_id": "p-wri",
|
||||
"name": "writer",
|
||||
"display_name": "Creative Writer",
|
||||
"description": "Prose-first writing partner",
|
||||
"base_prompt": "W",
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
)
|
||||
|
||||
def _coord_session(self, mock_openai_client: Any) -> ChatSession:
|
||||
return _session(
|
||||
mock_openai_client,
|
||||
kind=WorkstreamKind.COORDINATOR,
|
||||
user_id="u1",
|
||||
coord_client=MagicMock(),
|
||||
)
|
||||
|
||||
def test_task_agent_description_lists_personas(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = _session(mock_openai_client)
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
assert desc.startswith(_pristine_persona_desc("task_agent"))
|
||||
assert "Available personas:" in desc
|
||||
# Default first, then A→Z, each with its one-line description.
|
||||
assert desc.index("`engineer` (default)") < desc.index("`writer`")
|
||||
assert "Prose-first writing partner" in desc
|
||||
|
||||
def test_spawn_tools_list_personas_for_coordinators(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
for tool_name in ("spawn_workstream", "spawn_batch"):
|
||||
desc = _persona_desc(session, tool_name)
|
||||
assert desc.startswith(_pristine_persona_desc(tool_name))
|
||||
assert "Available personas:" in desc
|
||||
assert "`engineer` (default)" in desc
|
||||
|
||||
def test_coordinator_kind_personas_are_not_offered(self, tmp_db, mock_openai_client) -> None:
|
||||
# Children and sub-agents are always interactive-kind; a
|
||||
# coordinator-only persona in the list would be a guaranteed error.
|
||||
self._seed()
|
||||
get_storage().create_persona(
|
||||
{
|
||||
"persona_id": "p-exe",
|
||||
"name": "executive",
|
||||
"base_prompt": "X",
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
}
|
||||
)
|
||||
session = self._coord_session(mock_openai_client)
|
||||
assert "`executive`" not in _persona_desc(session, "spawn_workstream")
|
||||
|
||||
def test_storage_down_keeps_pristine_base(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
with patch("turnstone.core.storage.is_storage_initialized", return_value=False):
|
||||
session = _session(mock_openai_client)
|
||||
assert _persona_desc(session, "task_agent") == _pristine_persona_desc("task_agent")
|
||||
|
||||
def test_rerender_is_idempotent_and_tracks_archive(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = _session(mock_openai_client)
|
||||
session._render_agent_tool_descriptions()
|
||||
session._render_agent_tool_descriptions()
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
assert desc.count("Available personas:") == 1
|
||||
# Archive one persona; the next render must drop it, not append.
|
||||
storage = get_storage()
|
||||
writer = storage.get_persona_by_name("writer")
|
||||
assert writer is not None
|
||||
storage.update_persona(writer["persona_id"], enabled=False)
|
||||
session._render_agent_tool_descriptions()
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
assert "`writer`" not in desc
|
||||
assert desc.count("Available personas:") == 1
|
||||
|
||||
def test_large_shelf_drops_prose_keeps_every_name(self, tmp_db, mock_openai_client) -> None:
|
||||
storage = get_storage()
|
||||
for i in range(26):
|
||||
storage.create_persona(
|
||||
{
|
||||
"persona_id": f"p-{i:02d}",
|
||||
"name": f"persona-{i:02d}",
|
||||
"description": "UNIQUE-PROSE-MARKER",
|
||||
"base_prompt": "x",
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
)
|
||||
session = _session(mock_openai_client)
|
||||
desc = _persona_desc(session, "task_agent")
|
||||
for i in range(26):
|
||||
assert f"`persona-{i:02d}`" in desc
|
||||
assert "UNIQUE-PROSE-MARKER" not in desc
|
||||
|
||||
def test_spawn_forgives_case_and_display_name_but_stamps_slug(
|
||||
self, tmp_db, mock_openai_client
|
||||
) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
for variant in ("WRITER", "Writer", "Creative Writer"):
|
||||
item = session._prepare_spawn_workstream("c1", {"persona": variant})
|
||||
assert not item.get("error"), item.get("error")
|
||||
assert item["persona"] == "writer"
|
||||
|
||||
def test_spawn_batch_rows_land_on_canonical_slug(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
item = session._prepare_spawn_batch(
|
||||
"c1",
|
||||
{
|
||||
"children": [
|
||||
{"initial_message": "a", "persona": "WRITER"},
|
||||
{"initial_message": "b", "persona": "Creative Writer"},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert not item.get("error"), item.get("error")
|
||||
personas = [c["persona"] for c in item["children"] if "_error" not in c]
|
||||
assert personas == ["writer", "writer"]
|
||||
|
||||
def test_task_agent_prep_canonicalizes_header_and_stamp(
|
||||
self, tmp_db, mock_openai_client
|
||||
) -> None:
|
||||
self._seed()
|
||||
session = _session(mock_openai_client)
|
||||
item = session._prepare_task("t1", {"prompt": "go", "persona": "Writer"})
|
||||
assert not item.get("error"), item.get("error")
|
||||
assert item["persona"] == "writer"
|
||||
assert "persona: writer" in item["header"]
|
||||
|
||||
def test_unknown_persona_error_enumerates_live_names(self, tmp_db, mock_openai_client) -> None:
|
||||
self._seed()
|
||||
session = self._coord_session(mock_openai_client)
|
||||
item = session._prepare_spawn_workstream("c1", {"persona": "nope"})
|
||||
assert item.get("error")
|
||||
assert "Available for interactive: engineer (default), writer" in item["error"]
|
||||
|
||||
@@ -125,3 +125,191 @@ class TestConfigParsing:
|
||||
cfg["persona_memory"] = "True"
|
||||
with pytest.raises(ValueError, match="persona_memory"):
|
||||
snapshot_from_config(cfg)
|
||||
|
||||
|
||||
class _FakeStorage:
|
||||
"""Minimal storage double for resolve tests — exact-name index + list."""
|
||||
|
||||
def __init__(self, rows: list[dict]) -> None:
|
||||
self._rows = rows
|
||||
|
||||
def get_persona_by_name(self, name: str) -> dict | None:
|
||||
return next((dict(r) for r in self._rows if r["name"] == name), None)
|
||||
|
||||
def list_personas(self, include_disabled: bool = False) -> list[dict]:
|
||||
return [dict(r) for r in self._rows if include_disabled or r.get("enabled")]
|
||||
|
||||
|
||||
def _rows() -> list[dict]:
|
||||
return [
|
||||
{
|
||||
"name": "engineer",
|
||||
"display_name": "Engineer",
|
||||
"enabled": True,
|
||||
"is_default": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
{
|
||||
"name": "writer",
|
||||
"display_name": "Creative Writer",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
{
|
||||
"name": "executive",
|
||||
"display_name": "Executive",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
},
|
||||
{
|
||||
"name": "retired",
|
||||
"display_name": "Retired Persona",
|
||||
"enabled": False,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
class TestForgivingResolution:
|
||||
"""resolve_persona_for_kind — one shared rule, forgiving on all surfaces.
|
||||
|
||||
Exact slug first, then the lowercased input, then a UNIQUE
|
||||
case-insensitive display-name match; every failure enumerates the
|
||||
kind's live names (the self-correction path for stale tool
|
||||
descriptions), and callers stamp the returned row's canonical slug.
|
||||
"""
|
||||
|
||||
def _resolve(self, name: str, kind: str = "interactive", rows: list[dict] | None = None):
|
||||
from turnstone.core.personas import resolve_persona_for_kind
|
||||
|
||||
return resolve_persona_for_kind(_FakeStorage(rows or _rows()), name, kind)
|
||||
|
||||
def test_exact_slug_resolves(self) -> None:
|
||||
row, err = self._resolve("writer")
|
||||
assert err == "" and row is not None and row["name"] == "writer"
|
||||
|
||||
def test_case_variants_resolve_to_canonical_row(self) -> None:
|
||||
for variant in ("Writer", "WRITER", " writer "):
|
||||
row, err = self._resolve(variant)
|
||||
assert err == "" and row is not None and row["name"] == "writer"
|
||||
|
||||
def test_unique_display_name_resolves_to_slug(self) -> None:
|
||||
for variant in ("Creative Writer", "creative writer"):
|
||||
row, err = self._resolve(variant)
|
||||
assert err == "" and row is not None and row["name"] == "writer"
|
||||
|
||||
def test_ambiguous_display_name_names_the_candidates(self) -> None:
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "novelist",
|
||||
"display_name": "creative writer",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
]
|
||||
row, err = self._resolve("Creative Writer", rows=rows)
|
||||
assert row is None
|
||||
assert "more than one display name" in err
|
||||
assert "novelist" in err and "writer" in err
|
||||
assert "use the exact name" in err
|
||||
|
||||
def test_same_display_name_across_kinds_resolves_per_kind(self) -> None:
|
||||
# The label the caller saw came from a kind-filtered surface, so a
|
||||
# same-label persona of the OTHER kind must neither block (spurious
|
||||
# ambiguity) nor win (cross-kind resolution).
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "helper-coord",
|
||||
"display_name": "Helper",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
},
|
||||
{
|
||||
"name": "helper-int",
|
||||
"display_name": "Helper",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
]
|
||||
row, err = self._resolve("Helper", rows=rows)
|
||||
assert err == "" and row is not None and row["name"] == "helper-int"
|
||||
row, err = self._resolve("Helper", kind="coordinator", rows=rows)
|
||||
assert err == "" and row is not None and row["name"] == "helper-coord"
|
||||
|
||||
def test_wrong_kind_display_match_is_not_found_with_choices(self) -> None:
|
||||
# Display names are labels, not identifiers: a label that only exists
|
||||
# on another kind's persona reads as unknown for THIS kind (with the
|
||||
# kind's live choices attached) — never as a cross-kind resolution.
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "chief",
|
||||
"display_name": "The Chief",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
}
|
||||
]
|
||||
row, err = self._resolve("The Chief", rows=rows)
|
||||
assert row is None
|
||||
assert "not found or disabled" in err
|
||||
assert "Available for interactive: engineer (default), writer" in err
|
||||
|
||||
def test_whitespace_input_never_matches_blank_display_names(self) -> None:
|
||||
# display_name defaults to "" — a whitespace-only input (reachable via
|
||||
# CLI `--persona " "`) must read as unknown, never resolve to a
|
||||
# blank-labelled persona or report a bogus ambiguity.
|
||||
rows = _rows() + [
|
||||
{
|
||||
"name": "unlabelled",
|
||||
"display_name": "",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
{
|
||||
"name": "unlabelled-too",
|
||||
"display_name": " ",
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
},
|
||||
]
|
||||
for raw in ("", " ", " "):
|
||||
row, err = self._resolve(raw, rows=rows)
|
||||
assert row is None
|
||||
assert "not found or disabled" in err
|
||||
assert "more than one display name" not in err
|
||||
|
||||
def test_unknown_error_lists_kind_names_default_first(self) -> None:
|
||||
row, err = self._resolve("nope")
|
||||
assert row is None
|
||||
assert "Persona not found or disabled: 'nope'" in err
|
||||
assert "Available for interactive: engineer (default), writer" in err
|
||||
assert "executive" not in err # wrong kind
|
||||
assert "retired" not in err # disabled
|
||||
|
||||
def test_kind_mismatch_reports_canonical_slug_and_choices(self) -> None:
|
||||
row, err = self._resolve("Executive") # case-forgiven, then kind-refused
|
||||
assert row is None
|
||||
assert "'executive' does not apply to kind 'interactive'" in err
|
||||
assert "Available for interactive: engineer (default), writer" in err
|
||||
|
||||
def test_disabled_persona_is_not_resolvable_by_any_route(self) -> None:
|
||||
for variant in ("retired", "RETIRED", "Retired Persona"):
|
||||
row, err = self._resolve(variant)
|
||||
assert row is None
|
||||
assert "not found or disabled" in err
|
||||
|
||||
def test_storage_none_is_a_distinct_error(self) -> None:
|
||||
from turnstone.core.personas import resolve_persona_for_kind
|
||||
|
||||
row, err = resolve_persona_for_kind(None, "writer", "interactive")
|
||||
assert row is None and err == "persona storage unavailable"
|
||||
|
||||
def test_listing_failure_degrades_to_plain_error(self) -> None:
|
||||
class _Broken(_FakeStorage):
|
||||
def list_personas(self, include_disabled: bool = False) -> list[dict]:
|
||||
raise RuntimeError("db gone")
|
||||
|
||||
from turnstone.core.personas import resolve_persona_for_kind
|
||||
|
||||
row, err = resolve_persona_for_kind(_Broken(_rows()), "nope", "interactive")
|
||||
assert row is None
|
||||
assert "Persona not found or disabled: 'nope'" in err
|
||||
|
||||
@@ -179,10 +179,10 @@ def test_bulk_live_admin_bypass_returns_live(storage):
|
||||
|
||||
|
||||
def test_bulk_live_cluster_wide_visibility(storage):
|
||||
"""Trusted-team visibility: any ``admin.cluster.inspect`` caller
|
||||
sees every row in ``results``. ``denied`` is reserved for ids
|
||||
that don't correspond to a persisted workstream (no existence
|
||||
oracle for unknown ids)."""
|
||||
"""A project-less workstream has no tenancy to enforce, so any
|
||||
``admin.cluster.inspect`` caller sees it in ``results``. ``denied``
|
||||
is reserved for ids that don't correspond to a persisted workstream
|
||||
(no existence oracle for unknown ids)."""
|
||||
ws_id = "b" * 32
|
||||
_seed_workstream(storage, ws_id=ws_id, node_id="node-a", user_id="stranger")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage))
|
||||
@@ -196,6 +196,48 @@ def test_bulk_live_cluster_wide_visibility(storage):
|
||||
assert body["denied"] == []
|
||||
|
||||
|
||||
def test_bulk_live_private_project_row_routes_to_denied(storage):
|
||||
"""A workstream in a private project the caller isn't a member of
|
||||
routes to ``denied``, not ``results`` — a cluster admin gets no
|
||||
private-project oracle from the bulk surface either."""
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
ws_id = "c" * 32
|
||||
storage.register_workstream(ws_id, node_id="node-a", user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage))
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/live?ids={ws_id}",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["results"] == {}
|
||||
assert body["denied"] == [ws_id]
|
||||
|
||||
|
||||
def test_bulk_live_private_project_row_visible_to_member(storage):
|
||||
"""A project member sees the row (routes to ``results``); the live
|
||||
block is null only because the coordinator row isn't loaded."""
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
ws_id = "c" * 32
|
||||
storage.register_workstream(
|
||||
ws_id,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage))
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/live?ids={ws_id}",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert ws_id in body["results"]
|
||||
assert body["denied"] == []
|
||||
|
||||
|
||||
def test_bulk_live_unknown_ids_route_to_denied(storage):
|
||||
"""Unknown ids (not in storage) land in ``denied`` so the endpoint
|
||||
can't be used as an existence oracle."""
|
||||
@@ -228,34 +270,42 @@ def test_bulk_live_coordinator_row_uses_manager_snapshot(storage):
|
||||
live = body["results"][ws.id]
|
||||
assert live is not None
|
||||
assert "pending_approval" in live
|
||||
# New field always present on the wire — None when no approval
|
||||
# is pending so the JS can `key in row` without surprise.
|
||||
assert "pending_approval_detail" in live
|
||||
assert live["pending_approval_detail"] is None
|
||||
# The details list is always present on the wire — empty when no
|
||||
# approval is pending so the JS can `key in row` without surprise.
|
||||
# Replaces 1.6's singular ``pending_approval_detail`` null
|
||||
# (breaking, 1.7).
|
||||
assert "pending_approval_details" in live
|
||||
assert live["pending_approval_details"] == []
|
||||
|
||||
|
||||
def test_bulk_live_coordinator_row_includes_pending_approval_detail(storage):
|
||||
"""When _pending_approval is set on a coord UI, the live block
|
||||
surfaces the merged items + judge_verdict payload through the
|
||||
coord-pseudo-node path. End-to-end equivalent of the dashboard
|
||||
test in test_server_authz, but for the console live-bulk
|
||||
endpoint that the coord tree UI actually consumes."""
|
||||
def test_bulk_live_coordinator_row_includes_pending_approval_details(storage):
|
||||
"""When an approval cycle is live on a coord UI, the live block
|
||||
surfaces one detail entry per cycle with merged items +
|
||||
judge_verdict through the coord-pseudo-node path. End-to-end
|
||||
equivalent of the dashboard test in test_server_authz, but for
|
||||
the console live-bulk endpoint that the coord tree UI actually
|
||||
consumes."""
|
||||
from turnstone.core.session_ui_base import ApprovalCycle
|
||||
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
ws.ui._pending_approval = {
|
||||
items = [
|
||||
{
|
||||
"call_id": "c-99",
|
||||
"header": "spawn_workstream",
|
||||
"preview": "{...}",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
]
|
||||
card = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-99",
|
||||
"header": "spawn_workstream",
|
||||
"preview": "{...}",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
],
|
||||
"cycle_id": "cyc-99",
|
||||
"items": ws.ui._serialize_approval_items(items),
|
||||
"judge_pending": False,
|
||||
}
|
||||
ws.ui._register_approval_cycle(ApprovalCycle(items, card, None))
|
||||
ws.ui._llm_verdicts["c-99"] = {
|
||||
"recommendation": "approve",
|
||||
"risk_level": "low",
|
||||
@@ -269,8 +319,10 @@ def test_bulk_live_coordinator_row_includes_pending_approval_detail(storage):
|
||||
assert resp.status_code == 200
|
||||
live = resp.json()["results"][ws.id]
|
||||
assert live["pending_approval"] is True # boolean derived flag
|
||||
detail = live["pending_approval_detail"]
|
||||
assert detail is not None
|
||||
details = live["pending_approval_details"]
|
||||
assert len(details) == 1
|
||||
detail = details[0]
|
||||
assert detail["cycle_id"] == "cyc-99"
|
||||
assert detail["call_id"] == "c-99"
|
||||
assert detail["items"][0]["func_name"] == "spawn_workstream"
|
||||
assert detail["items"][0]["judge_verdict"]["recommendation"] == "approve"
|
||||
|
||||
@@ -127,10 +127,14 @@ class TestWsVisiblePredicate:
|
||||
assert storage.get_project.call_count == 1
|
||||
|
||||
def test_for_request_bypass_rules(self) -> None:
|
||||
# Only service scope bypasses (node→console machine plumbing,
|
||||
# re-filtered per-user at the console edge).
|
||||
assert WorkstreamProjectVisibility.for_request(
|
||||
_request_for("bob", scopes=("service",))
|
||||
)._bypass
|
||||
assert WorkstreamProjectVisibility.for_request(
|
||||
# admin.cluster.inspect gates the inspect *surfaces* but does NOT
|
||||
# bypass private-project tenancy — the admin filters as themselves.
|
||||
assert not WorkstreamProjectVisibility.for_request(
|
||||
_request_for("bob", permissions=("admin.cluster.inspect",))
|
||||
)._bypass
|
||||
assert not WorkstreamProjectVisibility.for_request(_request_for("bob"))._bypass
|
||||
@@ -214,15 +218,17 @@ class TestResolveWorkstreamOwnerProjectGate:
|
||||
assert err is None
|
||||
assert owner == "bob"
|
||||
|
||||
def test_admin_inspect_bypasses(self, tmp_db: str) -> None:
|
||||
def test_admin_inspect_does_not_bypass(self, tmp_db: str) -> None:
|
||||
# A permitted admin (admin.cluster.inspect) who isn't the owner /
|
||||
# creator / member of a private project is still 403'd at the row
|
||||
# gate — the permission gates the inspect surface, not the tenancy.
|
||||
from turnstone.core.web_helpers import resolve_workstream_owner
|
||||
|
||||
self._seed(member=False)
|
||||
owner, err = resolve_workstream_owner(
|
||||
_request_for("bob", permissions=("admin.cluster.inspect",)), "ws-priv"
|
||||
)
|
||||
assert err is None
|
||||
assert owner == "alice"
|
||||
assert err is not None and err.status_code == 403
|
||||
|
||||
def test_missing_ws_still_404s(self, tmp_db: str) -> None:
|
||||
from turnstone.core.web_helpers import resolve_workstream_owner
|
||||
|
||||
@@ -88,10 +88,9 @@ def _make_session(**kwargs):
|
||||
|
||||
|
||||
def _sys_content(session: ChatSession) -> str:
|
||||
"""Extract the system message content."""
|
||||
msgs = [m for m in session.system_messages if m["role"] == "system"]
|
||||
assert msgs
|
||||
return msgs[0]["content"]
|
||||
"""Full prompt prefix: identity system message + any skill context message."""
|
||||
assert session.system_messages
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
def _create_template(db, template_id, name, content, is_default=False, **kwargs):
|
||||
@@ -187,17 +186,23 @@ class TestDefaultTemplates:
|
||||
content = _sys_content(session)
|
||||
assert "Not default." not in content
|
||||
|
||||
def test_templates_before_instructions(self, tmp_db):
|
||||
def test_default_template_stays_in_identity_system_message(self, tmp_db):
|
||||
"""Default (always-on) templates are the standing baseline and never
|
||||
change mid-session, so they stay in the identity system message (with
|
||||
user instructions) — only a NAMED applied skill moves to a separate
|
||||
context message. Template guidance still precedes user instructions."""
|
||||
from turnstone.core.storage import get_storage
|
||||
|
||||
db = get_storage()
|
||||
_create_template(db, "t1", "tpl", "TEMPLATE_CONTENT", is_default=True)
|
||||
|
||||
session = _make_session(instructions="USER_INSTRUCTIONS")
|
||||
content = _sys_content(session)
|
||||
tpl_pos = content.index("TEMPLATE_CONTENT")
|
||||
instr_pos = content.index("USER_INSTRUCTIONS")
|
||||
assert tpl_pos < instr_pos
|
||||
msgs = session.system_messages
|
||||
# Both the default template and instructions live in the system message,
|
||||
# template first — and no default triggers a user-role context message.
|
||||
assert all(m["role"] == "system" for m in msgs)
|
||||
content = msgs[0]["content"]
|
||||
assert content.index("TEMPLATE_CONTENT") < content.index("USER_INSTRUCTIONS")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -12,6 +12,7 @@ toggle.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
import os
|
||||
import sys
|
||||
from typing import Any
|
||||
@@ -21,6 +22,7 @@ import pytest
|
||||
|
||||
from tests._session_helpers import make_session as _make_session
|
||||
from turnstone.core.providers._anthropic import AnthropicProvider
|
||||
from turnstone.core.providers._protocol import ModelCapabilities
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
@@ -166,6 +168,178 @@ class TestCompatWireShape:
|
||||
assert kwargs["thinking"] == {"type": "enabled", "budget_tokens": 2048}
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# TestCompatReasoningControl
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
class TestCompatReasoningControl:
|
||||
"""Session effort knob → ``chat_template_kwargs`` on the compat lane.
|
||||
|
||||
vLLM's ``/v1/messages`` ignores the native ``thinking`` param — the
|
||||
reasoning levers live in the chat template.
|
||||
``merge_reasoning_template_kwargs`` maps the knob onto
|
||||
``caps.thinking_param`` (manual: knob "none" = off, mirroring
|
||||
``_reasoning_params``; adaptive: always on) and ``caps.effort_param``
|
||||
(graded value for gpt-oss-style templates). Verified live against
|
||||
qwen3.6 on vLLM 2026-07-03: ``{"enable_thinking": false}`` disables
|
||||
thinking, unknown chat_template_kwargs keys are silently ignored.
|
||||
"""
|
||||
|
||||
_MANUAL_CAPS = ModelCapabilities(
|
||||
token_param="max_tokens",
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
)
|
||||
|
||||
def setup_method(self) -> None:
|
||||
self.provider = AnthropicProvider(compat=True)
|
||||
|
||||
def _stream_kwargs(
|
||||
self,
|
||||
caps: ModelCapabilities | None,
|
||||
reasoning_effort: str,
|
||||
extra_params: dict[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
client = _capture_client()
|
||||
with patch("turnstone.core.providers._anthropic._ensure_anthropic"):
|
||||
list(
|
||||
self.provider.create_streaming(
|
||||
client=client,
|
||||
model="qwen3.6-27b",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
temperature=0.6,
|
||||
reasoning_effort=reasoning_effort,
|
||||
extra_params=extra_params,
|
||||
capabilities=caps,
|
||||
)
|
||||
)
|
||||
return client.messages.stream.call_args[1]
|
||||
|
||||
def test_manual_toggle_on(self) -> None:
|
||||
"""Any non-none effort turns the toggle on AND carries the graded
|
||||
value under the fallback key — the user's effort setting always
|
||||
reaches the wire; a template that doesn't reference the kwarg
|
||||
ignores it."""
|
||||
kwargs = self._stream_kwargs(self._MANUAL_CAPS, "medium")
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "medium"}
|
||||
}
|
||||
assert "thinking" not in kwargs
|
||||
assert kwargs["temperature"] == 0.6 # never forced to 1.0 on compat
|
||||
|
||||
@pytest.mark.parametrize("knob", ["none", ""])
|
||||
def test_manual_toggle_off(self, knob: str) -> None:
|
||||
"""Effort "none"/empty disables thinking — native manual-mode parity;
|
||||
no effort key rides when thinking is off."""
|
||||
kwargs = self._stream_kwargs(self._MANUAL_CAPS, knob)
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
assert "thinking" not in kwargs
|
||||
|
||||
def test_adaptive_always_on(self) -> None:
|
||||
"""Adaptive never knob-disables — native-adaptive contract, no native
|
||||
dict; the graded value rides for on-positions only."""
|
||||
caps = dataclasses.replace(self._MANUAL_CAPS, thinking_mode="adaptive")
|
||||
kwargs = self._stream_kwargs(caps, "high")
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "high"}
|
||||
}
|
||||
assert "thinking" not in kwargs
|
||||
assert kwargs["temperature"] == 0.6
|
||||
kwargs = self._stream_kwargs(caps, "none")
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": True}}
|
||||
assert "thinking" not in kwargs
|
||||
assert kwargs["temperature"] == 0.6
|
||||
|
||||
def test_default_caps_inject_nothing(self) -> None:
|
||||
"""Untouched compat defaults (thinking_mode=none) keep today's wire."""
|
||||
kwargs = self._stream_kwargs(None, "medium")
|
||||
assert "extra_body" not in kwargs
|
||||
assert "thinking" not in kwargs
|
||||
|
||||
def test_effort_param_validated_against_values(self) -> None:
|
||||
"""Off-list knob rounds up onto the declared values (ceiling-capped),
|
||||
never sent raw — and never snaps DOWN to the default."""
|
||||
caps = dataclasses.replace(
|
||||
self._MANUAL_CAPS,
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "xhigh")
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "high"}
|
||||
}
|
||||
|
||||
def test_effort_param_freeform_without_values(self) -> None:
|
||||
"""No declared values → knob forwarded as-is, template is authority."""
|
||||
caps = ModelCapabilities(
|
||||
token_param="max_tokens",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "xhigh")
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"reasoning_effort": "xhigh"}}
|
||||
|
||||
def test_effort_param_omitted_on_none(self) -> None:
|
||||
"""Knob "none" sends no effort key (and toggles thinking off)."""
|
||||
caps = dataclasses.replace(
|
||||
self._MANUAL_CAPS,
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "none")
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
def test_operator_override_wins(self) -> None:
|
||||
"""server_compat chat_template_kwargs entries beat the knob mapping."""
|
||||
kwargs = self._stream_kwargs(
|
||||
self._MANUAL_CAPS,
|
||||
"none",
|
||||
extra_params={"chat_template_kwargs": {"enable_thinking": True}, "foo": 1},
|
||||
)
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True},
|
||||
"foo": 1,
|
||||
}
|
||||
|
||||
def test_caller_extra_params_not_mutated(self) -> None:
|
||||
"""The session's extra_params dict must never be written through."""
|
||||
extra = {"chat_template_kwargs": {"foo": 1}}
|
||||
self._stream_kwargs(self._MANUAL_CAPS, "medium", extra_params=extra)
|
||||
assert extra == {"chat_template_kwargs": {"foo": 1}}
|
||||
|
||||
def test_no_output_config_on_compat(self) -> None:
|
||||
"""supports_effort must not leak Anthropic output_config to vLLM."""
|
||||
caps = dataclasses.replace(
|
||||
self._MANUAL_CAPS,
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
)
|
||||
kwargs = self._stream_kwargs(caps, "high")
|
||||
assert "output_config" not in kwargs
|
||||
assert kwargs["extra_body"] == {
|
||||
"chat_template_kwargs": {"enable_thinking": True, "reasoning_effort": "high"}
|
||||
}
|
||||
|
||||
def test_create_completion_same_injection(self) -> None:
|
||||
"""The non-streaming path shares _build_thinking_and_kwargs."""
|
||||
client = MagicMock()
|
||||
final = MagicMock(content=[], stop_reason="end_turn")
|
||||
stream = MagicMock()
|
||||
stream.get_final_message.return_value = final
|
||||
client.messages.stream.return_value.__enter__.return_value = stream
|
||||
with patch("turnstone.core.providers._anthropic._ensure_anthropic"):
|
||||
self.provider.create_completion(
|
||||
client=client,
|
||||
model="qwen3.6-27b",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
reasoning_effort="none",
|
||||
capabilities=self._MANUAL_CAPS,
|
||||
)
|
||||
kwargs = client.messages.stream.call_args[1]
|
||||
assert kwargs["extra_body"] == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# TestCompatFactory
|
||||
# ===========================================================================
|
||||
|
||||
+200
-81
@@ -12,10 +12,12 @@ from turnstone.core.lowering import repair_wire_messages
|
||||
from turnstone.core.providers._openai import OpenAIProvider
|
||||
from turnstone.core.providers._openai_chat import OpenAIChatCompletionsProvider
|
||||
from turnstone.core.providers._openai_common import (
|
||||
OPENAI_COMPAT_DEFAULT,
|
||||
apply_cache_retention,
|
||||
apply_temperature_and_effort,
|
||||
apply_tool_search,
|
||||
format_citations,
|
||||
lookup_openai_capabilities,
|
||||
sanitize_messages,
|
||||
)
|
||||
from turnstone.core.providers._protocol import (
|
||||
@@ -153,51 +155,89 @@ class TestOpenAIProvider:
|
||||
def test_provider_name(self) -> None:
|
||||
assert self.provider.provider_name == "openai-compatible"
|
||||
|
||||
# -- _apply_thinking_mode -------------------------------------------------
|
||||
# -- reasoning template kwargs (_finalize_extra_body) ---------------------
|
||||
|
||||
def test_thinking_mode_none_does_nothing(self) -> None:
|
||||
"""No thinking params injected when thinking_mode is 'none'."""
|
||||
"""No toggle injected when thinking_mode is 'none'; operator keys pass."""
|
||||
caps = ModelCapabilities(thinking_mode="none")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert "enable_thinking" not in extra_body["chat_template_kwargs"]
|
||||
extra_params = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
eb = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert eb is not None
|
||||
assert "enable_thinking" not in eb["chat_template_kwargs"]
|
||||
assert eb["chat_template_kwargs"]["reasoning_effort"] == "medium"
|
||||
|
||||
def test_thinking_mode_manual_injects_param(self) -> None:
|
||||
"""Manual thinking mode injects enable_thinking into chat_template_kwargs."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is True
|
||||
assert extra_body["chat_template_kwargs"]["reasoning_effort"] == "medium"
|
||||
extra_params = {"chat_template_kwargs": {"reasoning_effort": "medium"}}
|
||||
eb = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert eb is not None
|
||||
assert eb["chat_template_kwargs"]["enable_thinking"] is True
|
||||
assert eb["chat_template_kwargs"]["reasoning_effort"] == "medium"
|
||||
|
||||
def test_thinking_mode_manual_knob_none_disables(self) -> None:
|
||||
"""Effort knob "none" turns the template toggle off, not just quiet."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
eb = self.provider._finalize_extra_body(None, caps, "none")
|
||||
assert eb == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
def test_thinking_mode_custom_param(self) -> None:
|
||||
"""Custom thinking_param (e.g. Granite's 'thinking') is used."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="thinking")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["thinking"] is True
|
||||
assert "enable_thinking" not in extra_body["chat_template_kwargs"]
|
||||
eb = self.provider._finalize_extra_body(None, caps, "medium")
|
||||
assert eb == {"chat_template_kwargs": {"thinking": True}}
|
||||
|
||||
def test_thinking_mode_does_not_override_explicit(self) -> None:
|
||||
"""If operator explicitly set the param to False, provider respects it."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is False
|
||||
extra_params = {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
eb = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert eb is not None
|
||||
assert eb["chat_template_kwargs"]["enable_thinking"] is False
|
||||
|
||||
def test_thinking_mode_creates_ctk_if_missing(self) -> None:
|
||||
"""Creates chat_template_kwargs dict if not present in extra_body."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_body: dict[str, Any] = {}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is True
|
||||
|
||||
def test_thinking_mode_adaptive(self) -> None:
|
||||
"""Adaptive thinking mode also injects the param."""
|
||||
def test_thinking_mode_adaptive_never_knob_disables(self) -> None:
|
||||
"""Adaptive = model self-regulates; knob "none" must not force false."""
|
||||
caps = ModelCapabilities(thinking_mode="adaptive")
|
||||
extra_body: dict[str, Any] = {"chat_template_kwargs": {}}
|
||||
OpenAIProvider._apply_thinking_mode(extra_body, caps)
|
||||
assert extra_body["chat_template_kwargs"]["enable_thinking"] is True
|
||||
for knob in ("high", "none", ""):
|
||||
eb = self.provider._finalize_extra_body(None, caps, knob)
|
||||
assert eb == {"chat_template_kwargs": {"enable_thinking": True}}
|
||||
|
||||
def test_effort_param_suppresses_flat_reasoning_effort(self) -> None:
|
||||
"""Declaring the ctk effort channel must not double-send the flat param."""
|
||||
from turnstone.core.providers._openai_common import apply_temperature_and_effort
|
||||
|
||||
caps = ModelCapabilities(
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
)
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, 0.5, "medium")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
# Without effort_param the flat param still flows (commercial path).
|
||||
flat_caps = ModelCapabilities(reasoning_effort_values=("low", "medium", "high"))
|
||||
kwargs = {}
|
||||
apply_temperature_and_effort(kwargs, flat_caps, 0.5, "medium")
|
||||
assert kwargs["reasoning_effort"] == "medium"
|
||||
|
||||
def test_effort_param_injects_knob_value(self) -> None:
|
||||
"""effort_param carries the knob into chat_template_kwargs (gpt-oss);
|
||||
a knob above the declared ceiling rides the ceiling, not the default."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eb = self.provider._finalize_extra_body(None, caps, "xhigh")
|
||||
assert eb == {"chat_template_kwargs": {"reasoning_effort": "high"}}
|
||||
assert self.provider._finalize_extra_body(None, caps, "none") is None
|
||||
|
||||
def test_caller_extra_params_not_mutated(self) -> None:
|
||||
"""The session dict and its ctk sub-dict survive injection untouched."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
extra_params = {"chat_template_kwargs": {"foo": 1}}
|
||||
self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
assert extra_params == {"chat_template_kwargs": {"foo": 1}}
|
||||
|
||||
# -- _sanitize_messages ---------------------------------------------------
|
||||
|
||||
@@ -1633,6 +1673,25 @@ class TestProviderFactory:
|
||||
assert openai_prov.provider_name == "openai"
|
||||
assert compat.provider_name == "openai-compatible"
|
||||
|
||||
def test_openai_compatible_never_consults_commercial_registry(self) -> None:
|
||||
"""Local-lane model ids are operator-chosen strings — a prefix
|
||||
collision with a cloud model id must not inherit that model's
|
||||
sampling/effort contract, on either API surface. Cloud lookups
|
||||
are unaffected."""
|
||||
from turnstone.core.providers import create_provider
|
||||
|
||||
compat = create_provider("openai-compatible")
|
||||
compat_responses = create_provider("openai-compatible", api_surface="responses")
|
||||
for name in ("gpt-5.5-my-finetune", "o3-distill", "deepseek-v4-flash", ""):
|
||||
assert compat.get_capabilities(name) is OPENAI_COMPAT_DEFAULT
|
||||
assert compat_responses.get_capabilities(name) is OPENAI_COMPAT_DEFAULT
|
||||
# The commercial lane keeps resolving its registry rows — through
|
||||
# the factory AND through the non-compat class default.
|
||||
cloud = create_provider("openai").get_capabilities("gpt-5.5")
|
||||
assert cloud.default_reasoning_effort == "medium"
|
||||
assert "xhigh" in cloud.reasoning_effort_values
|
||||
assert create_provider("openai") is not compat_responses
|
||||
|
||||
def test_create_provider_returns_singleton(self) -> None:
|
||||
from turnstone.core.providers import create_provider
|
||||
|
||||
@@ -1765,6 +1824,39 @@ class TestProviderFactory:
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
class TestGoogleEffortKnob:
|
||||
"""The session effort knob reaches Gemini as a flat reasoning_effort."""
|
||||
|
||||
def _create_kwargs(self, reasoning_effort: str) -> dict[str, Any]:
|
||||
from turnstone.core.providers._google import GoogleProvider
|
||||
|
||||
prov = GoogleProvider()
|
||||
client = MagicMock()
|
||||
client.chat.completions.create.return_value = iter([])
|
||||
list(
|
||||
prov.create_streaming(
|
||||
client=client,
|
||||
model="gemini-3-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
reasoning_effort=reasoning_effort,
|
||||
)
|
||||
)
|
||||
return client.chat.completions.create.call_args[1]
|
||||
|
||||
def test_knob_values_forward_verbatim(self) -> None:
|
||||
for knob in ("minimal", "low", "medium", "high"):
|
||||
assert self._create_kwargs(knob)["reasoning_effort"] == knob
|
||||
|
||||
def test_off_list_knob_snaps_to_high(self) -> None:
|
||||
"""xhigh/max are not in Gemini's vocabulary — snap down to high."""
|
||||
for knob in ("xhigh", "max"):
|
||||
assert self._create_kwargs(knob)["reasoning_effort"] == "high"
|
||||
|
||||
def test_none_omits_the_param(self) -> None:
|
||||
"""Knob none never sends "none" — 2.5 Pro / 3.x reject disabling."""
|
||||
assert "reasoning_effort" not in self._create_kwargs("none")
|
||||
|
||||
|
||||
class TestGoogleProviderFidelity:
|
||||
"""Tests for thought_signature round-trip via provider_blocks."""
|
||||
|
||||
@@ -2003,49 +2095,64 @@ class TestOpenAIParameterGating:
|
||||
def setup_method(self) -> None:
|
||||
self.provider = OpenAIProvider()
|
||||
|
||||
def test_unknown_model_no_reasoning_effort(self) -> None:
|
||||
"""Unknown/local models should NOT receive top-level reasoning_effort."""
|
||||
def test_local_model_effort_forwarded_verbatim(self) -> None:
|
||||
"""Local-lane models receive the session knob verbatim on the flat
|
||||
param (effort_passthrough) — the user's effort setting always
|
||||
reaches the wire; "none" stays omitted (nothing to disable
|
||||
beyond the template toggle)."""
|
||||
caps = self.provider.get_capabilities("my-local-model")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "medium"
|
||||
assert kwargs["temperature"] == 0.7
|
||||
kwargs = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
|
||||
def test_gpt5_no_temperature_has_reasoning_effort(self) -> None:
|
||||
"""GPT-5 base: no temperature, reasoning_effort sent."""
|
||||
caps = self.provider.get_capabilities("gpt-5")
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="high")
|
||||
assert "temperature" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_gpt51_temperature_when_effort_none(self) -> None:
|
||||
"""GPT-5.1: temperature only when reasoning_effort='none'."""
|
||||
caps = self.provider.get_capabilities("gpt-5.1")
|
||||
"""GPT-5.1: temperature only when reasoning_effort='none'; the
|
||||
declared "none" level is forwarded explicitly (knob = off)."""
|
||||
caps = lookup_openai_capabilities("gpt-5.1")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert kwargs["temperature"] == 0.7
|
||||
assert "reasoning_effort" not in kwargs # "none" is skipped
|
||||
assert kwargs["reasoning_effort"] == "none"
|
||||
|
||||
def test_gpt51_no_temperature_when_reasoning_active(self) -> None:
|
||||
"""GPT-5.1: no temperature when reasoning is active."""
|
||||
caps = self.provider.get_capabilities("gpt-5.1")
|
||||
caps = lookup_openai_capabilities("gpt-5.1")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="high")
|
||||
assert "temperature" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_o_series_no_temperature_no_reasoning_effort(self) -> None:
|
||||
"""O-series: no temperature, no reasoning_effort."""
|
||||
caps = self.provider.get_capabilities("o3")
|
||||
def test_o_series_no_temperature_but_effort_forwarded(self) -> None:
|
||||
"""O-series: no temperature; low/medium/high ARE valid effort
|
||||
values (all o-series except o1-mini) and the knob reaches them."""
|
||||
caps = lookup_openai_capabilities("o3")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "temperature" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "medium"
|
||||
|
||||
def test_o1_mini_has_no_effort_control(self) -> None:
|
||||
"""o1-mini is the one o-series model without reasoning_effort."""
|
||||
caps = lookup_openai_capabilities("o1-mini")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "reasoning_effort" not in kwargs
|
||||
|
||||
def test_gpt5_pro_unsupported_effort_falls_back(self) -> None:
|
||||
"""GPT-5 pro only supports 'high'; unsupported values fall back to default."""
|
||||
caps = self.provider.get_capabilities("gpt-5-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5-pro")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="medium")
|
||||
assert "temperature" not in kwargs
|
||||
@@ -2053,19 +2160,19 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt5_pro_supported_effort_passes_through(self) -> None:
|
||||
"""GPT-5 pro accepts 'high' directly."""
|
||||
caps = self.provider.get_capabilities("gpt-5-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5-pro")
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="high")
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_gpt54_1m_context_and_effort(self) -> None:
|
||||
"""GPT-5.4: 1M context, temperature when effort=none, xhigh supported."""
|
||||
caps = self.provider.get_capabilities("gpt-5.4")
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
assert caps.context_window == 1050000
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert kwargs["temperature"] == 0.7
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "none" # declared level, forwarded
|
||||
kwargs2: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs2, caps, temperature=0.7, reasoning_effort="xhigh")
|
||||
assert "temperature" not in kwargs2
|
||||
@@ -2073,7 +2180,7 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt54_pro_no_temperature_always_reasoning(self) -> None:
|
||||
"""GPT-5.4 pro: no temperature, medium/high/xhigh only."""
|
||||
caps = self.provider.get_capabilities("gpt-5.4-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5.4-pro")
|
||||
assert caps.context_window == 1050000
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="low")
|
||||
@@ -2082,14 +2189,14 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt55_1m_context_and_effort(self) -> None:
|
||||
"""GPT-5.5: 1M context, temperature when effort=none, xhigh supported."""
|
||||
caps = self.provider.get_capabilities("gpt-5.5")
|
||||
caps = lookup_openai_capabilities("gpt-5.5")
|
||||
assert caps.context_window == 1050000
|
||||
assert caps.supports_tool_search is True
|
||||
assert caps.supports_vision is True
|
||||
kwargs: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs, caps, temperature=0.7, reasoning_effort="none")
|
||||
assert kwargs["temperature"] == 0.7
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "none" # declared level, forwarded
|
||||
kwargs2: dict[str, Any] = {}
|
||||
apply_temperature_and_effort(kwargs2, caps, temperature=0.7, reasoning_effort="xhigh")
|
||||
assert "temperature" not in kwargs2
|
||||
@@ -2097,7 +2204,7 @@ class TestOpenAIParameterGating:
|
||||
|
||||
def test_gpt55_pro_no_temperature_always_reasoning(self) -> None:
|
||||
"""GPT-5.5 pro: no temperature, medium/high/xhigh only."""
|
||||
caps = self.provider.get_capabilities("gpt-5.5-pro")
|
||||
caps = lookup_openai_capabilities("gpt-5.5-pro")
|
||||
assert caps.context_window == 1050000
|
||||
assert caps.supports_tool_search is True
|
||||
kwargs: dict[str, Any] = {}
|
||||
@@ -2316,11 +2423,20 @@ class TestAnthropicReasoningNone:
|
||||
result = _map_reasoning_to_effort("xhigh", ("low", "medium", "high", "xhigh", "max"))
|
||||
assert result == "xhigh"
|
||||
|
||||
def test_map_xhigh_rejected_by_model_without_it(self) -> None:
|
||||
def test_map_xhigh_snaps_up_through_gap_to_max(self) -> None:
|
||||
"""Levels with a hole (no xhigh) round the knob UP to the next
|
||||
declared level rather than dropping output_config entirely."""
|
||||
from turnstone.core.providers._anthropic import _map_reasoning_to_effort
|
||||
|
||||
result = _map_reasoning_to_effort("xhigh", ("low", "medium", "high", "max"))
|
||||
assert result is None
|
||||
assert result == "max"
|
||||
|
||||
def test_map_above_ceiling_rides_ceiling(self) -> None:
|
||||
from turnstone.core.providers._anthropic import _map_reasoning_to_effort
|
||||
|
||||
assert _map_reasoning_to_effort("max", ("low", "medium", "high")) == "high"
|
||||
assert _map_reasoning_to_effort("minimal", ("low", "medium", "high")) == "low"
|
||||
assert _map_reasoning_to_effort("none", ("low", "medium", "high")) is None
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
@@ -2704,19 +2820,19 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_search_model_capability(self) -> None:
|
||||
"""Search models should have supports_web_search=True."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
assert caps.supports_web_search is True
|
||||
|
||||
def test_non_search_model_no_web_search(self) -> None:
|
||||
"""Regular models should not have supports_web_search."""
|
||||
caps = self.provider.get_capabilities("gpt-5")
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
assert caps.supports_web_search is False
|
||||
caps = self.provider.get_capabilities("gpt-5.2")
|
||||
caps = lookup_openai_capabilities("gpt-5.2")
|
||||
assert caps.supports_web_search is False
|
||||
|
||||
def test_apply_web_search_injects_options(self) -> None:
|
||||
"""For search models, web_search_options should be added to kwargs."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {"model": "gpt-5-search-api"}
|
||||
tools: list[dict[str, Any]] = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run bash"}},
|
||||
@@ -2733,7 +2849,7 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_apply_web_search_no_op_for_regular_models(self) -> None:
|
||||
"""For non-search models, no web_search_options, tools unchanged."""
|
||||
caps = self.provider.get_capabilities("gpt-5")
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
kwargs: dict[str, Any] = {"model": "gpt-5"}
|
||||
tools: list[dict[str, Any]] = [
|
||||
{"type": "function", "function": {"name": "web_search", "description": "Search"}},
|
||||
@@ -2744,7 +2860,7 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_apply_web_search_returns_none_when_only_web_search(self) -> None:
|
||||
"""If web_search was the only tool, return None after removing it."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {}
|
||||
tools: list[dict[str, Any]] = [
|
||||
{"type": "function", "function": {"name": "web_search", "description": "Search"}},
|
||||
@@ -2758,7 +2874,7 @@ class TestOpenAIWebSearch:
|
||||
toolset) must NOT gain native search — the option stays off and the
|
||||
tools pass through untouched. Contrast test_apply_web_search_with_
|
||||
no_tools, which covers the tool-less utility-call case."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
assert caps.supports_web_search is True
|
||||
kwargs: dict[str, Any] = {"model": "gpt-5-search-api"}
|
||||
tools: list[dict[str, Any]] = [
|
||||
@@ -2833,7 +2949,7 @@ class TestOpenAIWebSearch:
|
||||
visibility set, coordinator toolset, or a tool-less utility call —
|
||||
must not gain native search at the provider layer.
|
||||
"""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {}
|
||||
result = self.provider._apply_web_search(kwargs, caps, None)
|
||||
assert "web_search_options" not in kwargs
|
||||
@@ -2841,7 +2957,7 @@ class TestOpenAIWebSearch:
|
||||
|
||||
def test_apply_web_search_replaces_client_def(self) -> None:
|
||||
"""With the client def present, it is filtered and the option set."""
|
||||
caps = self.provider.get_capabilities("gpt-5-search-api")
|
||||
caps = lookup_openai_capabilities("gpt-5-search-api")
|
||||
kwargs: dict[str, Any] = {}
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}]
|
||||
result = self.provider._apply_web_search(kwargs, caps, tools)
|
||||
@@ -2867,6 +2983,10 @@ class TestOpenAIWebSearch:
|
||||
"function": {"name": "web_search", "description": "Search"},
|
||||
},
|
||||
],
|
||||
# The local lane resolves no commercial rows — the search
|
||||
# model's capabilities ride in explicitly, as the session
|
||||
# layer would pass them.
|
||||
capabilities=lookup_openai_capabilities("gpt-5-search-api"),
|
||||
)
|
||||
)
|
||||
call_kwargs = client.chat.completions.create.call_args[1]
|
||||
@@ -3287,22 +3407,18 @@ class TestAnthropicToolSearch:
|
||||
|
||||
|
||||
class TestOpenAIToolSearch:
|
||||
"""Test OpenAI provider tool search injection."""
|
||||
"""Test OpenAI tool search injection (registry rows + shared helper)."""
|
||||
|
||||
@pytest.fixture()
|
||||
def provider(self):
|
||||
return OpenAIProvider()
|
||||
|
||||
def test_tool_search_capability_on_gpt54(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5.4")
|
||||
def test_tool_search_capability_on_gpt54(self):
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
assert caps.supports_tool_search is True
|
||||
|
||||
def test_tool_search_not_supported_on_gpt5(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5")
|
||||
def test_tool_search_not_supported_on_gpt5(self):
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
assert caps.supports_tool_search is False
|
||||
|
||||
def test_apply_tool_search_marks_deferred(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5.4")
|
||||
def test_apply_tool_search_marks_deferred(self):
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
tools = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run commands"}},
|
||||
{
|
||||
@@ -3318,16 +3434,16 @@ class TestOpenAIToolSearch:
|
||||
# slack tool deferred
|
||||
assert result[1]["defer_loading"] is True
|
||||
|
||||
def test_apply_tool_search_no_op_without_deferred(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5.4")
|
||||
def test_apply_tool_search_no_op_without_deferred(self):
|
||||
caps = lookup_openai_capabilities("gpt-5.4")
|
||||
tools = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run commands"}},
|
||||
]
|
||||
result = apply_tool_search(caps, tools, None)
|
||||
assert result == tools
|
||||
|
||||
def test_apply_tool_search_no_op_on_unsupported_model(self, provider):
|
||||
caps = provider.get_capabilities("gpt-5")
|
||||
def test_apply_tool_search_no_op_on_unsupported_model(self):
|
||||
caps = lookup_openai_capabilities("gpt-5")
|
||||
tools = [
|
||||
{"type": "function", "function": {"name": "bash", "description": "Run commands"}},
|
||||
]
|
||||
@@ -3395,13 +3511,12 @@ class TestVisionCapabilities:
|
||||
assert caps.supports_vision is False
|
||||
|
||||
def test_openai_commercial_supports_vision(self) -> None:
|
||||
provider = OpenAIProvider()
|
||||
for model in ("gpt-5", "gpt-5-mini", "gpt-5.4", "o3", "o4-mini"):
|
||||
caps = provider.get_capabilities(model)
|
||||
caps = lookup_openai_capabilities(model)
|
||||
assert caps.supports_vision is True, f"{model} should support vision"
|
||||
|
||||
def test_openai_default_no_vision(self) -> None:
|
||||
"""Unknown models (local servers) default to no vision."""
|
||||
"""Local-lane models (any name) default to no vision."""
|
||||
provider = OpenAIProvider()
|
||||
caps = provider.get_capabilities("some-local-model")
|
||||
assert caps.supports_vision is False
|
||||
@@ -3626,8 +3741,9 @@ class TestAnthropicPromptCaching:
|
||||
)
|
||||
assert kwargs["output_config"] == {"effort": "xhigh"}
|
||||
|
||||
def test_xhigh_effort_not_applied_to_opus_4_6(self) -> None:
|
||||
"""xhigh is not a valid effort level for Opus 4.6 — should be ignored."""
|
||||
def test_xhigh_effort_snaps_to_max_on_opus_4_6(self) -> None:
|
||||
"""Opus 4.6 declares (low, medium, high, max) — a knob of xhigh
|
||||
rounds up to max instead of silently dropping output_config."""
|
||||
caps = self.provider.get_capabilities("claude-opus-4-6")
|
||||
kwargs = self.provider._build_thinking_and_kwargs(
|
||||
caps=caps,
|
||||
@@ -3640,7 +3756,7 @@ class TestAnthropicPromptCaching:
|
||||
model="claude-opus-4-6",
|
||||
tools=None,
|
||||
)
|
||||
assert "output_config" not in kwargs
|
||||
assert kwargs["output_config"] == {"effort": "max"}
|
||||
|
||||
@patch("turnstone.core.providers._anthropic._ensure_anthropic")
|
||||
def test_streaming_message_start_cache_metrics(self, mock_ensure: MagicMock) -> None:
|
||||
@@ -4173,7 +4289,10 @@ class TestResponsesParamBuilding:
|
||||
assert kwargs["reasoning"] == {"effort": "high"}
|
||||
assert "reasoning_effort" not in kwargs
|
||||
|
||||
def test_no_reasoning_when_none_effort(self) -> None:
|
||||
def test_none_effort_sends_declared_none_level(self) -> None:
|
||||
"""gpt-5.4 declares an explicit "none" level — the knob's off
|
||||
position forwards it rather than omitting (omission would leave
|
||||
the server default in charge on models like gpt-5.5)."""
|
||||
kwargs = self.provider._build_kwargs(
|
||||
model="gpt-5.4",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
@@ -4183,7 +4302,7 @@ class TestResponsesParamBuilding:
|
||||
reasoning_effort="none",
|
||||
deferred_names=None,
|
||||
)
|
||||
assert "reasoning" not in kwargs
|
||||
assert kwargs["reasoning"] == {"effort": "none"}
|
||||
|
||||
def test_store_is_false(self) -> None:
|
||||
kwargs = self.provider._build_kwargs(
|
||||
|
||||
+67
-21
@@ -57,6 +57,39 @@ def _auth(
|
||||
return {"Authorization": f"Bearer {_make_jwt(user, scopes=scopes, permissions=permissions)}"}
|
||||
|
||||
|
||||
class TestAssignableScopes:
|
||||
"""``service`` scope is a cross-tenant bypass and must never be
|
||||
GRANTED via a user-facing token mint (admin API or CLI) — otherwise an
|
||||
``admin.users`` holder could self-mint it and see every private
|
||||
project's workstreams. Both mint paths route through
|
||||
:func:`reject_unassignable_scopes`."""
|
||||
|
||||
def test_service_scope_rejected(self) -> None:
|
||||
from turnstone.core.auth import reject_unassignable_scopes
|
||||
|
||||
assert reject_unassignable_scopes("service") is not None
|
||||
assert reject_unassignable_scopes("read,service") is not None
|
||||
assert reject_unassignable_scopes("read,write,approve,service") is not None
|
||||
|
||||
def test_service_not_in_assignable_set(self) -> None:
|
||||
from turnstone.core.auth import ASSIGNABLE_SCOPES, VALID_SCOPES
|
||||
|
||||
assert "service" in VALID_SCOPES # still a valid runtime scope
|
||||
assert "service" not in ASSIGNABLE_SCOPES # but not user-assignable
|
||||
|
||||
def test_ordinary_scopes_accepted(self) -> None:
|
||||
from turnstone.core.auth import reject_unassignable_scopes
|
||||
|
||||
assert reject_unassignable_scopes("read") is None
|
||||
assert reject_unassignable_scopes("read,write,approve") is None
|
||||
|
||||
def test_empty_and_unknown_rejected(self) -> None:
|
||||
from turnstone.core.auth import reject_unassignable_scopes
|
||||
|
||||
assert reject_unassignable_scopes("") is not None
|
||||
assert reject_unassignable_scopes("bogus") is not None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FakeUI / FakeSession doubles — match the shape the create handler expects
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -88,18 +121,19 @@ class _FakeUI:
|
||||
self._ws_turn_tool_calls = 0
|
||||
self._llm_verdicts: dict[str, dict[str, Any]] = {}
|
||||
|
||||
def serialize_pending_approval_detail(self) -> dict[str, Any] | None:
|
||||
# Mirrors SessionUIBase.serialize_pending_approval_detail —
|
||||
def serialize_pending_approval_details(self) -> list[dict[str, Any]]:
|
||||
# Mirrors SessionUIBase.serialize_pending_approval_details —
|
||||
# the fake is monkeypatched in for ``WebUI`` and the dashboard
|
||||
# handler reads this method during projection. Real subclasses
|
||||
# inherit from ``SessionUIBase``; the fake replicates the
|
||||
# shape directly to stay decoupled.
|
||||
# iterate their approval-cycle registry (one entry per live
|
||||
# cycle); the fake models a single slot, so the list carries
|
||||
# zero or one entries.
|
||||
pending = self._pending_approval
|
||||
if pending is None:
|
||||
return None
|
||||
return []
|
||||
items = pending.get("items") or []
|
||||
if not items:
|
||||
return None
|
||||
return []
|
||||
call_ids = [item.get("call_id", "") for item in items]
|
||||
# Match the real impl's pattern (session_ui_base.py): snapshot
|
||||
# references under the lock, copy after release. Writers only
|
||||
@@ -133,11 +167,14 @@ class _FakeUI:
|
||||
# fake here keeps test-vs-prod behavioural drift from
|
||||
# masking a real-shape regression.
|
||||
primary = next((cid for cid in call_ids if cid), "")
|
||||
return {
|
||||
"call_id": primary,
|
||||
"judge_pending": bool(pending.get("judge_pending", False)),
|
||||
"items": serialized,
|
||||
}
|
||||
return [
|
||||
{
|
||||
"cycle_id": pending.get("cycle_id", ""),
|
||||
"call_id": primary,
|
||||
"judge_pending": bool(pending.get("judge_pending", False)),
|
||||
"items": serialized,
|
||||
}
|
||||
]
|
||||
|
||||
def serialize_recent_auto_approvals(self) -> list[dict[str, Any]]:
|
||||
# Empty buffer for tests that don't exercise the auto-approve
|
||||
@@ -671,28 +708,35 @@ class TestDashboardTrustedTeamVisibility:
|
||||
owners = {w["user_id"] for w in data["workstreams"]}
|
||||
assert {"user-a", "user-b"}.issubset(owners)
|
||||
|
||||
def test_dashboard_pending_approval_detail_default_none(self, app_client):
|
||||
"""No pending approval → field is explicitly null on the wire so
|
||||
consumers can distinguish "not present" from "absent key"."""
|
||||
def test_dashboard_pending_approval_details_default_empty(self, app_client):
|
||||
"""No pending approval → the list field is explicitly empty on
|
||||
the wire so consumers can distinguish "nothing pending" from
|
||||
"absent key". Replaces 1.6's singular ``pending_approval_detail``
|
||||
null (breaking, 1.7)."""
|
||||
client, _mgr = app_client
|
||||
client.post("/v1/api/workstreams/new", json={"name": "a"}, headers=_auth("user-a"))
|
||||
resp = client.get("/v1/api/dashboard", headers=_auth("user-a"))
|
||||
assert resp.status_code == 200
|
||||
rows = resp.json()["workstreams"]
|
||||
assert len(rows) == 1
|
||||
assert "pending_approval_detail" in rows[0]
|
||||
assert rows[0]["pending_approval_detail"] is None
|
||||
assert "pending_approval_details" in rows[0]
|
||||
assert rows[0]["pending_approval_details"] == []
|
||||
# The 1.6 singular field is GONE, not null — a consumer still
|
||||
# reading it should break loudly, not read None forever.
|
||||
assert "pending_approval_detail" not in rows[0]
|
||||
|
||||
def test_dashboard_pending_approval_detail_merges_judge_verdict(self, app_client):
|
||||
def test_dashboard_pending_approval_details_merge_judge_verdict(self, app_client):
|
||||
"""When _pending_approval is set on a ws's UI, /dashboard
|
||||
embeds the merged items + judge_verdict so coord live-bulk
|
||||
callers can render inline approve/deny buttons."""
|
||||
embeds one detail entry per live cycle with merged items +
|
||||
judge_verdict so coord live-bulk callers can render inline
|
||||
approve/deny buttons."""
|
||||
client, mgr = app_client
|
||||
client.post("/v1/api/workstreams/new", json={"name": "a"}, headers=_auth("user-a"))
|
||||
ws_id = next(iter(mgr.list_all())).id
|
||||
ui = mgr.get(ws_id).ui
|
||||
ui._pending_approval = {
|
||||
"type": "approve_request",
|
||||
"cycle_id": "cyc-1",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
@@ -714,8 +758,10 @@ class TestDashboardTrustedTeamVisibility:
|
||||
resp = client.get("/v1/api/dashboard", headers=_auth("user-a"))
|
||||
assert resp.status_code == 200
|
||||
row = next(w for w in resp.json()["workstreams"] if w["ws_id"] == ws_id)
|
||||
detail = row["pending_approval_detail"]
|
||||
assert detail is not None
|
||||
details = row["pending_approval_details"]
|
||||
assert len(details) == 1
|
||||
detail = details[0]
|
||||
assert detail["cycle_id"] == "cyc-1"
|
||||
assert detail["call_id"] == "c-1"
|
||||
assert detail["judge_pending"] is False
|
||||
item = detail["items"][0]
|
||||
|
||||
+41
-17
@@ -208,7 +208,9 @@ class TestMergeServerCompat:
|
||||
|
||||
|
||||
class TestEndToEndRequestShaping:
|
||||
"""Compose both layers — session builds extra_params, provider applies thinking."""
|
||||
"""Compose both layers — session builds extra_params, provider applies reasoning."""
|
||||
|
||||
provider = OpenAIChatCompletionsProvider()
|
||||
|
||||
def test_vllm_gemma_full_flow(self) -> None:
|
||||
"""Gemma now needs only the thinking param — no server workaround."""
|
||||
@@ -216,18 +218,27 @@ class TestEndToEndRequestShaping:
|
||||
server_compat = {"server_type": "vllm"}
|
||||
# Step 1: session forwards (no auto-injection of reasoning_effort).
|
||||
extra_params = merge_server_compat(None, server_compat)
|
||||
# Step 2: provider injects thinking param into chat_template_kwargs.
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
# Step 2: provider folds the effort knob into chat_template_kwargs.
|
||||
extra_body = self.provider._finalize_extra_body(extra_params, caps, "medium")
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"enable_thinking": True}}
|
||||
|
||||
def test_knob_none_disables_toggle(self) -> None:
|
||||
"""Session effort "none" turns the template toggle off dynamically."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, {"server_type": "vllm"}), caps, "none"
|
||||
)
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"enable_thinking": False}}
|
||||
|
||||
def test_server_workaround_composes_with_thinking(self) -> None:
|
||||
"""A top-level server workaround forwards alongside the injected thinking param."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
compat = {"server_type": "llama.cpp", "extra_body": {"reasoning_format": "auto"}}
|
||||
extra_body = dict(merge_server_compat(None, compat))
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, compat), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {
|
||||
"chat_template_kwargs": {"enable_thinking": True},
|
||||
@@ -237,20 +248,20 @@ class TestEndToEndRequestShaping:
|
||||
def test_granite_thinking_key(self) -> None:
|
||||
"""Granite uses 'thinking' instead of 'enable_thinking'."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="thinking")
|
||||
extra_params = merge_server_compat(None, {})
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, {}), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"thinking": True}}
|
||||
|
||||
def test_non_thinking_model_no_injection(self) -> None:
|
||||
"""Non-thinking model gets no chat_template_kwargs at all."""
|
||||
"""Non-thinking model gets no extra_body at all."""
|
||||
caps = ModelCapabilities() # thinking_mode="none"
|
||||
extra_params = merge_server_compat(None, {})
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, {}), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {}
|
||||
assert extra_body is None
|
||||
|
||||
def test_operator_reasoning_effort_passthrough(self) -> None:
|
||||
"""Operator-supplied reasoning_effort under chat_template_kwargs is preserved."""
|
||||
@@ -259,9 +270,9 @@ class TestEndToEndRequestShaping:
|
||||
"server_type": "vllm",
|
||||
"extra_body": {"chat_template_kwargs": {"reasoning_effort": "high"}},
|
||||
}
|
||||
extra_params = merge_server_compat(None, compat)
|
||||
extra_body = dict(extra_params)
|
||||
OpenAIChatCompletionsProvider._apply_thinking_mode(extra_body, caps)
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, compat), caps, "medium"
|
||||
)
|
||||
|
||||
assert extra_body == {
|
||||
"chat_template_kwargs": {
|
||||
@@ -270,6 +281,19 @@ class TestEndToEndRequestShaping:
|
||||
},
|
||||
}
|
||||
|
||||
def test_operator_pin_beats_effort_param(self) -> None:
|
||||
"""A pinned effort key wins over the knob mapping (setdefault)."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
compat = {"extra_body": {"chat_template_kwargs": {"reasoning_effort": "low"}}}
|
||||
extra_body = self.provider._finalize_extra_body(
|
||||
merge_server_compat(None, compat), caps, "high"
|
||||
)
|
||||
|
||||
assert extra_body == {"chat_template_kwargs": {"reasoning_effort": "low"}}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Probe integration: suggest_profile called from _detect_openai_compat
|
||||
|
||||
+395
-51
@@ -272,18 +272,19 @@ class TestChatSessionConstruction:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — _exec_task (optional skill substitutes the hardcoded identity)
|
||||
# Tests — _exec_task (identity from persona/default; skill = capability turn)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestTaskExec:
|
||||
"""Tests for _exec_task: optional skill= replaces the default persona,
|
||||
but operating guidance (one-shot, tool-use over narration, no follow-ups)
|
||||
is always preserved."""
|
||||
"""Tests for _exec_task: identity comes from ``persona=`` (or the default
|
||||
task-agent identity), NEVER the skill; a ``skill=`` rides a distinct
|
||||
capability turn. Operating guidance (one-shot, tool-use over narration,
|
||||
no follow-ups) always layers on top."""
|
||||
|
||||
@staticmethod
|
||||
def _capture_exec_messages(session, item):
|
||||
"""Run _exec_task with _run_agent patched; return system message text."""
|
||||
def _capture_exec_turns(session, item):
|
||||
"""Run _exec_task with _run_agent patched; return the turns list."""
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
@@ -292,27 +293,22 @@ class TestTaskExec:
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
return captured["messages"][0].text
|
||||
return captured["messages"]
|
||||
|
||||
def test_known_skill_renders_into_system_message(self, tmp_db) -> None:
|
||||
"""Validated skill content (with template vars resolved) replaces
|
||||
the default '# Task Agent' persona, but the operating guidance
|
||||
(the numbered list) is preserved — those are sub-agent semantics
|
||||
that a persona should layer on top of, not replace.
|
||||
|
||||
Covers the full prepare→exec round-trip so a future regression
|
||||
in either half (skill not stored on the item, or exec ignoring it)
|
||||
is caught."""
|
||||
def test_skill_delivered_as_capability_turn_not_identity(self, tmp_db) -> None:
|
||||
"""A skill= is CAPABILITY, not identity: its body (template vars
|
||||
resolved) rides a distinct turn AFTER the system message, while the
|
||||
default '# Task Agent' identity + operating guidance stay in the
|
||||
system message. Covers the full prepare→exec round-trip."""
|
||||
session = _make_session()
|
||||
skill = {
|
||||
"name": "research",
|
||||
"content": "# Research Agent\nws={{ws_id}} model={{model}} node={{node_id}}",
|
||||
"content": "# Research Skill\nws={{ws_id}} model={{model}} node={{node_id}}",
|
||||
}
|
||||
with patch("turnstone.core.session.get_skill_by_name", return_value=skill):
|
||||
item = session._prepare_task("c1", {"prompt": "investigate X", "skill": "research"})
|
||||
|
||||
# Item carries the minimized projection — name/content/risk_level
|
||||
# only — not the raw prompt_templates row.
|
||||
# Item carries the minimized projection — name/content/risk_level only.
|
||||
assert item["skill"] == {
|
||||
"name": "research",
|
||||
"content": skill["content"],
|
||||
@@ -321,35 +317,302 @@ class TestTaskExec:
|
||||
assert item.get("needs_approval") is True
|
||||
assert "skill: research" in item["header"]
|
||||
|
||||
sys_msg = self._capture_exec_messages(session, item)
|
||||
# Skill persona rendered with template vars resolved
|
||||
assert "# Research Agent" in sys_msg
|
||||
assert f"ws={session._ws_id}" in sys_msg
|
||||
assert f"model={session.model}" in sys_msg
|
||||
# Default persona is gone — skill substitutes for it.
|
||||
assert "# Task Agent" not in sys_msg
|
||||
assert "autonomous task agent with full tool access" not in sys_msg
|
||||
# Operating guidance survives regardless of skill.
|
||||
turns = self._capture_exec_turns(session, item)
|
||||
sys_msg = turns[0].text
|
||||
# Identity stays the DEFAULT — the skill does NOT become identity.
|
||||
assert ChatSession._TASK_DEFAULT_IDENTITY in sys_msg
|
||||
assert "# Task Agent" in sys_msg
|
||||
assert ChatSession._TASK_OPERATING_GUIDANCE in sys_msg
|
||||
# Skill body is NOT fused into the identity system message.
|
||||
assert "# Research Skill" not in sys_msg
|
||||
# It rides a distinct capability turn, template vars resolved.
|
||||
capability = turns[1].text
|
||||
assert ChatSession._TASK_SKILL_CAPABILITY_PREAMBLE in capability
|
||||
assert "# Research Skill" in capability
|
||||
assert f"ws={session._ws_id}" in capability
|
||||
assert f"model={session.model}" in capability
|
||||
# Task prompt is the final turn.
|
||||
assert turns[-1].text == "investigate X"
|
||||
|
||||
def test_omitted_skill_uses_hardcoded_identity(self, tmp_db) -> None:
|
||||
"""Regression guard: without skill=, the default '# Task Agent'
|
||||
persona AND the operating guidance both appear verbatim.
|
||||
|
||||
Pins the no-skill path so the substitution branch can't
|
||||
accidentally swallow the default case."""
|
||||
def test_omitted_skill_uses_default_identity(self, tmp_db) -> None:
|
||||
"""Without skill= or persona=, the default '# Task Agent' identity +
|
||||
operating guidance appear in the system message, and there is NO
|
||||
capability turn — just system + prompt."""
|
||||
session = _make_session()
|
||||
item = session._prepare_task("c1", {"prompt": "do x"})
|
||||
|
||||
assert item["skill"] is None
|
||||
assert item["persona"] == ""
|
||||
assert "skill:" not in item["header"]
|
||||
assert "persona:" not in item["header"]
|
||||
|
||||
sys_msg = self._capture_exec_messages(session, item)
|
||||
turns = self._capture_exec_turns(session, item)
|
||||
sys_msg = turns[0].text
|
||||
assert ChatSession._TASK_DEFAULT_IDENTITY in sys_msg
|
||||
assert ChatSession._TASK_OPERATING_GUIDANCE in sys_msg
|
||||
# Default-persona literals also present (sanity check on the constant).
|
||||
assert "# Task Agent" in sys_msg
|
||||
assert "autonomous task agent with full tool access" in sys_msg
|
||||
# No skill → no capability turn: just system + prompt.
|
||||
assert len(turns) == 2
|
||||
assert turns[-1].text == "do x"
|
||||
|
||||
def test_persona_sets_identity_skill_stays_capability(self, tmp_db) -> None:
|
||||
"""persona= sets the sub-agent identity (base prompt) in place of the
|
||||
default; a skill passed alongside stays a capability turn."""
|
||||
session = _make_session()
|
||||
persona_row = {
|
||||
"name": "engineer",
|
||||
"base_prompt": "# Engineer\nYou are an engineer.",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
skill = {"name": "research", "content": "# Research Skill"}
|
||||
with (
|
||||
patch("turnstone.core.session.get_skill_by_name", return_value=skill),
|
||||
patch("turnstone.core.session.get_storage") as gs,
|
||||
):
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task(
|
||||
"c1", {"prompt": "do x", "skill": "research", "persona": "engineer"}
|
||||
)
|
||||
|
||||
assert item.get("needs_approval") is True
|
||||
assert item["persona"] == "engineer"
|
||||
assert "persona: engineer" in item["header"]
|
||||
assert "skill: research" in item["header"]
|
||||
|
||||
turns = self._capture_exec_turns(session, item)
|
||||
sys_msg = turns[0].text
|
||||
# Identity = persona, not the default and not the skill.
|
||||
assert "# Engineer" in sys_msg
|
||||
assert ChatSession._TASK_DEFAULT_IDENTITY not in sys_msg
|
||||
assert "# Research Skill" not in sys_msg
|
||||
# Operating guidance still layers on the persona identity.
|
||||
assert ChatSession._TASK_OPERATING_GUIDANCE in sys_msg
|
||||
# Skill remains a capability turn.
|
||||
assert "# Research Skill" in turns[1].text
|
||||
|
||||
def test_unknown_persona_returns_error(self, tmp_db) -> None:
|
||||
"""Unknown persona name → clean error item, no approval."""
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = None
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "ghost"})
|
||||
assert item.get("needs_approval") is False
|
||||
assert "ghost" in item["error"]
|
||||
assert "Omit `persona`" in item["error"]
|
||||
|
||||
def test_persona_wrong_kind_returns_error(self, tmp_db) -> None:
|
||||
"""A coordinator-only persona can't serve as a task-agent identity."""
|
||||
session = _make_session()
|
||||
coord_row = {
|
||||
"name": "orchestrator",
|
||||
"base_prompt": "# Orchestrator",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["coordinator"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = coord_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "orchestrator"})
|
||||
assert item.get("needs_approval") is False
|
||||
assert "interactive" in item["error"]
|
||||
|
||||
def test_persona_tool_allowlist_restricts_sub_agent_tools(self, tmp_db) -> None:
|
||||
"""A restrictive persona caps the sub-agent's TOOLS (Principle 7 /
|
||||
review fix), not just its identity text — stated identity must match
|
||||
granted authority."""
|
||||
session = _make_session()
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "write_file"}},
|
||||
{"function": {"name": "bash"}},
|
||||
]
|
||||
persona_row = {
|
||||
"name": "readonly",
|
||||
"base_prompt": "# Readonly reviewer",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": ["read_file", "search"], # excludes write_file/bash
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "edit auth", "persona": "readonly"})
|
||||
assert item["persona_tools"] == frozenset({"read_file", "search"})
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
tool_names = {t["function"]["name"] for t in captured["tools"]}
|
||||
# write_file + bash dropped by the persona; read_file kept (search was
|
||||
# never in the task tool set to begin with).
|
||||
assert tool_names == {"read_file"}
|
||||
|
||||
def test_no_persona_keeps_full_task_tools(self, tmp_db) -> None:
|
||||
session = _make_session()
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "bash"}},
|
||||
]
|
||||
item = session._prepare_task("c1", {"prompt": "do x"})
|
||||
assert item["persona_tools"] is None
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file", "bash"}
|
||||
|
||||
def test_parent_persona_caps_sub_agent_tools(self, tmp_db) -> None:
|
||||
"""A restricted PARENT session must not escalate authority by spawning:
|
||||
the sub-agent's tools are capped by the parent's own persona grant even
|
||||
with NO child persona (Principle 7 — delegation narrows, never widens;
|
||||
whole-PR review fix)."""
|
||||
session = _make_session()
|
||||
session._tool_search = None
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "write_file"}},
|
||||
{"function": {"name": "bash"}},
|
||||
]
|
||||
# Parent runs under a read-only persona.
|
||||
session._persona_tools = frozenset({"read_file", "search"})
|
||||
|
||||
item = session._prepare_task("c1", {"prompt": "edit auth"})
|
||||
assert item["persona_tools"] is None # no CHILD persona
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
# Parent's read-only grant caps the sub-agent: write_file + bash dropped.
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file"}
|
||||
|
||||
def test_child_persona_mcp_off_drops_mcp_tools(self, tmp_db) -> None:
|
||||
"""A child persona with mcp_enabled=False hides MCP tools (``mcp__*`` and
|
||||
the MCP-access read_resource / use_prompt) from the sub-agent, even when
|
||||
tool_allowlist is null (unrestricted native tools) — the mcp lever must
|
||||
not silently no-op on the task_agent path (whole-PR review fix)."""
|
||||
session = _make_session()
|
||||
session._tool_search = None
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "read_resource"}},
|
||||
{"function": {"name": "use_prompt"}},
|
||||
{"function": {"name": "mcp__github__search"}},
|
||||
]
|
||||
persona_row = {
|
||||
"name": "sandboxed",
|
||||
"base_prompt": "# Sandboxed",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None, # null = unrestricted native tools
|
||||
"mcp_enabled": False, # but MCP is OFF
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "sandboxed"})
|
||||
assert item["persona_mcp"] is False
|
||||
assert item["persona_tools"] is None
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
# MCP tools shed; native read_file kept.
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file"}
|
||||
|
||||
def test_child_persona_memory_off_drops_memory_tool(self, tmp_db) -> None:
|
||||
"""A child persona with memory_enabled=False drops the memory tool from
|
||||
the sub-agent's hands (lever 4), matching a main session under the same
|
||||
persona (whole-PR review fix)."""
|
||||
session = _make_session()
|
||||
session._tool_search = None
|
||||
session._task_tools = [
|
||||
{"function": {"name": "read_file"}},
|
||||
{"function": {"name": "memory"}},
|
||||
]
|
||||
persona_row = {
|
||||
"name": "nomem",
|
||||
"base_prompt": "# No memory",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": False,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "nomem"})
|
||||
assert item["persona_memory"] is False
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
def fake_run_agent(messages, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return "done"
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
session._exec_task(item)
|
||||
assert {t["function"]["name"] for t in captured["tools"]} == {"read_file"}
|
||||
|
||||
def test_evaluate_intent_projects_persona_for_task_agent(self, tmp_db, monkeypatch) -> None:
|
||||
"""Judge/audit projection includes the persona name (review fix): a
|
||||
persona-driven identity shift must be visible to policy + audit, like
|
||||
spawn_workstream."""
|
||||
session = _make_session()
|
||||
fake_verdict = MagicMock()
|
||||
fake_verdict.to_dict.return_value = {"verdict_id": "v0", "tier": "heuristic"}
|
||||
fake_judge = MagicMock()
|
||||
fake_judge.evaluate.side_effect = lambda items, *_a, **_kw: [fake_verdict] * len(items)
|
||||
fake_judge.arg_budget_chars.return_value = 200_000
|
||||
monkeypatch.setattr(session, "_ensure_judge", lambda: fake_judge)
|
||||
|
||||
persona_row = {
|
||||
"name": "engineer",
|
||||
"base_prompt": "# Engineer",
|
||||
"base_prompt_file": None,
|
||||
"tool_allowlist": None,
|
||||
"mcp_enabled": True,
|
||||
"memory_enabled": True,
|
||||
"enabled": True,
|
||||
"applies_to_kinds": ["interactive"],
|
||||
}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_persona_by_name.return_value = persona_row
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "persona": "engineer"})
|
||||
session._evaluate_intent([item])
|
||||
assert item["func_args"]["persona"] == "engineer"
|
||||
|
||||
@pytest.mark.parametrize("skill_value", ["", " ", "\t\n"])
|
||||
def test_prepare_task_empty_or_whitespace_skill_treated_as_omitted(
|
||||
@@ -397,12 +660,11 @@ class TestTaskExec:
|
||||
# tell them apart at recovery time.
|
||||
assert "unknown skill" not in item["error"]
|
||||
|
||||
def test_prepare_task_high_risk_skill_surfaces_in_header(self, tmp_db, caplog) -> None:
|
||||
"""High/critical risk skills surface the tier in the approval header
|
||||
and emit a structured warning, mirroring the signal ``_load_skills``
|
||||
emits for session-level skills (session.py:1336)."""
|
||||
import logging
|
||||
|
||||
def test_prepare_task_denies_high_risk_skill(self, tmp_db) -> None:
|
||||
"""High/critical-risk skills are PRINCIPAL-load-only: task_agent(skill=…)
|
||||
DENIES them — the same gate skills(load) / spawn_* enforce, so a model
|
||||
cannot route around it by delegating activation to a sub-agent
|
||||
(whole-PR review fix — task_agent was the un-gated surface)."""
|
||||
session = _make_session()
|
||||
risky_skill = {
|
||||
"name": "danger",
|
||||
@@ -410,16 +672,15 @@ class TestTaskExec:
|
||||
"enabled": True,
|
||||
"risk_level": "critical",
|
||||
}
|
||||
with (
|
||||
caplog.at_level(logging.WARNING, logger="turnstone.core.session"),
|
||||
patch("turnstone.core.session.get_skill_by_name", return_value=risky_skill),
|
||||
):
|
||||
with patch("turnstone.core.session.get_skill_by_name", return_value=risky_skill):
|
||||
item = session._prepare_task("c1", {"prompt": "do x", "skill": "danger"})
|
||||
assert item.get("needs_approval") is True
|
||||
assert "skill: danger" in item["header"]
|
||||
assert "risk: critical" in item["header"]
|
||||
warning_seen = any("high_risk_skill" in r.getMessage() for r in caplog.records)
|
||||
assert warning_seen, "expected task_agent.high_risk_skill warning"
|
||||
assert item.get("needs_approval") is False
|
||||
assert "principal-load-only" in item["header"]
|
||||
assert "/skill danger" in item["error"]
|
||||
# Distinct from the unknown/disabled errors so the model's recovery
|
||||
# path can tell them apart.
|
||||
assert "unknown skill" not in item["error"]
|
||||
assert "disabled" not in item["error"]
|
||||
|
||||
def test_prepare_task_normal_risk_skill_omits_tier_from_header(self, tmp_db) -> None:
|
||||
"""Header only surfaces high/critical — low/medium/safe skills don't
|
||||
@@ -537,6 +798,89 @@ class TestTaskExec:
|
||||
captured[0](fake_verdict) # must not raise
|
||||
session.ui.on_intent_verdict.assert_not_called()
|
||||
|
||||
def test_evaluate_intent_agent_gate_owns_generation_off_the_main_slot(
|
||||
self, tmp_db, monkeypatch
|
||||
) -> None:
|
||||
"""Sub-agent gates run the SAME judge pipeline as the main loop
|
||||
but as their OWN generation (release blocker #1: task_agent
|
||||
calls used to reach the gate judge-blind). The main-loop
|
||||
supersede slot stays untouched — with parallel task agents,
|
||||
publishing into it would make every sibling's verdicts look
|
||||
stale to the previous sibling's callback — while the generation
|
||||
is stamped on the items for the UI's origin checks, registered
|
||||
for ``close()``'s sweep, delivered alongside the verdict, and
|
||||
grounded on the SUB-AGENT's trajectory (its task prompt is the
|
||||
delegation contract), not the parent conversation."""
|
||||
import threading
|
||||
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.trajectory import turns_from_dicts
|
||||
|
||||
class _GateUI(SessionUIBase):
|
||||
pass
|
||||
|
||||
session = _make_session()
|
||||
ui = _GateUI(ws_id="ws-gate", user_id="u1")
|
||||
ui.on_intent_verdict = MagicMock() # shadow: capture delivery kwargs
|
||||
session.ui = ui
|
||||
|
||||
captured: dict[str, Any] = {}
|
||||
fake_verdict = MagicMock()
|
||||
fake_verdict.to_dict.return_value = {"verdict_id": "v0", "call_id": "c1", "tier": "llm"}
|
||||
fake_judge = MagicMock()
|
||||
|
||||
def _eval(items, convo, **kw):
|
||||
captured["convo"] = convo
|
||||
captured["callback"] = kw.get("callback")
|
||||
captured["cancel_event"] = kw.get("cancel_event")
|
||||
captured["done"] = kw.get("done_callback")
|
||||
return [fake_verdict] * len(items)
|
||||
|
||||
fake_judge.evaluate.side_effect = _eval
|
||||
monkeypatch.setattr(session, "_ensure_judge", lambda: fake_judge)
|
||||
|
||||
main_slot = threading.Event()
|
||||
session._judge_cancel_event = main_slot
|
||||
agent_turns = turns_from_dicts([{"role": "user", "content": "Task: reindex the docs tree"}])
|
||||
item = {"call_id": "c1", "func_name": "bash", "needs_approval": True, "command": "ls"}
|
||||
|
||||
ev = session._evaluate_intent([item], conversation=agent_turns, agent_gate=True)
|
||||
|
||||
assert ev is not None and ev is not main_slot
|
||||
# Main-loop slot untouched by the sub-agent spawn.
|
||||
assert session._judge_cancel_event is main_slot
|
||||
# Generation stamped for the UI's origin checks + close() sweep,
|
||||
# and handed to the daemon as its cancel event.
|
||||
assert item["_judge_event"] is ev
|
||||
assert ev in session._judge_cancel_events
|
||||
assert captured["cancel_event"] is ev
|
||||
# Judge grounded on the sub-agent trajectory, not session.messages.
|
||||
assert any("reindex the docs tree" in str(m) for m in captured["convo"])
|
||||
# Delivery rides the generation into the UI.
|
||||
captured["callback"](fake_verdict)
|
||||
assert ui.on_intent_verdict.call_args.kwargs.get("judge_event") is ev
|
||||
# Daemon completion keeps the close()-sweep set exact.
|
||||
captured["done"]()
|
||||
assert ev not in session._judge_cancel_events
|
||||
|
||||
def test_close_fires_agent_gate_judge_generations(self, tmp_db, monkeypatch) -> None:
|
||||
"""``close()`` aborts EVERY in-flight judge daemon — including
|
||||
sub-agent generations that never touched the main slot — so a
|
||||
torn-down session can't leave daemons running against a dead
|
||||
UI."""
|
||||
session = _make_session()
|
||||
fake_verdict = MagicMock()
|
||||
fake_verdict.to_dict.return_value = {"verdict_id": "v0", "call_id": "c1", "tier": "llm"}
|
||||
fake_judge = MagicMock()
|
||||
fake_judge.evaluate.side_effect = lambda items, *_a, **_kw: [fake_verdict] * len(items)
|
||||
monkeypatch.setattr(session, "_ensure_judge", lambda: fake_judge)
|
||||
|
||||
item = {"call_id": "c1", "func_name": "bash", "needs_approval": True, "command": "ls"}
|
||||
ev = session._evaluate_intent([item], conversation=[], agent_gate=True)
|
||||
assert ev is not None and not ev.is_set()
|
||||
session.close()
|
||||
assert ev.is_set()
|
||||
|
||||
def _drive_gate(self, session, monkeypatch, *, cancel_on_approval: bool):
|
||||
"""Run one needs_approval bash item through ``_execute_tools`` with a
|
||||
stubbed judge + approval gate; return the cancel event the judge
|
||||
|
||||
+732
-180
File diff suppressed because it is too large
Load Diff
@@ -49,6 +49,7 @@ _ESM_BUNDLES = [
|
||||
_SHARED / "composer_queue.js",
|
||||
_SHARED / "interactive.js",
|
||||
_SHARED / "conversation.js",
|
||||
_SHARED / "redact_credentials.js",
|
||||
]
|
||||
|
||||
# Sink scan: everything except renderer.js — the one sanctioned HTML-string
|
||||
@@ -68,6 +69,7 @@ _ESM_NO_VAR_BUNDLES = [
|
||||
_SHARED / "auth.js",
|
||||
_SHARED / "interactive.js",
|
||||
_SHARED / "conversation.js",
|
||||
_SHARED / "redact_credentials.js",
|
||||
]
|
||||
|
||||
# The same unsafe DOM-write / dynamic-code sink set that ``test_app_js.py``
|
||||
@@ -444,7 +446,8 @@ def test_shell_bridges_setrowbadge_for_classic_subsystems() -> None:
|
||||
badge the same way the gear deletion did)."""
|
||||
body = _SHELL_JS.read_text(encoding="utf-8")
|
||||
assert 'setRowBadge } from "./rail.js"' in body, "shell must import setRowBadge from rail.js"
|
||||
assert "notifySessionClosed, setRowBadge }" in body, (
|
||||
ts_shell = body[body.index("window.TS_SHELL = {") :][:200]
|
||||
assert "setRowBadge" in ts_shell, (
|
||||
"TS_SHELL must expose setRowBadge for classic subsystems (the consent-badge bridge)"
|
||||
)
|
||||
|
||||
@@ -1011,7 +1014,8 @@ def test_shell_closes_pane_on_ws_closed() -> None:
|
||||
assert 'pm.getPane("interactive", wsId)' in shell
|
||||
assert "if (p) pm.close(p.id)" in shell, "ws_closed closes the pane, not mark-dead"
|
||||
assert "showDeadBanner" in shell, "the banner lane must survive for non-closed deaths"
|
||||
assert "window.TS_SHELL = { panes: pm, caps, notifySessionClosed, setRowBadge }" in shell, (
|
||||
ts_shell = shell[shell.index("window.TS_SHELL = {") :][:200]
|
||||
assert "panes: pm" in ts_shell and "notifySessionClosed" in ts_shell, (
|
||||
"the seam must be exported on TS_SHELL for the console's Tier-1 handler"
|
||||
)
|
||||
app = _CONSOLE_APP.read_text(encoding="utf-8")
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
"""Tests for turnstone.eval skill-adherence measurement mode.
|
||||
|
||||
Two levels, neither requires a live model:
|
||||
|
||||
* ``TestSkillComposition`` is the load-bearing plumbing proof — it seeds a
|
||||
named skill, builds ``HeadlessSession`` under natural composition, and
|
||||
asserts the skill body folds into ``system_messages`` for the treatment
|
||||
arm and is absent for the control arm. This is what makes the two arms
|
||||
measure different things.
|
||||
* ``TestAdherenceLift`` unit-tests ``run_skill_adherence``'s lift math with
|
||||
the per-arm runner stubbed out.
|
||||
"""
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
from collections.abc import Iterator
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from openai import OpenAI
|
||||
|
||||
from turnstone.core.storage import get_storage, init_storage, reset_storage
|
||||
from turnstone.eval import core
|
||||
from turnstone.eval.core import HeadlessSession, run_skill_adherence
|
||||
|
||||
_SKILL = {
|
||||
"name": "search-first",
|
||||
"content": (
|
||||
"# Search First\n\nBefore answering ANY question about where something "
|
||||
"lives in the codebase, you MUST call the `search` tool first. "
|
||||
"SENTINEL_SKILL_BODY_MARKER."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_storage() -> Iterator[None]:
|
||||
"""Fresh sqlite storage in a temp dir, torn down after the test."""
|
||||
workdir = tempfile.mkdtemp(prefix="turnstone_skill_test_")
|
||||
reset_storage()
|
||||
init_storage("sqlite", path=os.path.join(workdir, ".eval.db"), run_migrations=False)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
reset_storage()
|
||||
import shutil
|
||||
|
||||
shutil.rmtree(workdir, ignore_errors=True)
|
||||
|
||||
|
||||
def _seed_skill(skill: dict[str, str]) -> None:
|
||||
"""Seed a named skill exactly as the runner does."""
|
||||
get_storage().create_prompt_template(
|
||||
template_id="eval-skill",
|
||||
name=skill["name"],
|
||||
category="eval",
|
||||
content=skill["content"],
|
||||
variables="[]",
|
||||
is_default=False,
|
||||
org_id="",
|
||||
created_by="eval",
|
||||
activation="named",
|
||||
enabled=True,
|
||||
)
|
||||
|
||||
|
||||
def _system_text(session: HeadlessSession) -> str:
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
class TestSkillComposition:
|
||||
"""Prove the treatment/control arms compose different system messages."""
|
||||
|
||||
def test_treatment_folds_skill_into_system(self, temp_storage: None) -> None:
|
||||
_seed_skill(_SKILL)
|
||||
client = OpenAI(base_url="http://localhost:9/v1", api_key="dummy")
|
||||
session = HeadlessSession(client=client, model="test-model")
|
||||
try:
|
||||
# Treatment arm activates the seeded skill via the real path.
|
||||
session.set_skill(_SKILL["name"])
|
||||
assert "SENTINEL_SKILL_BODY_MARKER" in _system_text(session)
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_control_omits_skill(self, temp_storage: None) -> None:
|
||||
# Control arm: no skill seeded, no set_skill — natural default only.
|
||||
client = OpenAI(base_url="http://localhost:9/v1", api_key="dummy")
|
||||
session = HeadlessSession(client=client, model="test-model")
|
||||
try:
|
||||
assert "SENTINEL_SKILL_BODY_MARKER" not in _system_text(session)
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_no_system_prompt_override_in_skill_mode(self, temp_storage: None) -> None:
|
||||
# skill_mode must NOT override the base identity — a real base prompt
|
||||
# (persona / composed developer message) must survive, or we'd be
|
||||
# measuring an empty prompt instead of the identity under test.
|
||||
client = OpenAI(base_url="http://localhost:9/v1", api_key="dummy")
|
||||
session = HeadlessSession(client=client, model="test-model")
|
||||
try:
|
||||
assert _system_text(session).strip(), "expected a composed base prompt"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestAdherenceLift:
|
||||
"""Unit-test the lift math with the per-arm runner stubbed."""
|
||||
|
||||
def test_lift_treatment_over_control(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# Stub _run_iteration: treatment (skill != None) passes 3/3, control
|
||||
# (skill is None) passes 1/3. run_skill_adherence must report the
|
||||
# difference as the lift.
|
||||
def fake_run_iteration(**kwargs: Any) -> dict[str, Any]:
|
||||
rate = 1.0 if kwargs.get("skill") is not None else 1.0 / 3.0
|
||||
return {"aggregate": {"overall_pass_rate": rate}}
|
||||
|
||||
monkeypatch.setattr(core, "_run_iteration", fake_run_iteration)
|
||||
|
||||
cases = [
|
||||
{
|
||||
"id": "search-first",
|
||||
"skill": _SKILL,
|
||||
"user_prompt": "where is X?",
|
||||
"expected_actions": [{"tool": "search"}],
|
||||
}
|
||||
]
|
||||
result = run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=3,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
|
||||
assert len(result["cases"]) == 1
|
||||
row = result["cases"][0]
|
||||
assert row["case_id"] == "search-first"
|
||||
assert row["skill"] == "search-first"
|
||||
assert row["treatment_rate"] == pytest.approx(1.0)
|
||||
assert row["control_rate"] == pytest.approx(1.0 / 3.0)
|
||||
assert row["lift"] == pytest.approx(2.0 / 3.0)
|
||||
assert row["n_runs"] == 3
|
||||
assert result["mean_lift"] == pytest.approx(2.0 / 3.0)
|
||||
|
||||
def test_rejects_malformed_skill(self) -> None:
|
||||
# A skill missing 'content' (or 'name') fails fast with a clear error,
|
||||
# not a KeyError mid-run (Copilot review). Validation raises before any
|
||||
# arm runs, so no _run_iteration stub is needed.
|
||||
cases = [
|
||||
{
|
||||
"id": "bad-skill",
|
||||
"skill": {"name": "x"}, # missing 'content'
|
||||
"user_prompt": "do x",
|
||||
"expected_actions": [{"tool": "search"}],
|
||||
}
|
||||
]
|
||||
with pytest.raises(ValueError, match="non-empty 'name' and 'content'"):
|
||||
run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=1,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
|
||||
def test_skipped_when_no_skill(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# A case with no skill is not measurable — it must be skipped, not
|
||||
# crash, and must not contribute to the mean.
|
||||
def fake_run_iteration(**kwargs: Any) -> dict[str, Any]:
|
||||
return {"aggregate": {"overall_pass_rate": 1.0}}
|
||||
|
||||
monkeypatch.setattr(core, "_run_iteration", fake_run_iteration)
|
||||
|
||||
cases = [{"id": "no-skill", "user_prompt": "hi", "expected_actions": []}]
|
||||
result = run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=3,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
assert result["cases"] == []
|
||||
assert result["mean_lift"] == 0.0
|
||||
|
||||
def test_mean_lift_averages_multiple_cases(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# Two skill cases with different lifts average into mean_lift.
|
||||
rates = iter([1.0, 0.0, 1.0, 0.5]) # t1, c1, t2, c2 -> lifts 1.0, 0.5
|
||||
|
||||
def fake_run_iteration(**kwargs: Any) -> dict[str, Any]:
|
||||
return {"aggregate": {"overall_pass_rate": next(rates)}}
|
||||
|
||||
monkeypatch.setattr(core, "_run_iteration", fake_run_iteration)
|
||||
|
||||
cases = [
|
||||
{"id": "a", "skill": _SKILL, "user_prompt": "q", "expected_actions": []},
|
||||
{"id": "b", "skill": _SKILL, "user_prompt": "q", "expected_actions": []},
|
||||
]
|
||||
result = run_skill_adherence(
|
||||
client=None,
|
||||
base_url="http://localhost:9/v1",
|
||||
api_key="dummy",
|
||||
model="test-model",
|
||||
cases=cases,
|
||||
n_runs=2,
|
||||
temperature=0.7,
|
||||
max_tokens=1024,
|
||||
reasoning_effort="medium",
|
||||
context_window=8192,
|
||||
)
|
||||
assert [c["lift"] for c in result["cases"]] == pytest.approx([1.0, 0.5])
|
||||
assert result["mean_lift"] == pytest.approx(0.75)
|
||||
@@ -135,9 +135,8 @@ def _create_skill(db: Any, skill_id: str, name: str, content: str, **kw: Any) ->
|
||||
|
||||
|
||||
def _sys_content(session: ChatSession) -> str:
|
||||
msgs = [m for m in session.system_messages if m["role"] == "system"]
|
||||
assert msgs
|
||||
return msgs[0]["content"]
|
||||
assert session.system_messages
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,295 @@
|
||||
"""Skill substitution unification (SKILL.md subsystem refactor, step 1).
|
||||
|
||||
Pins the invariant that skill-body placeholder substitution is IDENTICAL
|
||||
across every invocation context. Interactive load, default skills, and
|
||||
``task_agent`` sub-agents all route through
|
||||
``ChatSession._render_skill_body`` — so a skill reading ``$ARGUMENTS`` or
|
||||
``${TURNSTONE_EFFORT}`` resolves the same everywhere, rather than
|
||||
rendering literally on the ``task_agent`` path (which previously ran
|
||||
``_render_template`` alone).
|
||||
|
||||
Also covers the two behaviours the unified path newly guarantees:
|
||||
|
||||
* ``${TURNSTONE_*}`` env vars (canonical) and their ``${CLAUDE_*}``
|
||||
back-compat aliases both resolve.
|
||||
* ``${TURNSTONE_SKILL_DIR}`` resolves to the concrete materialized bundle
|
||||
path in the rendered body, because resources are materialized BEFORE
|
||||
substitution (the ordering fix).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from tests._session_helpers import make_session
|
||||
from turnstone.core.storage._registry import get_storage
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import pytest
|
||||
|
||||
|
||||
def _create_skill(db: Any, skill_id: str, name: str, content: str, **kw: Any) -> None:
|
||||
db.create_prompt_template(
|
||||
template_id=skill_id,
|
||||
name=name,
|
||||
category=kw.get("category", "general"),
|
||||
content=content,
|
||||
variables="[]",
|
||||
is_default=kw.get("is_default", False),
|
||||
org_id="",
|
||||
created_by="test",
|
||||
origin="manual",
|
||||
mcp_server="",
|
||||
readonly=False,
|
||||
description="",
|
||||
tags="[]",
|
||||
source_url="",
|
||||
version="1.0.0",
|
||||
author="",
|
||||
activation=kw.get("activation", "named"),
|
||||
token_estimate=0,
|
||||
model="",
|
||||
auto_approve=False,
|
||||
temperature=None,
|
||||
reasoning_effort="",
|
||||
max_tokens=None,
|
||||
token_budget=0,
|
||||
agent_max_turns=None,
|
||||
notify_on_complete="{}",
|
||||
enabled=True,
|
||||
allowed_tools="[]",
|
||||
priority=0,
|
||||
)
|
||||
|
||||
|
||||
class TestRenderSkillBodySharedPath:
|
||||
"""``_render_skill_body`` is the single substitution path — the one
|
||||
``task_agent`` now calls. With ``substitute_args=True`` (arg-capable
|
||||
invocations: interactive /skill, skills(load)) the spec arg forms
|
||||
resolve; with ``substitute_args=False`` (capability contexts: defaults,
|
||||
task_agent) literal ``$N``/``$ARGUMENTS`` are left untouched. Env vars
|
||||
resolve either way."""
|
||||
|
||||
def test_env_vars_resolve(self, tmp_db: str) -> None:
|
||||
session = make_session(reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body(
|
||||
"id ${TURNSTONE_SESSION_ID} effort ${TURNSTONE_EFFORT}"
|
||||
)
|
||||
assert out == f"id {session._ws_id} effort high"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_claude_aliases_resolve(self, tmp_db: str) -> None:
|
||||
session = make_session(reasoning_effort="low")
|
||||
try:
|
||||
out = session._render_skill_body("id ${CLAUDE_SESSION_ID} effort ${CLAUDE_EFFORT}")
|
||||
assert out == f"id {session._ws_id} effort low"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_arg_capable_path_clears_bare_arguments_when_no_args(self, tmp_db: str) -> None:
|
||||
# Arg-capable invocation (interactive /skill, skills(load)) with no
|
||||
# args → bare $ARGUMENTS clears to empty, per the SKILL.md spec.
|
||||
session = make_session()
|
||||
try:
|
||||
assert session._render_skill_body("before $ARGUMENTS after") == "before after"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_curly_and_spec_passes_both_apply(self, tmp_db: str) -> None:
|
||||
# Legacy ``{{model}}`` AND spec ``${TURNSTONE_EFFORT}`` in one body —
|
||||
# both passes run through the shared path.
|
||||
session = make_session(model="my-model", reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body("model {{model}} effort ${TURNSTONE_EFFORT}")
|
||||
assert out == "model my-model effort high"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_arg_capable_path_clears_positional_tokens_when_no_args(self, tmp_db: str) -> None:
|
||||
# Arg-capable path with no args → positional forms clear to empty (spec).
|
||||
session = make_session()
|
||||
try:
|
||||
out = session._render_skill_body("step $1 / $0 / $ARGUMENTS[2] done")
|
||||
assert out == "step / / done"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_capability_context_preserves_literal_arg_tokens(self, tmp_db: str) -> None:
|
||||
# Capability contexts (task_agent, defaults) never receive invocation
|
||||
# args, so substitute_args=False leaves literal $ARGUMENTS/$N/$name
|
||||
# untouched (they are prose/shell text) while env vars still resolve.
|
||||
# Pins the review fix that stopped blanking such tokens for sub-agents.
|
||||
session = make_session(reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body(
|
||||
"run ./deploy.sh $1 $2 at ${TURNSTONE_EFFORT}; process $ARGUMENTS",
|
||||
substitute_args=False,
|
||||
)
|
||||
assert out == "run ./deploy.sh $1 $2 at high; process $ARGUMENTS"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_skill_dir_literal_on_sub_agent_path(self, tmp_db: str) -> None:
|
||||
# task_agent calls _render_skill_body with no skill_dir (sub-agent
|
||||
# bundles aren't materialized yet), so ${TURNSTONE_SKILL_DIR} stays
|
||||
# literal on this path — unchanged from before, resolved in a later
|
||||
# step. The env vars that DO have values still resolve.
|
||||
session = make_session(reasoning_effort="high")
|
||||
try:
|
||||
out = session._render_skill_body(
|
||||
"dir ${TURNSTONE_SKILL_DIR} effort ${TURNSTONE_EFFORT}"
|
||||
)
|
||||
assert out == "dir ${TURNSTONE_SKILL_DIR} effort high"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestSkillDirResolvesInBody:
|
||||
"""Materialize-before-substitute: ``${TURNSTONE_SKILL_DIR}`` in a skill
|
||||
body resolves to the concrete on-disk bundle path after a full load."""
|
||||
|
||||
def test_turnstone_skill_dir_in_body(self, tmp_db: str) -> None:
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "dir-skill", "Scripts under ${TURNSTONE_SKILL_DIR}/scripts.")
|
||||
db.create_skill_resource("r1", "s1", "scripts/go.py", "print('x')")
|
||||
|
||||
session = make_session(skill="dir-skill")
|
||||
try:
|
||||
base = session._skill_resources_dir
|
||||
assert base is not None
|
||||
assert session._skill_content == f"Scripts under {base}/scripts."
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_claude_skill_dir_not_aliased_in_body(self, tmp_db: str) -> None:
|
||||
# CLAUDE_SKILL_DIR is NOT a turnstone-owned alias: it stays a literal
|
||||
# placeholder even with a materialized bundle (that name belongs to the
|
||||
# host in bash; turnstone claims neither surface). The canonical
|
||||
# TURNSTONE_SKILL_DIR does resolve.
|
||||
db = get_storage()
|
||||
_create_skill(
|
||||
db,
|
||||
"s1",
|
||||
"dir-alias-skill",
|
||||
"Bundle at ${CLAUDE_SKILL_DIR} vs ${TURNSTONE_SKILL_DIR}",
|
||||
)
|
||||
db.create_skill_resource("r1", "s1", "references/a.md", "# a")
|
||||
|
||||
session = make_session(skill="dir-alias-skill")
|
||||
try:
|
||||
base = session._skill_resources_dir
|
||||
assert base is not None
|
||||
assert session._skill_content == f"Bundle at ${{CLAUDE_SKILL_DIR}} vs {base}"
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_skill_dir_literal_without_resources(self, tmp_db: str) -> None:
|
||||
# No bundled resources → no dir → placeholder stays literal
|
||||
# (graceful degradation), not an empty path.
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "no-res-skill", "Path ${TURNSTONE_SKILL_DIR} here.")
|
||||
|
||||
session = make_session(skill="no-res-skill")
|
||||
try:
|
||||
assert session._skill_resources_dir is None
|
||||
assert session._skill_content == "Path ${TURNSTONE_SKILL_DIR} here."
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestSkillResourceEnvAliases:
|
||||
"""Bash env exposes the materialized bundle dir under turnstone-owned
|
||||
names (``TURNSTONE_SKILL_DIR`` / ``SKILL_RESOURCES_DIR``) unconditionally,
|
||||
and never under the foreign ``CLAUDE_SKILL_DIR`` — that name is the host's,
|
||||
so turnstone leaves it untouched whether or not the host has set it."""
|
||||
|
||||
def test_turnstone_owned_names_present_claude_absent(
|
||||
self, tmp_db: str, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.delenv("CLAUDE_SKILL_DIR", raising=False)
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "env-alias-skill", "content")
|
||||
db.create_skill_resource("r1", "s1", "scripts/t.py", "code")
|
||||
|
||||
session = make_session(skill="env-alias-skill")
|
||||
try:
|
||||
env = session._skill_resource_env()
|
||||
d = session._skill_resources_dir
|
||||
assert env["SKILL_RESOURCES_DIR"] == d
|
||||
assert env["TURNSTONE_SKILL_DIR"] == d
|
||||
# turnstone never supplies CLAUDE_SKILL_DIR (the host's namespace),
|
||||
# even when the host hasn't set it.
|
||||
assert "CLAUDE_SKILL_DIR" not in env
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_host_claude_skill_dir_untouched(
|
||||
self, tmp_db: str, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# turnstone as a node inside Claude Code: the host's CLAUDE_SKILL_DIR
|
||||
# must survive. turnstone never injects CLAUDE_SKILL_DIR into the
|
||||
# extra-env, so scrubbed_env's passthrough keeps the host value.
|
||||
monkeypatch.setenv("CLAUDE_SKILL_DIR", "/host/claude/skill")
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "env-host-skill", "content")
|
||||
db.create_skill_resource("r1", "s1", "scripts/t.py", "code")
|
||||
|
||||
session = make_session(skill="env-host-skill")
|
||||
try:
|
||||
env = session._skill_resource_env()
|
||||
d = session._skill_resources_dir
|
||||
assert env["TURNSTONE_SKILL_DIR"] == d
|
||||
assert env["SKILL_RESOURCES_DIR"] == d
|
||||
assert "CLAUDE_SKILL_DIR" not in env
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
class TestSkillContextPlacement:
|
||||
"""Step 3: an applied skill's body rides its own capability context
|
||||
message (user role), separate from the identity system message — so it
|
||||
never occupies the cached identity prefix or reads as identity, and it
|
||||
does not leak into the task_agent base."""
|
||||
|
||||
def test_skill_body_in_context_message_not_identity(self, tmp_db: str) -> None:
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "place-skill", "PLACEMENT_MARKER body text")
|
||||
|
||||
session = make_session(skill="place-skill")
|
||||
try:
|
||||
msgs = session.system_messages
|
||||
# Identity system message is first, role=system, and skill-free.
|
||||
assert msgs[0]["role"] == "system"
|
||||
assert "PLACEMENT_MARKER" not in msgs[0]["content"]
|
||||
# Skill rides exactly one separate user-role capability message.
|
||||
skill_msgs = [m for m in msgs if m["role"] == "user"]
|
||||
assert len(skill_msgs) == 1
|
||||
assert "PLACEMENT_MARKER" in skill_msgs[0]["content"]
|
||||
# The intro names the active skill so the model knows what it is.
|
||||
assert "place-skill" in skill_msgs[0]["content"]
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_no_skill_no_context_message(self, tmp_db: str) -> None:
|
||||
# No applied skill and no defaults → only the identity system message.
|
||||
session = make_session()
|
||||
try:
|
||||
assert all(m["role"] == "system" for m in session.system_messages)
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
def test_agent_prefix_excludes_skill_context(self, tmp_db: str) -> None:
|
||||
# task_agent base = the identity system block only; the parent's
|
||||
# applied skill does NOT leak into the sub-agent prefix.
|
||||
db = get_storage()
|
||||
_create_skill(db, "s1", "leak-skill", "SHOULD_NOT_LEAK body")
|
||||
|
||||
session = make_session(skill="leak-skill")
|
||||
try:
|
||||
assert len(session._agent_system_messages) == 1
|
||||
assert session._agent_system_messages[0]["role"] == "system"
|
||||
assert "SHOULD_NOT_LEAK" not in session._agent_system_messages[0]["content"]
|
||||
finally:
|
||||
session.close()
|
||||
@@ -111,10 +111,9 @@ def _make_session(**kwargs):
|
||||
|
||||
|
||||
def _sys_content(session: ChatSession) -> str:
|
||||
"""Extract the system message content."""
|
||||
msgs = [m for m in session.system_messages if m["role"] == "system"]
|
||||
assert msgs
|
||||
return msgs[0]["content"]
|
||||
"""Full prompt prefix: identity system message + any skill context message."""
|
||||
assert session.system_messages
|
||||
return "\n".join(m["content"] for m in session.system_messages)
|
||||
|
||||
|
||||
def _create_template(db, template_id, name, content, **kwargs):
|
||||
|
||||
@@ -1444,3 +1444,98 @@ class TestSkillCatalogDisclosure:
|
||||
# human-facing path); the tool-facing path is the new
|
||||
# ``skills(action='find')`` flow.
|
||||
assert "/skill" in content
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — model-initiated load risk gate (design §5.5 / Principle 7)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestSkillsLoadRiskGate:
|
||||
"""A high/critical-risk skill is PRINCIPAL-load-only: the model cannot
|
||||
activate it through skills(action='load') — that path is denied outright,
|
||||
forcing an explicit operator /skill. Lower-risk skills load as before.
|
||||
The scanner-computed risk_level is the gate signal (it already escalates
|
||||
for the auto_approve + allowed_tools authority the create path warns about),
|
||||
so no new schema is needed. The principal path (set_skill via /skill or
|
||||
cli --skill) is not routed through _prepare_skills_load and is not gated."""
|
||||
|
||||
@staticmethod
|
||||
def _load_item(session, name: str, risk: str):
|
||||
row = {"name": name, "risk_level": risk, "enabled": True, "content": "x"}
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = row
|
||||
return session._prepare_skills("c1", {"action": "load", "name": name})
|
||||
|
||||
def test_high_risk_load_denied_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "danger", "high")
|
||||
assert item.get("needs_approval") is False
|
||||
assert "/skill danger" in item["error"]
|
||||
assert "high" in item["error"]
|
||||
|
||||
def test_critical_risk_load_denied_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "nuke", "critical")
|
||||
assert item.get("needs_approval") is False
|
||||
assert "critical" in item["error"]
|
||||
|
||||
def test_low_risk_load_allowed_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "safe", "low")
|
||||
assert item.get("needs_approval") is True
|
||||
assert item["action"] == "load"
|
||||
assert item["name"] == "safe"
|
||||
|
||||
def test_no_risk_load_allowed_for_model(self) -> None:
|
||||
item = self._load_item(_make_session(), "plain", "")
|
||||
assert item.get("needs_approval") is True
|
||||
|
||||
def test_missing_skill_falls_through_to_exec_not_found(self) -> None:
|
||||
# Unknown name → the gate finds no row and passes through; the
|
||||
# not-found error is the exec path's job, so prep still asks for
|
||||
# approval rather than erroring on the gate.
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = None
|
||||
item = session._prepare_skills("c1", {"action": "load", "name": "ghost"})
|
||||
assert item.get("needs_approval") is True
|
||||
|
||||
def test_shared_helper_denies_high_and_critical_only(self) -> None:
|
||||
# The one shared gate used by skills(load) AND the spawn paths, so a
|
||||
# child spawn cannot route around it.
|
||||
session = _make_session()
|
||||
for tier in ("high", "critical"):
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = {
|
||||
"name": "x",
|
||||
"risk_level": tier,
|
||||
}
|
||||
assert "/skill x" in session._high_risk_skill_denied("x")
|
||||
for row in (
|
||||
{"name": "y", "risk_level": "low"},
|
||||
{"name": "y", "risk_level": ""},
|
||||
None,
|
||||
):
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = row
|
||||
assert session._high_risk_skill_denied("y") == ""
|
||||
|
||||
def test_risk_gate_fails_closed_on_storage_error(self) -> None:
|
||||
# A risk gate that can't read the row must DENY, never wave the skill
|
||||
# through (fail closed). Returning a denial (not "") also keeps
|
||||
# spawn_batch's per-row partial-success intact under a storage blip.
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.side_effect = RuntimeError("db down")
|
||||
denial = session._high_risk_skill_denied("z")
|
||||
assert denial != ""
|
||||
assert "z" in denial
|
||||
assert "/skill z" in denial
|
||||
|
||||
def test_risk_gate_fails_closed_on_no_storage(self) -> None:
|
||||
# Storage-unavailable (get_storage() is None) is treated the same as a
|
||||
# lookup fault — DENY, not allow. A gate that can't verify the risk
|
||||
# tier must not wave the skill through (Copilot review nit).
|
||||
session = _make_session()
|
||||
with patch("turnstone.core.session.get_storage", return_value=None):
|
||||
denial = session._high_risk_skill_denied("z")
|
||||
assert denial != ""
|
||||
assert "/skill z" in denial
|
||||
|
||||
@@ -4,6 +4,7 @@ from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
import sqlalchemy as sa
|
||||
|
||||
from turnstone.core.storage._schema import workstreams
|
||||
@@ -576,6 +577,48 @@ class TestSearch:
|
||||
results = backend.search_history_recent(limit=1)
|
||||
assert len(results) == 1
|
||||
|
||||
def test_search_history_survives_oversized_row(self, backend):
|
||||
# A multi-MB row of mostly-unique words: on PostgreSQL its full
|
||||
# tsvector exceeds the 1MB hard limit, which used to abort every
|
||||
# search_history scan ("string is too long for tsvector") — one
|
||||
# giant tool dump silently killed history recall entirely.
|
||||
backend.register_workstream("s1")
|
||||
giant = "gargantuan beacon " + " ".join(f"w{i}" for i in range(300_000))
|
||||
assert len(giant) > 2_000_000
|
||||
backend.save_message("s1", "tool", giant)
|
||||
backend.save_message("s1", "user", "hello world")
|
||||
|
||||
results = backend.search_history("hello")
|
||||
assert any("hello" in str(r[3]) for r in results)
|
||||
|
||||
# The oversized row itself stays findable by its head.
|
||||
results = backend.search_history("gargantuan beacon")
|
||||
assert any("gargantuan" in str(r[3]) for r in results)
|
||||
|
||||
def test_search_history_fts_error_falls_back_to_ilike(self, request, backend, monkeypatch):
|
||||
# PostgreSQL only: a failed FTS statement aborts the connection's
|
||||
# autobegun transaction, and the ILIKE fallback runs on that same
|
||||
# connection — without a rollback first it dies with
|
||||
# InFailedSqlTransaction instead of returning results.
|
||||
if request.config.getoption("--storage-backend") != "postgresql":
|
||||
pytest.skip("exercises PostgreSQL aborted-transaction fallback")
|
||||
|
||||
backend.register_workstream("s1")
|
||||
backend.save_message("s1", "user", "hello fallback world")
|
||||
|
||||
real_execute = sa.engine.Connection.execute
|
||||
|
||||
def failing_fts_execute(self, statement, *args, **kwargs):
|
||||
if "to_tsvector" in str(statement):
|
||||
# A genuine server-side error, so the transaction is aborted
|
||||
# exactly as when to_tsvector rejects a row.
|
||||
return real_execute(self, sa.text("SELECT 1/0"))
|
||||
return real_execute(self, statement, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(sa.engine.Connection, "execute", failing_fts_execute)
|
||||
results = backend.search_history("fallback")
|
||||
assert any("fallback" in str(r[3]) for r in results)
|
||||
|
||||
|
||||
# -- Workstream operations -----------------------------------------------------
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user