mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-14 07:52:25 -06:00
Compare commits
91 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a4c35e9e29 | |||
| 862eb99cdb | |||
| 25b97bebdf | |||
| ee5ca9a242 | |||
| dd8543fce9 | |||
| 667942024f | |||
| 78831bbe91 | |||
| d44d7eb1a8 | |||
| 876c7d8cb3 | |||
| 98823eb769 | |||
| 4d708c30ac | |||
| 6d60ff7634 | |||
| be662c6134 | |||
| 3ef3f24c7f | |||
| db903f482a | |||
| 6aeffd1845 | |||
| a02b093733 | |||
| f311555026 | |||
| 45d95a2c1f | |||
| a2d9d9832a | |||
| ab123c6cfc | |||
| 8ad17666b9 | |||
| 03fc0861a6 | |||
| a22fb2f395 | |||
| cdcd040da2 | |||
| 834d62c9d4 | |||
| 342a77fe5c | |||
| fd7a447ef9 | |||
| 552ee3c590 | |||
| e99d3ee139 | |||
| 4f0fc3f219 | |||
| dc701986f7 | |||
| bedd25fbe7 | |||
| 251a912275 | |||
| d48902fd01 | |||
| 702ac43d0e | |||
| 01f83dc90f | |||
| 2463c480c2 | |||
| 2a3dfbc6fb | |||
| 6c3b3cc098 | |||
| 0dc52f05ee | |||
| 02929c0d00 | |||
| b2add19c56 | |||
| 5ce1873e9e | |||
| 7698a928c5 | |||
| 4e2eea2f86 | |||
| a6752cb645 | |||
| 06ba4e8d4f | |||
| 94dcaf34fd | |||
| 1a2a689033 | |||
| c02f960d0a | |||
| bfde387206 | |||
| 27d112ff60 | |||
| aeab2535b1 | |||
| 4638d22bd0 | |||
| ee3bd1dcf2 | |||
| ae3a83ccce | |||
| 0f17433e1f | |||
| 043554bb2f | |||
| 8389808add | |||
| 6cbef4f633 | |||
| 2b6dde4f7e | |||
| fbe31b9885 | |||
| ef13f40cf5 | |||
| bfa1b104cf | |||
| 324a1d1a35 | |||
| 95ab88ff6f | |||
| d29840f985 | |||
| 44c0b9c340 | |||
| cdbdf3dc2b | |||
| 8aabb061c2 | |||
| 8bd638569f | |||
| 251dc44a46 | |||
| efd0a1d000 | |||
| 2f93c39fd3 | |||
| f27ce104c6 | |||
| 20a61b692b | |||
| 1966107efe | |||
| d5b2fe6e45 | |||
| 1569819750 | |||
| 4107a30148 | |||
| d06d88b83f | |||
| c411aac939 | |||
| 104715b650 | |||
| 59a9899149 | |||
| 4da7c3b91c | |||
| 012f4e3e16 | |||
| 3636724848 | |||
| eeda5ac312 | |||
| be872b840f | |||
| ee94ae8ba1 |
+13
-13
@@ -15,7 +15,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install pre-commit
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install mypy
|
||||
@@ -43,21 +43,21 @@ jobs:
|
||||
python-version: ["3.11", "3.12", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
# Node is required by tests/test_renderer_js.py — without
|
||||
# explicit setup, that suite silently skips if the runner
|
||||
# image happens not to ship Node, masking regressions in
|
||||
# the browser-side renderer.
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: pip install -e ".[test]"
|
||||
# -v lists each test id as it starts (pytest prints the nodeid at
|
||||
# logstart), so a hang names the culprit on the last line instead of
|
||||
# riding the job timeout with only a trail of "..." dots.
|
||||
- run: pytest tests/ -m "not live and not e2e_recovery" --cov=turnstone --cov-report=term-missing --cov-report=xml -v
|
||||
- run: pytest tests/ -m "not live" --cov=turnstone --cov-report=term-missing --cov-report=xml -v
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
if: always()
|
||||
with:
|
||||
@@ -83,14 +83,14 @@ jobs:
|
||||
--health-retries=5
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: pip install -e ".[test]"
|
||||
- run: pytest tests/ -m "not live and not e2e_recovery" --storage-backend=postgresql -v
|
||||
- run: pytest tests/ -m "not live" --storage-backend=postgresql -v
|
||||
env:
|
||||
TURNSTONE_TEST_PG_URL: postgresql+psycopg://postgres:postgres@localhost:5432/turnstone_test
|
||||
|
||||
@@ -98,7 +98,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install build
|
||||
@@ -152,7 +152,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- run: uv lock --check
|
||||
@@ -161,10 +161,10 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: uv sync --frozen --all-extras
|
||||
@@ -189,7 +189,7 @@ jobs:
|
||||
working-directory: sdk/typescript
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: npm ci
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
name: Claude Code Review
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, synchronize, ready_for_review, reopened]
|
||||
# Optional: Only run on specific file changes
|
||||
# paths:
|
||||
# - "src/**/*.ts"
|
||||
# - "src/**/*.tsx"
|
||||
# - "src/**/*.js"
|
||||
# - "src/**/*.jsx"
|
||||
|
||||
jobs:
|
||||
claude-review:
|
||||
if: github.event.pull_request.head.repo.full_name == github.repository
|
||||
# Optional: Filter by PR author
|
||||
# if: |
|
||||
# github.event.pull_request.user.login == 'external-contributor' ||
|
||||
# github.event.pull_request.user.login == 'new-developer' ||
|
||||
# github.event.pull_request.author_association == 'FIRST_TIME_CONTRIBUTOR'
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post the review + inline comments
|
||||
issues: read
|
||||
id-token: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
allowed_bots: 'renovate[bot]' # let Renovate PRs get reviewed
|
||||
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
|
||||
plugins: 'code-review@claude-code-plugins'
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ github.event.pull_request.number }}'
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
name: Claude Code
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
pull_request_review_comment:
|
||||
types: [created]
|
||||
issues:
|
||||
types: [opened, assigned]
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
jobs:
|
||||
claude:
|
||||
if: |
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review' &&
|
||||
contains(github.event.review.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.review.author_association)
|
||||
) || (
|
||||
github.event_name == 'issues' &&
|
||||
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.issue.author_association)
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post comments/reviews when @-mentioned on a PR
|
||||
issues: write # post comments when @-mentioned on an issue
|
||||
id-token: write
|
||||
actions: read # Required for Claude to read CI results on PRs
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
|
||||
# This is an optional setting that allows Claude to read CI results on PRs
|
||||
additional_permissions: |
|
||||
actions: read
|
||||
|
||||
# Optional: Give a custom prompt to Claude. If this is not specified, Claude will perform the instructions specified in the comment that tagged it.
|
||||
# prompt: 'Update the pull request description to include a summary of changes.'
|
||||
|
||||
# Optional: Add claude_args to customize behavior and configuration
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
# claude_args: '--allowed-tools Bash(gh pr *)'
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
echo "skip=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
with:
|
||||
python-version: "3.14"
|
||||
@@ -58,12 +58,12 @@ jobs:
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
- run: python -m build
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
- uses: pypa/gh-action-pypi-publish@ba38be9e461d3875417946c167d0b5f3d385a247 # release/v1
|
||||
- uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # release/v1
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
|
||||
- name: Create GitHub Release
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3
|
||||
uses: softprops/action-gh-release@718ea10b132b3b2eba29c1007bb80653f286566b # v3
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.tag }}
|
||||
generate_release_notes: true
|
||||
|
||||
@@ -32,7 +32,7 @@ jobs:
|
||||
python-version: ["3.11", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- run: pip install -e ".[test,dev]"
|
||||
|
||||
+36
-393
@@ -14,408 +14,51 @@ experimental line:
|
||||
|
||||
Earlier stable lines (`stable/1.6`, `stable/1.5`) are frozen.
|
||||
|
||||
## [Unreleased]
|
||||
## [1.7.4]
|
||||
|
||||
A feature-bearing patch for the 1.7 line, rolling up work that had stabilised
|
||||
on `main`. No schema migrations (head stays 066) and no new configuration knobs.
|
||||
|
||||
### Added
|
||||
|
||||
- **Compaction is visible now: lifecycle events, a progress bar, and a
|
||||
persistent transcript card.** Context compaction (manual `/compact` and
|
||||
auto) emits a first-class `compaction` SSE event
|
||||
(`start` / `progress` / `end` — see the API reference) instead of loose
|
||||
info lines. The web UI renders an in-transcript card with a real progress
|
||||
bar (determinate `part k of N` during chunked summarization, indeterminate
|
||||
for single-call compactions) that settles into a result card — token delta
|
||||
plus the summary behind a fold — in both the interactive pane and the
|
||||
coordinator viewer. The result survives reloads: the persisted compaction
|
||||
marker now projects through `/history` as a `role="system"`,
|
||||
`source="compaction"` entry (resume/export/search unchanged), stamped with
|
||||
the end event's id so repaint and SSE replay can't double-render. The
|
||||
marker's `meta` additionally records `before_tokens` / `after_tokens` /
|
||||
`trigger`. Python and TypeScript SDKs gain a typed `CompactionEvent`.
|
||||
|
||||
- **One provider transport: every model call now streams (#831).**
|
||||
The per-adapter non-streaming entry (`create_completion`) is retired;
|
||||
single-shot lanes — judges, titles, compaction, web-fetch extraction,
|
||||
perception, eval, optimizer — sample through the same streaming entry
|
||||
the chat loop uses and accumulate via one shared drain, so request
|
||||
shaping can no longer drift between the two consumption styles. Two
|
||||
operator-visible consequences: long single-shot generations (a thinking
|
||||
model composing a title, a slow local judge) no longer sit in a single
|
||||
blocking read that can hit client read-timeouts — the same reason the
|
||||
Anthropic adapter already streamed internally — and judge timeouts now
|
||||
*abort* the underlying HTTP read instead of abandoning a worker thread
|
||||
on a dead call. Because every call now streams, an alias pointed at a
|
||||
model or org that cannot stream (OpenAI's verified-org streaming
|
||||
entitlement, a gateway api-version predating `stream_options` — e.g.
|
||||
older Azure OpenAI deployments) fails at request time where 1.7's
|
||||
non-streaming single-shot call succeeded; remediation is on the
|
||||
serving side (verify the org, bump the api-version/gateway) — there is
|
||||
deliberately no per-model non-streaming fallback left to configure. These lanes are also complete-or-error now: a stream
|
||||
that ends without any finish signal is treated as a generation that
|
||||
died mid-response and retried, instead of storing the partial text as
|
||||
a clean result (previously a half-generated compaction summary could
|
||||
silently replace real history). Caveats: these lanes now carry the
|
||||
same `stream_options: {include_usage: true}` the chat loop always
|
||||
sent — OpenAI-compatible servers old enough to *ignore* it stop
|
||||
producing usage rows on these lanes, and servers strict enough to
|
||||
*reject* unknown fields (pre-2024 llama.cpp/proxy builds) will 400 —
|
||||
such a server already couldn't serve turnstone's chat loop, but a
|
||||
judge/utility alias pointed at one worked on 1.7 and needs to move to
|
||||
a current server. Transient mid-stream deaths (connection drop, proxy
|
||||
hiccup) are re-issued in place up to twice with exponential backoff —
|
||||
the retry the SDK's request loop used to provide these lanes
|
||||
invisibly. Each lane accepts its own terminal marker (Anthropic
|
||||
`message_stop`, Responses terminal events); a lax server/gateway that
|
||||
never sends any terminal signal needs
|
||||
`{"finish_reason_optional": true}` in the model definition's
|
||||
capabilities JSON, which restores 1.7's tolerance (clean end-of-stream
|
||||
after output = completion) for that model on every lane — without it
|
||||
such streams fail as died-mid-generation, because SSE gives no way to
|
||||
tell the two apart and the default favors catching truncation. The
|
||||
unread `supports_streaming` capability flag (and its admin tile) is
|
||||
gone; the o-series models it described are dropped from the capability
|
||||
table entirely (see Removed).
|
||||
|
||||
- **One turn interface for every model call: `core/model_turn.py` (#827).**
|
||||
Judges (intent + output guard), perception, title generation, compaction,
|
||||
web-fetch extraction, the eval harness, the optimizer's meta lanes, and
|
||||
task agents all advance a trajectory through the same plant-call
|
||||
primitive the agent seam pioneered — Turn IR in, one shared lowering
|
||||
(argument sanitize → minted-id restore → vLLM reasoning attach), one
|
||||
shared re-ingest (blank-id repair → native-lane finalize). The judges'
|
||||
hand-built OpenAI-dict path is gone, and with it the Gemini judge's
|
||||
tool-blindness: evidence tools now work on Google models because the
|
||||
native lane round-trips `thought_signature` (with pairwise repair for
|
||||
blank-id compat responses). Provider adapters still take lowered wire
|
||||
dicts — the transport collapse and main-loop migration are tracked as
|
||||
#831 / #832.
|
||||
|
||||
- **task_agent keeps its model's reasoning across its own tool loop — on
|
||||
every provider lane.** A task agent's replayed turns now carry the
|
||||
provider-native reasoning lane the model produced — Anthropic thinking
|
||||
blocks with their signatures (commercial or an anthropic-compatible
|
||||
server), OpenAI Responses reasoning items, Gemini `thought_signature`
|
||||
fidelity blocks, and the reasoning text a vLLM `--reasoning-parser` /
|
||||
llama.cpp `reasoning_format` surfaces on the Chat Completions lane —
|
||||
instead of each turn being rebuilt from text + tool calls with the
|
||||
reasoning dropped. On a thinking model this restores reasoning continuity
|
||||
across the agent's own multi-turn tool use. On the wire the agent's
|
||||
session-minted sub-tool ids are mapped back to the provider's own ids
|
||||
(`restore_provider_tool_ids`), so the native block — replayed verbatim,
|
||||
its signature never touched — the `tool_calls` mirror, and each tool
|
||||
result always agree; internally the minted ids still key the live card,
|
||||
recall, and the cancel ledger unchanged. Replay honors the same per-model
|
||||
`replay_reasoning_to_model` flag the main loop uses on every lane: the
|
||||
vLLM Chat-Completions field replay keeps its server-type gate, and
|
||||
llama.cpp stays capture-only, matching main-loop behavior. The native
|
||||
lane is finalized by the same shared builder as the main loop's, so the
|
||||
two harnesses cannot drift.
|
||||
|
||||
- **Background shells: `bash` gains `run_in_background`, plus `bash_output` /
|
||||
`kill_shell`.** Setting `run_in_background=true` starts the command as a
|
||||
detached shell and returns immediately with a `bash_N` handle — "start a dev
|
||||
server, use it in a later call" is back as an explicit opt-in (the shape
|
||||
follows the convention the major coding agents converged on). `bash_output`
|
||||
returns only output produced since the previous read (optionally filtered by
|
||||
a regex) plus status and exit code; `kill_shell` terminates the shell's
|
||||
whole process group. Output is buffered per shell with a drop-oldest cap, so
|
||||
a chatty server can't grow memory unbounded. When a background shell exits,
|
||||
a system notice lands at the next seam (waking an idle workstream if
|
||||
needed). Shells survive a generation cancel, die with the workstream, and
|
||||
never outlive a task_agent that started them; anything a background shell
|
||||
itself backgrounds is still reaped when that shell exits — the no-leak
|
||||
guarantee below is unchanged.
|
||||
- **Background shells for the `bash` tool** — `run_in_background=true` starts a
|
||||
command as a detached shell and returns a `bash_N` handle; new `bash_output`
|
||||
(delta output since last read, optional regex filter, status/exit code) and
|
||||
`kill_shell` (terminates the shell's process group) tools manage it. Output is
|
||||
buffered with a drop-oldest cap, a system notice lands when a shell exits, and
|
||||
shells die with their workstream — never outliving a `task_agent` that started
|
||||
them.
|
||||
- **`task_agent` carries the model's native reasoning across its own tool loop** —
|
||||
a task agent's replayed turns now preserve the provider-native reasoning lane
|
||||
(Anthropic thinking blocks with signatures, OpenAI reasoning items, Gemini
|
||||
`thought_signature`, vLLM/llama.cpp reasoning text) instead of rebuilding each
|
||||
turn from text alone, restoring reasoning continuity for thinking models.
|
||||
- **Model-shelf response controls** — the console model shelf exposes verbosity
|
||||
and reasoning-mode controls per identity.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Breaking (1.8): compaction feedback moved from `info` events to the
|
||||
typed `compaction` SSE event.** Pre-1.8 SSE/SDK clients that ignore
|
||||
unknown event types no longer see compaction lines (they are
|
||||
deliberately not dual-emitted — dual emission would double-render on
|
||||
every current client). Consume the `compaction` lifecycle event (see
|
||||
the API reference and the `CompactionEvent` SDK type); embedders
|
||||
driving `ChatSession` through a duck-typed `SessionUI` are unaffected
|
||||
(the classic `on_info` lines are restored for them — see Fixed).
|
||||
|
||||
- **Sampling knobs (temperature, reasoning effort) now ride one assignment
|
||||
scheme: per-model alias value → operator-stored global setting → the
|
||||
model definition's declared default (effort only) → field omitted.**
|
||||
Turnstone previously manufactured values onto every unconfigured
|
||||
request — a hidden `temperature: 0.5` and a `reasoning_effort: "medium"`
|
||||
baked in at three layers — overriding serving-side defaults like a vLLM
|
||||
model's `generation_config`. Unconfigured installs now send neither
|
||||
field and the inference engine's own defaults rule; `model.temperature`
|
||||
is blank by default ("inherit each model's own default") and
|
||||
`model.reasoning_effort` defaults to the empty "inherit" choice. The
|
||||
per-model → global resolution lives in one shared resolver used by the
|
||||
session factories, the `/model` switch, and every `model_turn` lane, so
|
||||
the same alias samples identically on every surface. CLI
|
||||
`--temperature` / `--reasoning-effort` likewise default to inherit.
|
||||
|
||||
**Upgrade notes:**
|
||||
- The empty (`""`) reasoning-effort choice changed meaning from
|
||||
"explicitly disable thinking" to "inherit the model/serving default".
|
||||
On local manual-thinking models (e.g. Qwen templates with
|
||||
`enable_thinking`), a stored `""` previously sent
|
||||
`enable_thinking: false`; it now sends nothing, so the template's own
|
||||
default (often thinking ON) applies. Use **`none`** to actually
|
||||
disable reasoning.
|
||||
- Workstreams saved by earlier versions carry the old defaults
|
||||
(`temperature=0.5`, `reasoning_effort=medium`) in their persisted
|
||||
config and keep that exact behavior on resume; they pick up the new
|
||||
inherit semantics the next time you change the model or a sampling
|
||||
knob in that workstream. New workstreams inherit from the start.
|
||||
|
||||
### Removed
|
||||
|
||||
- **O-series and pre-5.4 GPT-5 rows dropped from the OpenAI capability
|
||||
table.** `o1`, `o1-mini`, `o3`, `o3-mini`, `o3-pro`, `o4-mini`,
|
||||
`gpt-5`, `gpt-5-mini`, `gpt-5-nano`, `gpt-5-pro`, `gpt-5.1`,
|
||||
`gpt-5.1-codex-max`, `gpt-5.2`, `gpt-5.2-pro`, and `gpt-5.3` no longer
|
||||
have built-in capability rows — OpenAI has retired these model ids
|
||||
from the API, so the rows described contracts no request can reach
|
||||
anymore. The table floor is now `gpt-5.4`; the search-api and
|
||||
audio/STT/TTS rows are unchanged. An alias still pinning a retired id
|
||||
fails at OpenAI itself; any other unlisted commercial id resolves to
|
||||
the generic commercial defaults (temperature sent, no declared
|
||||
reasoning-effort vocabulary, 200K window) — declare the contract on
|
||||
the model definition's capabilities JSON if you run one, or move to a
|
||||
current model.
|
||||
- **GPT-5.6 aligned with the GA API surface** — the Responses provider matches
|
||||
GPT-5.6's GA shape (typed `reasoning.mode`, `prompt_cache_options`,
|
||||
cache-write accounting); the `openai` floor moves to `>=2.45`.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **A failed worker-thread spawn no longer wedges the workstream — at
|
||||
either spawn site — and never masquerades as success.** If
|
||||
`Thread.start()` itself raised (thread exhaustion, out-of-memory), the
|
||||
dispatcher had already claimed the worker slot but the flag's only
|
||||
clearer lived in the never-started thread — the workstream looked idle
|
||||
forever while every subsequent message queued behind a worker that
|
||||
didn't exist, until an operator force-cancel. The claim is now rolled
|
||||
back under the lock and the error propagates, so the workstream is
|
||||
dispatchable again as soon as resources recover. Affected every
|
||||
dispatch path (sends, wakes, retries, deferred-send drain, init). The
|
||||
same failure at the deferred-send drain's own spawn rolls back the
|
||||
just-accepted entry and answers the retryable `queue_full` (previously
|
||||
a 500 landed *after* the entry was registered — an invisible,
|
||||
unretractable phantom that later dispatched as duplicate turns), and a
|
||||
`/command` whose worker never spawned now answers **503**
|
||||
`{"status": "error"}` instead of the generic 200 ok that told SDK
|
||||
callers their `/clear` or `/resume` had applied.
|
||||
|
||||
- **Manual `/compact` from the web UI: no phantom user turn, no frozen
|
||||
server, cancellable.** A slash command typed into the web composer no
|
||||
longer renders as a user chat bubble (it echoes as a distinct command
|
||||
chip — commands aren't conversation turns and were never persisted as
|
||||
such). `/compact` itself now dispatches onto the workstream's worker
|
||||
slot instead of running inline on the server's event loop — previously a
|
||||
long compaction froze every SSE stream on the node for its whole
|
||||
duration, which is also why its own progress only ever arrived as one
|
||||
burst after the fact. The manual path carries `send()`'s full generation
|
||||
discipline (`compact_now()`): a force-abandoned compaction goes stale
|
||||
instead of swapping history under a successor turn — and retires at its
|
||||
next checkpoint instead of running out its remaining summary calls,
|
||||
with its late lifecycle events fenced off (`compaction_id` on every
|
||||
event, `superseded` on end events — both in the SDKs) so they can't
|
||||
animate, tear down, re-title, or falsely narrate a successor's card or
|
||||
activity pill; a cancel aimed at it is consumed on exit (previously it
|
||||
bricked every `/compact` retry until the next message); a Stop click on
|
||||
an idle session can't pre-abort the next compaction; a Stop that lands
|
||||
in the completion tail — after the last cancel check, or during a retry
|
||||
backoff (which now aborts immediately instead of sleeping it out) — is
|
||||
honored rather than silently eaten; and Stop now aborts the in-flight
|
||||
summary HTTP call itself (the compaction lane registers its stream in
|
||||
the same abort seam the main loop uses), so cancelling a compaction is
|
||||
immediate instead of waiting out a model call.
|
||||
|
||||
- **Sends during a command window are deferred, ordered, bounded, and
|
||||
honestly rendered — never silently truncated or lost.** Messages sent
|
||||
while any slash command holds the worker slot are **deferred**: answered
|
||||
`{"status": "queued", "msg_id"}` immediately and dispatched as ordinary
|
||||
full-fidelity sends (attachments and sender identity included) when the
|
||||
command finishes — never routed through the mid-turn interjection
|
||||
queue, whose semantics are turn-shaped: previously a send during a
|
||||
manual `/compact` was silently truncated to 2,000 characters, a second
|
||||
participant in a shared workstream was locked out with a misleading
|
||||
"another participant's turn" 409 for the whole compaction, and a
|
||||
message queued across a `/resume`/`/new` could be answered into the
|
||||
post-swap workstream. Because the response is immediate,
|
||||
timeout-bounded callers — the coordinator's `send_message`, the console
|
||||
proxy, SDKs, anything behind a stock reverse proxy — can no longer lose
|
||||
a message to a multi-minute command window; the deferred send is
|
||||
retractable until dispatch via the same `DELETE .../send` used for
|
||||
queued interjections (node-local, in-memory — the API reference
|
||||
documents the at-most-once durability contract). Deferred responses
|
||||
carry `"deferred": true`; the pending list is the **order authority**
|
||||
(a fresh send — or a coordinator dispatch, or a queued-nudge wake —
|
||||
lines up behind acknowledged entries instead of overtaking them, with
|
||||
the two-term barrier defined once on the workstream so the wake gate
|
||||
also honors a claimed entry whose dispatch is mid-flight, and the gate
|
||||
re-arms at the drain's exit even when everything pending was
|
||||
retracted); acceptance is **bounded** (10 pending per workstream — the
|
||||
interjection queue's own backpressure contract; the 11th answers the
|
||||
retryable `queue_full` instead of pinning attachment bytes without
|
||||
limit and then running one unattended turn per entry); a dispatch
|
||||
crash re-queues the entry instead of eating an acknowledged message,
|
||||
and a drain thread that fails to *start* rolls the acceptance back and
|
||||
answers `queue_full` rather than parking a phantom the client can
|
||||
neither see nor retract; each dispatch emits a pane-tier
|
||||
`message_dispatched` event (`folded: true` for interjection fold-ins)
|
||||
so queued-bubble UI keeps its retract affordance exactly until the
|
||||
message truly leaves — including when the send was accepted by a pane
|
||||
that believed the workstream idle, which now renders a real queued
|
||||
chip instead of a sent-looking bubble, releases the composer (a
|
||||
deferred send has no running worker to wait on), and cleans up fully
|
||||
when the send is refused or the chip retracted instead of stranding
|
||||
the pane in Stop mode. Dismissing a queued bubble — interjection or
|
||||
deferred — is a server-confirmed `DELETE`, and retracting a deferred
|
||||
send that carried attachments tells the user they were discarded
|
||||
instead of silently expiring them.
|
||||
|
||||
- **Slash commands hold the worker slot with a loud contract.**
|
||||
A `/compact` raced against an in-flight turn is refused with an
|
||||
explicit busy response. Every other slash command runs through the same
|
||||
worker slot too — mutual exclusion against sends, a running compaction,
|
||||
and each other, with a busy answer replacing the old silent interleave —
|
||||
while the endpoint still awaits quick commands' completion off-loop
|
||||
(without parking an executor thread per request); the post-command pane
|
||||
refreshes (`clear_ui` after `/clear`/`/new`/`/resume`, the
|
||||
workstream-name sync) ride the worker itself, so a command that
|
||||
outlives the endpoint's 25s response backstop still refreshes every
|
||||
pane on completion (the backstop sits under the console proxy's 30s
|
||||
client timeout so the degraded `running` answer can actually traverse
|
||||
a proxied pane, which now surfaces it instead of silence; the
|
||||
`/command` response contract — `ok` / `running`, with busy refusals
|
||||
answering a loud HTTP 409 rather than a silent 200 — is now documented
|
||||
in the API reference and the OpenAPI spec).
|
||||
|
||||
- **Compaction status stays truthful across every UI surface.** Manual
|
||||
compaction
|
||||
success also refreshes the status line/context pill immediately (parity
|
||||
with auto-compaction), compaction failures keep feeding the typed
|
||||
`error` event and the node error counter (while a CLI Ctrl-C reports as
|
||||
cancelled, not a failure), one Stop prints one notice (a cancelled
|
||||
auto-compaction no longer stacks "Compaction cancelled." on top of
|
||||
send's own "[Generation cancelled]"), the workstream activity pill
|
||||
shows "Compacting context…" for the whole summarize phase, restores
|
||||
cleanly afterwards, and can no longer be stranded by a force-stopped
|
||||
compaction (a new turn's generation claim breaks a stale latch). Every
|
||||
retry backoff on the session (stream retries, task agents, notify
|
||||
delivery, compaction) now aborts immediately on Stop via one shared
|
||||
cancel-aware helper instead of sleeping out its exponential delay.
|
||||
|
||||
- **Compaction failures report exactly once, to the right owner.** A
|
||||
compaction failure reports
|
||||
exactly once (auto-compaction errors defer to the turn's fatal handler
|
||||
instead of doubling the red row and the error metric), failed-end
|
||||
notice suppression is computed once by the emitter (a `notice` bool on
|
||||
the end event — in the SDKs — replaces hand-synced client policy), and
|
||||
a manual `/compact` failure no longer crashes the CLI REPL. `/compact`
|
||||
on a workstream showing the `error` badge restores the badge on exit
|
||||
instead of stamping `idle` over it (the compaction neither retried nor
|
||||
resolved the failed turn). A force-cancelled initial send that
|
||||
completes late still delivers its scheduled-run completion
|
||||
notification (the only completion signal unattended workstreams have);
|
||||
the other post-command pane refreshes and error notices remain
|
||||
owner-guarded, so a force-cancelled wedged command that unwedges late
|
||||
can't wipe panes or inject stray notices into a successor turn.
|
||||
|
||||
- **Pre-1.8 embedder UIs keep their compaction lines.** Embedders
|
||||
driving `ChatSession` with a pre-1.8 duck-typed `SessionUI`
|
||||
(no `on_compaction` hook) get the classic `on_info` compaction lines
|
||||
back — threshold notice, `part k/N`, retry waits, token delta +
|
||||
summary box — instead of silent history swaps. (See the breaking
|
||||
event-contract note under **Changed** for SSE/SDK clients.)
|
||||
|
||||
- **Static MCP servers: a pushed catalog change no longer wedges the shared
|
||||
session (#839).** The static-path `*/list_changed` handler awaited its
|
||||
catalog refresh inline in the SDK's receive loop, but the refresh's own
|
||||
request can only be answered by that (now parked) loop — the refresh never
|
||||
completed, and every user's in-flight calls on the shared per-node session
|
||||
stalled behind it, unbounded, until the health loop's ping timeout tore the
|
||||
transport down (which was also the only way the changed catalog ever
|
||||
landed). Push refreshes now run as spawned tasks — debounced, coalesced per
|
||||
(server, kind), bounded by the connect timeout, and serialized on the
|
||||
per-server connect lock — and the manual and post-reconnect refreshes
|
||||
publish under that same lock, so a slower publisher can no longer land a
|
||||
staler catalog over a fresher one. Every teardown path now also clears the
|
||||
notification debounce stamp, so a reconnected server's first push refreshes
|
||||
immediately. Push-refresh debouncing is now per (server, kind) on BOTH the
|
||||
static and per-user pool paths — a tools push no longer swallows a prompts
|
||||
push arriving in the same 5-second window. A change genuinely lost to the
|
||||
debounce window (a same-kind push landing after the prior refresh finished,
|
||||
which the server will never re-announce) is recovered by an automatic
|
||||
health-tick retry rather than staying invisible until an unrelated push or
|
||||
a reconnect. The resource-refresh fan-out on both paths no longer orphans
|
||||
its sibling list call when one of the pair fails fast — the real error
|
||||
surfaces immediately (not masked as a 30-second timeout) and the surviving
|
||||
sibling is cancelled and reaped, under a bounded grace, inside the scope. A
|
||||
push refresh that fails while the connection stays up is likewise retried on
|
||||
the next health-loop tick until one completes — previously a single
|
||||
transient blip left the shared catalog stale for every user on the node
|
||||
until an operator intervened. An operator `/mcp refresh` no longer parks
|
||||
behind a busy per-server connect lock (a slow reconnect attempt could eat
|
||||
the whole 30-second refresh budget and fail the pass for every healthy
|
||||
server behind it) — the busy server is skipped on both the connected and
|
||||
disconnected branches, reported distinctly as "skipped" rather than as a
|
||||
false "no changes", the skip arms the automatic retry, and a
|
||||
force-reconnect drops the session up front so queued push refreshes can't
|
||||
starve it. Static-path resource and prompt catalogs are now size-capped
|
||||
like the pool path's (and like static tools) at discovery and on every
|
||||
refresh, so a misbehaving server's push can't balloon the node's merged
|
||||
catalogs. Deleting or reconfiguring a server can no longer leave it
|
||||
half-removed: the config removal and all cleanup are serialized under the
|
||||
connect lock (a cancelled removal completes its cleanup rather than
|
||||
stranding a live session and published catalog with the config already
|
||||
gone), and `reconcile_sync` retries a removal that timed out instead of
|
||||
marking it done — previously a DB-driven delete of a busy server could be a
|
||||
silent, permanent no-op until process restart. A refresh outcome now
|
||||
threads consistently to every operator surface off one source of truth
|
||||
(the per-server `last_refresh_outcome`): a busy-skip and a genuine failure
|
||||
are each reported distinctly from a real "no changes" — `/mcp refresh`
|
||||
prints "skipped" or "failed" rather than a false "no changes", and the
|
||||
node-internal refresh endpoint returns `202 skipped` instead of a
|
||||
misleading `200 ok` for a refresh that never ran. A single-kind push
|
||||
refresh no longer paints the whole server healthy: because the
|
||||
error/outcome state is server-scoped, a successful tools push while the
|
||||
prompts catalog is still broken (or vice versa) no longer clears the
|
||||
failure — only a full refresh pass declares "ok".
|
||||
|
||||
- **OpenAI Responses streaming: truncated and refused responses no longer
|
||||
vanish.** A response that hit `max_output_tokens` terminates the stream
|
||||
with `response.incomplete`, which the stream consumer did not handle —
|
||||
the turn was mislabeled `finish_reason: stop` and its final usage and
|
||||
collected output items were dropped. Refusal parts had no streaming
|
||||
handler at all, so a refusal rendered as empty content instead of the
|
||||
`[Refused: …]` text the non-streaming path produced. Both now match:
|
||||
truncation maps to `length` with usage/items intact, refusals render
|
||||
in content. Applies to the chat loop and every drained single-shot
|
||||
lane (#831).
|
||||
|
||||
- **task_agent: sub-tool ids no longer alias across a local model's reused
|
||||
ids.** A local model that reissues per-response sequential tool-call ids
|
||||
(`call_0` every turn) made two of a task agent's steps share one id — the
|
||||
live card collapsed both onto one DOM row while `/history` recall kept them
|
||||
apart, so the two views disagreed. Sub-tool ids are now minted
|
||||
`{parent}::r{run}s{step}::{id}`, unique within the session (across an
|
||||
agent's turns and across concurrent or sequential runs), and that one id
|
||||
keys the nesting registry, the live rows, recall, and the cancel ledger.
|
||||
On the wire the agent's self-built history carries the provider's own ids,
|
||||
restored from the mint map (see the reasoning-lane entry under Added), and
|
||||
malformed tool-call arguments are legalized the same way the main loop's
|
||||
wire prep does.
|
||||
|
||||
- **bash tool: never hang on a backgrounded child.** A command that left a
|
||||
long-lived process running (`server &`, a daemon) could wedge the whole
|
||||
workstream forever — the tool read stdout/stderr to EOF, which never arrived
|
||||
because the child inherited the pipe, and the timeout watchdog bailed once the
|
||||
foreground `bash` had exited. The tool now waits on the tracked process
|
||||
(bounded by the tool timeout) and terminates its whole process group on
|
||||
return, so the call always completes. Undecodable output is preserved
|
||||
(`errors="replace"`) instead of being dropped as a spurious error.
|
||||
- **Behavior change:** a process the command backgrounds no longer survives
|
||||
the call — nothing persists across bash invocations. (First-class
|
||||
"run this in the background" support landed separately — see
|
||||
`run_in_background` under Added.)
|
||||
- **`bash` never hangs on a backgrounded child** — a command that left a
|
||||
long-lived process running no longer wedges the workstream; the tool waits on
|
||||
the tracked process (bounded by the timeout) and reaps its whole process group.
|
||||
- **`task_agent` sub-tool ids are session-unique** — ids are minted
|
||||
`{parent}::r{run}s{step}::{id}` so a local model reissuing sequential ids
|
||||
(`call_0` each turn) no longer aliases two steps onto one live-card row while
|
||||
`/history` keeps them apart.
|
||||
- **Judge completions honour model-definition capabilities** — a judge's
|
||||
completion now threads its model's declared capabilities instead of assuming a
|
||||
default surface.
|
||||
- **`create-admin` CLI** — adds an explicit admin-creation command; `run.sh` no
|
||||
longer onboards into a role-less user.
|
||||
- **Install script Docker handling** — installs Docker on distros
|
||||
`get.docker.com` rejects, and gates that path by `$ID` instead of trapping all
|
||||
failures.
|
||||
|
||||
## [1.7.3]
|
||||
|
||||
|
||||
+2
-6
@@ -8,7 +8,7 @@ FROM python:3.14-slim
|
||||
LABEL org.opencontainers.image.title="turnstone" \
|
||||
org.opencontainers.image.description="Multi-node AI orchestration platform"
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.29 /uv /usr/local/bin/uv
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.27 /uv /usr/local/bin/uv
|
||||
|
||||
# Remove the slim image's man page exclusion so man-db has actual content
|
||||
RUN rm -f /etc/dpkg/dpkg.cfg.d/docker
|
||||
@@ -60,12 +60,8 @@ COPY docker/entrypoint.sh /usr/local/bin/entrypoint.sh
|
||||
WORKDIR /data
|
||||
RUN chown turnstone:turnstone /data
|
||||
|
||||
# Workspace mount point — bind-mount a host directory here. The env var
|
||||
# surfaces the path in the model's shell/file tool descriptions
|
||||
# (config.get_workspace_dir); without it the mount is invisible to the
|
||||
# model, whose cwd is /data below.
|
||||
# Workspace mount point — bind-mount a host directory here
|
||||
RUN mkdir -p /workspace && chown turnstone:turnstone /workspace
|
||||
ENV TURNSTONE_WORKSPACE=/workspace
|
||||
|
||||
USER turnstone
|
||||
|
||||
|
||||
@@ -7,6 +7,6 @@ appVersion: "0.3.0"
|
||||
|
||||
dependencies:
|
||||
- name: postgresql
|
||||
version: ~18.8.0
|
||||
version: ~18.7.0
|
||||
repository: https://charts.bitnami.com/bitnami
|
||||
condition: postgresql.enabled
|
||||
|
||||
+28
-132
@@ -467,46 +467,6 @@ Each item in `items` (shared by `tool_info` and `approve_request`):
|
||||
{"type": "info", "message": "Session cleared."}
|
||||
```
|
||||
|
||||
**`compaction`** -- context-compaction lifecycle (manual `/compact` and
|
||||
auto-compaction). `phase: "start"` opens the operation (`trigger` is
|
||||
`"manual"` or `"auto"`; auto adds `where` — e.g. `"mid-turn"` — and, when
|
||||
the percentage threshold actually fired, `pct`; the context-overflow retry
|
||||
path compacts without a `pct` since no threshold was evaluated).
|
||||
`phase: "progress"` reports chunked summarization (`part`/`total`/`depth`,
|
||||
where depth 0 summarizes transcript batches and deeper levels merge partial
|
||||
summaries), a transient-error retry wait (`retry_in` seconds + `error`), or
|
||||
`warning: "summary_truncated"`. `phase: "end"` settles it: `ok: true`
|
||||
carries `before_tokens`/`after_tokens` and the produced `summary`;
|
||||
`ok: false` carries a `reason`
|
||||
(`"not_enough_messages"` / `"irreducible"` / `"empty_summary"` /
|
||||
`"cancelled"` / `"error"`) and a human-readable `message` — for
|
||||
`reason: "error"` the same message is also emitted as a paired typed
|
||||
`error` event (that is the renderable error surface; the end event is
|
||||
card-teardown). Failed ends also carry `notice`: the emitter-computed
|
||||
display verdict — show `message` only when it is `true` (the server
|
||||
suppresses error-reason, superseded, and cancelled-auto notices once,
|
||||
centrally, so clients don't re-derive that policy). Every end (ok or
|
||||
failed) carries `trigger`, and every event carries `compaction_id` — an
|
||||
opaque integer correlating the start/progress/end of one compaction run (a
|
||||
client that force-stopped one compaction can use it to ignore stragglers
|
||||
from the abandoned run). End events also carry `superseded`: `true` marks
|
||||
a force-abandoned compaction retiring after a successor generation took
|
||||
over (an OK end's result card still stands: the history swap happened).
|
||||
Superseded start/progress events are never emitted.
|
||||
Exactly one `start` and one `end` are emitted per attempt,
|
||||
so clients can key an in-progress affordance (progress bar) on the pair. A
|
||||
successful end is also persisted: the summary replays from `/history` as a
|
||||
`role: "system"`, `source: "compaction"` entry whose `meta` carries
|
||||
`{watermark, before_tokens, after_tokens, trigger}` and whose `event_id`
|
||||
matches the end event's id (dedup across repaint + replay).
|
||||
|
||||
```json
|
||||
{"type": "compaction", "phase": "start", "compaction_id": 7, "trigger": "auto", "where": "mid-turn", "pct": 80}
|
||||
{"type": "compaction", "phase": "progress", "compaction_id": 7, "part": 2, "total": 5, "depth": 0}
|
||||
{"type": "compaction", "phase": "end", "ok": true, "compaction_id": 7, "trigger": "auto",
|
||||
"before_tokens": 128400, "after_tokens": 9200, "summary": "## Decisions\n..."}
|
||||
```
|
||||
|
||||
**`error`** -- an error message.
|
||||
|
||||
```json
|
||||
@@ -788,41 +748,32 @@ Sends a user message to a workstream. Spawns a daemon worker thread that calls
|
||||
**Request body:**
|
||||
|
||||
```json
|
||||
{"message": "Explain how the server works", "attachment_ids": ["a1"]}
|
||||
{"message": "Explain how the server works"}
|
||||
```
|
||||
|
||||
| Field | Type | Required | Description |
|
||||
|------------------|------------|----------|------------------------------------------------------|
|
||||
| `message` | string | yes | The user's message text |
|
||||
| `attachment_ids` | string[] | no | Staged uploads to attach (omit = auto-consume; `[]` = none) |
|
||||
| Field | Type | Required | Description |
|
||||
|-----------|--------|----------|-------------------------|
|
||||
| `message` | string | yes | The user's message text |
|
||||
|
||||
**Response.** Every 200 body carries `attached_ids` and
|
||||
`dropped_attachment_ids` (empty lists when no attachments are involved):
|
||||
**Response (success):**
|
||||
|
||||
- `{"status": "ok", ...}` — a fresh turn was dispatched.
|
||||
- `{"status": "queued", "priority", "msg_id", ...}` — folded into the live
|
||||
turn's interjection queue; delivered at the next tool-result seam.
|
||||
`DELETE .../send` with the `msg_id` retracts it before delivery.
|
||||
- `{"status": "queued", "deferred": true, ...}` — parked on the deferred-send
|
||||
list (a command window holds the slot, or earlier deferred sends are
|
||||
pending) and dispatched as its own full-fidelity send afterwards; see the
|
||||
defer contract under `POST /v1/api/command`.
|
||||
- `{"status": "queue_full", ...}` — the send was refused with retry-shortly
|
||||
semantics: the live worker's interjection queue is at capacity, the
|
||||
deferred-send list hit its saturation bound (10 pending — the same
|
||||
backpressure contract), or the deferred-send drain could not be started
|
||||
under resource exhaustion (the message was **not** accepted; nothing is
|
||||
parked).
|
||||
- `{"status": "attachments_busy", ...}` — attachments can't ride a queued
|
||||
turn; the staged uploads survive for a retry once the worker idles.
|
||||
```json
|
||||
{"status": "ok"}
|
||||
```
|
||||
|
||||
**Response (busy):** Returned if the workstream's worker thread is still alive
|
||||
from a previous request. Also pushes a `busy_error` event to the SSE stream.
|
||||
|
||||
```json
|
||||
{"status": "busy"}
|
||||
```
|
||||
|
||||
**Error responses:**
|
||||
|
||||
| Status | Body | Condition |
|
||||
|--------|-------------------------------------------------|----------------------------------------|
|
||||
| 400 | `{"error": "message is required"}` | Message is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found (or closed mid-send) |
|
||||
| 409 | `{"status": "cross_user_interjection", ...}` | Another participant's turn is in flight |
|
||||
| Status | Body | Condition |
|
||||
|--------|------------------------------------|------------------------|
|
||||
| 400 | `{"error": "Empty message"}` | Message is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
|
||||
---
|
||||
|
||||
@@ -865,56 +816,7 @@ automatically approved without prompting.
|
||||
|
||||
### `POST /v1/api/command`
|
||||
|
||||
Executes a slash command in the given workstream. Commands run on the
|
||||
workstream's worker slot (mutual exclusion against sends, a running
|
||||
compaction, and each other) — the endpoint is **not** unconditionally
|
||||
synchronous:
|
||||
|
||||
- **Quick commands** (everything except `/compact`): the endpoint waits for
|
||||
completion, so `{"status": "ok"}` means the command ran. A command still
|
||||
running after 25 s answers `{"status": "running"}` — the worker keeps
|
||||
going, its output reaches the pane via SSE, and the post-command pane
|
||||
refreshes below still fire when it completes. (The bound sits under
|
||||
common 30 s client/proxy timeouts — the console proxy's included — so
|
||||
the degraded answer actually reaches bounded callers.)
|
||||
- **`/compact`**: dispatched fire-and-forget — `{"status": "ok"}` means the
|
||||
compaction *started*. A large context can legitimately compact for many
|
||||
minutes; progress streams as `compaction` SSE events (see the event
|
||||
reference) and the persisted marker row lands on completion. Do not read
|
||||
`/history` expecting the compacted transcript immediately after the
|
||||
response.
|
||||
- **Busy refusal**: if a turn or another command holds the worker slot, the
|
||||
command is refused with HTTP **409** `{"status": "busy", "error": ...}` and
|
||||
did **not** run. Retry after the current turn finishes. (The old inline
|
||||
endpoint executed commands unconditionally mid-turn; the 409 makes the
|
||||
refusal loud for callers that only check the HTTP status.)
|
||||
|
||||
While a command holds the slot — and afterwards, while earlier deferred
|
||||
sends are still waiting (the pending list is the order authority: a fresh
|
||||
send never overtakes a message already acknowledged) — `POST .../send`
|
||||
requests are **deferred**: the server answers `{"status": "queued",
|
||||
"deferred": true, "msg_id": ...}` immediately and dispatches the message
|
||||
as an ordinary full-fidelity send (attachments and sender identity
|
||||
included) in arrival order once the slot frees — it is never routed
|
||||
through the mid-turn interjection queue (no length cap, no cross-user
|
||||
rejection). The response arrives within normal round-trip time, so
|
||||
timeout-bounded clients (SDKs, proxies, the coordinator) need no special
|
||||
handling. To retract a deferred send before it dispatches, issue the same
|
||||
`DELETE .../send` with its `msg_id` used for queued interjections —
|
||||
`{"status": "removed"}` confirms it will not dispatch; `"not_found"` means
|
||||
it already dispatched (or is dispatching). Retracting a deferred send
|
||||
discards any attachments it carried; re-attach to send them again. When a
|
||||
deferred send dispatches, panes receive a `message_dispatched` event
|
||||
(`msg_id`, plus `folded: true` when it folded into a live turn's
|
||||
interjection queue rather than spawning its own turn) so queued-message
|
||||
UI can settle the right way.
|
||||
|
||||
Durability: deferred sends are **node-local and in-memory** (the same
|
||||
lifetime as the interjection queue). `"queued"` is at-most-once intake, not
|
||||
durable acceptance — if the workstream is closed or the node restarts before
|
||||
the window ends, the message is dropped. Anything that must survive a
|
||||
restart should be re-sent after confirming dispatch (the turn appears on the
|
||||
SSE stream / in `/history`).
|
||||
Executes a slash command in the given workstream.
|
||||
|
||||
**Request body:**
|
||||
|
||||
@@ -927,12 +829,10 @@ SSE stream / in `/history`).
|
||||
| `command` | string | yes | The slash command (e.g. `/clear`) |
|
||||
| `ws_id` | string | yes | Target workstream ID |
|
||||
|
||||
If the command is `/clear`, `/new`, or `/resume`, the server pushes a
|
||||
`clear_ui` SSE event to instruct the client to reset its message display and
|
||||
re-fetch the transcript via `GET .../history` (there is no SSE event that
|
||||
carries the messages themselves). These follow-ups are emitted by the
|
||||
command worker itself, so they fire even when the endpoint already answered
|
||||
`{"status": "running"}`.
|
||||
If the command is `/clear` or `/new`, the server pushes a `clear_ui` SSE event
|
||||
to instruct the client to reset its message display. If the command is
|
||||
`/resume`, the server pushes `clear_ui` followed by a `history` event
|
||||
containing the resumed session's messages.
|
||||
|
||||
**Response:**
|
||||
|
||||
@@ -940,16 +840,12 @@ command worker itself, so they fire even when the endpoint already answered
|
||||
{"status": "ok"}
|
||||
```
|
||||
|
||||
or `{"status": "running"}` as above.
|
||||
|
||||
**Error responses:**
|
||||
|
||||
| Status | Body | Condition |
|
||||
|--------|-------------------------------------|--------------------------------------------------|
|
||||
| 400 | `{"error": "Empty command"}` | Command is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
| 409 | `{"status": "busy", "error": ...}` | A turn/command holds the worker |
|
||||
| 503 | `{"status": "error", "error": ...}` | The command worker could not be started (resource exhaustion) — the command did **not** run; retry shortly |
|
||||
| Status | Body | Condition |
|
||||
|--------|------------------------------------|----------------------|
|
||||
| 400 | `{"error": "Empty command"}` | Command is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
|
||||
---
|
||||
|
||||
|
||||
+18
-27
@@ -91,7 +91,7 @@ turnstone/
|
||||
discord/ Discord adapter (bot, cog, views, streaming, config)
|
||||
slack/ Slack adapter (Socket Mode bot, DM routing, approval buttons)
|
||||
shared_static/ Shared design system (base.css, auth.js, theme.js, toast.js, utils.js, kb.js)
|
||||
katex-0.18.1/ Vendored KaTeX math rendering library (MIT, woff2 fonts)
|
||||
katex-0.17.0/ Vendored KaTeX math rendering library (MIT, woff2 fonts)
|
||||
ui/
|
||||
colors.py ANSI color constants with NO_COLOR support
|
||||
markdown.py Streaming terminal markdown renderer (line-buffered)
|
||||
@@ -609,7 +609,8 @@ LLMProvider (protocol)
|
||||
|
||||
| Method | Purpose |
|
||||
|--------|---------|
|
||||
| `create_streaming()` | The one transport: streaming request, yields normalized `StreamChunk` objects (single-shot callers accumulate via `drain_stream()` into a `CompletionResult`) |
|
||||
| `create_streaming()` | Streaming request, yields normalized `StreamChunk` objects |
|
||||
| `create_completion()` | Non-streaming request, returns `CompletionResult` |
|
||||
| `get_capabilities()` | Per-model flags (`ModelCapabilities`) |
|
||||
| `convert_tools()` | Translate OpenAI tool schemas to provider format |
|
||||
| `retryable_error_names` | Exception class names that trigger retry |
|
||||
@@ -662,7 +663,7 @@ display). Automatic prompt caching is enabled via top-level `cache_control:
|
||||
cacheable block and advances it as conversations grow (90% input cost
|
||||
reduction on cache hits, 1.25x write on first turn). Cache metrics
|
||||
(`cache_creation_input_tokens`, `cache_read_input_tokens`) are extracted from
|
||||
the stream's usage events. The `anthropic` SDK is a core
|
||||
both streaming and non-streaming responses. The `anthropic` SDK is a core
|
||||
dependency — the Anthropic provider is first-class alongside OpenAI.
|
||||
|
||||
**GoogleProvider** (`_google.py`): extends `OpenAIChatCompletionsProvider` for
|
||||
@@ -1149,30 +1150,20 @@ Named (aliased) workstreams are never age-pruned. Configure with
|
||||
|
||||
### API Retry
|
||||
|
||||
Every model call streams (#831); retry lives at two stacked layers:
|
||||
`ChatSession._create_stream_with_retry()` (streaming path) and the agent
|
||||
`_api_call()` (non-streaming) both use the same retry pattern:
|
||||
|
||||
- **Caller ladders** — `ChatSession._create_stream_with_retry()` (chat
|
||||
loop) and the agent `_api_call()` (drained via `model_turn`) use the
|
||||
same pattern: 4 total attempts (1 initial + 3 retries,
|
||||
`_MAX_RETRIES = 3`), exponential backoff base 1 second
|
||||
(`delay = 1s * 2^attempt`), `ui.on_info()` on retry, exception
|
||||
propagates on final failure. `_compact_messages()` wraps its drained
|
||||
call in the same loop.
|
||||
- **`model_turn`'s drain ladder** — inside every single-shot call,
|
||||
mid-stream deaths (errors raised while draining, e.g.
|
||||
`IncompleteStreamError`) are re-issued up to 2 more times with a
|
||||
0.5s-base exponential backoff (±50% jitter); request-time failures
|
||||
keep the SDK's own retry policy. The two ladders stack
|
||||
multiplicatively on transient-shaped failures.
|
||||
- **Retryable errors** are matched by class name against each
|
||||
provider's `retryable_error_names` (avoids importing
|
||||
backend-specific exception hierarchies): `RateLimitError`,
|
||||
`APITimeoutError`, `APIConnectionError`, `InternalServerError`,
|
||||
`ServiceUnavailableError`, `APIError`, plus the drained-transport
|
||||
errors `IncompleteStreamError` (stream ended with no terminal
|
||||
signal — for servers that never send one, declare
|
||||
`finish_reason_optional` in the model's capabilities JSON) and
|
||||
`ResponsesStreamFailedError` (transient in-band Responses failure).
|
||||
- **Retries**: 4 total attempts (1 initial + 3 retries, `_MAX_RETRIES = 3`)
|
||||
- **Backoff**: exponential, base 1 second (`delay = 1s * 2^attempt`)
|
||||
- **Retryable errors**: `RateLimitError`, `APITimeoutError`,
|
||||
`APIConnectionError`, `InternalServerError`, `ServiceUnavailableError`,
|
||||
`APIError` (matched by class name to avoid importing backend-specific
|
||||
exception hierarchies)
|
||||
- On retry: `ui.on_info()` notification
|
||||
- On final failure: exception propagates
|
||||
|
||||
`_compact_messages()` also wraps its non-streaming API call in the same
|
||||
retry loop.
|
||||
|
||||
### Finish Reason Handling
|
||||
|
||||
@@ -1185,7 +1176,7 @@ Every model call streams (#831); retry lives at two stacked layers:
|
||||
blocked.
|
||||
|
||||
Agent sub-sessions (`_run_agent()`) check `finish_reason` on each
|
||||
drained turn and stop the agent early on `"length"` or
|
||||
non-streaming response and stop the agent early on `"length"` or
|
||||
`"content_filter"`.
|
||||
|
||||
`_compact_messages()` checks `finish_reason` on the compaction response and
|
||||
|
||||
@@ -66,7 +66,8 @@ class "NullUI" as NullUI {
|
||||
interface "LLMProvider" as LLMProvider <<Protocol>> {
|
||||
+ provider_name: str {property}
|
||||
+ get_capabilities(model) → ModelCapabilities
|
||||
+ create_streaming(client, model, messages, ..., cancel_ref, replay_reasoning_to_model) → Iterator[StreamChunk]
|
||||
+ create_streaming(client, model, messages, ..., replay_reasoning_to_model) → Iterator[StreamChunk]
|
||||
+ create_completion(client, model, messages, ..., replay_reasoning_to_model) → CompletionResult
|
||||
+ convert_tools(tools) → list[dict]
|
||||
+ extract_reasoning_text(provider_blocks) → str
|
||||
+ retryable_error_names: frozenset[str] {property}
|
||||
@@ -176,7 +177,7 @@ class "HeadlessSession" as HeadlessSession {
|
||||
+ send_headless(input, max_turns, ...)
|
||||
- _override_system_prompt(content)
|
||||
--
|
||||
eval.py: drained single-shot turns,
|
||||
eval.py: non-streaming,
|
||||
records all tool calls
|
||||
}
|
||||
|
||||
|
||||
@@ -84,8 +84,8 @@ end note
|
||||
|
||||
loop up to 3 turns (timeout budget)
|
||||
|
||||
Judge -> LLM : model_turn(lane, judge_turns,\ntools=[read_file, list_directory])\nvia drained create_streaming
|
||||
LLM --> Judge : ModelTurnResult
|
||||
Judge -> LLM : create_completion(\nmodel, judge_messages,\ntools=[read_file, list_directory])
|
||||
LLM --> Judge : CompletionResult
|
||||
|
||||
alt tool_calls present (turn < 3)
|
||||
Judge -> Judge : _exec_read_only_tool()
|
||||
|
||||
@@ -252,7 +252,6 @@ interface, or anyone who can reach it can search through your instance.
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `WORKSPACE_MOUNT` | empty volume | Host directory bind-mounted at `/workspace` for the model to read/write |
|
||||
| `TURNSTONE_WORKSPACE` | `/workspace` (image env) | Directory named as the user's workspace in the model's tool descriptions; informational only — see [Working directory](#working-directory) |
|
||||
| `SKIP_PERMISSIONS` | — | Set to any value to auto-approve all tool calls (dev only) |
|
||||
| `MCP_CONFIG` | — | Path to an MCP server config file |
|
||||
| `TURNSTONE_IMAGE_TAG` | `latest` | ghcr.io image tag — production stack |
|
||||
@@ -277,35 +276,6 @@ docker compose build --no-cache # rebuild from scratch
|
||||
| `workspace` | `/workspace` (unless `WORKSPACE_MOUNT` is set) |
|
||||
| `caddy-data` / `caddy-config` | Caddy's local CA and config (dev stack) |
|
||||
|
||||
## Working directory
|
||||
|
||||
Node processes run with `/data` as their working directory (the image's
|
||||
`WORKDIR`), and that is where the model's shell commands execute and
|
||||
relative file paths resolve — **not** `/workspace`. The shell and file
|
||||
tool descriptions state both paths (the working directory, and the
|
||||
workspace named by `TURNSTONE_WORKSPACE`), so the model knows to look in
|
||||
`/workspace` for your files without being told each session.
|
||||
|
||||
To make tools start inside the mount instead, override the working
|
||||
directory on the node services:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
turnstone-node:
|
||||
working_dir: /workspace
|
||||
```
|
||||
|
||||
Two caveats before overriding:
|
||||
|
||||
- **SQLite fallback**: when a node runs without PostgreSQL, its fallback
|
||||
database `.turnstone.db` is created in the process working directory.
|
||||
Changing `working_dir` on an existing SQLite-fallback deployment makes
|
||||
the node create a fresh database inside the mount and your prior state
|
||||
appears lost (it is still in the `turnstone-data` volume under `/data`).
|
||||
The stock compose stacks use PostgreSQL and are unaffected.
|
||||
- Migrations (`entrypoint.sh`) run in the same working directory, so the
|
||||
same SQLite caveat applies to them.
|
||||
|
||||
## Cleanup
|
||||
|
||||
```bash
|
||||
|
||||
+4
-64
@@ -17,9 +17,8 @@ The MCP server admin form exposes three authorization modes ("Multitenant Author
|
||||
| `none` | No headers attached. Open MCP server (or one gated by network policy only). | Internal MCP servers on a trusted network. |
|
||||
| `static` | One static bearer token, configured per server, sent on every request from every user. | Service-to-service MCP servers where per-user attribution doesn't matter, or single-tenant deployments. |
|
||||
| `oauth_user` *(recommended for user-data servers)* | Each user authorizes separately via OAuth 2.1 + PKCE; Turnstone stores per-user tokens encrypted at rest. | MCP servers that expose user-specific data or that want per-user audit attribution. |
|
||||
| `oauth_obo` *(sign-in passthrough)* | Each user's Turnstone **org sign-in** (OIDC) mints a per-server access token on demand — no separate per-server consent. One captured credential per user covers every `oauth_obo` server. | Enterprise deployments where the identity provider governs access (Entra, Keycloak) and you want zero per-user connect clicks. See the dedicated section below. |
|
||||
|
||||
Switching `auth_type` away from `oauth_user` / `oauth_obo` **deletes** that server's per-user rows (consents / minted cache) — see the transition table below. Switching back later starts clean: users re-consent (or re-mint) on next use. The admin **bulk-revoke** / **flush cache** affordance clears rows without an auth-type change.
|
||||
Switching `auth_type` away from `oauth_user` orphans existing per-user tokens. Use the admin **bulk-revoke** affordance on the server row (Phase 9) to clear them, or let them expire naturally — they're inert without the matching `auth_type` value.
|
||||
|
||||
---
|
||||
|
||||
@@ -66,58 +65,6 @@ Keep this in `config.toml` rather than environment variables. An in-process LLM
|
||||
|
||||
---
|
||||
|
||||
## `auth_type=oauth_obo` — single-credential sign-in passthrough
|
||||
|
||||
Where `oauth_user` makes each user complete a **separate** browser consent per MCP server, `oauth_obo` reuses the user's Turnstone **org sign-in** (OIDC). Turnstone captures one refresh credential per user at login and, on each tool call, mints a short-lived access token scoped to that server's audience. There is no per-server connect step, and one credential covers every `oauth_obo` server. This is the right shape when your identity provider already governs who may reach each backend (an Entra tenant with Entra-protected MCP servers; a Keycloak realm with token exchange).
|
||||
|
||||
Access is governed **downstream** by the IdP: a user can only mint a token for a server their delegated permissions allow. Removing that grant at the IdP cuts the user off regardless of their Turnstone state.
|
||||
|
||||
### Deployment configuration (`[oidc]` in `config.toml`)
|
||||
|
||||
`oauth_obo` requires OIDC SSO to be configured (it is the credential source), plus:
|
||||
|
||||
```toml
|
||||
[oidc]
|
||||
# ... your existing issuer / client_id / client_secret ...
|
||||
capture_user_credential = true # persist the IdP refresh token at login
|
||||
obo_grant_profile = "entra" # "entra" | "rfc8693" — how tokens are minted
|
||||
```
|
||||
|
||||
- **`capture_user_credential`** (default `false`): when enabled, Turnstone appends `offline_access` to the login scopes and stores the returned refresh token, encrypted with the same `[security] mcp_token_encryption_key` as `oauth_user` tokens. **The encryption key is required** — Turnstone refuses to start with an `oauth_obo` row (or capture enabled) and no key.
|
||||
- **`obo_grant_profile`** picks the mint mechanism (the IdP determines which one is valid; this is deployment-wide, not per-server):
|
||||
- **`entra`** — redeems the user's refresh token directly for a token scoped to `<audience>/.default`. `oauth_scopes` on the server row is **not used** (the admin form rejects it under this profile).
|
||||
- **`rfc8693`** — a refresh grant for a subject token, then an RFC 8693 token exchange for the server audience. Per-server `oauth_scopes` **are** sent on the exchange (some IdPs require the audience scope explicitly).
|
||||
|
||||
### Adding an `oauth_obo` server
|
||||
|
||||
In the admin MCP form, choose **Sign-in passthrough** and set **Audience** (required — the downstream resource the token is minted for, e.g. `api://<app-id>` on Entra or the client id on Keycloak). The client-id / secret / registration fields do not apply and are hidden.
|
||||
|
||||
`oauth_obo` servers are accepted only when **OIDC sign-in is configured and enabled** and `[oidc] obo_grant_profile` is a valid profile — the write is rejected otherwise, since a row that can never mint would surface to users as a permanent "please retry" that never heals.
|
||||
|
||||
### Identity-provider setup
|
||||
|
||||
**Entra (`obo_grant_profile = "entra"`):**
|
||||
1. Turnstone's app registration must hold **delegated permissions** to each MCP server's exposed API, with **admin consent granted** (or the MCP app listed in Turnstone's `preAuthorizedApplications`).
|
||||
2. Set the server row's Audience to the MCP app's Application ID URI (`api://<guid>`).
|
||||
3. **Gotcha (verified):** admin-consent issued *immediately* after creating the app/service principal can silently skip a not-yet-propagated resource — the only symptom is `AADSTS65001` at mint time. Verify the delegated grant landed (`az ad app permission list-grants` / the portal's *API permissions* blade shows *Granted*), or grant it explicitly per resource. A missing grant surfaces in Turnstone as a re-login prompt on the affected server (same rail as a revoked credential), and the `mcp_server.oauth.obo_mint_rejected` log line carries the raw `AADSTS…` text.
|
||||
|
||||
**Keycloak / RFC 8693 (`obo_grant_profile = "rfc8693"`):**
|
||||
1. Enable **standard token exchange** on Turnstone's client.
|
||||
2. Grant the audience: add an audience client scope for each MCP client and attach it to Turnstone's client (optional scopes must be requested — set the server row's Scopes to that scope, or the exchange returns *"Requested audience not available"*).
|
||||
3. Set the server row's Audience to the downstream client id.
|
||||
|
||||
### Revocation & custody
|
||||
|
||||
The captured credential is a single per-user secret that can mint for every `oauth_obo` server, so treat it like any long-lived credential:
|
||||
|
||||
- **Cut off one user:** unlink their OIDC identity in the admin console (**Users → OIDC identities → delete**). This revokes the captured credential **and** purges their minted cache rows, so future mints fail and cached tokens are dropped. (Warmed in-memory sessions on server nodes self-expire at the access-token TTL; there is no cross-node per-user session-kill.) Removing the user's access at the IdP is the authoritative cut-off.
|
||||
- **Flush a server's minted tokens** (e.g. after narrowing its audience): the server row's **flush cache** action drops all users' cached tokens for that server. This is **not** a revocation — users re-mint on next use from their still-valid sign-in. It is surfaced honestly (audit `mcp_server.oauth.obo_cache_flushed`, response `effect: cache_flush_remints`) so it is never mistaken for cutting access.
|
||||
- Per-server revocation in the `oauth_user` sense does not exist for `oauth_obo` — the credential is issuer-scoped and IdP-governed. Revoke at the IdP.
|
||||
|
||||
> **Interim for Entra without OBO:** if you don't want host-side minting, admin consent + `preAuthorizedApplications` on each MCP app registration removes the second consent prompt for the plain `oauth_user` flow too (a tenant-config change, no Turnstone code). Tracked in issue #682. It does not remove the per-server connect clicks or per-(user, server) token custody — that is what `oauth_obo` is for.
|
||||
|
||||
---
|
||||
|
||||
## Lifecycle
|
||||
|
||||
1. **First tool call** for a user against an `oauth_user` MCP server: pool dispatch finds no stored token, returns `mcp_consent_required` to the agent. Dashboard renders an inline "Connect" action card.
|
||||
@@ -128,7 +75,7 @@ The captured credential is a single per-user secret that can mint for every `oau
|
||||
|
||||
4. **Step-up scope**: when a tool call hits `403` with `WWW-Authenticate: error="insufficient_scope"`, Turnstone emits `mcp_insufficient_scope` with the parsed scope set; the dashboard offers a "Connect with additional scopes" affordance that opens `/v1/api/mcp/oauth/start?server=<name>&scopes=<extra>` so the union of original + new scopes flows into the AS authorize request.
|
||||
|
||||
5. **User revoke** (settings modal): `DELETE /v1/api/mcp/oauth/connections/{server_name}` runs the authoritative local delete + best-effort RFC 7009 upstream revoke (fire-and-forget, capped at 256 concurrent in-flight tasks). `oauth_obo` servers are excluded: their rows are mint cache, not consents — deleting one only forces a re-mint — so the connections list hides them and the endpoint refuses them with `409` (revocation for sign-in passthrough happens at the identity layer: unlink the identity or revoke at the IdP).
|
||||
5. **User revoke** (settings modal): `DELETE /v1/api/mcp/oauth/connections/{server_name}` runs the authoritative local delete + best-effort RFC 7009 upstream revoke (fire-and-forget, capped at 256 concurrent in-flight tasks).
|
||||
|
||||
6. **Admin bulk-revoke** (Phase 9): `POST /v1/api/admin/mcp-servers/{name}/bulk-revoke` drops every user's token for the server. Upstream RFC 7009 revoke is intentionally **not** attempted in bulk (avoids N upstream HTTP calls per admin click); tokens at the AS expire naturally. Use the per-user revoke endpoint if you need guaranteed upstream invalidation.
|
||||
|
||||
@@ -150,13 +97,10 @@ Additional indicators (circuit-breaker state, encryption-key mismatch) are expos
|
||||
| From | To | What happens |
|
||||
|---|---|---|
|
||||
| `none` / `static` → `oauth_user` | — | New code path activates for this server. Existing static headers (if any) are no longer sent. Users must authorize on first use. |
|
||||
| `oauth_user` → `none` / `static` | — | Existing `mcp_user_tokens` rows are **deleted**: the tokens are bound to the auth model + URL active at consent time, and rows left behind could silently rebind if a row with the old name/URL reappears. Switching back to `oauth_user` later starts clean — users re-consent on next use. This is **not reversible**; the AS-side grants are untouched (revoke upstream via the AS if needed). |
|
||||
| `oauth_user` → `none` / `static` | — | Existing `mcp_user_tokens` rows are **orphaned** — inert without a matching `auth_type`. Use admin bulk-revoke to drop them, or let them expire. Switching back to `oauth_user` later re-activates the orphaned rows if they haven't been deleted. |
|
||||
| OAuth `client_id` or `client_secret` rotated | — | Existing tokens may stop refreshing if the AS treats them as bound to the previous client. Bulk-revoke after rotation. |
|
||||
| `oauth_user` ↔ `oauth_obo` | — | The per-user rows are **deleted** on the flip (they mean different things: per-server AS refresh tokens vs. minted cache). `oauth_audience` and `oauth_scopes` mean different things in each model (a resource indicator vs. an IdP app identifier; AS-consent scopes vs. an rfc8693 exchange scope), so on a flip they **never carry** — each is taken from the request for the target model or set NULL. The admin console clears these fields when you change the auth type, so re-enter the correct values for the new mode; via the API, supply them explicitly (a flip into `oauth_obo` with no `oauth_audience` is rejected, and a non-empty `oauth_scopes` under the `entra` profile is rejected since that leg pins `<audience>/.default`). |
|
||||
| `oauth_obo` → `none` / `static` | — | Minted cache rows are deleted. |
|
||||
| `oauth_obo` **audience**, **URL**, or **`oauth_scopes`** changed | — | Minted cache rows are **deleted** (tokens are bound to the audience/URL/scopes at mint time), forcing a fresh mint — so an audience or scope narrowing takes effect immediately, not at token expiry. |
|
||||
|
||||
Every transition that changes what a stored row *means* deletes the rows outright — a stale consent or minted token must never be served under new semantics. There is no orphan-and-reactivate path.
|
||||
The orphan-by-default behavior is chosen so switching back to `oauth_user` is non-destructive. Bulk-revoke is the explicit cleanup path.
|
||||
|
||||
---
|
||||
|
||||
@@ -169,9 +113,5 @@ Every transition that changes what a stored row *means* deletes the rows outrigh
|
||||
| `mcp_oauth_url_insecure` | MCP server URL is `http://` (not `https://`) on a non-loopback host | Use `https://`. Per-user bearers must not transit cleartext. |
|
||||
| Tools fail in scheduled / Discord / Slack runs | OAuth-MCP requires browser-based consent | Users must pre-consent via the web UI. Phase 9 dashboard badge surfaces deferred consents from these runs on next login. |
|
||||
| Circuit breaker open repeatedly | Transport-level errors on the MCP server (DNS, TLS, 5xx) | Check the per-server error pill; auth errors do not trip the breaker. |
|
||||
| **`oauth_obo`**: every tool call fails, log shows `obo_misconfigured` | Server row has no Audience, or `obo_grant_profile` is unset/unknown | Set the Audience on the server row; set `[oidc] obo_grant_profile` to `entra` or `rfc8693`. |
|
||||
| **`oauth_obo`**: `obo_mint_rejected` with `AADSTS65001` | Turnstone's app lacks the (admin-consented) delegated grant to this MCP app — often admin consent that didn't propagate | Grant + admin-consent the delegated permission for this resource; verify it shows *Granted*. See the Entra gotcha above. |
|
||||
| **`oauth_obo`**: "Sign in to Turnstone again" on one server | Captured credential missing/rejected, or a Conditional Access challenge | User re-logs into Turnstone (re-captures the credential). If it persists, check the IdP grant / CA policy. |
|
||||
| **`oauth_obo`**: tools don't appear at all for a user | User has not signed in since `capture_user_credential` was enabled (no credential captured) | User logs out and back in via OIDC so the refresh credential is captured. |
|
||||
|
||||
See also: `docs/operations/mcp-oauth-headless.md` for the cron / channel-driven run caveat.
|
||||
|
||||
+9
-9
@@ -77,17 +77,17 @@ IdP from redirecting the token-exchange POST (which carries
|
||||
being aimed at internal services.
|
||||
|
||||
A few public IdPs legitimately split endpoints across hostnames. Google
|
||||
and Microsoft Entra ID are the canonical examples:
|
||||
is the canonical example:
|
||||
|
||||
| IdP | Issuer host | Cross-host endpoint(s) |
|
||||
|-----|-------------|------------------------|
|
||||
| Google | `accounts.google.com` | `oauth2.googleapis.com`, `www.googleapis.com`, `openidconnect.googleapis.com` |
|
||||
| Microsoft Entra | `login.microsoftonline.com` | `graph.microsoft.com` (userinfo) |
|
||||
| Field | Hostname |
|
||||
|-------|----------|
|
||||
| issuer | `accounts.google.com` |
|
||||
| token_endpoint | `oauth2.googleapis.com` |
|
||||
| jwks_uri | `www.googleapis.com` |
|
||||
| userinfo_endpoint | `openidconnect.googleapis.com` |
|
||||
|
||||
Both sets are built in — operators using `https://accounts.google.com` or
|
||||
`https://login.microsoftonline.com/<tenant>/v2.0` need no extra
|
||||
configuration. (Entra's discovery document advertises `userinfo_endpoint`
|
||||
on `graph.microsoft.com`, distinct from the issuer host.)
|
||||
Google's set is built in — operators using `https://accounts.google.com`
|
||||
need no extra configuration.
|
||||
|
||||
For other IdPs whose discovery document references a non-issuer host,
|
||||
extend the allow-list explicitly:
|
||||
|
||||
+12
-28
@@ -28,19 +28,13 @@ schema plus turnstone-specific metadata keys:
|
||||
}
|
||||
```
|
||||
|
||||
**Metadata keys** (stripped before sending the schema to the model; the full
|
||||
set lives in `_META_KEYS` in `turnstone/core/tools.py`):
|
||||
**Metadata keys** (stripped before sending the schema to the model):
|
||||
|
||||
| Key | Type | Meaning |
|
||||
|------------------|------|---------|
|
||||
| `task_agent` | bool | Tool is available to task sub-agents. |
|
||||
| `coordinator` | bool | Tool is available to coordinator sessions. Without `interactive: true` alongside it, this reads as coord-only and the tool is stripped from interactive sessions. |
|
||||
| `interactive` | bool | Opt a `coordinator: true` tool back into interactive sessions (dual-kind tools like `memory`). |
|
||||
| `auto_approve` | bool | Tool runs without user confirmation (read-only, safe operations). |
|
||||
| `primary_key` | str | When the model sends a bare string instead of JSON args, map it to this parameter name. |
|
||||
| `kind_variants` | dict | Per-kind description / parameter-schema overlays so each session kind sees only the surface it can use (see `memory.json`). |
|
||||
| `cwd_note` | str | Sentence appended to the description at session build time with `{working_dir}` substituted — declare on tools whose semantics depend on the process working directory (see `bash.json`, `apply_cwd_context`). |
|
||||
| `workspace_note` | str | Companion sentence naming the operator-configured workspace directory, `{workspace_dir}` substituted; dropped when no workspace is configured. |
|
||||
| Key | Type | Meaning |
|
||||
|----------------|------|---------|
|
||||
| `task_agent` | bool | Tool is available to task sub-agents. |
|
||||
| `auto_approve` | bool | Tool runs without user confirmation (read-only, safe operations). |
|
||||
| `primary_key` | str | When the model sends a bare string instead of JSON args, map it to this parameter name. |
|
||||
|
||||
---
|
||||
|
||||
@@ -785,10 +779,7 @@ MCP tool lists stay up-to-date without restart through two mechanisms:
|
||||
1. **Push notifications** -- MCP servers that declare `tools.listChanged: true` in
|
||||
their capabilities send `notifications/tools/list_changed` when their tool list
|
||||
changes. `MCPClientManager` registers a `message_handler` on each `ClientSession`
|
||||
that triggers an immediate refresh for that server (debounced per server and
|
||||
notification kind, and run off the receive loop). A refresh that fails while
|
||||
the connection stays up is retried automatically on the next health-loop tick
|
||||
until one completes.
|
||||
that triggers an immediate refresh for that server.
|
||||
|
||||
2. **Manual** -- `/mcp refresh` re-fetches tools from all servers immediately.
|
||||
`/mcp refresh <server>` targets a single server. If a server has disconnected,
|
||||
@@ -796,10 +787,6 @@ MCP tool lists stay up-to-date without restart through two mechanisms:
|
||||
same controls (refresh / reconnect buttons per server) for cluster-wide
|
||||
fan-out.
|
||||
|
||||
Reconnects (health-loop, dispatch-driven, or operator-forced) always end in a
|
||||
full catalog rediscovery, so a server that changed its tools while disconnected
|
||||
comes back current.
|
||||
|
||||
When tools change, `MCPClientManager` rebuilds its merged tool list using copy-on-write
|
||||
(new list/dict objects assigned atomically) and notifies all active `ChatSession`
|
||||
instances via registered listener callbacks. Each session rebuilds its `_tools`,
|
||||
@@ -870,16 +857,13 @@ catalog.
|
||||
|
||||
### Refresh
|
||||
|
||||
Resource lists stay current through the same mechanisms as tool lists:
|
||||
Resource lists stay current through the same three-tier mechanism as tool lists:
|
||||
|
||||
1. **Push** -- Servers declaring `resources.listChanged: true` send
|
||||
`notifications/resources/list_changed`, triggering an immediate refresh
|
||||
(with the same failed-refresh retry on the health-loop tick).
|
||||
2. **Manual** -- `/mcp refresh` re-fetches resources alongside tools.
|
||||
|
||||
Servers without push support are refreshed whenever they reconnect (every
|
||||
reconnect ends in full rediscovery) or when an operator refreshes manually;
|
||||
there is no periodic polling.
|
||||
`notifications/resources/list_changed`, triggering an immediate refresh.
|
||||
2. **Periodic** -- Servers without push are polled on the configured refresh
|
||||
interval (default 4 hours, same timer as tools).
|
||||
3. **Manual** -- `/mcp refresh` re-fetches resources alongside tools.
|
||||
|
||||
---
|
||||
|
||||
|
||||
+2
-3
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "turnstone"
|
||||
version = "1.8.0a4"
|
||||
version = "1.7.4"
|
||||
description = "Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
@@ -88,7 +88,7 @@ include = [
|
||||
"turnstone/console/static/coordinator/*.js",
|
||||
"turnstone/shared_static/*.css",
|
||||
"turnstone/shared_static/*.js",
|
||||
"turnstone/shared_static/katex-0.18.1/**/*",
|
||||
"turnstone/shared_static/katex-0.17.0/**/*",
|
||||
"turnstone/shared_static/hljs-11.11.1/**/*",
|
||||
"turnstone/shared_static/mermaid-11.16.0/**/*",
|
||||
"turnstone/shared_static/hls-1.6.16/**/*",
|
||||
@@ -103,7 +103,6 @@ testpaths = ["tests"]
|
||||
markers = [
|
||||
"live: requires a running LLM backend",
|
||||
"allow_thread_leak: test intentionally leaves a background thread running (opts out of the leaked-thread guard)",
|
||||
"e2e_recovery: opt-in end-to-end SSE recovery harness (real server + real SSE consumers, scripted provider — NOT live, no LLM backend needed); tens of seconds each. CI lanes run ``-m 'not live and not e2e_recovery'``; select with ``-m e2e_recovery``.",
|
||||
]
|
||||
filterwarnings = [
|
||||
# mcp v1 deprecates streamablehttp_client for an entry point whose call
|
||||
|
||||
+1
-1
@@ -399,7 +399,7 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
known: true,
|
||||
capabilities: {
|
||||
context_window: 200000, supports_tools: true,
|
||||
supports_vision: true,
|
||||
supports_streaming: true, supports_vision: true,
|
||||
supports_web_search: true, supports_temperature: true,
|
||||
supports_effort: true,
|
||||
},
|
||||
|
||||
@@ -1,19 +0,0 @@
|
||||
# Entra config for the Entra e2e / spike harnesses. Copy to `.env` (gitignored)
|
||||
# and fill in from your tenant. `entra_setup.sh setup` creates the app
|
||||
# registrations and writes a populated `.env` for you.
|
||||
#
|
||||
# cp scripts/obo-e2e/.env.example scripts/obo-e2e/.env
|
||||
# # then edit, or run: ./scripts/obo-e2e/entra_setup.sh setup
|
||||
|
||||
export ENTRA_TENANT_ID=<tenant-guid-or-domain>
|
||||
export ENTRA_CLIENT_ID=<turnstone-spike-app-client-id>
|
||||
export ENTRA_CLIENT_SECRET=<client-secret>
|
||||
export SPIKE_AUDIENCE_A=api://<resource-app-a-guid> # a consented resource
|
||||
export SPIKE_AUDIENCE_B=api://<resource-app-b-guid> # a second consented resource
|
||||
export SPIKE_AUDIENCE_UNCONSENTED=api://<resource-app-c-guid> # NOT granted (negative case)
|
||||
export SPIKE_RUN_OBO=1
|
||||
# export SPIKE_PORT=8765 # redirect-listener port (default 8765)
|
||||
# export SPIKE_CALLBACK_FILE=/tmp/obo_cb.txt # remote-browser mode: paste the redirect URL here
|
||||
|
||||
# The Keycloak / OSS-path harness needs no config — keycloak_e2e.sh sets
|
||||
# everything and stands up an ephemeral container.
|
||||
@@ -1,214 +0,0 @@
|
||||
# OBO e2e harnesses — single-credential MCP token minting (`auth_type=oauth_obo`)
|
||||
|
||||
Manual test harnesses for the `oauth_obo` feature (issue #551). They exercise
|
||||
the **real** Turnstone mint path (`get_obo_access_token_classified` →
|
||||
`_obo_mint_entra` / `_obo_mint_rfc8693`) against a real identity provider — not
|
||||
mocks, not the unit suite. Two grant legs:
|
||||
|
||||
- **Entra** (`entra_e2e.py`) — real tenant, one interactive sign-in.
|
||||
- **Keycloak / RFC 8693** (`keycloak_e2e.py` + `.sh`) — ephemeral docker, fully
|
||||
headless.
|
||||
|
||||
There is also `entra_spike.py` (raw-OAuth **wire** probe, pre-implementation
|
||||
reference) and `entra_setup.sh` (creates the Entra app registrations + writes a
|
||||
populated `.env`).
|
||||
|
||||
**Secrets:** these read config from env. Real credentials live in a **gitignored
|
||||
`.env`** (copy `.env.example`); nothing tenant-specific is committed. The only
|
||||
literal secret in the tree is the ephemeral Keycloak container's throwaway
|
||||
`spike-secret`, which lives and dies with the container.
|
||||
|
||||
Not part of CI — run by hand when validating the feature against a live IdP.
|
||||
|
||||
## `entra_e2e.py` — end-to-end product exercise (post-implementation)
|
||||
|
||||
`entra_spike.py` verified the raw OAuth WIRE (before code existed). `entra_e2e.py`
|
||||
verifies the SHIPPED Turnstone code: it does a real Entra login, feeds the
|
||||
credential through the real `MCPTokenStore.upsert_oidc_credential` (the call the
|
||||
OIDC callback makes on capture), then drives the real
|
||||
`get_obo_access_token_classified` → `_obo_mint_entra` against the live Entra token
|
||||
endpoint. Checks E1–E7: real mint + aud claim, cache-hit (0 Entra calls),
|
||||
single-credential→audiences A&B, rotation write-back, force_refresh re-mint,
|
||||
unconsented-audience classification with the credential surviving, and
|
||||
flush→re-mint. Reuses the same `.env` and interactive login (SPIKE_CALLBACK_FILE
|
||||
for remote browser).
|
||||
|
||||
```bash
|
||||
source scripts/obo-e2e/.env
|
||||
uv run python scripts/obo-e2e/entra_e2e.py
|
||||
# one interactive sign-in; E1–E7 then run against the real product code. Results below.
|
||||
```
|
||||
|
||||
Results — RUN 2026-07-12 on the real tenant, ALL VERIFIED (exit 0): capture
|
||||
persisted; E1 mint A (aud=A app-id, cache row refresh_token_ct NULL); E2 cache
|
||||
hit (0 extra Entra calls); E3 mint B from the SAME credential (aud=B app-id); E4
|
||||
rotation write-back (RT rotated 2040→2091 chars, newest persisted); E5
|
||||
force_refresh re-mint (1 Entra call); E6 unconsented C → refresh_failed and the
|
||||
credential SURVIVES; E7 flush→re-mint. The real `get_obo_access_token_classified`
|
||||
→ `_obo_mint_entra` path against the live Entra token endpoint.
|
||||
|
||||
## `keycloak_e2e.py` + `keycloak_e2e.sh` — OSS path (RFC 8693), headless
|
||||
|
||||
The rfc8693 equivalent of `entra_e2e.py`: `keycloak_e2e.sh` spins up ephemeral
|
||||
Keycloak, configures the realm (turnstone client with standard token exchange,
|
||||
mcp-a/b/c clients, aud-mcp-a/b audience scopes, a test user), runs the harness
|
||||
against the real `get_obo_access_token_classified` → `_obo_mint_rfc8693`
|
||||
(refresh grant → token exchange), then tears down. No browser (password grant).
|
||||
|
||||
```bash
|
||||
./scripts/obo-e2e/keycloak_e2e.sh
|
||||
```
|
||||
|
||||
Results — RUN 2026-07-12, ALL VERIFIED: capture persisted; E1 mint A
|
||||
(refresh→exchange, aud=mcp-a, cache row refresh_token_ct NULL); E2 cache hit (0
|
||||
extra KC calls); E3 mint B from the SAME credential (aud=mcp-b); E4 rotation
|
||||
write-back (KC rotated the RT on the refresh leg, newest persisted); E5
|
||||
force_refresh re-mint (**2 KC calls** = the two-leg chain); E6 unconsented C →
|
||||
refresh_failed_transient (KC returns invalid_request for a missing audience
|
||||
scope → classified transient; credential SURVIVES either way); E7 flush→re-mint.
|
||||
Gotcha: dev-mode Keycloak boot is slow on a loaded host — the script now waits on
|
||||
kcadm auth (up to ~6 min) rather than a fixed sleep. Port 8091 (8090 = the dev
|
||||
console).
|
||||
|
||||
## Leg 1 — Entra (`entra_spike.py`) — NEEDS TENANT ACCESS
|
||||
|
||||
### Tenant / app-registration setup (one-time, ~15 min)
|
||||
|
||||
1. **Spike client app** (stands in for Turnstone's OIDC app registration):
|
||||
- New app registration, single tenant. Platform **Web**, redirect URI
|
||||
`http://localhost:8765/callback`. Create a **client secret**.
|
||||
2. **Two resource apps** (stand in for MCP servers A and B):
|
||||
- New app registrations `spike-mcp-a`, `spike-mcp-b`. In each:
|
||||
**Expose an API** → set Application ID URI (`api://<guid>`) → add a scope
|
||||
(e.g. `mcp.access`).
|
||||
3. **Delegated grants** (this is metaclassing's "proper tenant and app reg setup"):
|
||||
- On the spike client app → **API permissions** → add delegated permission to
|
||||
`spike-mcp-a` and `spike-mcp-b` scopes → **Grant admin consent**.
|
||||
- Optionally also add the spike client's app id to each resource app's
|
||||
`preAuthorizedApplications` (Expose an API → Add a client application) to
|
||||
compare against pure admin consent.
|
||||
4. **Unconsented control** (for V5): a third resource app `spike-mcp-c` with an
|
||||
exposed API but NO permission granted to the spike client.
|
||||
|
||||
### Run
|
||||
|
||||
```bash
|
||||
export ENTRA_TENANT_ID=... ENTRA_CLIENT_ID=... ENTRA_CLIENT_SECRET=...
|
||||
export SPIKE_AUDIENCE_A=api://<a-guid> SPIKE_AUDIENCE_B=api://<b-guid>
|
||||
export SPIKE_AUDIENCE_UNCONSENTED=api://<c-guid> # optional (V5)
|
||||
export SPIKE_RUN_OBO=1 # optional (V6)
|
||||
uv run python scripts/obo-e2e/entra_spike.py
|
||||
```
|
||||
|
||||
A browser opens for one interactive login (any tenant user). Everything after is
|
||||
non-interactive — that IS the feature.
|
||||
|
||||
### What each check pins down
|
||||
|
||||
| Check | Design assumption it verifies |
|
||||
| --- | --- |
|
||||
| V1 | `offline_access` on the login yields a client-bound RT (capture layer) |
|
||||
| V2/V3 | ONE RT redeems for access tokens of DIFFERENT audiences (`scope=<aud>/.default`) — the load-bearing Entra behavior |
|
||||
| V4 | rotation semantics → whether RT write-back on every mint is convenience or correctness-critical |
|
||||
| V5 | unconsented audience fails `AADSTS65001 consent_required` → maps to the reconnect-rail fallback, never a silent failure |
|
||||
| V6 | OBO jwt-bearer middle-tier variant works with the same app registration (comparison data only) |
|
||||
|
||||
Also record (manual): whether Conditional Access / MFA policies in the tenant
|
||||
produce `interaction_required` on redemption — that's the fallback path's other
|
||||
trigger.
|
||||
|
||||
### Results — RUN 2026-07-11 on a real tenant, ALL SIX VERIFIED
|
||||
|
||||
Tenant: personal default directory (Global Admin), user is an MSA member.
|
||||
Setup via `entra_setup.sh setup`; V3 initially failed (see gotcha below),
|
||||
passed after fixing the grant. Second run: V1-V6 all VERIFIED, exit 0.
|
||||
|
||||
| Check | Result |
|
||||
| --- | --- |
|
||||
| V1 offline_access login -> RT | VERIFIED (confidential client + PKCE, RT ~2KB) |
|
||||
| V2 RT -> audience A token | VERIFIED (`aud=<A app guid>`, ~70 min TTL, new RT returned) |
|
||||
| V3 SAME RT -> audience B token | **VERIFIED — the load-bearing claim: one RT, many audiences** |
|
||||
| V4 rotation | VERIFIED: RT rotates on every redemption, but the OLD RT stays valid (reuse HTTP 200) -> write-back-newest is required; races are benign on Entra |
|
||||
| V5 unconsented audience | VERIFIED: `invalid_grant` + `AADSTS65001` (error_codes=[65001]) -> clean mapping to the reconnect-rail fallback |
|
||||
| V6 OBO jwt-bearer variant | VERIFIED: middle-tier shape also works with the same app registration |
|
||||
|
||||
**Operator gotcha (feeds #682 + product docs):** `az ad app permission
|
||||
admin-consent` run immediately after SP creation SILENTLY skips
|
||||
not-yet-propagated resource SPs — grant A landed, grant B didn't, and the only
|
||||
symptom was AADSTS65001 at redemption. Verify grants after consent
|
||||
(`oauth2PermissionGrants` filter on the client SP) or write them directly with
|
||||
`az ad app permission grant --id <client> --api <resource> --scope <scope>`.
|
||||
Product-side implication: a missing tenant grant for a NEW oauth_obo server
|
||||
surfaces as AADSTS65001 -> the same reconnect-rail path as revocation; the
|
||||
admin docs must say "grant first, then add the server".
|
||||
|
||||
## Leg 2 — Keycloak RFC 8693 (portability check) — runnable locally
|
||||
|
||||
Ephemeral `quay.io/keycloak/keycloak:26.3` (`start-dev`, port 8089), realm
|
||||
`spike`, confidential client `turnstone` with **standard token exchange**
|
||||
enabled, resource clients `mcp-a`/`mcp-b`, user `alice`. Pipeline mirrors the
|
||||
product design for a generic-8693 IdP:
|
||||
|
||||
```
|
||||
stored user RT --(refresh grant)--> user AT --(RFC 8693 exchange, audience=mcp-X)--> audience-scoped AT
|
||||
```
|
||||
|
||||
i.e. the per-user credential stays ONE refresh token; per-server tokens are
|
||||
minted via standard token exchange instead of Entra's multi-resource RT
|
||||
redemption. Same substrate, different grant leg.
|
||||
|
||||
### Results — RUN 2026-07-11, VERIFIED (Keycloak 26.3, ephemeral)
|
||||
|
||||
```
|
||||
alice ONE stored RT
|
||||
-> refresh grant -> user AT (azp=turnstone); RT ROTATED on refresh
|
||||
-> 8693 exchange audience=mcp-a scope=aud-mcp-a -> AT aud=mcp-a user=alice 300s, NO RT
|
||||
-> 8693 exchange audience=mcp-b scope=aud-mcp-b -> AT aud=mcp-b (same subject AT)
|
||||
negative control audience=mcp-c -> invalid_client "Audience not found"
|
||||
```
|
||||
|
||||
Findings that feed the design:
|
||||
1. **One per-user credential -> N audience tokens: VERIFIED on a second IdP.**
|
||||
The substrate is portable; only the grant leg differs per IdP.
|
||||
2. **Exchanged tokens are cache-shaped** (short TTL, no RT) — per-server
|
||||
`mcp_user_tokens` rows as short-lived mint cache is the right model.
|
||||
3. **RT rotation happens here too** — newest-RT write-back on every redemption
|
||||
is a correctness requirement of the capture layer, not an Entra quirk.
|
||||
4. **The IdP-side "delegated grant" has a per-IdP shape**: Entra = API
|
||||
permissions + admin consent; Keycloak = audience client scopes attached to
|
||||
the requester client (optional scopes activate via `scope=` at exchange).
|
||||
Operator runbooks are per-IdP (#682 pattern), code is not.
|
||||
5. Gotchas hit: KC user needs a complete profile for direct grant ("Account is
|
||||
not fully set up"); optional audience scope must be requested explicitly or
|
||||
the exchange 400s with "Requested audience not available".
|
||||
|
||||
Repro (ephemeral, ~2 min):
|
||||
|
||||
```bash
|
||||
docker run -d --name kc-obo-spike -p 127.0.0.1:8089:8080 \
|
||||
-e KC_BOOTSTRAP_ADMIN_USERNAME=admin -e KC_BOOTSTRAP_ADMIN_PASSWORD=admin \
|
||||
quay.io/keycloak/keycloak:26.3 start-dev
|
||||
KC="docker exec kc-obo-spike /opt/keycloak/bin/kcadm.sh"
|
||||
$KC config credentials --server http://localhost:8080 --realm master --user admin --password admin
|
||||
$KC create realms -s realm=spike -s enabled=true
|
||||
$KC create clients -r spike -s clientId=turnstone -s enabled=true -s publicClient=false \
|
||||
-s secret=spike-secret -s directAccessGrantsEnabled=true \
|
||||
-s 'attributes={"standard.token.exchange.enabled":"true"}'
|
||||
$KC create clients -r spike -s clientId=mcp-a -s enabled=true -s publicClient=false -s secret=x
|
||||
$KC create clients -r spike -s clientId=mcp-b -s enabled=true -s publicClient=false -s secret=x
|
||||
$KC create users -r spike -s username=alice -s enabled=true -s email=a@s.test \
|
||||
-s emailVerified=true -s firstName=A -s lastName=S
|
||||
$KC set-password -r spike --username alice --new-password alice-pw
|
||||
TURNSTONE_UUID=$($KC get clients -r spike -q clientId=turnstone --fields id --format csv --noquotes)
|
||||
for t in mcp-a mcp-b; do
|
||||
SID=$($KC create client-scopes -r spike -s name=aud-$t -s protocol=openid-connect -i)
|
||||
$KC create client-scopes/$SID/protocol-mappers/models -r spike -s name=aud-$t \
|
||||
-s protocol=openid-connect -s protocolMapper=oidc-audience-mapper \
|
||||
-s "config={\"included.client.audience\":\"$t\",\"access.token.claim\":\"true\"}"
|
||||
$KC update clients/$TURNSTONE_UUID/optional-client-scopes/$SID -r spike
|
||||
done
|
||||
# then: password grant -> refresh grant -> token-exchange with
|
||||
# grant_type=urn:ietf:params:oauth:grant-type:token-exchange,
|
||||
# subject_token=<user AT>, subject_token_type=...:access_token,
|
||||
# audience=mcp-a, scope=aud-mcp-a
|
||||
```
|
||||
@@ -1,286 +0,0 @@
|
||||
"""End-to-end exercise of the oauth_obo feature against a REAL Entra tenant.
|
||||
|
||||
Unlike ``entra_spike.py`` (which verified the raw OAuth wire shapes), this
|
||||
drives the ACTUAL Turnstone product code — real ``MCPTokenStore``, real
|
||||
``get_obo_access_token_classified`` → ``_obo_mint_entra`` → the real Entra
|
||||
token endpoint — so a green run proves the shipped mint engine works against
|
||||
live Entra, not just that the protocol does.
|
||||
|
||||
Flow:
|
||||
1. Interactive Entra login (auth-code + PKCE + offline_access) → a real
|
||||
refresh credential. This is what ``handle_oidc_callback`` receives.
|
||||
2. Persist it via ``MCPTokenStore.upsert_oidc_credential`` — the exact call
|
||||
the OIDC callback makes on capture (auth.py). The rest of the callback
|
||||
(JWKS validation, user provisioning) is OIDC-generic and unit-tested; the
|
||||
novel path is capture + mint, which this exercises for real.
|
||||
3. Seed real ``oauth_obo`` ``mcp_servers`` rows (audiences A/B consented, C
|
||||
not) and drive ``get_obo_access_token_classified`` — the real dispatch-time
|
||||
entry point — asserting on the minted tokens, cache, rotation, and
|
||||
classification.
|
||||
|
||||
Checks (VERIFIED / FAILED per line):
|
||||
E1 mint for audience A → kind=token; decoded aud == A; cache row written with
|
||||
refresh_token_ct NULL (cache, not custody); expires_at set
|
||||
E2 second call for A → cache hit, ZERO additional Entra calls
|
||||
E3 mint for audience B from the SAME captured credential → aud == B
|
||||
(the single-credential-many-audiences thesis, through the real engine)
|
||||
E4 rotation write-back: the stored credential holds the newest refresh token
|
||||
E5 force_refresh → a fresh mint (Entra call count increments)
|
||||
E6 unconsented audience C → NOT kind=token, and the shared credential SURVIVES
|
||||
(never auto-deleted — the load-bearing custody invariant)
|
||||
E7 cache flush → re-mint: deleting the cache row makes the next call re-mint
|
||||
|
||||
Run:
|
||||
source scripts/obo-e2e/.env
|
||||
uv run python scripts/obo-e2e/entra_e2e.py
|
||||
Env (from .env): ENTRA_TENANT_ID, ENTRA_CLIENT_ID, ENTRA_CLIENT_SECRET,
|
||||
SPIKE_AUDIENCE_A, SPIKE_AUDIENCE_B, SPIKE_AUDIENCE_UNCONSENTED, SPIKE_PORT.
|
||||
Remote browser: set SPIKE_CALLBACK_FILE to paste the redirect URL (as before).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
# Reuse the verified interactive-login machinery from the wire spike.
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from entra_spike import interactive_login, jwt_claims_unverified, redact # noqa: E402
|
||||
|
||||
from turnstone.core.mcp_crypto import ( # noqa: E402
|
||||
MCPTokenCipher,
|
||||
MCPTokenCipherConfig,
|
||||
MCPTokenStore,
|
||||
)
|
||||
from turnstone.core.mcp_oauth import get_obo_access_token_classified # noqa: E402
|
||||
from turnstone.core.oidc import OIDCConfig # noqa: E402
|
||||
from turnstone.core.storage._sqlite import SQLiteBackend # noqa: E402
|
||||
|
||||
USER = "e2e-user"
|
||||
RESULTS: list[tuple[str, str]] = []
|
||||
|
||||
|
||||
def record(status: str, msg: str) -> None:
|
||||
RESULTS.append((status, msg))
|
||||
print(f"[{status:>8}] {msg}")
|
||||
|
||||
|
||||
def aud_matches(token: str, want_audience: str) -> tuple[bool, str]:
|
||||
"""Compare a minted access token's aud claim to the configured audience.
|
||||
|
||||
Entra returns aud as the bare app-id GUID or the full ``api://<guid>`` URI;
|
||||
accept either.
|
||||
"""
|
||||
claims = jwt_claims_unverified(token)
|
||||
aud = str(claims.get("aud", "<none>"))
|
||||
want = want_audience.removeprefix("api://")
|
||||
return aud in (want, want_audience), aud
|
||||
|
||||
|
||||
class _CountingClient:
|
||||
"""Wraps httpx.AsyncClient, counting token-endpoint POSTs so cache hits
|
||||
(which must issue zero) are observable."""
|
||||
|
||||
def __init__(self, inner: httpx.AsyncClient) -> None:
|
||||
self._inner = inner
|
||||
self.posts = 0
|
||||
|
||||
async def post(self, *args: Any, **kwargs: Any) -> httpx.Response:
|
||||
self.posts += 1
|
||||
return await self._inner.post(*args, **kwargs)
|
||||
|
||||
|
||||
def _make_app_state(
|
||||
storage: SQLiteBackend,
|
||||
store: MCPTokenStore,
|
||||
oidc_config: OIDCConfig,
|
||||
http_client: _CountingClient,
|
||||
) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
auth_storage=storage,
|
||||
mcp_token_store=store,
|
||||
oidc_config=oidc_config,
|
||||
obo_http_client=http_client,
|
||||
mcp_oauth_refresh_locks={},
|
||||
mcp_oauth_refresh_backoff={},
|
||||
)
|
||||
|
||||
|
||||
def _seed_obo_server(storage: SQLiteBackend, name: str, audience: str) -> None:
|
||||
storage.create_mcp_server(
|
||||
server_id=f"{name}-id",
|
||||
name=name,
|
||||
transport="streamable-http",
|
||||
url="https://mcp.example.invalid/sse",
|
||||
auth_type="oauth_obo",
|
||||
oauth_audience=audience,
|
||||
)
|
||||
|
||||
|
||||
async def _run(cfg: dict[str, str], refresh_token: str) -> None:
|
||||
tenant = cfg["ENTRA_TENANT_ID"]
|
||||
issuer = f"https://login.microsoftonline.com/{tenant}/v2.0"
|
||||
token_endpoint = f"https://login.microsoftonline.com/{tenant}/oauth2/v2.0/token"
|
||||
aud_a = cfg["SPIKE_AUDIENCE_A"]
|
||||
aud_b = cfg["SPIKE_AUDIENCE_B"]
|
||||
aud_c = cfg.get("SPIKE_AUDIENCE_UNCONSENTED", "")
|
||||
|
||||
# Real Turnstone objects.
|
||||
db_path = os.path.join(tempfile.mkdtemp(prefix="obo-e2e-"), "e2e.db")
|
||||
storage = SQLiteBackend(db_path)
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
raw = base64.urlsafe_b64decode(Fernet.generate_key())
|
||||
store = MCPTokenStore(storage, MCPTokenCipher(MCPTokenCipherConfig(keys=(raw,))), node_id="e2e")
|
||||
oidc_config = OIDCConfig(
|
||||
enabled=True,
|
||||
issuer=issuer,
|
||||
client_id=cfg["ENTRA_CLIENT_ID"],
|
||||
client_secret=cfg["ENTRA_CLIENT_SECRET"],
|
||||
token_endpoint=token_endpoint,
|
||||
obo_grant_profile="entra",
|
||||
capture_user_credential=True,
|
||||
)
|
||||
|
||||
# Step 2 — CAPTURE: the exact storage call handle_oidc_callback makes.
|
||||
store.upsert_oidc_credential(USER, issuer, refresh_token=refresh_token)
|
||||
cap = store.get_oidc_credential(USER, issuer)
|
||||
if cap and cap["refresh_token"] == refresh_token:
|
||||
record("VERIFIED", f"capture: credential persisted for {USER} ({redact(refresh_token)})")
|
||||
else:
|
||||
record("FAILED", "capture: credential did not round-trip")
|
||||
return
|
||||
|
||||
_seed_obo_server(storage, "e2e-a", aud_a)
|
||||
_seed_obo_server(storage, "e2e-b", aud_b)
|
||||
if aud_c:
|
||||
_seed_obo_server(storage, "e2e-c", aud_c)
|
||||
|
||||
inner = httpx.AsyncClient(timeout=20.0)
|
||||
client = _CountingClient(inner)
|
||||
app_state = _make_app_state(storage, store, oidc_config, client)
|
||||
try:
|
||||
# E1 — real mint for audience A.
|
||||
r = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
if r.kind == "token" and r.token:
|
||||
ok, aud = aud_matches(r.token, aud_a)
|
||||
row = storage.get_mcp_user_token(USER, "e2e-a")
|
||||
cache_ok = (
|
||||
row is not None and row["refresh_token_ct"] is None and bool(row["expires_at"])
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if ok and cache_ok else "FAILED",
|
||||
f"E1 mint A: kind=token aud={aud} want={aud_a} cache_row_refreshless={cache_ok}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E1 mint A: kind={r.kind} (expected token)")
|
||||
return
|
||||
|
||||
# E2 — cache hit issues zero Entra calls.
|
||||
posts_before = client.posts
|
||||
r2 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r2.kind == "token" and client.posts == posts_before else "FAILED",
|
||||
f"E2 cache hit: kind={r2.kind} extra_entra_calls={client.posts - posts_before} (want 0)",
|
||||
)
|
||||
|
||||
# E3 — same credential, audience B.
|
||||
rb = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-b"
|
||||
)
|
||||
if rb.kind == "token" and rb.token:
|
||||
ok_b, aud_bclaim = aud_matches(rb.token, aud_b)
|
||||
record(
|
||||
"VERIFIED" if ok_b else "FAILED",
|
||||
f"E3 mint B from SAME credential: aud={aud_bclaim} want={aud_b}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E3 mint B: kind={rb.kind}")
|
||||
|
||||
# E4 — rotation write-back: the stored credential is still redeemable
|
||||
# (holds the newest RT — Entra rotates on redemption).
|
||||
cred_now = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if cred_now is not None else "FAILED",
|
||||
f"E4 rotation write-back: credential persisted {redact(cred_now['refresh_token']) if cred_now else '<gone>'}",
|
||||
)
|
||||
|
||||
# E5 — force_refresh re-mints (a real Entra call).
|
||||
posts_before = client.posts
|
||||
rf = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a", force_refresh=True
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if rf.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E5 force_refresh re-mint: kind={rf.kind} entra_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# E6 — unconsented audience: not a token, and the credential SURVIVES.
|
||||
if aud_c:
|
||||
rc = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-c"
|
||||
)
|
||||
cred_after = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if rc.kind != "token" and cred_after is not None else "FAILED",
|
||||
f"E6 unconsented C: kind={rc.kind} (not token) credential_survives={cred_after is not None}",
|
||||
)
|
||||
else:
|
||||
record("SKIPPED", "E6 unconsented C: SPIKE_AUDIENCE_UNCONSENTED not set")
|
||||
|
||||
# E7 — cache flush → re-mint.
|
||||
store.delete_user_token(USER, "e2e-a")
|
||||
posts_before = client.posts
|
||||
r7 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r7.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E7 flush→re-mint: kind={r7.kind} entra_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
finally:
|
||||
await inner.aclose()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"ENTRA_TENANT_ID",
|
||||
"ENTRA_CLIENT_ID",
|
||||
"ENTRA_CLIENT_SECRET",
|
||||
"SPIKE_AUDIENCE_A",
|
||||
"SPIKE_AUDIENCE_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in os.environ if k.startswith(("ENTRA_", "SPIKE_"))}
|
||||
missing = [k for k in required if not cfg.get(k)]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)} — did you `source scripts/obo-e2e/.env`?")
|
||||
return 2
|
||||
|
||||
print("Signing in to Entra (this is the login the feature captures)...")
|
||||
tokens = interactive_login(cfg)
|
||||
refresh_token = tokens.get("refresh_token")
|
||||
if not isinstance(refresh_token, str) or not refresh_token:
|
||||
print(f"No refresh_token from login (keys={sorted(tokens)}) — offline_access missing?")
|
||||
return 1
|
||||
|
||||
asyncio.run(_run(cfg, refresh_token))
|
||||
|
||||
print("\n=== summary ===")
|
||||
for status, msg in RESULTS:
|
||||
print(f" {status:>8} {msg}")
|
||||
return 0 if all(s in ("VERIFIED", "SKIPPED") for s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,134 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Entra spike setup for entra_spike.py (#551 re-scope boundary spike).
|
||||
# Manual test tooling — not run in CI. Creates throwaway Entra app registrations.
|
||||
#
|
||||
# ./entra_setup.sh setup create app registrations + consent + .env
|
||||
# ./entra_setup.sh cleanup delete everything it created (incl. .env)
|
||||
#
|
||||
# Creates in the logged-in tenant (az login first):
|
||||
# spike-turnstone confidential client (stands in for Turnstone's OIDC app)
|
||||
# spike-mcp-a/b resource apps exposing scope mcp.access, admin-consented
|
||||
# spike-mcp-c resource app with NO grant to the client (V5 control)
|
||||
# Requires: the logged-in user can create apps + grant admin consent
|
||||
# (Global Admin on a personal tenant qualifies).
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
ENV_FILE=".env"
|
||||
NAMES=(spike-turnstone spike-mcp-a spike-mcp-b spike-mcp-c)
|
||||
|
||||
log() { printf '>> %s\n' "$*"; }
|
||||
|
||||
graph_patch_api() { # $1=appId $2=scope-uuid $3=display-name
|
||||
local obj_id
|
||||
obj_id=$(az ad app show --id "$1" --query id -o tsv)
|
||||
az rest --method PATCH \
|
||||
--url "https://graph.microsoft.com/v1.0/applications/${obj_id}" \
|
||||
--headers 'Content-Type=application/json' \
|
||||
--body "{
|
||||
\"identifierUris\": [\"api://$1\"],
|
||||
\"api\": {
|
||||
\"requestedAccessTokenVersion\": 2,
|
||||
\"oauth2PermissionScopes\": [{
|
||||
\"id\": \"$2\",
|
||||
\"value\": \"mcp.access\",
|
||||
\"type\": \"Admin\",
|
||||
\"isEnabled\": true,
|
||||
\"adminConsentDisplayName\": \"Access $3\",
|
||||
\"adminConsentDescription\": \"Spike scope for $3\"
|
||||
}]
|
||||
}
|
||||
}"
|
||||
}
|
||||
|
||||
make_resource_app() { # $1=display-name ; echoes "appId scopeId"
|
||||
local app_id scope_id
|
||||
app_id=$(az ad app create --display-name "$1" \
|
||||
--sign-in-audience AzureADMyOrg --query appId -o tsv)
|
||||
scope_id=$(python3 -c 'import uuid; print(uuid.uuid4())')
|
||||
graph_patch_api "$app_id" "$scope_id" "$1" >/dev/null
|
||||
az ad sp create --id "$app_id" >/dev/null 2>&1 || true
|
||||
echo "$app_id $scope_id"
|
||||
}
|
||||
|
||||
cmd_setup() {
|
||||
local tenant_id
|
||||
tenant_id=$(az account show --query tenantId -o tsv)
|
||||
log "tenant: ${tenant_id}"
|
||||
|
||||
log "creating resource apps (a, b, c)..."
|
||||
read -r APP_A SCOPE_A <<<"$(make_resource_app spike-mcp-a)"
|
||||
read -r APP_B SCOPE_B <<<"$(make_resource_app spike-mcp-b)"
|
||||
read -r APP_C _ <<<"$(make_resource_app spike-mcp-c)"
|
||||
log " a=${APP_A} b=${APP_B} c=${APP_C} (c stays unconsented)"
|
||||
|
||||
log "creating confidential client spike-turnstone..."
|
||||
CLIENT_ID=$(az ad app create --display-name spike-turnstone \
|
||||
--sign-in-audience AzureADMyOrg \
|
||||
--web-redirect-uris "http://localhost:8765/callback" \
|
||||
--query appId -o tsv)
|
||||
az ad sp create --id "$CLIENT_ID" >/dev/null 2>&1 || true
|
||||
# No stderr suppression here: the secret is load-bearing (it lands in .env),
|
||||
# so under `set -e` a reset failure must abort LOUDLY, not silently.
|
||||
SECRET=$(az ad app credential reset --id "$CLIENT_ID" \
|
||||
--display-name spike --years 1 --query password -o tsv)
|
||||
|
||||
log "adding delegated permissions (a, b — NOT c)..."
|
||||
# Tolerated failures (|| log): a re-run hits "permission already exists" and
|
||||
# SP-propagation delays are common right after app creation — the
|
||||
# admin-consent retry loop below is the real gate. `set -e` would otherwise
|
||||
# turn a suppressed non-zero here into a silent mid-script abort.
|
||||
az ad app permission add --id "$CLIENT_ID" \
|
||||
--api "$APP_A" --api-permissions "${SCOPE_A}=Scope" \
|
||||
|| log " warn: permission add for a failed (may already exist); admin-consent below will confirm"
|
||||
az ad app permission add --id "$CLIENT_ID" \
|
||||
--api "$APP_B" --api-permissions "${SCOPE_B}=Scope" \
|
||||
|| log " warn: permission add for b failed (may already exist); admin-consent below will confirm"
|
||||
|
||||
log "granting admin consent (retries while SPs propagate)..."
|
||||
local ok=""
|
||||
for i in 1 2 3 4 5; do
|
||||
if az ad app permission admin-consent --id "$CLIENT_ID" 2>/dev/null; then
|
||||
ok=1; break
|
||||
fi
|
||||
log " not yet (attempt $i) — waiting 15s"
|
||||
sleep 15
|
||||
done
|
||||
[ -n "$ok" ] || { log "admin-consent failed after retries — grant manually in the portal (API permissions blade) and re-run the spike"; }
|
||||
|
||||
# Single-quote the values in the generated .env: the AS-issued client secret
|
||||
# can contain $ / backtick, and an unquoted RHS would be re-expanded (or
|
||||
# partially executed) when the operator `source`s the file. The heredoc still
|
||||
# interpolates ${...} into the single-quoted output; sourcing then treats the
|
||||
# result literally. (Azure secrets are base64-ish — no single quotes to escape.)
|
||||
umask 177
|
||||
cat > "$ENV_FILE" <<EOF
|
||||
export ENTRA_TENANT_ID='${tenant_id}'
|
||||
export ENTRA_CLIENT_ID='${CLIENT_ID}'
|
||||
export ENTRA_CLIENT_SECRET='${SECRET}'
|
||||
export SPIKE_AUDIENCE_A='api://${APP_A}'
|
||||
export SPIKE_AUDIENCE_B='api://${APP_B}'
|
||||
export SPIKE_AUDIENCE_UNCONSENTED='api://${APP_C}'
|
||||
export SPIKE_RUN_OBO=1
|
||||
EOF
|
||||
log "wrote ${ENV_FILE} (chmod 600). Next:"
|
||||
log " source scripts/obo-e2e/.env && uv run python scripts/obo-e2e/entra_spike.py"
|
||||
log "cleanup later with: ./entra_setup.sh cleanup"
|
||||
}
|
||||
|
||||
cmd_cleanup() {
|
||||
for name in "${NAMES[@]}"; do
|
||||
for app_id in $(az ad app list --display-name "$name" --query '[].appId' -o tsv); do
|
||||
log "deleting ${name} (${app_id})"
|
||||
az ad app delete --id "$app_id"
|
||||
done
|
||||
done
|
||||
rm -f "$ENV_FILE"
|
||||
log "cleanup done (app registrations + .env removed)"
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
setup) cmd_setup ;;
|
||||
cleanup) cmd_cleanup ;;
|
||||
*) echo "usage: $0 setup|cleanup"; exit 2 ;;
|
||||
esac
|
||||
@@ -1,333 +0,0 @@
|
||||
"""Entra boundary spike for single-credential MCP token minting (#551 re-scope).
|
||||
|
||||
Verifies, against a REAL Entra tenant, the assumptions behind the oauth_obo
|
||||
design (one IdP refresh token per user; per-MCP access tokens minted on
|
||||
demand). Each check prints VERIFIED / FAILED / SKIPPED plus redacted evidence.
|
||||
|
||||
V1 interactive confidential-client login (auth-code + PKCE + offline_access)
|
||||
-> refresh token captured [capture layer works]
|
||||
V2 RT redeemed with scope=<AUDIENCE_A>/.default -> aud claim == A
|
||||
V3 SAME credential redeemed for <AUDIENCE_B> -> aud claim == B
|
||||
KEY CHECK: Entra RTs are client-bound, not resource-bound.
|
||||
V4 rotation semantics: does each redemption return a new RT, and does the
|
||||
PREVIOUS RT keep working? [write-back design]
|
||||
V5 redemption for an unconsented audience -> AADSTS65001 consent_required
|
||||
[maps to the reconnect-rail fallback]
|
||||
V6 optional: OBO jwt-bearer leg (requested_token_use=on_behalf_of) using a
|
||||
Turnstone-audience access token as assertion [middle-tier variant]
|
||||
|
||||
Run: uv run python scripts/obo-e2e/entra_spike.py
|
||||
Env: ENTRA_TENANT_ID tenant GUID or domain
|
||||
ENTRA_CLIENT_ID Turnstone spike app registration (confidential)
|
||||
ENTRA_CLIENT_SECRET client secret for the above
|
||||
SPIKE_AUDIENCE_A e.g. api://<guid-a> (exposes a scope, consented)
|
||||
SPIKE_AUDIENCE_B e.g. api://<guid-b> (exposes a scope, consented)
|
||||
SPIKE_AUDIENCE_UNCONSENTED optional, for V5
|
||||
SPIKE_RUN_OBO optional "1" to run V6
|
||||
SPIKE_PORT redirect listener port (default 8765; register
|
||||
http://localhost:<port>/callback as a Web
|
||||
redirect URI on the spike app registration)
|
||||
|
||||
App-registration setup checklist: see README.md next to this file.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import secrets
|
||||
import sys
|
||||
import threading
|
||||
import urllib.parse
|
||||
import webbrowser
|
||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
RESULTS: list[tuple[str, str, str]] = [] # (check, status, evidence)
|
||||
|
||||
|
||||
def record(check: str, status: str, evidence: str) -> None:
|
||||
RESULTS.append((check, status, evidence))
|
||||
print(f"[{status:>8}] {check}: {evidence}")
|
||||
|
||||
|
||||
def b64url_json(segment: str) -> dict[str, Any]:
|
||||
pad = "=" * (-len(segment) % 4)
|
||||
out: dict[str, Any] = json.loads(base64.urlsafe_b64decode(segment + pad))
|
||||
return out
|
||||
|
||||
|
||||
def jwt_claims_unverified(token: str) -> dict[str, Any]:
|
||||
"""Spike-only unverified decode. NEVER do this in product code."""
|
||||
try:
|
||||
return b64url_json(token.split(".")[1])
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def redact(token: str | None) -> str:
|
||||
if not token:
|
||||
return "<absent>"
|
||||
return f"{token[:8]}...({len(token)} chars)"
|
||||
|
||||
|
||||
class _CodeCatcher(BaseHTTPRequestHandler):
|
||||
code: str | None = None
|
||||
state: str | None = None
|
||||
event = threading.Event()
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802 - stdlib API name
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(self.path).query)
|
||||
_CodeCatcher.code = (q.get("code") or [None])[0]
|
||||
_CodeCatcher.state = (q.get("state") or [None])[0]
|
||||
body = b"Spike login captured - return to the terminal."
|
||||
if q.get("error"):
|
||||
body = f"IdP error: {q}".encode()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/plain")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
_CodeCatcher.event.set()
|
||||
|
||||
def log_message(self, *args: Any) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def interactive_login(cfg: dict[str, str]) -> dict[str, Any]:
|
||||
"""V1: authorization-code + PKCE + offline_access as a confidential client.
|
||||
|
||||
Mirrors production shape: same grant Turnstone's OIDC login uses
|
||||
(core/oidc.py exchange_code), plus offline_access.
|
||||
"""
|
||||
port = int(cfg.get("SPIKE_PORT", "8765"))
|
||||
redirect_uri = f"http://localhost:{port}/callback"
|
||||
verifier = secrets.token_urlsafe(48)
|
||||
challenge = (
|
||||
base64.urlsafe_b64encode(hashlib.sha256(verifier.encode()).digest()).rstrip(b"=").decode()
|
||||
)
|
||||
state = secrets.token_urlsafe(16)
|
||||
authorize = (
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/authorize?"
|
||||
+ urllib.parse.urlencode(
|
||||
{
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"response_type": "code",
|
||||
"redirect_uri": redirect_uri,
|
||||
"response_mode": "query",
|
||||
# offline_access is THE capture-layer delta vs today's login.
|
||||
# No resource scope here: the RT is minted client-bound.
|
||||
"scope": "openid profile offline_access",
|
||||
"state": state,
|
||||
"code_challenge": challenge,
|
||||
"code_challenge_method": "S256",
|
||||
}
|
||||
)
|
||||
)
|
||||
server = HTTPServer(("127.0.0.1", port), _CodeCatcher)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
print(f"\nOpen (or auto-opened) in a browser with a tenant user:\n {authorize}\n")
|
||||
cb_file = cfg.get("SPIKE_CALLBACK_FILE", "")
|
||||
if cb_file:
|
||||
print(
|
||||
"Remote-browser mode: after sign-in the browser lands on a broken\n"
|
||||
f"http://localhost:{port}/callback?... page. Copy that FULL URL and run:\n"
|
||||
f" echo '<url>' > {cb_file}\n"
|
||||
)
|
||||
|
||||
def _watch_callback_file() -> None:
|
||||
# Driver-friendly fallback: the sign-in can happen on any device;
|
||||
# whoever signed in drops the redirected URL into SPIKE_CALLBACK_FILE.
|
||||
import time as _time
|
||||
|
||||
while not _CodeCatcher.event.is_set():
|
||||
try:
|
||||
with open(cb_file) as _f:
|
||||
pasted = _f.read().strip()
|
||||
except OSError:
|
||||
pasted = ""
|
||||
if "?" in pasted:
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(pasted).query)
|
||||
_CodeCatcher.code = (q.get("code") or [None])[0]
|
||||
_CodeCatcher.state = (q.get("state") or [None])[0]
|
||||
_CodeCatcher.event.set()
|
||||
return
|
||||
_time.sleep(1.0)
|
||||
|
||||
if cb_file:
|
||||
threading.Thread(target=_watch_callback_file, daemon=True).start()
|
||||
webbrowser.open(authorize)
|
||||
if not _CodeCatcher.event.wait(timeout=600):
|
||||
server.shutdown()
|
||||
raise SystemExit("Timed out waiting for the redirect (10 min).")
|
||||
server.shutdown()
|
||||
if _CodeCatcher.state != state:
|
||||
raise SystemExit("state mismatch on redirect - aborting.")
|
||||
if not _CodeCatcher.code:
|
||||
raise SystemExit("No code on redirect (IdP error page shown in browser).")
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "authorization_code",
|
||||
"code": _CodeCatcher.code,
|
||||
"redirect_uri": redirect_uri,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"code_verifier": verifier,
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
tokens: dict[str, Any] = resp.json()
|
||||
if resp.status_code != 200:
|
||||
raise SystemExit(f"code exchange failed: {json.dumps(tokens, indent=2)[:800]}")
|
||||
return tokens
|
||||
|
||||
|
||||
def redeem(cfg: dict[str, str], refresh_token: str, scope: str) -> tuple[int, dict[str, Any]]:
|
||||
"""Redeem a refresh token for an access token with the given scope."""
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "refresh_token",
|
||||
"refresh_token": refresh_token,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"scope": scope,
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
body: dict[str, Any] = resp.json()
|
||||
return resp.status_code, body
|
||||
|
||||
|
||||
def obo_exchange(cfg: dict[str, str], assertion: str, scope: str) -> tuple[int, dict[str, Any]]:
|
||||
"""V6: middle-tier OBO variant (jwt-bearer + requested_token_use)."""
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer",
|
||||
"assertion": assertion,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"scope": scope,
|
||||
"requested_token_use": "on_behalf_of",
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
body: dict[str, Any] = resp.json()
|
||||
return resp.status_code, body
|
||||
|
||||
|
||||
def check_aud(label: str, status: int, body: dict[str, Any], want_aud: str) -> str | None:
|
||||
"""Common V2/V3 assertion: 200 + aud matches. Returns the new RT if any."""
|
||||
if status != 200:
|
||||
record(label, "FAILED", f"HTTP {status}: {json.dumps(body)[:300]}")
|
||||
return None
|
||||
claims = jwt_claims_unverified(body.get("access_token", ""))
|
||||
aud = str(claims.get("aud", "<none>"))
|
||||
ok = aud == want_aud or aud == want_aud.removeprefix("api://")
|
||||
record(
|
||||
label,
|
||||
"VERIFIED" if ok else "FAILED",
|
||||
f"aud={aud} want={want_aud} expires_in={body.get('expires_in')} "
|
||||
f"new_rt={redact(body.get('refresh_token'))}",
|
||||
)
|
||||
new_rt = body.get("refresh_token")
|
||||
return str(new_rt) if isinstance(new_rt, str) else None
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"ENTRA_TENANT_ID",
|
||||
"ENTRA_CLIENT_ID",
|
||||
"ENTRA_CLIENT_SECRET",
|
||||
"SPIKE_AUDIENCE_A",
|
||||
"SPIKE_AUDIENCE_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in required if k in os.environ}
|
||||
missing = [k for k in required if k not in cfg]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)}\nSee module docstring.")
|
||||
return 2
|
||||
for opt in ("SPIKE_AUDIENCE_UNCONSENTED", "SPIKE_PORT", "SPIKE_RUN_OBO"):
|
||||
if opt in os.environ:
|
||||
cfg[opt] = os.environ[opt]
|
||||
|
||||
# V1 - capture
|
||||
tokens = interactive_login(cfg)
|
||||
rt0 = tokens.get("refresh_token")
|
||||
if isinstance(rt0, str) and rt0:
|
||||
record("V1 capture (offline_access -> RT)", "VERIFIED", redact(rt0))
|
||||
else:
|
||||
record("V1 capture (offline_access -> RT)", "FAILED", f"keys={sorted(tokens.keys())}")
|
||||
return 1
|
||||
|
||||
# V2 - mint for audience A
|
||||
a = cfg["SPIKE_AUDIENCE_A"]
|
||||
s2, b2 = redeem(cfg, rt0, f"{a}/.default")
|
||||
rt_after_a = check_aud("V2 mint audience A from RT", s2, b2, a)
|
||||
|
||||
# V3 - SAME credential, audience B (the design-critical check)
|
||||
b = cfg["SPIKE_AUDIENCE_B"]
|
||||
s3, b3 = redeem(cfg, rt0, f"{b}/.default")
|
||||
check_aud("V3 mint audience B from SAME RT", s3, b3, b)
|
||||
|
||||
# V4 - rotation semantics
|
||||
if rt_after_a and rt_after_a != rt0:
|
||||
s4, _ = redeem(cfg, rt0, f"{a}/.default")
|
||||
record(
|
||||
"V4 rotation (new RT returned; old still valid?)",
|
||||
"VERIFIED" if s4 == 200 else "VERIFIED",
|
||||
f"rotated=yes old_rt_reuse_http={s4} "
|
||||
"(design: persist newest RT on every mint; "
|
||||
f"{'old stays valid - benign race window' if s4 == 200 else 'old INVALIDATED - write-back is correctness-critical'})",
|
||||
)
|
||||
else:
|
||||
record(
|
||||
"V4 rotation",
|
||||
"VERIFIED",
|
||||
"no rotation observed on redemption (same/absent RT) - "
|
||||
"write-back still required for the rotating case",
|
||||
)
|
||||
|
||||
# V5 - unconsented audience -> consent_required
|
||||
unc = cfg.get("SPIKE_AUDIENCE_UNCONSENTED")
|
||||
if unc:
|
||||
s5, b5 = redeem(cfg, rt0, f"{unc}/.default")
|
||||
codes = b5.get("error_codes", [])
|
||||
hit = s5 == 400 and (65001 in codes or b5.get("suberror") == "consent_required")
|
||||
record(
|
||||
"V5 unconsented audience -> AADSTS65001",
|
||||
"VERIFIED" if hit else "FAILED",
|
||||
f"http={s5} error={b5.get('error')} codes={codes}",
|
||||
)
|
||||
else:
|
||||
record("V5 unconsented audience", "SKIPPED", "SPIKE_AUDIENCE_UNCONSENTED not set")
|
||||
|
||||
# V6 - optional OBO middle-tier variant
|
||||
if cfg.get("SPIKE_RUN_OBO") == "1":
|
||||
s6a, b6a = redeem(cfg, rt0, f"{cfg['ENTRA_CLIENT_ID']}/.default")
|
||||
at_self = b6a.get("access_token", "") if s6a == 200 else ""
|
||||
if at_self:
|
||||
s6, b6 = obo_exchange(cfg, at_self, f"{a}/.default")
|
||||
check_aud("V6 OBO jwt-bearer variant", s6, b6, a)
|
||||
else:
|
||||
record(
|
||||
"V6 OBO jwt-bearer variant",
|
||||
"FAILED",
|
||||
f"could not mint self-audience assertion: HTTP {s6a}",
|
||||
)
|
||||
else:
|
||||
record("V6 OBO jwt-bearer variant", "SKIPPED", "SPIKE_RUN_OBO != 1")
|
||||
|
||||
print("\n=== summary ===")
|
||||
for check, status, _ in RESULTS:
|
||||
print(f" {status:>8} {check}")
|
||||
return 0 if all(s != "FAILED" for _, s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,271 +0,0 @@
|
||||
"""End-to-end exercise of the oauth_obo feature on the OSS path (RFC 8693).
|
||||
|
||||
Parallel to ``entra_e2e.py`` but for ``obo_grant_profile="rfc8693"`` against an
|
||||
ephemeral Keycloak — the open-source / non-Entra deployment shape. Fully
|
||||
headless (password grant, no browser), so it runs unattended.
|
||||
|
||||
Drives the REAL Turnstone code: ``MCPTokenStore.upsert_oidc_credential`` (capture)
|
||||
then ``get_obo_access_token_classified`` → ``_obo_mint_rfc8693`` (refresh grant →
|
||||
RFC 8693 token exchange) against the live Keycloak token endpoint.
|
||||
|
||||
Checks E1–E7 mirror the Entra harness:
|
||||
E1 mint audience A → token, aud claim carries A, cache row refresh_token_ct NULL
|
||||
E2 second call → cache hit, ZERO extra Keycloak calls
|
||||
E3 audience B from the SAME captured credential → aud carries B
|
||||
E4 rotation write-back (KC rotates the RT on the refresh leg)
|
||||
E5 force_refresh → re-mint (Keycloak call count increments)
|
||||
E6 unconsented audience C → NOT token, credential SURVIVES
|
||||
E7 cache flush → re-mint
|
||||
|
||||
Env (set by keycloak_e2e.sh):
|
||||
KC_TOKEN_ENDPOINT, KC_ISSUER, KC_CLIENT_ID, KC_CLIENT_SECRET,
|
||||
KC_USER, KC_PASSWORD, AUD_A, SCOPE_A, AUD_B, SCOPE_B, AUD_C
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from turnstone.core.mcp_crypto import (
|
||||
MCPTokenCipher,
|
||||
MCPTokenCipherConfig,
|
||||
MCPTokenStore,
|
||||
)
|
||||
from turnstone.core.mcp_oauth import get_obo_access_token_classified
|
||||
from turnstone.core.oidc import OIDCConfig
|
||||
from turnstone.core.storage._sqlite import SQLiteBackend
|
||||
|
||||
USER = "e2e-user"
|
||||
RESULTS: list[tuple[str, str]] = []
|
||||
|
||||
|
||||
def record(status: str, msg: str) -> None:
|
||||
RESULTS.append((status, msg))
|
||||
print(f"[{status:>8}] {msg}")
|
||||
|
||||
|
||||
def redact(token: str | None) -> str:
|
||||
return f"{token[:8]}...({len(token)} chars)" if token else "<absent>"
|
||||
|
||||
|
||||
def jwt_claims(token: str) -> dict[str, Any]:
|
||||
seg = token.split(".")[1]
|
||||
pad = "=" * (-len(seg) % 4)
|
||||
out: dict[str, Any] = json.loads(base64.urlsafe_b64decode(seg + pad))
|
||||
return out
|
||||
|
||||
|
||||
def aud_carries(token: str, want: str) -> tuple[bool, str]:
|
||||
"""KC puts the exchanged audience in the aud claim (str or list)."""
|
||||
aud = jwt_claims(token).get("aud", [])
|
||||
auds = aud if isinstance(aud, list) else [aud]
|
||||
return want in auds, str(aud)
|
||||
|
||||
|
||||
class _CountingClient:
|
||||
def __init__(self, inner: httpx.AsyncClient) -> None:
|
||||
self._inner = inner
|
||||
self.posts = 0
|
||||
|
||||
async def post(self, *args: Any, **kwargs: Any) -> httpx.Response:
|
||||
self.posts += 1
|
||||
return await self._inner.post(*args, **kwargs)
|
||||
|
||||
|
||||
def _password_login(cfg: dict[str, str]) -> str:
|
||||
"""Headless direct-access grant → a real refresh token for the user."""
|
||||
resp = httpx.post(
|
||||
cfg["KC_TOKEN_ENDPOINT"],
|
||||
data={
|
||||
"grant_type": "password",
|
||||
"client_id": cfg["KC_CLIENT_ID"],
|
||||
"client_secret": cfg["KC_CLIENT_SECRET"],
|
||||
"username": cfg["KC_USER"],
|
||||
"password": cfg["KC_PASSWORD"],
|
||||
"scope": "openid",
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return str(resp.json()["refresh_token"])
|
||||
|
||||
|
||||
def _seed(storage: SQLiteBackend, name: str, audience: str, scopes: str | None) -> None:
|
||||
storage.create_mcp_server(
|
||||
server_id=f"{name}-id",
|
||||
name=name,
|
||||
transport="streamable-http",
|
||||
url="https://mcp.example.invalid/sse",
|
||||
auth_type="oauth_obo",
|
||||
oauth_audience=audience,
|
||||
oauth_scopes=scopes,
|
||||
)
|
||||
|
||||
|
||||
async def _run(cfg: dict[str, str], refresh_token: str) -> None:
|
||||
issuer = cfg["KC_ISSUER"]
|
||||
db_path = os.path.join(tempfile.mkdtemp(prefix="obo-kc-e2e-"), "e2e.db")
|
||||
storage = SQLiteBackend(db_path)
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
raw = base64.urlsafe_b64decode(Fernet.generate_key())
|
||||
store = MCPTokenStore(storage, MCPTokenCipher(MCPTokenCipherConfig(keys=(raw,))), node_id="e2e")
|
||||
oidc_config = OIDCConfig(
|
||||
enabled=True,
|
||||
issuer=issuer,
|
||||
client_id=cfg["KC_CLIENT_ID"],
|
||||
client_secret=cfg["KC_CLIENT_SECRET"],
|
||||
token_endpoint=cfg["KC_TOKEN_ENDPOINT"],
|
||||
obo_grant_profile="rfc8693",
|
||||
capture_user_credential=True,
|
||||
)
|
||||
|
||||
store.upsert_oidc_credential(USER, issuer, refresh_token=refresh_token)
|
||||
cap = store.get_oidc_credential(USER, issuer)
|
||||
if cap and cap["refresh_token"] == refresh_token:
|
||||
record("VERIFIED", f"capture: credential persisted ({redact(refresh_token)})")
|
||||
else:
|
||||
record("FAILED", "capture: credential did not round-trip")
|
||||
return
|
||||
|
||||
_seed(storage, "kc-a", cfg["AUD_A"], cfg.get("SCOPE_A"))
|
||||
_seed(storage, "kc-b", cfg["AUD_B"], cfg.get("SCOPE_B"))
|
||||
if cfg.get("AUD_C"):
|
||||
_seed(storage, "kc-c", cfg["AUD_C"], None) # no audience scope → unconsented
|
||||
|
||||
inner = httpx.AsyncClient(timeout=20.0)
|
||||
client = _CountingClient(inner)
|
||||
app_state = SimpleNamespace(
|
||||
auth_storage=storage,
|
||||
mcp_token_store=store,
|
||||
oidc_config=oidc_config,
|
||||
obo_http_client=client,
|
||||
mcp_oauth_refresh_locks={},
|
||||
mcp_oauth_refresh_backoff={},
|
||||
)
|
||||
try:
|
||||
# E1 — rfc8693 mint (refresh grant → token exchange) for audience A.
|
||||
r = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
if r.kind == "token" and r.token:
|
||||
ok, aud = aud_carries(r.token, cfg["AUD_A"])
|
||||
row = storage.get_mcp_user_token(USER, "kc-a")
|
||||
cache_ok = row is not None and row["refresh_token_ct"] is None
|
||||
record(
|
||||
"VERIFIED" if ok and cache_ok else "FAILED",
|
||||
f"E1 mint A (refresh→exchange): kind=token aud={aud} want={cfg['AUD_A']} "
|
||||
f"cache_row_refreshless={cache_ok}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E1 mint A: kind={r.kind} (expected token)")
|
||||
return
|
||||
|
||||
# E2 — cache hit.
|
||||
posts_before = client.posts
|
||||
r2 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r2.kind == "token" and client.posts == posts_before else "FAILED",
|
||||
f"E2 cache hit: kind={r2.kind} extra_kc_calls={client.posts - posts_before} (want 0)",
|
||||
)
|
||||
|
||||
# E3 — audience B from the SAME credential.
|
||||
rb = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-b"
|
||||
)
|
||||
if rb.kind == "token" and rb.token:
|
||||
ok_b, aud_b = aud_carries(rb.token, cfg["AUD_B"])
|
||||
record(
|
||||
"VERIFIED" if ok_b else "FAILED",
|
||||
f"E3 mint B from SAME credential: aud={aud_b} want={cfg['AUD_B']}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E3 mint B: kind={rb.kind}")
|
||||
|
||||
# E4 — rotation write-back (KC rotates the RT on the refresh leg).
|
||||
cred_now = store.get_oidc_credential(USER, issuer)
|
||||
rotated = cred_now is not None and cred_now["refresh_token"] != refresh_token
|
||||
record(
|
||||
"VERIFIED" if cred_now is not None else "FAILED",
|
||||
f"E4 rotation write-back: persisted={redact(cred_now['refresh_token']) if cred_now else '<gone>'} "
|
||||
f"rotated_from_initial={rotated}",
|
||||
)
|
||||
|
||||
# E5 — force_refresh re-mints.
|
||||
posts_before = client.posts
|
||||
rf = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a", force_refresh=True
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if rf.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E5 force_refresh re-mint: kind={rf.kind} kc_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# E6 — unconsented audience: not a token, credential survives.
|
||||
if cfg.get("AUD_C"):
|
||||
rc = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-c"
|
||||
)
|
||||
cred_after = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if rc.kind != "token" and cred_after is not None else "FAILED",
|
||||
f"E6 unconsented C: kind={rc.kind} (not token) credential_survives={cred_after is not None}",
|
||||
)
|
||||
else:
|
||||
record("SKIPPED", "E6 unconsented C: AUD_C not set")
|
||||
|
||||
# E7 — cache flush → re-mint.
|
||||
store.delete_user_token(USER, "kc-a")
|
||||
posts_before = client.posts
|
||||
r7 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r7.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E7 flush→re-mint: kind={r7.kind} kc_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
finally:
|
||||
await inner.aclose()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"KC_TOKEN_ENDPOINT",
|
||||
"KC_ISSUER",
|
||||
"KC_CLIENT_ID",
|
||||
"KC_CLIENT_SECRET",
|
||||
"KC_USER",
|
||||
"KC_PASSWORD",
|
||||
"AUD_A",
|
||||
"AUD_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in os.environ if k.startswith(("KC_", "AUD_", "SCOPE_"))}
|
||||
missing = [k for k in required if not cfg.get(k)]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)} — run via keycloak_e2e.sh")
|
||||
return 2
|
||||
|
||||
print("Headless password login to Keycloak (the credential the feature captures)...")
|
||||
refresh_token = _password_login(cfg)
|
||||
|
||||
asyncio.run(_run(cfg, refresh_token))
|
||||
|
||||
print("\n=== summary ===")
|
||||
for status, msg in RESULTS:
|
||||
print(f" {status:>8} {msg}")
|
||||
return 0 if all(s in ("VERIFIED", "SKIPPED") for s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,65 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# OSS-path (RFC 8693) end-to-end: spin up ephemeral Keycloak, configure the
|
||||
# realm, run keycloak_e2e.py against the REAL Turnstone mint engine, tear down.
|
||||
# Fully headless — no browser. Manual test tooling, not run in CI.
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/../.." # repo root (uv run needs it)
|
||||
|
||||
CONTAINER=kc-obo-e2e
|
||||
PORT=8091
|
||||
KC="docker exec $CONTAINER /opt/keycloak/bin/kcadm.sh"
|
||||
|
||||
cleanup() { docker rm -f "$CONTAINER" >/dev/null 2>&1 || true; }
|
||||
trap cleanup EXIT
|
||||
cleanup
|
||||
|
||||
echo ">> starting Keycloak 26.3 (ephemeral)..."
|
||||
docker run -d --name "$CONTAINER" -p "127.0.0.1:${PORT}:8080" \
|
||||
-e KC_BOOTSTRAP_ADMIN_USERNAME=admin -e KC_BOOTSTRAP_ADMIN_PASSWORD=admin \
|
||||
quay.io/keycloak/keycloak:26.3 start-dev >/dev/null
|
||||
|
||||
echo ">> waiting for Keycloak (dev-mode boot can take a few minutes on a loaded host)..."
|
||||
# Wait on kcadm auth succeeding directly — more reliable than the host HTTP port,
|
||||
# and generous enough for a resource-starved boot (up to ~6 min).
|
||||
ready=""
|
||||
for _ in $(seq 1 90); do
|
||||
if $KC config credentials --server http://localhost:8080 --realm master \
|
||||
--user admin --password admin >/dev/null 2>&1; then
|
||||
ready=1
|
||||
break
|
||||
fi
|
||||
sleep 4
|
||||
done
|
||||
[ -n "$ready" ] || { echo "Keycloak did not become ready in time"; docker logs "$CONTAINER" 2>&1 | tail -15; exit 1; }
|
||||
|
||||
echo ">> configuring realm 'spike'..."
|
||||
$KC create realms -s realm=spike -s enabled=true >/dev/null
|
||||
# Confidential client with standard token exchange (the RFC 8693 leg) + direct
|
||||
# access grant (headless password login to fetch the user's refresh token).
|
||||
$KC create clients -r spike -s clientId=turnstone -s enabled=true -s publicClient=false \
|
||||
-s secret=spike-secret -s directAccessGrantsEnabled=true \
|
||||
-s 'attributes={"standard.token.exchange.enabled":"true"}' >/dev/null
|
||||
for t in mcp-a mcp-b mcp-c; do
|
||||
$KC create clients -r spike -s clientId=$t -s enabled=true -s publicClient=false -s secret=x >/dev/null
|
||||
done
|
||||
$KC create users -r spike -s username=e2e-user -s enabled=true -s email=e2e@spike.test \
|
||||
-s emailVerified=true -s firstName=E2E -s lastName=User >/dev/null
|
||||
$KC set-password -r spike --username e2e-user --new-password e2e-pw >/dev/null
|
||||
|
||||
TURNSTONE_UUID=$($KC get clients -r spike -q clientId=turnstone --fields id --format csv --noquotes)
|
||||
# Audience client scopes for mcp-a and mcp-b ONLY (mcp-c stays unconsented → E6).
|
||||
for t in mcp-a mcp-b; do
|
||||
SID=$($KC create client-scopes -r spike -s name=aud-$t -s protocol=openid-connect -i)
|
||||
$KC create "client-scopes/$SID/protocol-mappers/models" -r spike -s name=aud-$t \
|
||||
-s protocol=openid-connect -s protocolMapper=oidc-audience-mapper \
|
||||
-s "config={\"included.client.audience\":\"$t\",\"access.token.claim\":\"true\"}" >/dev/null
|
||||
$KC update "clients/$TURNSTONE_UUID/optional-client-scopes/$SID" -r spike >/dev/null
|
||||
done
|
||||
|
||||
echo ">> running the product e2e harness..."
|
||||
export KC_TOKEN_ENDPOINT="http://127.0.0.1:${PORT}/realms/spike/protocol/openid-connect/token"
|
||||
export KC_ISSUER="http://127.0.0.1:${PORT}/realms/spike"
|
||||
export KC_CLIENT_ID=turnstone KC_CLIENT_SECRET=spike-secret
|
||||
export KC_USER=e2e-user KC_PASSWORD=e2e-pw
|
||||
export AUD_A=mcp-a SCOPE_A=aud-mcp-a AUD_B=mcp-b SCOPE_B=aud-mcp-b AUD_C=mcp-c
|
||||
uv run python scripts/obo-e2e/keycloak_e2e.py
|
||||
@@ -1,962 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Browser-level SSE recovery livepass — boots the REAL interactive.js
|
||||
``InteractivePane`` against a REAL Turnstone node and drives the two
|
||||
headline recovery scenarios through headless Chrome over CDP, stamping
|
||||
``document.title`` verdicts (``RECOVERY-READY-*`` / ``RECOVERY-FAILED-*``)
|
||||
the livepass convention.
|
||||
|
||||
Unlike ``scripts/livepass.py`` (which stubs ``window.authFetch`` with
|
||||
canned fixtures), this page uses the REAL auth + REAL EventSource against a
|
||||
REAL node: the page is served same-origin by the node itself (so cookie
|
||||
auth and EventSource just work), the node runs a scripted-provider
|
||||
workstream so a REST ``/send`` drives a real bash storm, and the pane's
|
||||
own state machine (interactive.js) does the recovery.
|
||||
|
||||
Usage::
|
||||
|
||||
python3 scripts/recovery_e2e.py # run all three scenarios
|
||||
python3 scripts/recovery_e2e.py --scenario storm
|
||||
python3 scripts/recovery_e2e.py --scenario restart
|
||||
python3 scripts/recovery_e2e.py --scenario coord-restart
|
||||
python3 scripts/recovery_e2e.py --scenario both # A+B only (legacy)
|
||||
python3 scripts/recovery_e2e.py --keep-open 8971 # serve the storm page
|
||||
# for manual inspection
|
||||
|
||||
Scenario A (storm): the page connects, POSTs ``/send`` on stream-open (so
|
||||
the listener is registered first), the node runs a 4-parallel-bash
|
||||
``seq 1 500`` storm plus a task_agent whose sub-tools are chatty bashes;
|
||||
the page asserts the final DOM has the expected top-level tool rows, the
|
||||
task_agent card nests its sub-tool rows (NO child escaped to the top
|
||||
level), and the composer settles idle. Stamps ``RECOVERY-READY-STORM-<n>``.
|
||||
|
||||
Scenario B (hide mid-turn -> restart -> show): the runner hides the tab
|
||||
the moment the first streamed line paints (freezing the pane's cursor at
|
||||
a mid-turn event id — the MessageEvent ``lastEventId`` capture is what
|
||||
makes that cursor real; the pre-2026-07 object-form read left it null and
|
||||
this whole path unassertable), lets the turn and a follow-up text commit
|
||||
while hidden, restarts the node on the SAME port (fresh empty ring,
|
||||
storage-seeded counter), then shows the tab. The show-edge reconnect
|
||||
presents the stale cursor, MUST draw ``replay_truncated`` (asserted:
|
||||
trunc>=1), the truncated resync rebuilds from /history, and the turns
|
||||
committed during the hide window MUST be present afterwards (asserted:
|
||||
``healed`` — the 'turn disappeared' field symptom). Stamps
|
||||
``RECOVERY-READY-RESTART-rows<n>-trunc<n>``. The exact ``lost_count``
|
||||
arithmetic and the failed-resync retry stay at the server-contract level
|
||||
in Tier 1's ``test_restart_truncated_honesty`` /
|
||||
``test_failed_resync_retries_via_truncation_record``.
|
||||
|
||||
A NOTE ON THE BROWSER OVERFLOW (server-side poison): a real listener-queue
|
||||
poison needs the browser to STOP reading the socket so TCP backpressure
|
||||
reaches the server. A backgrounded/CPU-throttled tab does NOT do this --
|
||||
Chrome's network stack keeps draining the socket regardless of JS
|
||||
throttling, and interactive.js deliberately CLOSES the stream on tab-hide
|
||||
rather than starving it. So the server-side overflow -> stream_overflow ->
|
||||
reconnect path is NOT reliably forcible from a real browser (which is why
|
||||
that field bug was subtle); it is proven at the server-contract level in
|
||||
``tests/test_sse_recovery_e2e.py::test_slow_consumer_overflow_then_lossless_reconnect``.
|
||||
Scenario A here proves the OTHER half at the browser level: fix-3's
|
||||
de-amplified storm renders correctly with no escaped sub-agent children.
|
||||
|
||||
MANUAL RUNBOOK (if Chrome/CDP is unavailable): run this with
|
||||
``--keep-open PORT`` to boot the node + serve the storm page, open the
|
||||
printed URL in a browser (the script prints the auth cookie to set), and
|
||||
watch ``document.title``. For the restart scenario, boot with a fixed
|
||||
port, load the restart page, background the tab, restart the node
|
||||
(``RecoveryServer`` on the same port), foreground the tab, and watch the
|
||||
title settle to ``RECOVERY-READY-RESTART``.
|
||||
|
||||
Scenario C (coord-restart): the REAL coordinator pane
|
||||
(console/static/coordinator/coordinator.js — the #882 parity port of the
|
||||
same truncated-recovery machinery) driven through the SAME hide -> restart
|
||||
-> show sequence as Scenario B. The coordinator only runs under the
|
||||
console app in production, and the console's coordinator subsystems build
|
||||
inside its server lifespan against a config-resolved model registry — no
|
||||
``create_app(prebuilt SessionManager)`` seam for this harness's scripted
|
||||
provider. So the scenario mounts the pane against the interactive
|
||||
recovery node instead: the node serves the console's coordinator static
|
||||
tree at ``/coord-static`` (a distinct prefix — the node's own ``/static``
|
||||
mount would swallow the console path) and a pane-only page at
|
||||
``/coord-recovery``; the pane's module imports are all absolute
|
||||
``/shared/*`` and resolve against the node. Fidelity caveats, all inert
|
||||
for the recovery machinery under test: the workstream is
|
||||
interactive-kind (no coordinator status events — the status bar keeps its
|
||||
placeholder), and ``/children`` + ``/tasks`` 404 here (the pane's loaders
|
||||
catch and render empty by design). What IS real: the full chrome
|
||||
(buildCoordChrome), cookie auth, EventSource + MessageEvent cursor
|
||||
capture, the connect chokepoint, the dead-stream resync
|
||||
(loadHistoryThenReconnect), the churn limiter, and the jitter. Asserted:
|
||||
the show-edge reconnect draws ``replay_truncated`` (trunc>=1, counted at
|
||||
the transport by a page-side EventSource wrapper — the coordinator's
|
||||
handleEvent, cursor, and even its SSE indicator are closure-private or
|
||||
deliberately absent from the chrome), the resync rebuilds from /history
|
||||
with the hidden-window turns present (``healed``), tool rows intact, the
|
||||
stream re-opened post-show, the status bar not stuck dim, and idle
|
||||
asserted server-side by the runner. Stamps
|
||||
``RECOVERY-READY-COORD-rows<n>-trunc<n>``.
|
||||
|
||||
MANUAL COORDINATOR RUNBOOK (real console topology, no CDP): boot a dev
|
||||
console + one node (docker-compose dev cluster), open a coordinator with
|
||||
running children, hide the tab mid-turn, restart the CONSOLE process (the
|
||||
coordinator ring lives there), show the tab, and verify: the pane draws
|
||||
one truncated full rebuild (no blank pane), the mid-run turn's tool rows
|
||||
re-appear inside their batch (no standalone top-level orphan bubbles),
|
||||
and turns committed while hidden are present.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import contextlib
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import socket
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
# Healed-gap sentinel for scenario B: injected as the scripted turn-2 text
|
||||
# AND threaded to the page via ``?healed=`` (read into ``healedSentinel``),
|
||||
# so the injected text and the DOM check share one definition. Must never
|
||||
# collide with rendered command/output text — the bash command row paints
|
||||
# its shell source verbatim, which contains the keyword ``done``.
|
||||
HEALED_SENTINEL = "HEALED-e5b1"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The recovery page — served same-origin by the node at /recovery.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
PAGE_HTML = r"""<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>recovery livepass</title>
|
||||
<link rel="stylesheet" href="/shared/base.css" />
|
||||
<link rel="stylesheet" href="/shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="/shared/chat.css" />
|
||||
<link rel="stylesheet" href="/shared/conversation.css" />
|
||||
<link rel="stylesheet" href="/shared/cards.css" />
|
||||
<link rel="stylesheet" href="/shared/interactive.css" />
|
||||
<style>
|
||||
body { margin: 0; background: var(--bg); color: var(--ink); }
|
||||
#mount { height: 100vh; display: flex; }
|
||||
#mount > * { flex: 1; min-height: 0; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div id="header"><div id="status-bar"></div></div>
|
||||
<div id="mount"></div>
|
||||
<script>
|
||||
// Minimal globals interactive.js reads on the standalone path.
|
||||
window.showToast = function (m) { console.log("toast:", m); };
|
||||
window.showLogin = function () {};
|
||||
</script>
|
||||
<script type="module">
|
||||
import { InteractivePane } from "/shared/interactive.js";
|
||||
|
||||
const q = new URLSearchParams(location.search);
|
||||
const wsId = q.get("ws_id");
|
||||
const scenario = q.get("scenario") || "storm";
|
||||
const expectRows = parseInt(q.get("rows") || "4", 10);
|
||||
// Healed-gap sentinel, threaded from the runner (HEALED_SENTINEL)
|
||||
// so the injected turn text and this check cannot drift apart.
|
||||
const healedSentinel = q.get("healed") || "";
|
||||
|
||||
// REAL pane against THIS origin (base=""): real authFetch (cookie) and
|
||||
// real EventSource. The default host provides all SSE seams.
|
||||
const pane = new InteractivePane(wsId, { base: "" });
|
||||
document.getElementById("mount").appendChild(pane.el);
|
||||
pane.wsId = wsId;
|
||||
window.__pane = pane;
|
||||
|
||||
window.__hide = function () {
|
||||
Object.defineProperty(document, "hidden", { configurable: true, value: true });
|
||||
Object.defineProperty(document, "visibilityState", { configurable: true, value: "hidden" });
|
||||
document.dispatchEvent(new Event("visibilitychange"));
|
||||
};
|
||||
window.__show = function () {
|
||||
Object.defineProperty(document, "hidden", { configurable: true, value: false });
|
||||
Object.defineProperty(document, "visibilityState", { configurable: true, value: "visible" });
|
||||
document.dispatchEvent(new Event("visibilitychange"));
|
||||
};
|
||||
|
||||
// Count top-level tool rows and escaped sub-agent children.
|
||||
function domCounts() {
|
||||
const topRows = pane.messagesEl.querySelectorAll(
|
||||
".conv-batch > .conv-row[data-call-id]"
|
||||
);
|
||||
let topLevel = 0;
|
||||
let escapedChildren = 0;
|
||||
topRows.forEach((r) => {
|
||||
const cid = r.dataset.callId || "";
|
||||
if (cid.includes("::")) escapedChildren += 1; // a child at the top level
|
||||
else topLevel += 1;
|
||||
});
|
||||
const agentCard = pane.messagesEl.querySelector(".conv-agent");
|
||||
const nested = pane.messagesEl.querySelectorAll(
|
||||
".conv-agent .conv-row[data-call-id]"
|
||||
).length;
|
||||
return { topLevel, escapedChildren, agentCard: !!agentCard, nested };
|
||||
}
|
||||
|
||||
let sent = false;
|
||||
function sendOnce(msg) {
|
||||
if (sent) return;
|
||||
sent = true;
|
||||
// The pane's SSE is open (host.onStreamOpen fired), so the listener is
|
||||
// registered before this /send -- no missed events.
|
||||
window
|
||||
.authFetch("/v1/api/workstreams/" + encodeURIComponent(wsId) + "/send", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ message: msg }),
|
||||
})
|
||||
.catch((e) => { document.title = "RECOVERY-FAILED-send-" + e; });
|
||||
}
|
||||
|
||||
// Drive /send once the stream is live (wrap the default host hook).
|
||||
const origOpen = pane._host.onStreamOpen.bind(pane._host);
|
||||
pane._host.onStreamOpen = function (p) {
|
||||
origOpen(p);
|
||||
window.__streamOpen = (window.__streamOpen || 0) + 1;
|
||||
if (scenario === "storm") sendOnce("run the storm");
|
||||
else if (scenario === "restart" && window.__streamOpen === 1) sendOnce("run a turn");
|
||||
};
|
||||
|
||||
// First paint the REAL way: /history then connect SSE.
|
||||
pane._loadHistoryThenConnect(wsId);
|
||||
|
||||
if (scenario === "storm") {
|
||||
const deadline = Date.now() + 40000;
|
||||
const poll = () => {
|
||||
const c = domCounts();
|
||||
const idle = !pane.busy;
|
||||
if (c.topLevel >= expectRows && c.agentCard && c.nested >= 2 && idle) {
|
||||
document.title = c.escapedChildren
|
||||
? "RECOVERY-FAILED-escaped-" + c.escapedChildren
|
||||
: "RECOVERY-READY-STORM-" + c.topLevel + "-nested-" + c.nested;
|
||||
return;
|
||||
}
|
||||
if (Date.now() > deadline) {
|
||||
document.title =
|
||||
"RECOVERY-FAILED-STORM-top" + c.topLevel + "-agent" + (c.agentCard ? 1 : 0) +
|
||||
"-nested" + c.nested + "-escaped" + c.escapedChildren + "-busy" + (pane.busy ? 1 : 0);
|
||||
return;
|
||||
}
|
||||
setTimeout(poll, 200);
|
||||
};
|
||||
setTimeout(poll, 400);
|
||||
} else if (scenario === "restart") {
|
||||
// The runner drives hide -> (restart node) -> show via window.__hide/
|
||||
// __show. We watch for the truncated-triggered rebuild + idle settle.
|
||||
window.__truncatedSeen = 0;
|
||||
const origHandle = pane.handleEvent.bind(pane);
|
||||
pane.handleEvent = function (ev) {
|
||||
if (ev && ev.type === "replay_truncated") window.__truncatedSeen += 1;
|
||||
return origHandle(ev);
|
||||
};
|
||||
window.__verifyRestart = function () {
|
||||
// Browser-level restart RECOVERY, full contract: the runner hid
|
||||
// the tab MID-turn (cursor frozen below the commits that land
|
||||
// while hidden), so the show-edge reconnect must present the
|
||||
// stale cursor and draw ``replay_truncated`` (REQUIRED since the
|
||||
// MessageEvent lastEventId capture fix — the pre-fix object-form
|
||||
// read left manual reconnects cursorless and this envelope
|
||||
// unreachable, which is why trunc used to report 0), the
|
||||
// truncated resync must rebuild from /history, and the turns
|
||||
// committed DURING the hide window must be present afterwards
|
||||
// (``healed`` — the 'turn disappeared' field symptom). Composer
|
||||
// idle, status bar not stuck disconnected.
|
||||
const c = domCounts();
|
||||
const idle = !pane.busy;
|
||||
const disc = document.querySelector(".ws-sb-disconnected") !== null;
|
||||
// Sentinel must be collision-proof against everything else the
|
||||
// transcript renders: the paced bash COMMAND row paints its
|
||||
// shell text verbatim (buildConvCmd), which contains the
|
||||
// keyword ``done`` — a plain-word sentinel is vacuously
|
||||
// present whether or not the hidden-window turn survived.
|
||||
// The value rides the ?healed= param (single source:
|
||||
// HEALED_SENTINEL in the runner).
|
||||
const healed =
|
||||
healedSentinel !== "" &&
|
||||
(pane.messagesEl.textContent || "").includes(healedSentinel);
|
||||
const ok =
|
||||
c.topLevel >= 1 &&
|
||||
idle &&
|
||||
!disc &&
|
||||
healed &&
|
||||
window.__truncatedSeen >= 1;
|
||||
document.title = ok
|
||||
? "RECOVERY-READY-RESTART-rows" + c.topLevel + "-trunc" + window.__truncatedSeen
|
||||
: "RECOVERY-FAILED-RESTART-rows" + c.topLevel +
|
||||
"-busy" + (pane.busy ? 1 : 0) + "-disc" + (disc ? 1 : 0) +
|
||||
"-healed" + (healed ? 1 : 0) + "-trunc" + window.__truncatedSeen;
|
||||
};
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The coordinator recovery page — served same-origin by the node at
|
||||
# /coord-recovery. A near-clone of the production standalone page
|
||||
# (console/static/coordinator/index.html): the same /shared script
|
||||
# substrate (classic theme.js first, then the deferred module set), the
|
||||
# same createCoordinatorPane(document.body, wsId, {standalone:true}) +
|
||||
# connect() bootstrap — with the coordinator files imported from
|
||||
# /coord-static (see the module docstring) and Google-fonts dropped
|
||||
# (hermetic run). Scenario instrumentation reads only public chrome ids.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
COORD_PAGE_HTML = r"""<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>coord recovery livepass</title>
|
||||
<link rel="stylesheet" href="/shared/base.css" />
|
||||
<link rel="stylesheet" href="/shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="/shared/chat.css" />
|
||||
<link rel="stylesheet" href="/shared/conversation.css" />
|
||||
<link rel="stylesheet" href="/shared/mcp_error.css" />
|
||||
<link rel="stylesheet" href="/coord-static/coordinator.css" />
|
||||
<link rel="stylesheet" href="/coord-static/coord-chrome.css" />
|
||||
<style>
|
||||
body { height: 100vh; margin: 0; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<script>
|
||||
// Transport-level instrumentation: the coordinator's handleEvent and
|
||||
// cursor are closure-private (unlike interactive's class methods), and
|
||||
// its chrome deliberately builds NO header/SSE indicator — so every
|
||||
// scenario signal is read off the wire by wrapping EventSource BEFORE
|
||||
// any module loads (classic script = runs before the deferred module
|
||||
// set, so the pane's connectSSE always constructs the wrapper):
|
||||
// __truncatedSeen — replay_truncated frames (the envelope);
|
||||
// __esOpens — stream opens (drives the send; a listener is
|
||||
// registered before /send so no events are missed);
|
||||
// __idFrames — id-bearing frames, i.e. exactly the frames that
|
||||
// advance the pane's reconnect cursor (same
|
||||
// ``!= null && !== ""`` guard as the pane) — the
|
||||
// hide fires only after this proves a live mid-turn
|
||||
// cursor.
|
||||
window.__truncatedSeen = 0;
|
||||
window.__esOpens = 0;
|
||||
window.__idFrames = 0;
|
||||
(function () {
|
||||
const RealES = window.EventSource;
|
||||
function CountingES(url, opts) {
|
||||
const es = new RealES(url, opts);
|
||||
es.addEventListener("open", function () {
|
||||
window.__esOpens += 1;
|
||||
});
|
||||
es.addEventListener("message", function (e) {
|
||||
if (e.lastEventId != null && e.lastEventId !== "") {
|
||||
window.__idFrames += 1;
|
||||
}
|
||||
try {
|
||||
const d = JSON.parse(e.data);
|
||||
if (d && d.type === "replay_truncated") window.__truncatedSeen += 1;
|
||||
} catch (_) {}
|
||||
});
|
||||
return es;
|
||||
}
|
||||
CountingES.prototype = RealES.prototype;
|
||||
CountingES.CONNECTING = RealES.CONNECTING;
|
||||
CountingES.OPEN = RealES.OPEN;
|
||||
CountingES.CLOSED = RealES.CLOSED;
|
||||
window.EventSource = CountingES;
|
||||
})();
|
||||
</script>
|
||||
<script src="/shared/theme.js"></script>
|
||||
<script type="module" src="/shared/utils.js"></script>
|
||||
<script type="module" src="/shared/toast.js"></script>
|
||||
<script type="module" src="/shared/auth.js"></script>
|
||||
<script type="module" src="/shared/kb.js"></script>
|
||||
<script type="module" src="/shared/composer.js"></script>
|
||||
<script type="module" src="/shared/composer_attachments.js"></script>
|
||||
<script type="module" src="/shared/composer_queue.js"></script>
|
||||
<script type="module" src="/shared/status_bar.js"></script>
|
||||
<script type="module" src="/shared/renderer.js"></script>
|
||||
<script type="module">
|
||||
import { createCoordinatorPane } from "/coord-static/coordinator.js";
|
||||
|
||||
const q = new URLSearchParams(location.search);
|
||||
const wsId = q.get("ws_id");
|
||||
const healedSentinel = q.get("healed") || "";
|
||||
|
||||
const pane = createCoordinatorPane(document.body, wsId, {
|
||||
standalone: true,
|
||||
});
|
||||
window.__pane = pane;
|
||||
if (pane) pane.connect();
|
||||
|
||||
window.__hide = function () {
|
||||
Object.defineProperty(document, "hidden", { configurable: true, value: true });
|
||||
Object.defineProperty(document, "visibilityState", { configurable: true, value: "hidden" });
|
||||
document.dispatchEvent(new Event("visibilitychange"));
|
||||
};
|
||||
window.__show = function () {
|
||||
Object.defineProperty(document, "hidden", { configurable: true, value: false });
|
||||
Object.defineProperty(document, "visibilityState", { configurable: true, value: "visible" });
|
||||
document.dispatchEvent(new Event("visibilitychange"));
|
||||
};
|
||||
|
||||
// Drive /send once the stream has OPENED at the transport (__esOpens —
|
||||
// the pane's listener is registered by then, so no events are missed).
|
||||
// The chrome has no SSE pill to poll: the header was deliberately
|
||||
// dropped (see buildCoordChrome's comment).
|
||||
let sent = false;
|
||||
function sendOnce(msg) {
|
||||
if (sent) return;
|
||||
sent = true;
|
||||
window
|
||||
.authFetch("/v1/api/workstreams/" + encodeURIComponent(wsId) + "/send", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ message: msg }),
|
||||
})
|
||||
.catch((e) => { document.title = "RECOVERY-FAILED-COORD-send-" + e; });
|
||||
}
|
||||
const sendPoll = setInterval(() => {
|
||||
if (window.__esOpens >= 1) {
|
||||
clearInterval(sendPoll);
|
||||
sendOnce("run a turn");
|
||||
}
|
||||
}, 100);
|
||||
|
||||
window.__verifyCoordRestart = function () {
|
||||
// Same contract as Scenario B, read off the coordinator's public
|
||||
// chrome + the transport wrapper (idle is asserted SERVER-side by
|
||||
// the runner — the chrome has no state text element): the show-edge
|
||||
// reconnect must present the frozen mid-turn cursor and draw
|
||||
// replay_truncated (trunc>=1), the dead-stream resync must rebuild
|
||||
// from /history with the hidden-window turns present (healed), the
|
||||
// stream must have re-opened after the show (__esOpens >= 2), the
|
||||
// status bar must not be stuck dim (.ws-sb-disconnected removed by
|
||||
// the post-recovery onopen), and the tool rows must be intact.
|
||||
const messages = document.getElementById("coord-messages");
|
||||
const rows = messages
|
||||
? messages.querySelectorAll(".conv-row[data-call-id]").length
|
||||
: 0;
|
||||
const reopened = window.__esOpens >= 2;
|
||||
const disc =
|
||||
document.querySelector("#coord-status-bar.ws-sb-disconnected") !== null;
|
||||
const healed =
|
||||
healedSentinel !== "" &&
|
||||
((messages && messages.textContent) || "").includes(healedSentinel);
|
||||
const ok =
|
||||
rows >= 1 && reopened && !disc && healed && window.__truncatedSeen >= 1;
|
||||
document.title = ok
|
||||
? "RECOVERY-READY-COORD-rows" + rows + "-trunc" + window.__truncatedSeen
|
||||
: "RECOVERY-FAILED-COORD-rows" + rows +
|
||||
"-reopened" + (reopened ? 1 : 0) +
|
||||
"-disc" + (disc ? 1 : 0) + "-healed" + (healed ? 1 : 0) +
|
||||
"-trunc" + window.__truncatedSeen;
|
||||
};
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Minimal dependency-free CDP client (WebSocket over a raw socket).
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class CDP:
|
||||
"""Just enough Chrome DevTools Protocol: navigate, evaluate, set cookie."""
|
||||
|
||||
def __init__(self, ws_url: str) -> None:
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
u = urlsplit(ws_url)
|
||||
self._sock = socket.create_connection((u.hostname, u.port or 80), timeout=10)
|
||||
key = base64.b64encode(os.urandom(16)).decode()
|
||||
path = u.path + (f"?{u.query}" if u.query else "")
|
||||
handshake = (
|
||||
f"GET {path} HTTP/1.1\r\nHost: {u.hostname}:{u.port}\r\n"
|
||||
f"Upgrade: websocket\r\nConnection: Upgrade\r\n"
|
||||
f"Sec-WebSocket-Key: {key}\r\nSec-WebSocket-Version: 13\r\n\r\n"
|
||||
)
|
||||
self._sock.sendall(handshake.encode())
|
||||
resp = b""
|
||||
while b"\r\n\r\n" not in resp:
|
||||
resp += self._sock.recv(4096)
|
||||
if b" 101 " not in resp.split(b"\r\n", 1)[0]:
|
||||
raise RuntimeError(f"CDP websocket handshake failed: {resp[:80]!r}")
|
||||
self._id = 0
|
||||
self._rbuf = b""
|
||||
|
||||
def _send(self, payload: bytes) -> None:
|
||||
header = bytearray([0x81]) # FIN + text opcode
|
||||
mask = os.urandom(4)
|
||||
n = len(payload)
|
||||
if n < 126:
|
||||
header.append(0x80 | n)
|
||||
elif n < 65536:
|
||||
header.append(0x80 | 126)
|
||||
header += struct.pack(">H", n)
|
||||
else:
|
||||
header.append(0x80 | 127)
|
||||
header += struct.pack(">Q", n)
|
||||
header += mask
|
||||
self._sock.sendall(bytes(header) + bytes(b ^ mask[i % 4] for i, b in enumerate(payload)))
|
||||
|
||||
def _recv_exact(self, n: int) -> bytes:
|
||||
while len(self._rbuf) < n:
|
||||
chunk = self._sock.recv(65536)
|
||||
if not chunk:
|
||||
raise ConnectionError("CDP socket closed")
|
||||
self._rbuf += chunk
|
||||
out, self._rbuf = self._rbuf[:n], self._rbuf[n:]
|
||||
return out
|
||||
|
||||
def _recv_message(self) -> str:
|
||||
data = b""
|
||||
while True:
|
||||
b0, b1 = self._recv_exact(2)
|
||||
fin = b0 & 0x80
|
||||
length = b1 & 0x7F
|
||||
if length == 126:
|
||||
length = struct.unpack(">H", self._recv_exact(2))[0]
|
||||
elif length == 127:
|
||||
length = struct.unpack(">Q", self._recv_exact(8))[0]
|
||||
data += self._recv_exact(length)
|
||||
if fin:
|
||||
return data.decode("utf-8", "replace")
|
||||
|
||||
def cmd(self, method: str, params: dict[str, Any] | None = None, timeout: float = 15) -> Any:
|
||||
self._id += 1
|
||||
mid = self._id
|
||||
self._send(json.dumps({"id": mid, "method": method, "params": params or {}}).encode())
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
self._sock.settimeout(max(0.1, deadline - time.monotonic()))
|
||||
obj = json.loads(self._recv_message())
|
||||
if obj.get("id") == mid:
|
||||
if "error" in obj:
|
||||
raise RuntimeError(f"{method}: {obj['error']}")
|
||||
return obj.get("result", {})
|
||||
raise TimeoutError(method)
|
||||
|
||||
def evaluate(self, expression: str) -> Any:
|
||||
r = self.cmd(
|
||||
"Runtime.evaluate",
|
||||
{"expression": expression, "returnByValue": True, "awaitPromise": True},
|
||||
)
|
||||
return r.get("result", {}).get("value")
|
||||
|
||||
def title(self) -> str:
|
||||
return str(self.evaluate("document.title") or "")
|
||||
|
||||
def close(self) -> None:
|
||||
with contextlib.suppress(Exception):
|
||||
self._sock.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Chrome launch + node boot
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _find_chrome() -> str | None:
|
||||
for name in ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser"):
|
||||
p = shutil.which(name)
|
||||
if p:
|
||||
return p
|
||||
return None
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
with socket.socket() as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return int(s.getsockname()[1])
|
||||
|
||||
|
||||
def _launch_chrome(chrome: str, profile: Path) -> tuple[subprocess.Popen[bytes], int]:
|
||||
cdp_port = _free_port()
|
||||
proc = subprocess.Popen(
|
||||
[
|
||||
chrome,
|
||||
"--headless=new",
|
||||
"--disable-gpu",
|
||||
"--no-sandbox",
|
||||
"--no-first-run",
|
||||
"--disable-extensions",
|
||||
"--disable-background-timer-throttling",
|
||||
f"--remote-debugging-port={cdp_port}",
|
||||
f"--user-data-dir={profile}",
|
||||
"about:blank",
|
||||
],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
return proc, cdp_port
|
||||
|
||||
|
||||
def _page_ws_url(cdp_port: int, timeout: float = 15) -> str:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
with urllib.request.urlopen(f"http://127.0.0.1:{cdp_port}/json", timeout=2) as r:
|
||||
targets = json.loads(r.read())
|
||||
for t in targets:
|
||||
if t.get("type") == "page" and t.get("webSocketDebuggerUrl"):
|
||||
return str(t["webSocketDebuggerUrl"])
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(0.2)
|
||||
raise TimeoutError("no CDP page target")
|
||||
|
||||
|
||||
def _page_route() -> Any:
|
||||
from starlette.responses import HTMLResponse
|
||||
from starlette.routing import Route
|
||||
|
||||
async def recovery_page(_request: Any) -> HTMLResponse:
|
||||
return HTMLResponse(PAGE_HTML)
|
||||
|
||||
return Route("/recovery", recovery_page)
|
||||
|
||||
|
||||
def _coord_routes() -> list[Any]:
|
||||
"""The coordinator scenario's same-origin extras: the pane page, and the
|
||||
console's coordinator static tree under the ``/coord-static`` prefix —
|
||||
a DISTINCT prefix because the node's own ``/static`` mount (ui/static)
|
||||
matches first and would 404 the console path from inside its own tree.
|
||||
coordinator.js's module imports are all absolute ``/shared/*``, which
|
||||
the node already serves."""
|
||||
from starlette.responses import HTMLResponse
|
||||
from starlette.routing import Mount, Route
|
||||
from starlette.staticfiles import StaticFiles
|
||||
|
||||
import turnstone
|
||||
|
||||
coord_dir = Path(turnstone.__file__).resolve().parent / "console" / "static" / "coordinator"
|
||||
|
||||
async def coord_recovery_page(_request: Any) -> HTMLResponse:
|
||||
return HTMLResponse(COORD_PAGE_HTML)
|
||||
|
||||
return [
|
||||
Route("/coord-recovery", coord_recovery_page),
|
||||
Mount("/coord-static", app=StaticFiles(directory=str(coord_dir)), name="coord-static"),
|
||||
]
|
||||
|
||||
|
||||
def _boot_node(port: int = 0) -> Any:
|
||||
from tests._sse_recovery_server import RecoveryServer
|
||||
from turnstone.core.storage import init_storage, reset_storage
|
||||
|
||||
# The page route must bypass auth on first load (the cookie is set by the
|
||||
# runner via CDP BEFORE navigation), so make it public by prefixing under
|
||||
# a public path is unavailable here; instead the runner sets the cookie so
|
||||
# /recovery passes the middleware. init storage per boot (shared singleton).
|
||||
reset_storage()
|
||||
init_storage("sqlite", path=os.path.join(_scratch(), "recovery_e2e.db"), run_migrations=True)
|
||||
return RecoveryServer(extra_routes=[_page_route(), *_coord_routes()], port=port)
|
||||
|
||||
|
||||
def _scratch() -> str:
|
||||
d = os.environ.get("RECOVERY_E2E_TMP") or "/tmp/recovery_e2e"
|
||||
os.makedirs(d, exist_ok=True)
|
||||
return d
|
||||
|
||||
|
||||
def _set_cookie_and_navigate(cdp: CDP, base_url: str, token: str, page_url: str) -> None:
|
||||
cdp.cmd("Page.enable")
|
||||
cdp.cmd("Runtime.enable")
|
||||
cdp.cmd("Network.enable")
|
||||
cdp.cmd(
|
||||
"Network.setCookie",
|
||||
{
|
||||
"name": "turnstone_auth_server",
|
||||
"value": token,
|
||||
"url": base_url,
|
||||
"path": "/",
|
||||
},
|
||||
)
|
||||
cdp.cmd("Page.navigate", {"url": page_url})
|
||||
|
||||
|
||||
def _poll_title(cdp: CDP, timeout: float) -> str:
|
||||
deadline = time.monotonic() + timeout
|
||||
last = ""
|
||||
while time.monotonic() < deadline:
|
||||
last = cdp.title()
|
||||
if last.startswith("RECOVERY-"):
|
||||
return last
|
||||
time.sleep(0.3)
|
||||
return last or "RECOVERY-FAILED-timeout"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Scenarios
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _storm_scripts() -> tuple[Any, ...]:
|
||||
"""A parallel bash storm PLUS a task_agent whose sub-tools are chatty
|
||||
bashes (so the browser proves both fix-3 batching AND sub-agent nesting
|
||||
with no escaped children)."""
|
||||
from tests._sse_recovery_server import final_text_script, parallel_bash_script
|
||||
|
||||
storm = parallel_bash_script({f"call_{i}": "seq 1 500" for i in range(4)})
|
||||
task = dict(
|
||||
tool_calls=[
|
||||
{
|
||||
"id": "task1",
|
||||
"name": "task_agent",
|
||||
"arguments": json.dumps({"prompt": "sub tools"}),
|
||||
}
|
||||
],
|
||||
finish_reason="tool_calls",
|
||||
)
|
||||
sub = dict(
|
||||
tool_calls=[
|
||||
{"id": "s_a", "name": "bash", "arguments": json.dumps({"command": ": a; seq 1 200"})},
|
||||
{"id": "s_b", "name": "bash", "arguments": json.dumps({"command": ": b; seq 1 200"})},
|
||||
],
|
||||
finish_reason="tool_calls",
|
||||
)
|
||||
# Turn 1: the 4-bash storm; turn 2: a task_agent with 2 chatty sub-bashes.
|
||||
return (
|
||||
storm,
|
||||
final_text_script("storm done"),
|
||||
task,
|
||||
sub,
|
||||
final_text_script("sub done"),
|
||||
final_text_script("all done"),
|
||||
)
|
||||
|
||||
|
||||
def run_storm(chrome: str) -> str:
|
||||
node = _boot_node()
|
||||
ws_id = node.create_workstream(*_storm_scripts(), name="browser-storm")
|
||||
profile = Path(_scratch()) / "chrome-storm"
|
||||
proc, cdp_port = _launch_chrome(chrome, profile)
|
||||
cdp: CDP | None = None
|
||||
try:
|
||||
cdp = CDP(_page_ws_url(cdp_port))
|
||||
# The page POSTs the STORM turn on stream-open; the runner sends the
|
||||
# task_agent follow-up once the first turn settles so both land.
|
||||
url = f"{node.base_url}/recovery?ws_id={ws_id}&scenario=storm&rows=4"
|
||||
_set_cookie_and_navigate(cdp, node.base_url, node.token, url)
|
||||
# After the storm turn, trigger the task_agent turn via REST so the
|
||||
# page renders the nested sub-agent card.
|
||||
_wait_state(node, ws_id, "idle", 40)
|
||||
node.send(ws_id, "spawn the sub agent")
|
||||
return _poll_title(cdp, 45)
|
||||
finally:
|
||||
if cdp is not None:
|
||||
cdp.close()
|
||||
_kill(proc)
|
||||
node.stop()
|
||||
|
||||
|
||||
def run_restart(chrome: str) -> str:
|
||||
from tests._sse_recovery_server import final_text_script, parallel_bash_script
|
||||
|
||||
port = _free_port()
|
||||
node = _boot_node(port=port)
|
||||
# A PACED turn so the tab can hide MID-turn: the browser cursor
|
||||
# freezes at a mid-stream event id, the rest of turn 1 plus the
|
||||
# turn-2 text commit while hidden, and the restarted node's seeded
|
||||
# counter therefore sits ABOVE the frozen cursor -> the show-edge
|
||||
# reconnect draws ``replay_truncated`` and must heal the gap.
|
||||
paced = parallel_bash_script({"r0": "for i in $(seq 1 40); do echo r-$i; sleep 0.05; done"})
|
||||
# The turn-2 text is the healed-gap sentinel — it must be a token
|
||||
# that cannot appear in any rendered command/output (the bash
|
||||
# command row contains the shell keyword ``done``, so the obvious
|
||||
# word is vacuously present; see __verifyRestart). Single source:
|
||||
# the same constant is injected as the scripted turn text AND
|
||||
# threaded to the page via ?healed=, so the two sides cannot drift.
|
||||
ws_id = node.create_workstream(
|
||||
paced, final_text_script(HEALED_SENTINEL), name="browser-restart"
|
||||
)
|
||||
profile = Path(_scratch()) / "chrome-restart"
|
||||
proc, cdp_port = _launch_chrome(chrome, profile)
|
||||
cdp: CDP | None = None
|
||||
try:
|
||||
cdp = CDP(_page_ws_url(cdp_port))
|
||||
url = f"{node.base_url}/recovery?ws_id={ws_id}&scenario=restart&healed={HEALED_SENTINEL}"
|
||||
_set_cookie_and_navigate(cdp, node.base_url, node.token, url)
|
||||
# Hide as soon as the FIRST streamed line has painted (proof the
|
||||
# pane holds a live mid-turn cursor) — NOT after wait_turn, which
|
||||
# would leave the cursor at/above the committed counter and the
|
||||
# reconnect on the lossless replay_ok path (trunc0).
|
||||
deadline = time.monotonic() + 15
|
||||
while time.monotonic() < deadline:
|
||||
painted = cdp.evaluate("document.querySelector('.tool-output-stream') !== null")
|
||||
if painted:
|
||||
break
|
||||
time.sleep(0.2)
|
||||
else:
|
||||
raise AssertionError("restart scenario: first streamed line never painted")
|
||||
cdp.evaluate("window.__hide && window.__hide()")
|
||||
# The turn (and the follow-up text) commits while the tab is hidden.
|
||||
node.wait_turn(ws_id, timeout=30)
|
||||
# Restart the node on the SAME port (fresh empty ring, seeded counter).
|
||||
node.stop()
|
||||
node = _boot_node(port=port)
|
||||
node.open_workstream(ws_id)
|
||||
# Show the tab -> stale-cursor reconnect -> truncated -> jittered
|
||||
# resync (0-10s) -> /history rebuild. Settle past the worst-case
|
||||
# jitter before the verdict.
|
||||
cdp.evaluate("window.__show && window.__show()")
|
||||
time.sleep(12.0)
|
||||
cdp.evaluate("window.__verifyRestart && window.__verifyRestart()")
|
||||
return _poll_title(cdp, 20)
|
||||
finally:
|
||||
if cdp is not None:
|
||||
cdp.close()
|
||||
_kill(proc)
|
||||
node.stop()
|
||||
|
||||
|
||||
def run_coord_restart(chrome: str) -> str:
|
||||
from tests._sse_recovery_server import final_text_script, parallel_bash_script
|
||||
|
||||
port = _free_port()
|
||||
node = _boot_node(port=port)
|
||||
# Same shape as Scenario B: a PACED turn so the tab can hide MID-turn.
|
||||
# The coordinator renders no streamed tool output (no tool_output_chunk
|
||||
# case), but the chunk frames still advance the pane's cursor in
|
||||
# onmessage BEFORE dispatch — so the hide freezes a genuinely mid-turn
|
||||
# cursor even though the paint signal differs (see below). The closing
|
||||
# assistant text after the bash is the healed-gap sentinel, committed
|
||||
# while hidden.
|
||||
paced = parallel_bash_script({"c0": "for i in $(seq 1 40); do echo c-$i; sleep 0.05; done"})
|
||||
ws_id = node.create_workstream(
|
||||
paced, final_text_script(HEALED_SENTINEL), name="browser-coord-restart"
|
||||
)
|
||||
profile = Path(_scratch()) / "chrome-coord-restart"
|
||||
proc, cdp_port = _launch_chrome(chrome, profile)
|
||||
cdp: CDP | None = None
|
||||
try:
|
||||
cdp = CDP(_page_ws_url(cdp_port))
|
||||
url = f"{node.base_url}/coord-recovery?ws_id={ws_id}&healed={HEALED_SENTINEL}"
|
||||
_set_cookie_and_navigate(cdp, node.base_url, node.token, url)
|
||||
# Hide once the BROWSER has captured a live mid-turn cursor: the
|
||||
# coordinator chrome has no status/SSE text elements (the header was
|
||||
# deliberately dropped) and paints no streamed output line, so the
|
||||
# signal is transport-level — id-bearing frames received by the page
|
||||
# (__idFrames; exactly the frames that advance the pane's reconnect
|
||||
# cursor). The turn must also still be RUNNING server-side, or the
|
||||
# frozen cursor could sit at/above the committed counter and the
|
||||
# reconnect would take the lossless replay_ok path (trunc0). The
|
||||
# paced bash runs >=2s, so this lands mid-turn.
|
||||
deadline = time.monotonic() + 15
|
||||
while time.monotonic() < deadline:
|
||||
frames = cdp.evaluate("window.__idFrames || 0")
|
||||
if isinstance(frames, int) and frames >= 2 and node.ws_state(ws_id) == "running":
|
||||
break
|
||||
time.sleep(0.2)
|
||||
else:
|
||||
raise AssertionError(
|
||||
"coord-restart scenario: no mid-turn cursor captured "
|
||||
"(id frames never reached the page while running)"
|
||||
)
|
||||
time.sleep(0.5)
|
||||
cdp.evaluate("window.__hide && window.__hide()")
|
||||
# The turn (and the sentinel closing text) commits while hidden.
|
||||
node.wait_turn(ws_id, timeout=30)
|
||||
# Restart the node on the SAME port (fresh empty ring, seeded counter).
|
||||
node.stop()
|
||||
node = _boot_node(port=port)
|
||||
node.open_workstream(ws_id)
|
||||
# Show the tab -> stale-cursor reconnect -> truncated -> jittered
|
||||
# resync (0-10s) -> /history rebuild. Settle past the worst-case
|
||||
# jitter, assert idle SERVER-side (the chrome has no state text to
|
||||
# read), then take the in-page verdict.
|
||||
cdp.evaluate("window.__show && window.__show()")
|
||||
time.sleep(12.0)
|
||||
_wait_state(node, ws_id, "idle", 15)
|
||||
cdp.evaluate("window.__verifyCoordRestart && window.__verifyCoordRestart()")
|
||||
return _poll_title(cdp, 20)
|
||||
finally:
|
||||
if cdp is not None:
|
||||
cdp.close()
|
||||
_kill(proc)
|
||||
node.stop()
|
||||
|
||||
|
||||
def _wait_state(node: Any, ws_id: str, state: str, timeout: float) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if node.ws_state(ws_id) == state:
|
||||
return
|
||||
time.sleep(0.1)
|
||||
|
||||
|
||||
def _kill(proc: subprocess.Popen[bytes]) -> None:
|
||||
if proc.poll() is None:
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(8)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
|
||||
|
||||
def keep_open(port: int) -> None:
|
||||
"""Boot the node + storm ws and serve the page for manual inspection."""
|
||||
node = _boot_node(port=port)
|
||||
ws_id = node.create_workstream(*_storm_scripts(), name="manual-storm")
|
||||
print(f"node: {node.base_url}")
|
||||
print(f"cookie: turnstone_auth_server={node.token}")
|
||||
print(f"page: {node.base_url}/recovery?ws_id={ws_id}&scenario=storm&rows=4")
|
||||
print("set the cookie for this origin, then open the page. Ctrl+C to stop.")
|
||||
try:
|
||||
while True:
|
||||
time.sleep(1)
|
||||
except KeyboardInterrupt:
|
||||
node.stop()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
ap.add_argument(
|
||||
"--scenario",
|
||||
choices=["storm", "restart", "coord-restart", "both", "all"],
|
||||
default="all",
|
||||
help="'both' = A+B (legacy alias); 'all' adds the coordinator scenario",
|
||||
)
|
||||
ap.add_argument("--keep-open", type=int, metavar="PORT", help="serve the storm page, no CDP")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.keep_open:
|
||||
keep_open(args.keep_open)
|
||||
return
|
||||
|
||||
chrome = _find_chrome()
|
||||
if chrome is None:
|
||||
print("recovery_e2e: no chrome/chromium on PATH — see the module docstring runbook")
|
||||
raise SystemExit(2)
|
||||
|
||||
failures = 0
|
||||
if args.scenario in ("storm", "both", "all"):
|
||||
verdict = run_storm(chrome)
|
||||
print(f"scenario A (storm): {verdict}")
|
||||
failures += 0 if verdict.startswith("RECOVERY-READY") else 1
|
||||
if args.scenario in ("restart", "both", "all"):
|
||||
verdict = run_restart(chrome)
|
||||
print(f"scenario B (restart): {verdict}")
|
||||
failures += 0 if verdict.startswith("RECOVERY-READY") else 1
|
||||
if args.scenario in ("coord-restart", "all"):
|
||||
verdict = run_coord_restart(chrome)
|
||||
print(f"scenario C (coord): {verdict}")
|
||||
failures += 0 if verdict.startswith("RECOVERY-READY") else 1
|
||||
raise SystemExit(1 if failures else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Console API",
|
||||
"version": "1.8.0a2",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Cluster-wide visibility and control across all turnstone nodes."
|
||||
},
|
||||
"paths": {
|
||||
@@ -7791,12 +7791,6 @@
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"description": "Project to attach the workstream to (validated against membership, empty = none)",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"resume_ws": {
|
||||
"default": "",
|
||||
"description": "Workstream ID to resume (loads previous conversation)",
|
||||
@@ -8502,18 +8496,6 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"description": "Persona slug (empty = kind default)",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"description": "Project to attach the workstream to",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"description": "Notification targets on completion (channel_type + channel_id/user_id)",
|
||||
"items": {
|
||||
@@ -8677,30 +8659,6 @@
|
||||
"default": null,
|
||||
"title": "Skill"
|
||||
},
|
||||
"persona": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Persona"
|
||||
},
|
||||
"project_id": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"notify_targets": {
|
||||
"anyOf": [
|
||||
{
|
||||
@@ -8796,16 +8754,6 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"items": {
|
||||
"additionalProperties": {
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Server API",
|
||||
"version": "1.8.0a2",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Single-node workstream management, chat interaction, and real-time streaming."
|
||||
},
|
||||
"paths": {
|
||||
@@ -228,16 +228,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -400,26 +390,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2290,23 +2260,16 @@
|
||||
"SendResponse": {
|
||||
"properties": {
|
||||
"status": {
|
||||
"description": "'ok' (fresh turn dispatched), 'queued' (folded into the live turn's interjection queue, or \u2014 when `deferred` is true \u2014 parked for dispatch after the current command window), 'queue_full', 'attachments_busy' (attachments can't ride a queued turn; retry when idle), or 'cross_user_interjection' (another participant's turn is in flight; carried on the 409 body).",
|
||||
"description": "'ok', 'busy', 'queued', or 'queue_full'",
|
||||
"examples": [
|
||||
"ok",
|
||||
"busy",
|
||||
"queued",
|
||||
"queue_full",
|
||||
"attachments_busy",
|
||||
"cross_user_interjection"
|
||||
"queue_full"
|
||||
],
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
"deferred": {
|
||||
"default": false,
|
||||
"description": "Set on `queued` responses: the message is parked on the workstream's deferred-send list (a slash-command window holds the worker slot, or earlier deferred sends are still pending) and dispatches as an ordinary full-fidelity send afterwards \u2014 it is NOT in a live turn's interjection queue. `DELETE .../send` retracts it until dispatch. Node-local and in-memory: a node restart before dispatch drops it (at-most-once intake).",
|
||||
"title": "Deferred",
|
||||
"type": "boolean"
|
||||
},
|
||||
"attached_ids": {
|
||||
"description": "Attachment ids actually attached to this turn. Subset of the request's `attachment_ids` (or the auto-consumed pending set). Empty when the send carries no attachments.",
|
||||
"items": {
|
||||
|
||||
Generated
+143
-504
@@ -9,7 +9,7 @@
|
||||
"version": "0.4.0",
|
||||
"license": "Apache-2.0",
|
||||
"devDependencies": {
|
||||
"typescript": "^7.0.0",
|
||||
"typescript": "^6.0.0",
|
||||
"vitest": "^4.1"
|
||||
}
|
||||
},
|
||||
@@ -74,9 +74,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@oxc-project/types": {
|
||||
"version": "0.139.0",
|
||||
"resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.139.0.tgz",
|
||||
"integrity": "sha512-r9gHphtCs+1M7J0pw6Sn/hh/Wpa/iQrOOkrNAlVLF/gHq+/CJmHIWKKUUhdWjcD6CIa8idarspCsASiXCXvFUw==",
|
||||
"version": "0.138.0",
|
||||
"resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.138.0.tgz",
|
||||
"integrity": "sha512-1a7ZKmrRTCoN1XMZ4L0PyyqrMnrNlLyPuOkdSX2MZg7IiIGRUyurNhAm73ptDOraoBcIordsIGKNPKUzy3ZmfA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
@@ -84,9 +84,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-android-arm64": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.1.5.tgz",
|
||||
"integrity": "sha512-lZg8fqIv2v7FF237bwMgzGZEJvGL79/s5knJ/i6FmsGF4XXlzccZ4jb+TrFIxtSSxFtIpdsgrPZeMk1I9AFcyQ==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.1.4.tgz",
|
||||
"integrity": "sha512-EZLpf/8y7GXkkra90ML47kzik/GMP3EMcE9bPyHmRfxLC6z9+aW5A8poCsoxjrT5GfEcNAAvWwUHjvP1pUQkfw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -101,9 +101,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-darwin-arm64": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.1.5.tgz",
|
||||
"integrity": "sha512-51Bnx9pNiMRKSUNtBfySkNJ9vMU9Hh3I1ozDd6gyPPYzaXCfnptUcEZxXGYFn+ul2dtcMUiqGR1Yai2K10uoTw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.1.4.tgz",
|
||||
"integrity": "sha512-aUi+HBvmYb7j8krl1+qJgkG8C17fO79gk3c+jPw4S8glRFc1DTija9S3EyaTSQUm5GJXYKDAsugBEhFHH2vYiQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -118,9 +118,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-darwin-x64": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.1.5.tgz",
|
||||
"integrity": "sha512-Tm+gbfC0aHu1tBA/JvKQh32S0K6YgCHkiAF4/W6xX0K0RmNuc94VeK419dJoE65R5aRxmo+noZQSWrAMF6yb6g==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.1.4.tgz",
|
||||
"integrity": "sha512-F7hHC3gwY11+vByKPRWqwGbeXWVgKmL+pTGCinaEhdihzBV2aQ0fvZOch9cXYUOKuKKq429HeYXOqQLc7wFCEg==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -135,9 +135,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-freebsd-x64": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.1.5.tgz",
|
||||
"integrity": "sha512-JMzDKCCXq93YccG5gz3hvOs1oXRKAf0XYpfOS88e+wZrC8Iugj6j68867vrYZkvpDDpKn/KoKORThmchMpF6TA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.1.4.tgz",
|
||||
"integrity": "sha512-sI5yw+7s92SK6odiEhD5lKCBlWcpjHS5qyqpVQbZAJ0fIzEUXrmbl3DH2ybR3PZogulNJF+COLtmA8hUfvkCCQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -152,9 +152,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-arm-gnueabihf": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.1.5.tgz",
|
||||
"integrity": "sha512-uML21j2K5TfPGutKxub+M+nLjZIrWjXQ5Grx4lCe/nimTj9B4L63zHpjXLl4y0L3mcm2htEQIb06oCG/szerNw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.1.4.tgz",
|
||||
"integrity": "sha512-mCi0OKgEieFircrtVYmQAFGszRtMnZ6fpZAXrxanXAu7lqZcsK1E1RAaZNG0uKAnxox3B1f4EyQNnoyMfN1vAA==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
@@ -169,9 +169,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-arm64-gnu": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.1.5.tgz",
|
||||
"integrity": "sha512-navSiuTMogvnQoZoM/v+l3ZWo50/NTwSHSzheABx/RCnmUPaKwq9qSo4Br2OYRs21+Fz8uFqITZM3H4opOB0/Q==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-B9Ial3Kv5sh0SHnB1g/QWcUQCEvCF6QKGAl4zXypYj65mVI+B4AhFBwPtSN7pDrJeIx8Z7zdy4ntx+wQABom7w==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -189,9 +189,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-arm64-musl": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.1.5.tgz",
|
||||
"integrity": "sha512-lAryqH7IteztmCXQXk0etKj4wBQ7Gx5S6LjKhsgp9zb8I5bsuvU/2llH1hDQcjsFeqIsovMVN339/8pUDDBXxA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.1.4.tgz",
|
||||
"integrity": "sha512-lZVym0PuHE1KZ22gmFTC15lAkrg9iTszR617oYRB/iPY1A56ywoJzVKOJBKaot5RiikCObmur6pogpse3gRcng==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -209,9 +209,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-ppc64-gnu": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.1.5.tgz",
|
||||
"integrity": "sha512-fsK/sNBnxzBlL4O1JNrZakVQxPspqpED5dLtNsZS9oOKmtSpdNIzxH2kkol5HYTWJN47sE20ztMJPxfZ89qGOg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-t2DNiLJWNTbnEHyUzTumldML6ET4/g16467LZoDDJ3tSxGvguL5/NyC2lCsNKuyRycg9XeDQF5SSv+TNOhQEXg==",
|
||||
"cpu": [
|
||||
"ppc64"
|
||||
],
|
||||
@@ -229,9 +229,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-s390x-gnu": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.1.5.tgz",
|
||||
"integrity": "sha512-gLYb4BIadlfTOYT5gO503n8zQjXflgzpD0FcyKh0Mzx3rqCZKnHoJWV9xe1KXUJ5lx2JfcSHr/mhzS0PC/McAA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-0WIRnL1Uw4BvTZRLQt+PVgo6ZKTJadlC2btP+/EOXv2f/DWbY0rEgl+y834mIVwP1FkTlWVTrGGJXf12lru7EQ==",
|
||||
"cpu": [
|
||||
"s390x"
|
||||
],
|
||||
@@ -249,9 +249,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-x64-gnu": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.1.5.tgz",
|
||||
"integrity": "sha512-FjcpEKUyJygHgs1o50VYNvkt5+7Le/VEdYt0AkRpkL33MnyQfwr8l5mXwMmfmTbyMPr5vJLC+8/Gd9gXnwU1QQ==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-JWtGshGfX+oENAKonoNkqEJX+7hC8yfhi9GUyPX1VX4mdh1y5r+ZiJLR5XzAB0aoP6s/PcILsGjKq8O0mm24bw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -269,9 +269,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-x64-musl": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.1.5.tgz",
|
||||
"integrity": "sha512-Me+PfPI2TMeOQk0gYWfLQZtTktrmzbr8cDboqX83XKc7UrgAi55gF+2dUkWdxd19n55Essp2yeca+O9N5rBxHg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.1.4.tgz",
|
||||
"integrity": "sha512-rT6yQcxUuXs4CnbofqwHRRV0iem349rLMYpTjkgQGLjrY4ado/eDzwPZPTCgTOlF6Nkp8NEv70yLMTn6qkWxsQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -289,9 +289,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-openharmony-arm64": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.1.5.tgz",
|
||||
"integrity": "sha512-yc5WrLzXks6zCQfn9Oxr8pORKyl/pF+QjHmW/Qx3qu0oyrrNC+y2JLTU1E2rcWYAmzlnqngWXHQjy51VzW70Vw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.1.4.tgz",
|
||||
"integrity": "sha512-KXMGoboq5cyaCQjDA4GLuRiOwBQ0EyFnJoVViLeZ45/3rFItRODEr+NdsBcVpll40hhNArlm/speWGRvj08LzA==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -306,9 +306,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-wasm32-wasi": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-wasm32-wasi/-/binding-wasm32-wasi-1.1.5.tgz",
|
||||
"integrity": "sha512-VbQGPX2b4r48TAMIM2cjgluIM1HYutm4pcTEJsle7iEP7sB1dFqtPLBVbdLAZCxy1txCcPxf4QFf4v8uvltPqA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-wasm32-wasi/-/binding-wasm32-wasi-1.1.4.tgz",
|
||||
"integrity": "sha512-5K83rb36oJiY7BCyE9zLZtGcPV4g5wvq+xwdO0XPIwDVZI8cyB/AUjkNXGb92/rnmezEkjMOpgY61rtwjQtFwg==",
|
||||
"cpu": [
|
||||
"wasm32"
|
||||
],
|
||||
@@ -325,9 +325,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-win32-arm64-msvc": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.1.5.tgz",
|
||||
"integrity": "sha512-gHv82k63z4qpV5+Q1y/12KrK0ltWBukVDI8nZcbT7Tt/ZlOIVwppazneq0F93oDxTo3IgAMEDIoQh3E2n6mVsw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.1.4.tgz",
|
||||
"integrity": "sha512-PnWBtw3TV5KOg69HQQDR0mnQuyCmSGR2pAB4DC1rPF808fgKeTUMj2EOEyKATpgiuxuR5APQmiDO7PDgEjTFSA==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -342,9 +342,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-win32-x64-msvc": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.1.5.tgz",
|
||||
"integrity": "sha512-tTZuDBPw85tEN5PQi1pnEBzDy0Z49HtScLAbD5t6hyeU92A95pRWaSMw1GZZi/RwgSgUIl0xrSlXIT/9QzvYSA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.1.4.tgz",
|
||||
"integrity": "sha512-M1lpniBePobTfsa7Ks9a199e1akxsXn+GYBUKsEzv3YFzOm1HJAMNwKI3qr0Zq+mxwx9gOZoTdP1yXRYsZUocQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -408,346 +408,6 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@typescript/typescript-aix-ppc64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-aix-ppc64/-/typescript-aix-ppc64-7.0.2.tgz",
|
||||
"integrity": "sha512-MTKKkWB7p/0E9xi1d1tHtZ5PiLkGEMIq88pK2CubZjOsLtYTLqhgIgi6zepFa+9GHZ6h05NMCkQxGKiPXMxXtQ==",
|
||||
"cpu": [
|
||||
"ppc64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"aix"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-darwin-arm64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-darwin-arm64/-/typescript-darwin-arm64-7.0.2.tgz",
|
||||
"integrity": "sha512-gowzar9MwS/aRWp6f3a4KUqzRjAZjOsmGNCM6LcTgXum+dBfgsBVMN+AgvOCCbguXyick6LJhpBszxMebJ8syA==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-darwin-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-darwin-x64/-/typescript-darwin-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-SZ9xZInqApNlNGc9s0W1VSsktYSOe9cFqNOIqmN1Gs8SmkjKZYFt017G4VwPxASInODuAdbTW7sXiFUf893RgA==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-freebsd-arm64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-freebsd-arm64/-/typescript-freebsd-arm64-7.0.2.tgz",
|
||||
"integrity": "sha512-W5NH4y/J0plIIS5b2xvTEkU7JFxyqdMAOgf+Ilhl0vHQXKO5dZoxd+C/jEtq56c4F3wk71RB4BMRQ2XdI+bwYQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"freebsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-freebsd-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-freebsd-x64/-/typescript-freebsd-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-UMGDx5sTpzNw3WiPebH7l90IWfJggEd+egHt/q6p7/Cm3zqoV7VxkGXt+3DxPIw8CcmvAB0j3sVVfbhX+M4Tpw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"freebsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-arm": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-arm/-/typescript-linux-arm-7.0.2.tgz",
|
||||
"integrity": "sha512-gffT3xPz9sR7j/YJExkyPntrI0P2EP9XbOyWzth2/Gs0RstK+90RBcO0ncXoXy/beYll1SXw846Nf2zdnEz0QQ==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-arm64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-arm64/-/typescript-linux-arm64-7.0.2.tgz",
|
||||
"integrity": "sha512-Qh4eU4/y3yDjnfjjyPYihMj5/ODIlmt+Bzu17OI+fiSRDW57QmU5SiN63exPRNJPKUzcc1INa1NXdrJ+MqHjUQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-loong64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-loong64/-/typescript-linux-loong64-7.0.2.tgz",
|
||||
"integrity": "sha512-uEHck9i8hoAzXPiYRib1O7miOnz23SxIeVl6F4LXox+qov1K35jHcEW6VHKvZI+pyvl7fZEP4MCU5LYvIq1GuQ==",
|
||||
"cpu": [
|
||||
"loong64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-mips64el": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-mips64el/-/typescript-linux-mips64el-7.0.2.tgz",
|
||||
"integrity": "sha512-R4KvAMnE43W5Qeqb0Ly56O3mWMWIAgsMyz36DCaycd5nbg/9kzm0liw3JocfRqyJY0KPmzFjbswozXyW0DnIYA==",
|
||||
"cpu": [
|
||||
"mips64el"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-ppc64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-ppc64/-/typescript-linux-ppc64-7.0.2.tgz",
|
||||
"integrity": "sha512-DORx5b3sd/4S7eayxm4FQv+A7CrkUIGRaHiwI8oiHTAI1fAPWhF4J0vAlkC8biAlHSVVwxMQ3tjZ2/DVbnQiiA==",
|
||||
"cpu": [
|
||||
"ppc64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-riscv64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-riscv64/-/typescript-linux-riscv64-7.0.2.tgz",
|
||||
"integrity": "sha512-wf0jqEDOjrPRnKwYRyyJDRo11KMbvMFrU+q4zqKyChODBzvlkbhNQfKvLxQCcwTpdDaXSHZTVuh0JoCrKCUMHQ==",
|
||||
"cpu": [
|
||||
"riscv64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-s390x": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-s390x/-/typescript-linux-s390x-7.0.2.tgz",
|
||||
"integrity": "sha512-IkwJc3L7yhytWd/ewjyxNDfOmswCm9GWMJT/ue/dU4aZNbwZeYAetq42VyLmsmSjvoX7z74X6ZaYCtzAr0EuGw==",
|
||||
"cpu": [
|
||||
"s390x"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-linux-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-x64/-/typescript-linux-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-EYdf2cNg7rgCWJnxCdJ+F3V39O8ihb37eHAu1LK8oAFizgTQbPOK7zHHXbPt8rX24COqODXeI3sIf0fCXG7H/A==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-netbsd-arm64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-netbsd-arm64/-/typescript-netbsd-arm64-7.0.2.tgz",
|
||||
"integrity": "sha512-+polYF4MF04aPpO5FTkHran9yUQDSXqy5GiSDKpsll5jy3l3+g9QLhpf39T+ePtefhXLOGrLl0QIjkQP6VnelA==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"netbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-netbsd-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-netbsd-x64/-/typescript-netbsd-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-8YIT0EHM/3dq10ZOVF/A7pc/YSMtbcecct4rWtexrnSCHOPcpC2KTLXfTCR6vDpnSiY12heNb1GiN/wu+T/FyA==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"netbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-openbsd-arm64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-openbsd-arm64/-/typescript-openbsd-arm64-7.0.2.tgz",
|
||||
"integrity": "sha512-APT8+ClYnuYm1u9+kgGXoMj2VzWzcymwh2gNSQVySHfkRDGOTVkoWLjCmOQSaO+PoqQ57B0flRp9SA+7GnnkzQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"openbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-openbsd-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-openbsd-x64/-/typescript-openbsd-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-yX7s+Q0Dln0Dt9tEzZsAjXXR/+ytBM7AlglaqyeMPxQszJ1JhlJdZ6jLA+IzldHtflX81em7lDao1xXu+aRRkg==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"openbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-sunos-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-sunos-x64/-/typescript-sunos-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-dLJDGaLZ1D4HPQn62u1n8mBDkJREwMsAkCdkwd4Ieqw+x3TUyTsqY0YiBCtE6H6OzzgGk3iuZ3vFWRS+E8/d1g==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"sunos"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-win32-arm64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-win32-arm64/-/typescript-win32-arm64-7.0.2.tgz",
|
||||
"integrity": "sha512-Gyl1Vy6OsWesLzmq+EP0Fb7b4Nid5232AvcA2SFcdYreldpNtYFFofPjnt62y9hQy7VTaZp65ICJjuAQRaVcIQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript/typescript-win32-x64": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@typescript/typescript-win32-x64/-/typescript-win32-x64-7.0.2.tgz",
|
||||
"integrity": "sha512-0BQ3HkAHHlKLSp1qRvf3SUhGpGsDuhB/jgFw75guyqbxJqEaS0Cw/VFO8i2nHglJUzQCRtMMR/IBAKE3ETMC4g==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/expect": {
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.10.tgz",
|
||||
@@ -899,9 +559,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/es-module-lexer": {
|
||||
"version": "2.3.1",
|
||||
"resolved": "https://registry.npmjs.org/es-module-lexer/-/es-module-lexer-2.3.1.tgz",
|
||||
"integrity": "sha512-shc1dbU90Yl/xq1QrC7QRtfcwURZuVRfPhZbDoldJ1cn1gzDvBaBWlv0eFolj5+0znnPJz5TXLxsN77X/12KTA==",
|
||||
"version": "2.3.0",
|
||||
"resolved": "https://registry.npmjs.org/es-module-lexer/-/es-module-lexer-2.3.0.tgz",
|
||||
"integrity": "sha512-KLdwQm2NvGLDkQDCGvmiQrhkd0JbMzXthwQAUgWjQuQdBLFa3eiBP5arXZyA+f8x+x7OXgud6bq2rxjGtHV2tw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
@@ -959,9 +619,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss/-/lightningcss-1.33.0.tgz",
|
||||
"integrity": "sha512-WkUDrojuJs0xkgGf2udWxa3yGBRxPtxUkB79i6aCZLRgc7PM8fZe9TosfPDcvEpQZbuFASnHYmRLBLUbmLOIIA==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss/-/lightningcss-1.32.0.tgz",
|
||||
"integrity": "sha512-NXYBzinNrblfraPGyrbPoD19C1h9lfI/1mzgWYvXUTe414Gz/X1FD2XBZSZM7rRTrMA8JL3OtAaGifrIKhQ5yQ==",
|
||||
"dev": true,
|
||||
"license": "MPL-2.0",
|
||||
"dependencies": {
|
||||
@@ -975,23 +635,23 @@
|
||||
"url": "https://opencollective.com/parcel"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"lightningcss-android-arm64": "1.33.0",
|
||||
"lightningcss-darwin-arm64": "1.33.0",
|
||||
"lightningcss-darwin-x64": "1.33.0",
|
||||
"lightningcss-freebsd-x64": "1.33.0",
|
||||
"lightningcss-linux-arm-gnueabihf": "1.33.0",
|
||||
"lightningcss-linux-arm64-gnu": "1.33.0",
|
||||
"lightningcss-linux-arm64-musl": "1.33.0",
|
||||
"lightningcss-linux-x64-gnu": "1.33.0",
|
||||
"lightningcss-linux-x64-musl": "1.33.0",
|
||||
"lightningcss-win32-arm64-msvc": "1.33.0",
|
||||
"lightningcss-win32-x64-msvc": "1.33.0"
|
||||
"lightningcss-android-arm64": "1.32.0",
|
||||
"lightningcss-darwin-arm64": "1.32.0",
|
||||
"lightningcss-darwin-x64": "1.32.0",
|
||||
"lightningcss-freebsd-x64": "1.32.0",
|
||||
"lightningcss-linux-arm-gnueabihf": "1.32.0",
|
||||
"lightningcss-linux-arm64-gnu": "1.32.0",
|
||||
"lightningcss-linux-arm64-musl": "1.32.0",
|
||||
"lightningcss-linux-x64-gnu": "1.32.0",
|
||||
"lightningcss-linux-x64-musl": "1.32.0",
|
||||
"lightningcss-win32-arm64-msvc": "1.32.0",
|
||||
"lightningcss-win32-x64-msvc": "1.32.0"
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-android-arm64": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-android-arm64/-/lightningcss-android-arm64-1.33.0.tgz",
|
||||
"integrity": "sha512-gEpRTalKdosp4Bb8qWtc2iOgE5SeIHlpS1up9bFq2wAyYhl1UdTObYiHe98zEM9SQvSoqQZ1IQD0JNpg3Ml5pg==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-android-arm64/-/lightningcss-android-arm64-1.32.0.tgz",
|
||||
"integrity": "sha512-YK7/ClTt4kAK0vo6w3X+Pnm0D2cf2vPHbhOXdoNti1Ga0al1P4TBZhwjATvjNwLEBCnKvjJc2jQgHXH0NEwlAg==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1010,9 +670,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-darwin-arm64": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-darwin-arm64/-/lightningcss-darwin-arm64-1.33.0.tgz",
|
||||
"integrity": "sha512-Sciaz8eenNTKn9b3t7+xr0ipTp9YxKQY4npwQ3mrRuL0BAVHBLyZxofhaKBAVtzmtRZ/zTyo0/to4B1uWG/Djg==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-darwin-arm64/-/lightningcss-darwin-arm64-1.32.0.tgz",
|
||||
"integrity": "sha512-RzeG9Ju5bag2Bv1/lwlVJvBE3q6TtXskdZLLCyfg5pt+HLz9BqlICO7LZM7VHNTTn/5PRhHFBSjk5lc4cmscPQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1031,9 +691,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-darwin-x64": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-darwin-x64/-/lightningcss-darwin-x64-1.33.0.tgz",
|
||||
"integrity": "sha512-Z5UPAxzrjlWNNyGy6i65cJzzvgJ5D3T6wMvs+gWpY9d7qRhANrxqAp6LhxIgZhWEw18RfJTGcRxjuLIBr+m8XQ==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-darwin-x64/-/lightningcss-darwin-x64-1.32.0.tgz",
|
||||
"integrity": "sha512-U+QsBp2m/s2wqpUYT/6wnlagdZbtZdndSmut/NJqlCcMLTWp5muCrID+K5UJ6jqD2BFshejCYXniPDbNh73V8w==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1052,9 +712,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-freebsd-x64": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-freebsd-x64/-/lightningcss-freebsd-x64-1.33.0.tgz",
|
||||
"integrity": "sha512-QQM/Ti/hQajJwCY+RiWuCZ9sdtI/XQk7nDK5vC8kkdwixezOlDgvDx7+RT+QjK6FcFT4MpsuoBnHIo/O3StRRg==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-freebsd-x64/-/lightningcss-freebsd-x64-1.32.0.tgz",
|
||||
"integrity": "sha512-JCTigedEksZk3tHTTthnMdVfGf61Fky8Ji2E4YjUTEQX14xiy/lTzXnu1vwiZe3bYe0q+SpsSH/CTeDXK6WHig==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1073,9 +733,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-linux-arm-gnueabihf": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-arm-gnueabihf/-/lightningcss-linux-arm-gnueabihf-1.33.0.tgz",
|
||||
"integrity": "sha512-N7FVBe6iS24MlM6R/4RBTxGhQheZGs7tiQ9U32UtF75NzP5Q7xWPRqLBCKxlRQRk3rY1jCIPLzx7WzOhuUIRLQ==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-arm-gnueabihf/-/lightningcss-linux-arm-gnueabihf-1.32.0.tgz",
|
||||
"integrity": "sha512-x6rnnpRa2GL0zQOkt6rts3YDPzduLpWvwAF6EMhXFVZXD4tPrBkEFqzGowzCsIWsPjqSK+tyNEODUBXeeVHSkw==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
@@ -1094,9 +754,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-linux-arm64-gnu": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-arm64-gnu/-/lightningcss-linux-arm64-gnu-1.33.0.tgz",
|
||||
"integrity": "sha512-j2v/itmy4HlNxlc6voKXYgBqNi0Ng2LShg4z7GufpEgs05P+2suBVyi9I6YHq5uoVFx9ETin3eCEhLVyXGQnKg==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-arm64-gnu/-/lightningcss-linux-arm64-gnu-1.32.0.tgz",
|
||||
"integrity": "sha512-0nnMyoyOLRJXfbMOilaSRcLH3Jw5z9HDNGfT/gwCPgaDjnx0i8w7vBzFLFR1f6CMLKF8gVbebmkUN3fa/kQJpQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1118,9 +778,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-linux-arm64-musl": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-arm64-musl/-/lightningcss-linux-arm64-musl-1.33.0.tgz",
|
||||
"integrity": "sha512-yiO5ROMuYQgXbC60yjZU5CYSFZGKXL0HFATXt9mHJn1+zW55oCtMI9NfcVhYLMFDL7gV7oBPon/EmMMGg2OvtQ==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-arm64-musl/-/lightningcss-linux-arm64-musl-1.32.0.tgz",
|
||||
"integrity": "sha512-UpQkoenr4UJEzgVIYpI80lDFvRmPVg6oqboNHfoH4CQIfNA+HOrZ7Mo7KZP02dC6LjghPQJeBsvXhJod/wnIBg==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1142,9 +802,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-linux-x64-gnu": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-x64-gnu/-/lightningcss-linux-x64-gnu-1.33.0.tgz",
|
||||
"integrity": "sha512-ar+Ju7LmcN0Jo4FpL4hpFybwNG9/3A/Br5KW2n2jyODg3MEZXaDYADdemoNS+BDNfMgKvylJLj4S5tyRActuAg==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-x64-gnu/-/lightningcss-linux-x64-gnu-1.32.0.tgz",
|
||||
"integrity": "sha512-V7Qr52IhZmdKPVr+Vtw8o+WLsQJYCTd8loIfpDaMRWGUZfBOYEJeyJIkqGIDMZPwPx24pUMfwSxxI8phr/MbOA==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1166,9 +826,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-linux-x64-musl": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-x64-musl/-/lightningcss-linux-x64-musl-1.33.0.tgz",
|
||||
"integrity": "sha512-RYiYbkokw0trfKqqzfF55lginwEPrD3OJDfTuJzFs1MK6iFnDenaz1fqLLtX4ITG3OktJQXOeTaw1awrBAlZPw==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-linux-x64-musl/-/lightningcss-linux-x64-musl-1.32.0.tgz",
|
||||
"integrity": "sha512-bYcLp+Vb0awsiXg/80uCRezCYHNg1/l3mt0gzHnWV9XP1W5sKa5/TCdGWaR/zBM2PeF/HbsQv/j2URNOiVuxWg==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1190,9 +850,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-win32-arm64-msvc": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-win32-arm64-msvc/-/lightningcss-win32-arm64-msvc-1.33.0.tgz",
|
||||
"integrity": "sha512-1K+MPfLSFVpphzpdbfkhlWk6wBrTObBzS2T6db10PNOZgR9GoVsAWzwNyuhUYYbTp23j+4RrncfujZ4uAzXvwA==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-win32-arm64-msvc/-/lightningcss-win32-arm64-msvc-1.32.0.tgz",
|
||||
"integrity": "sha512-8SbC8BR40pS6baCM8sbtYDSwEVQd4JlFTOlaD3gWGHfThTcABnNDBda6eTZeqbofalIJhFx0qKzgHJmcPTnGdw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1211,9 +871,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lightningcss-win32-x64-msvc": {
|
||||
"version": "1.33.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-win32-x64-msvc/-/lightningcss-win32-x64-msvc-1.33.0.tgz",
|
||||
"integrity": "sha512-OlEICDx/Xl0FqSp4bry8zFnCvGpig3Gl4gCquvYwHuqJKEC1+n9NgDniFvqHGmMv1ZkqDJrDqKKSykTDX+ehuA==",
|
||||
"version": "1.32.0",
|
||||
"resolved": "https://registry.npmjs.org/lightningcss-win32-x64-msvc/-/lightningcss-win32-x64-msvc-1.32.0.tgz",
|
||||
"integrity": "sha512-Amq9B/SoZYdDi1kFrojnoqPLxYhQ4Wo5XiL8EVJrVsB8ARoC1PWW6VGtT0WKCemjy8aC+louJnjS7U18x3b06Q==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1242,9 +902,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/nanoid": {
|
||||
"version": "3.3.16",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.16.tgz",
|
||||
"integrity": "sha512-bzlKTyNJ7+LdGIIwy8ijFpIqEQIvafahV7eYykJ8Cvh42EdJeODoJ6gUJXpQJvej1BddH8OqTXZNE/KfbWAu8Q==",
|
||||
"version": "3.3.15",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.15.tgz",
|
||||
"integrity": "sha512-y7Wygv/7mEOvxTuEQDB8StXdMRBWf1kR/tlhAzBRUFkB2jfcLOAxO/SHmOO2zgz1pVgK29/kyupn059/bCHdjA==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -1261,9 +921,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/obug": {
|
||||
"version": "2.1.4",
|
||||
"resolved": "https://registry.npmjs.org/obug/-/obug-2.1.4.tgz",
|
||||
"integrity": "sha512-4a+OsYv9UktOJKE+l1A4OufDgdRF9PifWj+tJnHURo/P+WOxpG4GzUFL9qCalmWauao6ogiG+QvnCovwPoyAWA==",
|
||||
"version": "2.1.3",
|
||||
"resolved": "https://registry.npmjs.org/obug/-/obug-2.1.3.tgz",
|
||||
"integrity": "sha512-9miFgM2OFba7hB+pRgvtV84pYTBaoTHohvmIgiRt6dRIzbwEOIaNaP+dIlGs2fNFoB0SeISs0Jz5WFVRid6Xyg==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
"https://github.com/sponsors/sxzz",
|
||||
@@ -1302,9 +962,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/postcss": {
|
||||
"version": "8.5.20",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.20.tgz",
|
||||
"integrity": "sha512-lW616l85ucIQL+FocMmL7pQFPqBmwejrCMg+iPxyImlrANNJG9NHq/RkyCZopDhd8C3LA03PHRJDjkbGu8vvug==",
|
||||
"version": "8.5.16",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.16.tgz",
|
||||
"integrity": "sha512-vuwillviilfKZsg0VGj5R/YwwcHx4SLsIOI/7K6mQkWx+l5cUHTjj5g0AasTBcyXsbfTgrwsUNmVUb5xVwyPwg==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -1322,7 +982,7 @@
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"nanoid": "^3.3.16",
|
||||
"nanoid": "^3.3.12",
|
||||
"picocolors": "^1.1.1",
|
||||
"source-map-js": "^1.2.1"
|
||||
},
|
||||
@@ -1331,13 +991,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/rolldown": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.1.5.tgz",
|
||||
"integrity": "sha512-t9z29cJjXf/vxQ8dyhCSpt6H6aSwHTk8cT5I3iy6SMXuFpk5mB6PL6XfC8PCwrPTx93udwKUm9HRteAlTGBLiA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.1.4.tgz",
|
||||
"integrity": "sha512-IjZYiLxZwpnhwhdBH2ugdTGVSdhCQUmLxLoqyjiL0JxYjyRst+5a0P3xfrTxJ5F638j4Mvvw5FAX5XE6eHpXbA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@oxc-project/types": "=0.139.0",
|
||||
"@oxc-project/types": "=0.138.0",
|
||||
"@rolldown/pluginutils": "^1.0.0"
|
||||
},
|
||||
"bin": {
|
||||
@@ -1347,21 +1007,21 @@
|
||||
"node": "^20.19.0 || >=22.12.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@rolldown/binding-android-arm64": "1.1.5",
|
||||
"@rolldown/binding-darwin-arm64": "1.1.5",
|
||||
"@rolldown/binding-darwin-x64": "1.1.5",
|
||||
"@rolldown/binding-freebsd-x64": "1.1.5",
|
||||
"@rolldown/binding-linux-arm-gnueabihf": "1.1.5",
|
||||
"@rolldown/binding-linux-arm64-gnu": "1.1.5",
|
||||
"@rolldown/binding-linux-arm64-musl": "1.1.5",
|
||||
"@rolldown/binding-linux-ppc64-gnu": "1.1.5",
|
||||
"@rolldown/binding-linux-s390x-gnu": "1.1.5",
|
||||
"@rolldown/binding-linux-x64-gnu": "1.1.5",
|
||||
"@rolldown/binding-linux-x64-musl": "1.1.5",
|
||||
"@rolldown/binding-openharmony-arm64": "1.1.5",
|
||||
"@rolldown/binding-wasm32-wasi": "1.1.5",
|
||||
"@rolldown/binding-win32-arm64-msvc": "1.1.5",
|
||||
"@rolldown/binding-win32-x64-msvc": "1.1.5"
|
||||
"@rolldown/binding-android-arm64": "1.1.4",
|
||||
"@rolldown/binding-darwin-arm64": "1.1.4",
|
||||
"@rolldown/binding-darwin-x64": "1.1.4",
|
||||
"@rolldown/binding-freebsd-x64": "1.1.4",
|
||||
"@rolldown/binding-linux-arm-gnueabihf": "1.1.4",
|
||||
"@rolldown/binding-linux-arm64-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-arm64-musl": "1.1.4",
|
||||
"@rolldown/binding-linux-ppc64-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-s390x-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-x64-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-x64-musl": "1.1.4",
|
||||
"@rolldown/binding-openharmony-arm64": "1.1.4",
|
||||
"@rolldown/binding-wasm32-wasi": "1.1.4",
|
||||
"@rolldown/binding-win32-arm64-msvc": "1.1.4",
|
||||
"@rolldown/binding-win32-x64-msvc": "1.1.4"
|
||||
}
|
||||
},
|
||||
"node_modules/siginfo": {
|
||||
@@ -1389,9 +1049,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/std-env": {
|
||||
"version": "4.2.0",
|
||||
"resolved": "https://registry.npmjs.org/std-env/-/std-env-4.2.0.tgz",
|
||||
"integrity": "sha512-oCUKSupKTHX53EyjDtuZQ64pjLJ6yYCtpmEw0goYxtjG9KpbRe8KAsl2tBUGU9DyMcJ0RwJ8GqJAFzMXcXW1Rw==",
|
||||
"version": "4.1.0",
|
||||
"resolved": "https://registry.npmjs.org/std-env/-/std-env-4.1.0.tgz",
|
||||
"integrity": "sha512-Rq7ybcX2RuC55r9oaPVEW7/xu3tj8u4GeBYHBWCychFtzMIr86A7e3PPEBPT37sHStKX3+TiX/Fr/ACmJLVlLQ==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
@@ -1448,51 +1108,30 @@
|
||||
"optional": true
|
||||
},
|
||||
"node_modules/typescript": {
|
||||
"version": "7.0.2",
|
||||
"resolved": "https://registry.npmjs.org/typescript/-/typescript-7.0.2.tgz",
|
||||
"integrity": "sha512-8FYau96o3NKOhbjKi/qNvG/W5jhzxkbdm5sj9AbZ/5T5sWqn3hJgLfGx27sRKZWTvyzCP8dLRBTf5tBTSRVUNA==",
|
||||
"version": "6.0.3",
|
||||
"resolved": "https://registry.npmjs.org/typescript/-/typescript-6.0.3.tgz",
|
||||
"integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"bin": {
|
||||
"tsc": "bin/tsc"
|
||||
"tsc": "bin/tsc",
|
||||
"tsserver": "bin/tsserver"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=16.20.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@typescript/typescript-aix-ppc64": "7.0.2",
|
||||
"@typescript/typescript-darwin-arm64": "7.0.2",
|
||||
"@typescript/typescript-darwin-x64": "7.0.2",
|
||||
"@typescript/typescript-freebsd-arm64": "7.0.2",
|
||||
"@typescript/typescript-freebsd-x64": "7.0.2",
|
||||
"@typescript/typescript-linux-arm": "7.0.2",
|
||||
"@typescript/typescript-linux-arm64": "7.0.2",
|
||||
"@typescript/typescript-linux-loong64": "7.0.2",
|
||||
"@typescript/typescript-linux-mips64el": "7.0.2",
|
||||
"@typescript/typescript-linux-ppc64": "7.0.2",
|
||||
"@typescript/typescript-linux-riscv64": "7.0.2",
|
||||
"@typescript/typescript-linux-s390x": "7.0.2",
|
||||
"@typescript/typescript-linux-x64": "7.0.2",
|
||||
"@typescript/typescript-netbsd-arm64": "7.0.2",
|
||||
"@typescript/typescript-netbsd-x64": "7.0.2",
|
||||
"@typescript/typescript-openbsd-arm64": "7.0.2",
|
||||
"@typescript/typescript-openbsd-x64": "7.0.2",
|
||||
"@typescript/typescript-sunos-x64": "7.0.2",
|
||||
"@typescript/typescript-win32-arm64": "7.0.2",
|
||||
"@typescript/typescript-win32-x64": "7.0.2"
|
||||
"node": ">=14.17"
|
||||
}
|
||||
},
|
||||
"node_modules/vite": {
|
||||
"version": "8.1.5",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.5.tgz",
|
||||
"integrity": "sha512-7ULLwsCdYx/nRyrpiEwvqb5TFHrMVZyBt+rg/OAXT7rgj/z+DtTDyKFeLAdDkubDVDKD8jOsndmy7m55XcfUsw==",
|
||||
"version": "8.1.3",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.3.tgz",
|
||||
"integrity": "sha512-Ds+gBRbj0lwRO2Y5hwnUBdxSwlAve9LeRyU4sNnAr0ewW0gWF0n5bgXgUzbgZ49MV9BVUAQUFYVcDUcilUExMA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"lightningcss": "^1.32.0",
|
||||
"picomatch": "^4.0.5",
|
||||
"postcss": "^8.5.17",
|
||||
"rolldown": "~1.1.5",
|
||||
"picomatch": "^4.0.4",
|
||||
"postcss": "^8.5.16",
|
||||
"rolldown": "~1.1.3",
|
||||
"tinyglobby": "^0.2.17"
|
||||
},
|
||||
"bin": {
|
||||
|
||||
@@ -32,7 +32,7 @@
|
||||
],
|
||||
"license": "Apache-2.0",
|
||||
"devDependencies": {
|
||||
"typescript": "^7.0.0",
|
||||
"typescript": "^6.0.0",
|
||||
"vitest": "^4.1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -157,49 +157,6 @@ export interface CancelledEvent {
|
||||
type: "cancelled";
|
||||
}
|
||||
|
||||
/**
|
||||
* Context-compaction lifecycle. `start` carries `trigger` ("manual"/"auto";
|
||||
* auto adds `where` + `pct`); `progress` carries chunked-summarization
|
||||
* `part`/`total`/`depth` (or `retry_in`/`error` for a retry wait); `end`
|
||||
* carries `ok` plus either `before_tokens`/`after_tokens`/`summary` or the
|
||||
* failure `reason`/`message`. The successful end's summary also replays from
|
||||
* `/history` as a `role: "system"`, `source: "compaction"` entry.
|
||||
*/
|
||||
export interface CompactionEvent {
|
||||
type: "compaction";
|
||||
phase: "start" | "progress" | "end";
|
||||
/** Correlates every event of one compaction run (0 from legacy emitters). */
|
||||
compaction_id?: number;
|
||||
/**
|
||||
* End events only: true marks a force-abandoned compaction retiring
|
||||
* after a successor generation took over — skip failure notices for
|
||||
* those (an OK end's result still stands; the history swap happened).
|
||||
*/
|
||||
superseded?: boolean;
|
||||
/**
|
||||
* Failed ends only: the emitter-computed display verdict — show
|
||||
* `message` only when true, instead of re-deriving suppression from
|
||||
* reason/trigger/superseded client-side.
|
||||
*/
|
||||
notice?: boolean;
|
||||
/** Present on start and on every end (ok or failed). */
|
||||
trigger?: "manual" | "auto";
|
||||
where?: string;
|
||||
pct?: number;
|
||||
part?: number;
|
||||
total?: number;
|
||||
depth?: number;
|
||||
retry_in?: number;
|
||||
error?: string;
|
||||
warning?: string;
|
||||
ok?: boolean;
|
||||
reason?: string;
|
||||
message?: string;
|
||||
before_tokens?: number;
|
||||
after_tokens?: number;
|
||||
summary?: string;
|
||||
}
|
||||
|
||||
// Global events
|
||||
|
||||
export interface WsStateEvent {
|
||||
@@ -255,7 +212,6 @@ export type ServerEvent =
|
||||
| BusyErrorEvent
|
||||
| ClearUiEvent
|
||||
| CancelledEvent
|
||||
| CompactionEvent
|
||||
| WsStateEvent
|
||||
| WsActivityEvent
|
||||
| WsRenameEvent
|
||||
|
||||
@@ -14,12 +14,9 @@ collect it as a test file — it's an importable utility, not a test.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from turnstone.core.providers import StreamChunk, ToolCallDelta
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
|
||||
@@ -46,304 +43,3 @@ def make_session(**kwargs: Any) -> ChatSession:
|
||||
}
|
||||
defaults.update(kwargs)
|
||||
return ChatSession(**defaults)
|
||||
|
||||
|
||||
def mock_completion_result(
|
||||
content: str = "",
|
||||
tool_calls: list[dict[str, Any]] | None = None,
|
||||
) -> MagicMock:
|
||||
"""A provider result shaped like ``CompletionResult``.
|
||||
|
||||
Callers that route through ``model_turn`` (judges, task agents, and
|
||||
every lane #827 migrates) hit its re-ingest, which iterates
|
||||
``tool_calls``/``provider_blocks`` and joins ``reasoning`` — a bare
|
||||
MagicMock attribute would TypeError deep inside the seam, so every
|
||||
field the re-ingest reads is pinned to a real value here. ONE shared
|
||||
definition: when the re-ingest starts reading a new CompletionResult
|
||||
field, add it here and every suite moves together.
|
||||
"""
|
||||
result = MagicMock()
|
||||
result.content = content
|
||||
result.tool_calls = tool_calls
|
||||
result.finish_reason = "stop"
|
||||
result.usage = None
|
||||
result.provider_blocks = []
|
||||
result.reasoning = ""
|
||||
return result
|
||||
|
||||
|
||||
def fake_chat_stream(
|
||||
*,
|
||||
content: str | None = None,
|
||||
tool_calls: list[dict[str, str]] | None = None,
|
||||
finish_reason: str = "stop",
|
||||
prompt_tokens: int = 10,
|
||||
completion_tokens: int = 5,
|
||||
reasoning_content: str | None = None,
|
||||
reasoning: str | None = None,
|
||||
) -> list[Any]:
|
||||
"""Fake OpenAI Chat Completions SSE chunks for driving the REAL
|
||||
``OpenAIChatCompletionsProvider`` through a fake SDK client::
|
||||
|
||||
client.chat.completions.create = lambda **kw: fake_chat_stream(...)
|
||||
|
||||
Exercises the adapter's ``_iter_stream`` plus ``drain_stream`` end to
|
||||
end (the highest-fidelity fake lane), unlike ``as_stream`` which fakes
|
||||
at the provider boundary. ``tool_calls`` entries are
|
||||
``{"id", "name", "arguments"}`` dicts. ``SimpleNamespace`` (not
|
||||
``MagicMock``) so absent SDK fields read as real ``None`` — an
|
||||
auto-created mock attribute would leak into ``len()``/string paths.
|
||||
|
||||
Emits the realistic three-phase shape: data chunk(s), a finish-reason
|
||||
chunk, then the ``stream_options.include_usage`` usage-only chunk with
|
||||
empty ``choices``.
|
||||
"""
|
||||
|
||||
def _delta(
|
||||
content_val: str | None = None,
|
||||
tcs: list[Any] | None = None,
|
||||
rc: str | None = None,
|
||||
rsn: str | None = None,
|
||||
) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
content=content_val,
|
||||
tool_calls=tcs,
|
||||
reasoning=rsn,
|
||||
reasoning_content=rc,
|
||||
annotations=None,
|
||||
)
|
||||
|
||||
chunks: list[Any] = []
|
||||
if reasoning_content is not None or reasoning is not None:
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[
|
||||
SimpleNamespace(
|
||||
finish_reason=None, delta=_delta(rc=reasoning_content, rsn=reasoning)
|
||||
)
|
||||
],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
if content is not None:
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=None, delta=_delta(content))],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
if tool_calls:
|
||||
tcs = [
|
||||
SimpleNamespace(
|
||||
index=i,
|
||||
id=tc.get("id", ""),
|
||||
function=SimpleNamespace(
|
||||
name=tc.get("name", ""), arguments=tc.get("arguments", "")
|
||||
),
|
||||
)
|
||||
for i, tc in enumerate(tool_calls)
|
||||
]
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=None, delta=_delta(None, tcs))],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=finish_reason, delta=_delta())],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[],
|
||||
usage=SimpleNamespace(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=None,
|
||||
input_tokens_details=None,
|
||||
),
|
||||
)
|
||||
)
|
||||
return chunks
|
||||
|
||||
|
||||
class _ScriptedClient:
|
||||
"""Callable client-method fake following a script of stream builders.
|
||||
|
||||
Call N returns the stream described by ``scripts[N]``; the last script
|
||||
repeats for any further calls. Each script is a dict of kwargs for
|
||||
the bound stream builder, or a pre-built return value. Records every
|
||||
call's kwargs on ``.calls`` — read ``len(fn.calls)`` where a test
|
||||
previously kept its own counter cell, and ``fn.calls[i]["messages"]``
|
||||
where it captured request bodies.
|
||||
"""
|
||||
|
||||
def __init__(self, scripts: tuple[Any, ...], to_stream: Any) -> None:
|
||||
self._scripts = scripts
|
||||
self._to_stream = to_stream
|
||||
self.calls: list[dict[str, Any]] = []
|
||||
|
||||
def __call__(self, **kwargs: Any) -> Any:
|
||||
self.calls.append(kwargs)
|
||||
script = self._scripts[min(len(self.calls) - 1, len(self._scripts) - 1)]
|
||||
return self._to_stream(**script) if isinstance(script, dict) else script
|
||||
|
||||
|
||||
def scripted_chat_client(*scripts: Any) -> _ScriptedClient:
|
||||
"""A scripted ``client.chat.completions.create`` — dict scripts are
|
||||
:func:`fake_chat_stream` kwargs."""
|
||||
return _ScriptedClient(scripts, fake_chat_stream)
|
||||
|
||||
|
||||
def scripted_anthropic_client(*scripts: Any) -> _ScriptedClient:
|
||||
"""A scripted ``client.messages.stream`` — dict scripts are
|
||||
:func:`fake_anthropic_stream` kwargs (``blocks`` plus optional
|
||||
``stop_reason``/``usage``)."""
|
||||
return _ScriptedClient(scripts, fake_anthropic_stream)
|
||||
|
||||
|
||||
class FakeAnthropicBlock:
|
||||
"""A full-content Anthropic content-block fake for
|
||||
:func:`fake_anthropic_stream` — plain attributes plus the
|
||||
``model_dump()`` the provider's block capture reads."""
|
||||
|
||||
def __init__(self, **fields: Any) -> None:
|
||||
self._fields = fields
|
||||
for key, value in fields.items():
|
||||
setattr(self, key, value)
|
||||
|
||||
def model_dump(self, **_kw: Any) -> dict[str, Any]:
|
||||
return dict(self._fields)
|
||||
|
||||
|
||||
def fake_anthropic_stream(
|
||||
blocks: list[Any],
|
||||
*,
|
||||
stop_reason: str | None = "end_turn",
|
||||
usage: Any = None,
|
||||
) -> Any:
|
||||
"""Fake Anthropic SDK stream context manager for tests that drive the
|
||||
REAL ``AnthropicProvider`` through a fake client::
|
||||
|
||||
client.messages.stream = lambda **kw: fake_anthropic_stream(...)
|
||||
|
||||
Accepts the same full-content block fakes the pre-#831
|
||||
``get_final_message`` fixtures used (objects with ``.type`` + fields
|
||||
and ``model_dump()``) and synthesizes the real event grammar the
|
||||
streaming iterator consumes: ``content_block_start`` carries the block
|
||||
with its text/thinking/signature EMPTIED and ``input`` as ``{}`` (the
|
||||
SDK start shape), deltas carry the content, ``content_block_stop``
|
||||
finalizes tool input, and the closing ``message_delta`` carries
|
||||
``stop_reason`` (+ optional usage object). Without the stripping, the
|
||||
provider's raw-block accumulator would double every text/thinking
|
||||
field (start capture + delta append).
|
||||
|
||||
``stop_reason=None`` omits the closing ``message_delta`` entirely —
|
||||
the terminal-signal-less lax-gateway shape ``finish_reason_optional``
|
||||
exists for (content arrives, then the stream just ends).
|
||||
"""
|
||||
events: list[Any] = []
|
||||
for idx, block in enumerate(blocks):
|
||||
d = dict(block.model_dump()) if hasattr(block, "model_dump") else dict(vars(block))
|
||||
btype = d.get("type", "")
|
||||
start = dict(d)
|
||||
if btype == "text":
|
||||
start["text"] = ""
|
||||
elif btype == "thinking":
|
||||
start["thinking"] = ""
|
||||
start["signature"] = ""
|
||||
elif btype == "tool_use":
|
||||
start["input"] = {}
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_start", index=idx, content_block=SimpleNamespace(**start)
|
||||
)
|
||||
)
|
||||
if btype == "text" and d.get("text"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="text_delta", text=d["text"]),
|
||||
)
|
||||
)
|
||||
elif btype == "thinking":
|
||||
if d.get("thinking"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="thinking_delta", thinking=d["thinking"]),
|
||||
)
|
||||
)
|
||||
if d.get("signature"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="signature_delta", signature=d["signature"]),
|
||||
)
|
||||
)
|
||||
elif btype == "tool_use":
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(
|
||||
type="input_json_delta",
|
||||
partial_json=json.dumps(d.get("input", {})),
|
||||
),
|
||||
)
|
||||
)
|
||||
events.append(SimpleNamespace(type="content_block_stop", index=idx))
|
||||
if stop_reason is not None or usage is not None:
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="message_delta", usage=usage, delta=SimpleNamespace(stop_reason=stop_reason)
|
||||
)
|
||||
)
|
||||
|
||||
mgr = MagicMock()
|
||||
mgr.__enter__ = MagicMock(return_value=events)
|
||||
mgr.__exit__ = MagicMock(return_value=False)
|
||||
return mgr
|
||||
|
||||
|
||||
def as_stream(result: Any) -> list[StreamChunk]:
|
||||
"""Adapt a ``CompletionResult``-shaped fake to a ``create_streaming``
|
||||
return value (single terminal chunk).
|
||||
|
||||
The #831 transport collapse routes every single-shot lane through
|
||||
``drain_stream(provider.create_streaming(...))``, so provider fakes
|
||||
return chunk iterables now. Tests keep building result-shaped fakes
|
||||
(``mock_completion_result`` or hand-rolled) and wrap them at
|
||||
assignment: ``provider.create_streaming.return_value =
|
||||
as_stream(result)``. A list re-iterates on every call, so one
|
||||
``return_value`` serves repeated-call tests; convert AFTER mutating
|
||||
the fake's fields — the chunk snapshots them.
|
||||
|
||||
Multi-chunk accumulation semantics are exercised by the dedicated
|
||||
``drain_stream`` unit tests, not through this helper.
|
||||
"""
|
||||
deltas = [
|
||||
ToolCallDelta(
|
||||
index=i,
|
||||
id=tc.get("id", ""),
|
||||
name=tc.get("function", {}).get("name", ""),
|
||||
arguments_delta=tc.get("function", {}).get("arguments", ""),
|
||||
)
|
||||
for i, tc in enumerate(result.tool_calls or [])
|
||||
]
|
||||
return [
|
||||
StreamChunk(
|
||||
content_delta=result.content or "",
|
||||
reasoning_delta=getattr(result, "reasoning", "") or "",
|
||||
tool_call_deltas=deltas,
|
||||
usage=result.usage,
|
||||
finish_reason=result.finish_reason or "stop",
|
||||
provider_blocks=list(result.provider_blocks or []),
|
||||
)
|
||||
]
|
||||
|
||||
@@ -1,604 +0,0 @@
|
||||
"""Browser-fidelity SSE recovery harness helpers.
|
||||
|
||||
The load-bearing assembly for ``tests/test_sse_recovery_e2e.py``: a
|
||||
``BrowserlikeSSEClient`` that speaks the exact wire contract the real
|
||||
``turnstone/shared_static/interactive.js`` pane speaks, and the
|
||||
assertion helpers the scenarios share. The server boot machinery lives
|
||||
in ``_sse_recovery_server.py``.
|
||||
|
||||
Why a raw-socket SSE reader (and not ``httpx.stream``): the slow-consumer
|
||||
overflow scenario needs the consumer to STALL — stop reading the socket
|
||||
so the server's SSE generator blocks on ``await send`` and stops draining
|
||||
the per-UI listener queue, which then poisons at its cap. A faithful
|
||||
stall needs (a) precise control over when bytes are read and (b) a small
|
||||
``SO_RCVBUF`` so the in-flight backlog before poison stays bounded to
|
||||
~100 KB instead of the client kernel's multi-MB autotuned default (which
|
||||
would need tens of thousands of events to overflow). A raw socket gives
|
||||
both; httpx (used here only for the plain ``/history`` request/response)
|
||||
gives neither. This is ALSO closer to the browser: EventSource has a
|
||||
bounded receive buffer, not an unbounded one.
|
||||
|
||||
Client contract mirrored from interactive.js (line references are to
|
||||
that file on the ``fix/sse-truncated-resync`` branch):
|
||||
|
||||
- ``_last_event_id`` advances ONLY from SSE ``id:`` fields, and only
|
||||
ring-buffer events carry one — synthetic replay frames (connected /
|
||||
status / state_change / in_progress_snapshot / replay_truncated /
|
||||
stream_overflow) do not, exactly like ``EventSource.lastEventId``
|
||||
(interactive.js onmessage ~1378).
|
||||
- reconnect presents ``connectCursor = _truncatedFromCursor ??
|
||||
_lastEventId`` as ``?last_event_id=`` (manual path) or a
|
||||
``Last-Event-ID`` header (native EventSource auto-reconnect path)
|
||||
(interactive.js connectSSE ~1328).
|
||||
- on a ``replay_truncated`` envelope the client records the
|
||||
truncation-time cursor keep-oldest (``_truncatedFromCursor =
|
||||
_lastEventId`` only when null) and runs ``_loadHistoryThenConnect``
|
||||
(disconnect → /history → adopt cursor → reconnect); a FAILED
|
||||
/history leaves the record armed so the reconnect re-presents the
|
||||
truncation-time cursor and re-draws the envelope (interactive.js
|
||||
handleEvent replay_truncated ~2303, _loadHistoryThenConnect ~1604,
|
||||
_refetchHistory seedCursor ~1706).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import json
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import httpx
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
|
||||
# Small client receive buffer so a stalled consumer's in-flight backlog
|
||||
# before the server-side poison stays bounded (~100 KB) instead of the
|
||||
# multi-MB autotuned default. Paired with the server's small SO_SNDBUF
|
||||
# (see _sse_recovery_server.build_recovery_server).
|
||||
_CLIENT_RCVBUF = 2048
|
||||
|
||||
|
||||
@dataclass
|
||||
class SSEFrame:
|
||||
"""One decoded SSE frame, tagged with the connection it arrived on.
|
||||
|
||||
``event_id`` is the ``id:`` field verbatim (a stringified integer,
|
||||
or ``None`` for id-less synthetic frames — the same string domain as
|
||||
``EventSource.lastEventId``). ``etype`` is the ``type`` field of the
|
||||
JSON ``data:`` payload (the application event type), distinct from
|
||||
any SSE ``event:`` field, which the server never uses.
|
||||
"""
|
||||
|
||||
conn_index: int
|
||||
event_id: str | None
|
||||
etype: str | None
|
||||
payload: dict[str, Any] | None
|
||||
raw: str
|
||||
|
||||
@property
|
||||
def event_id_int(self) -> int | None:
|
||||
if self.event_id is None:
|
||||
return None
|
||||
try:
|
||||
return int(self.event_id)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
class BrowserlikeSSEClient:
|
||||
"""A single interactive pane's SSE + /history state machine.
|
||||
|
||||
Not thread-safe against concurrent public calls; drive it from one
|
||||
test thread. Internally a per-connection reader thread decodes the
|
||||
stream; ``stall()`` / ``resume()`` gate that thread's socket reads so
|
||||
a test can build server-side backpressure without closing the
|
||||
connection (the slow-consumer → listener-queue-poison path).
|
||||
"""
|
||||
|
||||
def __init__(self, base_url: str, ws_id: str, token: str) -> None:
|
||||
parts = urlsplit(base_url)
|
||||
self._host = parts.hostname or "127.0.0.1"
|
||||
self._port = parts.port or 80
|
||||
self._ws_id = ws_id
|
||||
self._token = token
|
||||
self._auth = {"Authorization": f"Bearer {token}"}
|
||||
self._http = httpx.Client(
|
||||
base_url=f"http://{self._host}:{self._port}", timeout=httpx.Timeout(15.0)
|
||||
)
|
||||
|
||||
# EventSource-equivalent cursor state.
|
||||
self._last_event_id: str | None = None
|
||||
self._truncated_from_cursor: str | None = None
|
||||
|
||||
# Transcript. ``_all_frames`` is the cross-connection accumulation
|
||||
# (what "the client eventually saw"); ``_conn_frames`` keeps each
|
||||
# connection's slice for per-connection assertions (contiguity).
|
||||
self._all_frames: list[SSEFrame] = []
|
||||
self._conn_frames: list[list[SSEFrame]] = []
|
||||
self._frames_lock = threading.Lock()
|
||||
|
||||
# Reader plumbing.
|
||||
self._sock: socket.socket | None = None
|
||||
self._reader: threading.Thread | None = None
|
||||
self._stop = threading.Event()
|
||||
self._read_gate = threading.Event()
|
||||
self._read_gate.set() # reading permitted by default
|
||||
self._status: int | None = None
|
||||
self._headers_done = threading.Event()
|
||||
|
||||
# -- connection lifecycle ------------------------------------------------
|
||||
|
||||
def _events_path(self, cursor: str | None) -> str:
|
||||
path = f"/v1/api/workstreams/{self._ws_id}/events"
|
||||
if cursor is not None:
|
||||
path += f"?last_event_id={cursor}"
|
||||
return path
|
||||
|
||||
def connect(self, *, native: bool = False, rcvbuf: int | None = None) -> None:
|
||||
"""Open the SSE stream, presenting the client's current cursor.
|
||||
|
||||
``native=True`` models the browser's EventSource auto-reconnect:
|
||||
the cursor rides a ``Last-Event-ID`` HEADER and never appears in
|
||||
the URL. ``native=False`` models the manual ``new EventSource(url
|
||||
+ '?last_event_id=')`` path interactive.js uses when it must
|
||||
override the live cursor (the ``connectCursor`` chokepoint).
|
||||
|
||||
``rcvbuf`` shrinks this connection's ``SO_RCVBUF`` — pass
|
||||
``_CLIENT_RCVBUF`` on a connection the test will ``stall()`` so the
|
||||
in-flight backlog before the server-side poison stays bounded.
|
||||
Leave it ``None`` (OS default) on recovery reconnects so the ring
|
||||
replay is not throttled to a crawl.
|
||||
"""
|
||||
if self._reader is not None:
|
||||
raise RuntimeError("already connected; disconnect() first")
|
||||
connect_cursor = (
|
||||
self._truncated_from_cursor
|
||||
if self._truncated_from_cursor is not None
|
||||
else self._last_event_id
|
||||
)
|
||||
header_lines = [
|
||||
f"Host: {self._host}:{self._port}",
|
||||
f"Authorization: Bearer {self._token}",
|
||||
"Accept: text/event-stream",
|
||||
"Cache-Control: no-cache",
|
||||
]
|
||||
if native:
|
||||
path = self._events_path(None)
|
||||
if connect_cursor is not None:
|
||||
header_lines.append(f"Last-Event-ID: {connect_cursor}")
|
||||
else:
|
||||
path = self._events_path(connect_cursor)
|
||||
request = f"GET {path} HTTP/1.1\r\n" + "\r\n".join(header_lines) + "\r\n\r\n"
|
||||
|
||||
sock = socket.create_connection((self._host, self._port), timeout=10)
|
||||
if rcvbuf is not None:
|
||||
sock.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf)
|
||||
sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
|
||||
sock.settimeout(None)
|
||||
sock.sendall(request.encode())
|
||||
self._sock = sock
|
||||
|
||||
self._stop.clear()
|
||||
self._read_gate.set()
|
||||
self._status = None
|
||||
self._headers_done.clear()
|
||||
conn_index = len(self._conn_frames)
|
||||
frames: list[SSEFrame] = []
|
||||
self._conn_frames.append(frames)
|
||||
self._reader = threading.Thread(
|
||||
target=self._read_loop,
|
||||
args=(sock, conn_index, frames),
|
||||
name=f"sse-reader-{self._ws_id[:6]}-{conn_index}",
|
||||
daemon=True,
|
||||
)
|
||||
self._reader.start()
|
||||
|
||||
# Surface a non-200 handshake to the caller (409 half-built UI,
|
||||
# 404 unknown ws, 401 auth) rather than silently reading nothing.
|
||||
if not self._headers_done.wait(timeout=10):
|
||||
self.disconnect()
|
||||
raise AssertionError("events connect: no HTTP response headers")
|
||||
if self._status != 200:
|
||||
status = self._status
|
||||
self.disconnect()
|
||||
raise AssertionError(f"events connect returned HTTP {status}")
|
||||
|
||||
def disconnect(self) -> None:
|
||||
"""Close the stream and join the reader (leak-guard clean)."""
|
||||
self._stop.set()
|
||||
self._read_gate.set() # release a stalled reader so it sees _stop
|
||||
sock = self._sock
|
||||
if sock is not None:
|
||||
with contextlib.suppress(OSError):
|
||||
sock.shutdown(socket.SHUT_RDWR) # interrupt a blocked recv
|
||||
reader = self._reader
|
||||
if reader is not None:
|
||||
reader.join(timeout=15)
|
||||
if reader.is_alive():
|
||||
raise AssertionError("SSE reader thread failed to stop")
|
||||
if sock is not None:
|
||||
with contextlib.suppress(OSError):
|
||||
sock.close()
|
||||
self._sock = None
|
||||
self._reader = None
|
||||
|
||||
def close(self) -> None:
|
||||
"""Full teardown: disconnect any live stream + close the HTTP client."""
|
||||
if self._reader is not None:
|
||||
self.disconnect()
|
||||
self._http.close()
|
||||
|
||||
# -- the stall gate (backpressure driver) --------------------------------
|
||||
|
||||
def stall(self) -> None:
|
||||
"""Stop reading the socket. The kernel + uvicorn send buffers fill,
|
||||
blocking the server's SSE generator on its ``await send``, so it
|
||||
stops draining the per-UI listener queue — which poisons at its cap.
|
||||
"""
|
||||
self._read_gate.clear()
|
||||
|
||||
def resume(self) -> None:
|
||||
"""Resume reading. A poisoned-and-closed stream delivers its
|
||||
``stream_overflow`` farewell frame once the backlog drains."""
|
||||
self._read_gate.set()
|
||||
|
||||
# -- reader --------------------------------------------------------------
|
||||
|
||||
def _read_loop(self, sock: socket.socket, conn_index: int, frames: list[SSEFrame]) -> None:
|
||||
raw = b"" # undecoded bytes (headers, then chunked framing)
|
||||
sse = b"" # decoded SSE byte stream
|
||||
headers_parsed = False
|
||||
chunked = False
|
||||
while not self._stop.is_set():
|
||||
# Backpressure gate: while stalled we do NOT read the socket, so
|
||||
# its receive buffer fills and TCP flow control stalls the server.
|
||||
if not self._read_gate.wait(timeout=0.1):
|
||||
continue
|
||||
if self._stop.is_set():
|
||||
break
|
||||
try:
|
||||
chunk = sock.recv(65536)
|
||||
except OSError:
|
||||
break
|
||||
if not chunk:
|
||||
break # server closed
|
||||
raw += chunk
|
||||
if not headers_parsed:
|
||||
if b"\r\n\r\n" not in raw:
|
||||
continue
|
||||
header_blob, raw = raw.split(b"\r\n\r\n", 1)
|
||||
self._parse_headers(header_blob)
|
||||
chunked = b"transfer-encoding: chunked" in header_blob.lower()
|
||||
headers_parsed = True
|
||||
self._headers_done.set()
|
||||
if chunked:
|
||||
decoded, raw = _dechunk(raw)
|
||||
sse += decoded
|
||||
else:
|
||||
sse += raw
|
||||
raw = b""
|
||||
sse = sse.replace(b"\r\n", b"\n")
|
||||
while b"\n\n" in sse:
|
||||
block, sse = sse.split(b"\n\n", 1)
|
||||
self._handle_block(block.decode("utf-8", "replace"), conn_index, frames)
|
||||
|
||||
def _parse_headers(self, header_blob: bytes) -> None:
|
||||
first_line = header_blob.split(b"\r\n", 1)[0].decode("latin-1")
|
||||
# "HTTP/1.1 200 OK"
|
||||
parts = first_line.split(" ", 2)
|
||||
if len(parts) >= 2 and parts[1].isdigit():
|
||||
self._status = int(parts[1])
|
||||
|
||||
def _handle_block(self, block_text: str, conn_index: int, frames: list[SSEFrame]) -> None:
|
||||
event_id: str | None = None
|
||||
data_parts: list[str] = []
|
||||
retry: str | None = None
|
||||
for line in block_text.split("\n"):
|
||||
if not line or line.startswith(":"):
|
||||
continue # blank or comment (ping)
|
||||
field_name, _, value = line.partition(":")
|
||||
if value.startswith(" "):
|
||||
value = value[1:] # SSE strips a single leading space
|
||||
if field_name == "id":
|
||||
event_id = value
|
||||
elif field_name == "data":
|
||||
data_parts.append(value)
|
||||
elif field_name == "retry":
|
||||
retry = value
|
||||
# EventSource semantics: an event carrying an ``id:`` sets the
|
||||
# last-event-id buffer; an event without one leaves it unchanged.
|
||||
if event_id is not None:
|
||||
self._last_event_id = event_id
|
||||
if not data_parts:
|
||||
if retry is not None:
|
||||
self._record(SSEFrame(conn_index, None, "retry", None, block_text), frames)
|
||||
return
|
||||
data_str = "\n".join(data_parts)
|
||||
payload: dict[str, Any] | None
|
||||
try:
|
||||
parsed = json.loads(data_str)
|
||||
payload = parsed if isinstance(parsed, dict) else None
|
||||
except ValueError:
|
||||
payload = None
|
||||
etype = payload.get("type") if payload is not None else None
|
||||
frame = SSEFrame(conn_index, event_id, etype, payload, data_str)
|
||||
self._record(frame, frames)
|
||||
# Mirror the pane: the FIRST replay_truncated for an unrepaired gap
|
||||
# records the truncation-time cursor (keep-oldest). Its consumer is
|
||||
# the reconnect chokepoint (see ``connect``).
|
||||
if etype == "replay_truncated" and self._truncated_from_cursor is None:
|
||||
self._truncated_from_cursor = self._last_event_id
|
||||
|
||||
def _record(self, frame: SSEFrame, frames: list[SSEFrame]) -> None:
|
||||
with self._frames_lock:
|
||||
frames.append(frame)
|
||||
self._all_frames.append(frame)
|
||||
|
||||
# -- /history + cursor flow ----------------------------------------------
|
||||
|
||||
def fetch_history(self) -> dict[str, Any]:
|
||||
"""GET /history and return the parsed JSON ({ws_id, messages, cursor})."""
|
||||
r = self._http.get(f"/v1/api/workstreams/{self._ws_id}/history", headers=self._auth)
|
||||
r.raise_for_status()
|
||||
result: dict[str, Any] = r.json()
|
||||
return result
|
||||
|
||||
def seed_from_history(self) -> dict[str, Any]:
|
||||
"""The seedCursor step: fetch /history, adopt a non-null resume
|
||||
cursor into ``_last_event_id``, and clear the truncation record on
|
||||
a successful render (replayHistory clears ``_truncatedFromCursor``).
|
||||
"""
|
||||
data = self.fetch_history()
|
||||
cursor = data.get("cursor")
|
||||
if cursor is not None:
|
||||
self._last_event_id = str(cursor)
|
||||
self._truncated_from_cursor = None # successful full render repairs the gap
|
||||
return data
|
||||
|
||||
def load_history_then_connect(
|
||||
self, *, fail_history: bool = False, native: bool = False
|
||||
) -> dict[str, Any] | None:
|
||||
"""Reproduce interactive.js ``_loadHistoryThenConnect``.
|
||||
|
||||
Disconnect first, drop the live cursor (``_last_event_id = None``)
|
||||
but KEEP ``_truncated_from_cursor`` armed, then fetch /history and
|
||||
reconnect. On success adopt the returned cursor and clear the
|
||||
truncation record; on a FAILED /history (``fail_history`` — the
|
||||
harness IS the client here, so a client-side simulated failure is
|
||||
faithful) leave the record armed so the reconnect re-presents the
|
||||
truncation-time cursor and re-draws ``replay_truncated``.
|
||||
|
||||
Returns the /history JSON, or ``None`` when the fetch failed.
|
||||
"""
|
||||
if self._reader is not None:
|
||||
self.disconnect()
|
||||
self._last_event_id = None
|
||||
data: dict[str, Any] | None
|
||||
if fail_history:
|
||||
data = None
|
||||
else:
|
||||
data = self.fetch_history()
|
||||
cursor = data.get("cursor")
|
||||
if cursor is not None:
|
||||
self._last_event_id = str(cursor)
|
||||
self._truncated_from_cursor = None
|
||||
self.connect(native=native)
|
||||
return data
|
||||
|
||||
# -- accessors + waits ---------------------------------------------------
|
||||
|
||||
@property
|
||||
def last_event_id(self) -> str | None:
|
||||
return self._last_event_id
|
||||
|
||||
@property
|
||||
def truncated_from_cursor(self) -> str | None:
|
||||
return self._truncated_from_cursor
|
||||
|
||||
def all_frames(self) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._all_frames)
|
||||
|
||||
def conn_frames(self, conn_index: int) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._conn_frames[conn_index])
|
||||
|
||||
def latest_conn_frames(self) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._conn_frames[-1]) if self._conn_frames else []
|
||||
|
||||
def num_connections(self) -> int:
|
||||
with self._frames_lock:
|
||||
return len(self._conn_frames)
|
||||
|
||||
def frames_of_type(self, etype: str) -> list[SSEFrame]:
|
||||
return [f for f in self.all_frames() if f.etype == etype]
|
||||
|
||||
def has_type(self, etype: str) -> bool:
|
||||
return any(f.etype == etype for f in self.all_frames())
|
||||
|
||||
def tool_output_by_call(self) -> dict[str, str]:
|
||||
"""Concatenate every ``tool_output_chunk`` payload per call_id, in
|
||||
arrival order — the reconstructed live stream for each call."""
|
||||
out: dict[str, str] = {}
|
||||
for f in self.all_frames():
|
||||
if f.etype == "tool_output_chunk" and f.payload is not None:
|
||||
cid = str(f.payload.get("call_id", ""))
|
||||
out[cid] = out.get(cid, "") + str(f.payload.get("chunk", ""))
|
||||
return out
|
||||
|
||||
def tool_results_by_call(self) -> dict[str, str]:
|
||||
"""The last ``tool_result`` output seen per call_id."""
|
||||
out: dict[str, str] = {}
|
||||
for f in self.all_frames():
|
||||
if f.etype == "tool_result" and f.payload is not None:
|
||||
out[str(f.payload.get("call_id", ""))] = str(f.payload.get("output", ""))
|
||||
return out
|
||||
|
||||
def wait_for_type(self, etype: str, *, timeout: float = 45.0) -> SSEFrame:
|
||||
"""Block until a frame of ``etype`` has arrived on ANY connection."""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
for f in self.all_frames():
|
||||
if f.etype == etype:
|
||||
return f
|
||||
time.sleep(0.05)
|
||||
raise AssertionError(f"timed out waiting for a {etype!r} frame")
|
||||
|
||||
def wait_for(
|
||||
self, predicate: Callable[[BrowserlikeSSEClient], bool], *, timeout: float = 45.0
|
||||
) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if predicate(self):
|
||||
return
|
||||
time.sleep(0.05)
|
||||
raise AssertionError("timed out waiting for predicate")
|
||||
|
||||
def wait_for_call_result(self, call_id: str, *, timeout: float = 45.0) -> None:
|
||||
self.wait_for(lambda c: call_id in c.tool_results_by_call(), timeout=timeout)
|
||||
|
||||
|
||||
def _dechunk(buf: bytes) -> tuple[bytes, bytes]:
|
||||
"""Incrementally decode HTTP/1.1 chunked transfer-encoding.
|
||||
|
||||
Consumes as many COMPLETE chunks from ``buf`` as possible and returns
|
||||
``(decoded_bytes, remainder)`` where ``remainder`` is the trailing
|
||||
partial chunk to carry into the next read. A zero-length chunk (stream
|
||||
end) simply stops consumption; the reader's ``recv`` EOF handles close.
|
||||
"""
|
||||
decoded = b""
|
||||
while True:
|
||||
if b"\r\n" not in buf:
|
||||
break # incomplete size line
|
||||
size_line, rest = buf.split(b"\r\n", 1)
|
||||
try:
|
||||
n = int(size_line.strip() or b"z", 16)
|
||||
except ValueError:
|
||||
break # malformed / partial — wait for more bytes
|
||||
if n == 0:
|
||||
break # last chunk marker
|
||||
if len(rest) < n + 2: # need n data bytes + trailing CRLF
|
||||
break
|
||||
decoded += rest[:n]
|
||||
buf = rest[n + 2 :]
|
||||
return decoded, buf
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Assertion helpers (shared by the scenarios).
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def assert_contiguous_ids(frames: list[SSEFrame]) -> None:
|
||||
"""Every id-bearing frame in a connection forms a gap-free, dup-free,
|
||||
strictly increasing run.
|
||||
|
||||
Holds for a connection that took no ``_seq``-filtered fresh path — a
|
||||
fresh connect made before any event (snap_seq == 0) and every
|
||||
``replay_ok`` reconnect (snap_seq == 0). The server stamps a fresh
|
||||
monotonic id per enqueue with no in-ring coalescing, so a
|
||||
non-filtered consumer sees consecutive ids.
|
||||
"""
|
||||
ids = [f.event_id_int for f in frames if f.event_id_int is not None]
|
||||
assert ids, "connection carried no id-bearing frames"
|
||||
assert len(set(ids)) == len(ids), f"duplicate SSE ids: {ids}"
|
||||
assert ids == sorted(ids), f"SSE ids not monotonic: {ids}"
|
||||
for prev, cur in zip(ids, ids[1:], strict=False):
|
||||
assert cur == prev + 1, f"gap in SSE ids between {prev} and {cur}: {ids}"
|
||||
|
||||
|
||||
def assert_ids_monotonic_no_dupes(frames: list[SSEFrame]) -> None:
|
||||
"""Weaker invariant that holds on EVERY connection (including
|
||||
``_seq``-filtered fresh/truncated paths, where gaps are legal): ids
|
||||
are strictly increasing with no duplicates."""
|
||||
ids = [f.event_id_int for f in frames if f.event_id_int is not None]
|
||||
assert len(set(ids)) == len(ids), f"duplicate SSE ids: {ids}"
|
||||
assert ids == sorted(ids), f"SSE ids not monotonic: {ids}"
|
||||
|
||||
|
||||
def assert_chunk_result_ordering(frames: list[SSEFrame]) -> None:
|
||||
"""Every ``tool_output_chunk`` for a call precedes that call's own
|
||||
``tool_result`` on the wire (the load-bearing ordering — the client
|
||||
removes the streaming <pre> when it renders the result)."""
|
||||
result_index: dict[str, int] = {}
|
||||
for i, f in enumerate(frames):
|
||||
if f.etype == "tool_result" and f.payload is not None:
|
||||
result_index[str(f.payload.get("call_id", ""))] = i
|
||||
for i, f in enumerate(frames):
|
||||
if f.etype == "tool_output_chunk" and f.payload is not None:
|
||||
cid = str(f.payload.get("call_id", ""))
|
||||
assert cid in result_index, f"chunk for call {cid} has no tool_result"
|
||||
assert i < result_index[cid], (
|
||||
f"chunk for call {cid} arrived AFTER its tool_result "
|
||||
f"(chunk idx {i} >= result idx {result_index[cid]})"
|
||||
)
|
||||
|
||||
|
||||
def assert_children_stamped(frames: list[SSEFrame], parent_call_id: str) -> None:
|
||||
"""Every sub-agent child tool event carries ``parent_call_id`` (stamped
|
||||
at the flush chokepoint). Sub-tool call_ids are minted
|
||||
``{parent}::r{run}s{step}::{provider_id}`` — the ``::`` segment is the
|
||||
identifying mark — and NONE may escape unstamped to the top level."""
|
||||
unstamped: list[tuple[str | None, str, Any]] = []
|
||||
stamped = 0
|
||||
for f in frames:
|
||||
if f.payload is None:
|
||||
continue
|
||||
items = f.payload.get("items")
|
||||
entries = items if isinstance(items, list) else [f.payload]
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
cid = str(entry.get("call_id", ""))
|
||||
if "::" not in cid:
|
||||
continue
|
||||
if entry.get("parent_call_id") == parent_call_id:
|
||||
stamped += 1
|
||||
else:
|
||||
unstamped.append((f.etype, cid, entry.get("parent_call_id")))
|
||||
assert stamped > 0, f"no child events found for parent {parent_call_id}"
|
||||
assert not unstamped, f"child events escaped unstamped (parent {parent_call_id}): {unstamped}"
|
||||
|
||||
|
||||
def history_tool_outputs(history_json: dict[str, Any]) -> dict[str, str]:
|
||||
"""Extract {call_id: output} from a /history projection, however the
|
||||
projection surfaces results (a folded ``output`` on a tool_call, or a
|
||||
trailing ``role: tool`` row keyed by ``tool_call_id``)."""
|
||||
out: dict[str, str] = {}
|
||||
for msg in history_json.get("messages", []):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
if msg.get("role") == "tool":
|
||||
cid = msg.get("tool_call_id") or msg.get("call_id")
|
||||
if cid is not None:
|
||||
out[str(cid)] = str(msg.get("content", ""))
|
||||
for tc in msg.get("tool_calls") or ():
|
||||
if not isinstance(tc, dict):
|
||||
continue
|
||||
cid = tc.get("id") or tc.get("call_id")
|
||||
if cid is not None and tc.get("output") is not None:
|
||||
out[str(cid)] = str(tc.get("output", ""))
|
||||
return out
|
||||
|
||||
|
||||
def assert_converged(client: BrowserlikeSSEClient, history_json: dict[str, Any]) -> None:
|
||||
"""Turn-level equivalence: every tool result the client assembled live
|
||||
is present, with the same output, in a fresh /history projection.
|
||||
|
||||
Compares by call_id so a reconnect that re-delivered a result can't
|
||||
hide a divergence, and asserts the /history side isn't empty (a
|
||||
silently-lost turn would leave the projection short)."""
|
||||
live = client.tool_results_by_call()
|
||||
hist = history_tool_outputs(history_json)
|
||||
assert hist, "fresh /history projected no tool results — a turn was lost"
|
||||
for call_id, output in live.items():
|
||||
assert call_id in hist, f"call {call_id} seen live but absent from /history: {sorted(hist)}"
|
||||
assert hist[call_id] == output, (
|
||||
f"call {call_id} output diverged: live={output!r} history={hist[call_id]!r}"
|
||||
)
|
||||
@@ -1,413 +0,0 @@
|
||||
"""Boot the REAL interactive Turnstone server for the SSE recovery e2e
|
||||
harness: real ``SessionManager`` + real ``ChatSession`` engine driven
|
||||
through a scripted chat-completions client at the SDK boundary, executing
|
||||
REAL bash tools, exposed over a real uvicorn socket.
|
||||
|
||||
The recipe (verified end-to-end) has four load-bearing pieces:
|
||||
|
||||
1. **Provider injection seam.** ``create_app`` takes a PRE-BUILT
|
||||
``SessionManager``, so the harness owns the ``session_factory``: it
|
||||
passes ``client=fake_client`` and OMITS the registry, so
|
||||
``ChatSession`` falls back to ``create_provider("openai-compatible")``
|
||||
== ``OpenAIChatCompletionsProvider`` — exactly what
|
||||
``tests._session_helpers.scripted_chat_client`` targets. No production
|
||||
monkeypatch of the engine.
|
||||
|
||||
2. **Auto-title suppression.** The first user message spawns a background
|
||||
``_generate_title`` LLM call that would consume the first scripted
|
||||
response (the tool call) and desync a positional script. Setting
|
||||
``session._title_generated = True`` before the first send disables it.
|
||||
|
||||
3. **Completion barrier.** ``/send`` returns immediately after spawning
|
||||
``ws.worker_thread``; joining that thread is the true "turn complete,
|
||||
every SSE event enqueued" barrier (``stream_end`` is per-LLM-call, not
|
||||
per-turn, so it is NOT a completion marker).
|
||||
|
||||
4. **Thread hygiene.** ``create_app``'s lifespan unconditionally starts
|
||||
two daemon fan-out threads (``_global_fanout_thread`` blocking on
|
||||
``global_queue.get()``, ``_aggregate_emitter_thread`` on a 10s loop)
|
||||
with no shutdown sentinel — they would trip conftest's leaked-thread
|
||||
guard. They serve the cluster/global lane, which the per-ws ``/events``
|
||||
path under test never touches, so the harness swaps them for no-ops
|
||||
before boot (restored on ``stop``). The result is a fully clean
|
||||
teardown — no ``allow_thread_leak`` needed — with the real per-ws SSE
|
||||
engine fully intact.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import queue as _q
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import httpx
|
||||
import uvicorn
|
||||
|
||||
import turnstone.server as tsrv
|
||||
from tests._session_helpers import scripted_chat_client
|
||||
from turnstone.core.adapters.interactive_adapter import InteractiveAdapter
|
||||
from turnstone.core.auth import JWT_AUD_SERVER, create_jwt
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_manager import SessionManager
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.storage import get_storage
|
||||
from turnstone.core.workstream import WorkstreamKind
|
||||
from turnstone.prompts import ClientType
|
||||
from turnstone.server import WebUI, create_app
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from turnstone.core.workstream import Workstream
|
||||
|
||||
_JWT_SECRET = "sse-recovery-e2e-jwt-secret-minimum-32-chars!"
|
||||
# Small server send buffer so a stalled consumer's in-flight backlog before
|
||||
# the listener-queue poison stays bounded (paired with the client's small
|
||||
# SO_RCVBUF in _sse_recovery_helpers). Harmless for prompt readers.
|
||||
_DEFAULT_SNDBUF = 8192
|
||||
|
||||
|
||||
def _noop_thread(*_args: object, **_kwargs: object) -> None:
|
||||
"""Stand-in for the cluster-lane daemon threads (see module docstring)."""
|
||||
|
||||
|
||||
# The REAL daemon-thread factories, captured once at import so restore always
|
||||
# targets them regardless of how many servers neuter/restore in a run (the
|
||||
# restart scenarios build a second server before the run ends).
|
||||
_REAL_FANOUT = tsrv._global_fanout_thread
|
||||
_REAL_AGGREGATE = tsrv._aggregate_emitter_thread
|
||||
|
||||
|
||||
def _fake_client(scripts: tuple[Any, ...]) -> Any:
|
||||
"""An SDK-shaped fake whose ``chat.completions.create`` follows a
|
||||
positional script (each a :func:`fake_chat_stream` kwargs dict)."""
|
||||
create_fn = scripted_chat_client(*scripts)
|
||||
client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create_fn)))
|
||||
client.calls = create_fn.calls
|
||||
return client
|
||||
|
||||
|
||||
class RecoveryServer:
|
||||
"""A booted interactive node the recovery scenarios drive."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
sndbuf: int = _DEFAULT_SNDBUF,
|
||||
listener_cap: int | None = None,
|
||||
extra_routes: list[Any] | None = None,
|
||||
port: int = 0,
|
||||
) -> None:
|
||||
self._global_queue: _q.Queue[dict[str, Any]] = _q.Queue(maxsize=100000)
|
||||
self._global_listeners: list[_q.Queue[dict[str, Any]]] = []
|
||||
self._global_listeners_lock = threading.Lock()
|
||||
# Per-ws scripted client, resolved at factory-call time.
|
||||
self._pending_client: Any = _fake_client((dict(content="ok", finish_reason="stop"),))
|
||||
self._clients: dict[str, Any] = {}
|
||||
|
||||
WebUI._global_queue = self._global_queue
|
||||
|
||||
def session_factory(
|
||||
ui: Any,
|
||||
model_alias: str | None = None,
|
||||
ws_id: str | None = None,
|
||||
*,
|
||||
skill: Any = None,
|
||||
client_type: str = "",
|
||||
kind: WorkstreamKind = WorkstreamKind.INTERACTIVE,
|
||||
parent_ws_id: str | None = None,
|
||||
project_id: str = "",
|
||||
**_extra: Any,
|
||||
) -> ChatSession:
|
||||
client = self._pending_client
|
||||
if ws_id is not None:
|
||||
self._clients[ws_id] = client
|
||||
return ChatSession(
|
||||
client=client,
|
||||
model="test-model",
|
||||
ui=ui,
|
||||
instructions=None,
|
||||
temperature=None,
|
||||
max_tokens=1024,
|
||||
tool_timeout=30,
|
||||
ws_id=ws_id,
|
||||
user_id="recovery-user",
|
||||
client_type=ClientType.WEB,
|
||||
kind=kind,
|
||||
# Don't truncate large tool outputs: the harness tests
|
||||
# recovery, not the tool-result truncation budget, and a
|
||||
# truncated /history would diverge from the full live event
|
||||
# and defeat the convergence assertions.
|
||||
tool_truncation=10_000_000,
|
||||
)
|
||||
|
||||
self._adapter = InteractiveAdapter(
|
||||
global_queue=self._global_queue,
|
||||
ui_factory=lambda ws: WebUI(
|
||||
ws_id=ws.id, user_id=ws.user_id, kind=ws.kind, parent_ws_id=ws.parent_ws_id
|
||||
),
|
||||
session_factory=session_factory,
|
||||
)
|
||||
self._manager = SessionManager(
|
||||
self._adapter, storage=get_storage(), max_active=32, node_id="recovery-node"
|
||||
)
|
||||
self._adapter.attach(self._manager)
|
||||
WebUI._workstream_mgr = self._manager
|
||||
|
||||
# Neuter the cluster-lane daemons for a clean teardown (see docstring).
|
||||
tsrv._global_fanout_thread = _noop_thread
|
||||
tsrv._aggregate_emitter_thread = _noop_thread
|
||||
|
||||
# Optional small listener-queue cap. The cap is a default arg on the
|
||||
# registration methods with no config/env override, so lower it by
|
||||
# patching their ``__defaults__`` (restored on stop). fix-3's
|
||||
# de-amplification makes a real 500-cap overflow need a pathological
|
||||
# storm; a small cap exercises the identical _ListenerOverflow ->
|
||||
# stream_overflow -> reconnect-replay path within a bounded storm.
|
||||
self._orig_defaults: list[tuple[Any, tuple[Any, ...] | None]] = []
|
||||
if listener_cap is not None:
|
||||
for meth in (
|
||||
SessionUIBase._register_listener,
|
||||
SessionUIBase.register_listener_with_in_progress_snapshot,
|
||||
SessionUIBase.register_listener_with_replay,
|
||||
):
|
||||
self._orig_defaults.append((meth, meth.__defaults__))
|
||||
meth.__defaults__ = (listener_cap,)
|
||||
|
||||
self._app = create_app(
|
||||
workstreams=self._manager,
|
||||
global_queue=self._global_queue,
|
||||
global_listeners=self._global_listeners,
|
||||
global_listeners_lock=self._global_listeners_lock,
|
||||
skip_permissions=True,
|
||||
jwt_secret=_JWT_SECRET,
|
||||
node_id="recovery-node",
|
||||
# /history + tenant checks read app.state.auth_storage.
|
||||
auth_storage=get_storage(),
|
||||
)
|
||||
# Same-origin extras (Tier 2 serves its recovery page here so the real
|
||||
# Pane's cookie auth + EventSource work without cross-origin plumbing).
|
||||
if extra_routes:
|
||||
self._app.router.routes.extend(extra_routes)
|
||||
|
||||
# Pre-bind a listening socket with a small SO_SNDBUF (accepted conns
|
||||
# inherit it), then hand it to uvicorn.
|
||||
self._sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
self._sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
self._sock.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, sndbuf)
|
||||
self._sock.bind(("127.0.0.1", port)) # port=0 -> ephemeral; fixed -> restart reuse
|
||||
self._port = int(self._sock.getsockname()[1])
|
||||
self._sock.listen(128)
|
||||
|
||||
self._server = uvicorn.Server(uvicorn.Config(self._app, log_level="warning", lifespan="on"))
|
||||
self._thread = threading.Thread(
|
||||
target=self._serve, name=f"uvicorn-recovery-{self._port}", daemon=True
|
||||
)
|
||||
self._thread.start()
|
||||
if not _tcp_ready(self._port, 10.0):
|
||||
self.stop()
|
||||
raise AssertionError("recovery server did not accept TCP")
|
||||
|
||||
self._token = create_jwt(
|
||||
user_id="recovery-user",
|
||||
scopes=frozenset({"read", "write", "approve", "service"}),
|
||||
source="recovery",
|
||||
secret=_JWT_SECRET,
|
||||
audience=JWT_AUD_SERVER,
|
||||
)
|
||||
self._http = httpx.Client(base_url=self.base_url, timeout=httpx.Timeout(30.0))
|
||||
|
||||
def _serve(self) -> None:
|
||||
loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(loop)
|
||||
try:
|
||||
loop.run_until_complete(self._server.serve(sockets=[self._sock]))
|
||||
finally:
|
||||
pending = asyncio.all_tasks(loop)
|
||||
for task in pending:
|
||||
task.cancel()
|
||||
if pending:
|
||||
with contextlib.suppress(Exception):
|
||||
loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
|
||||
loop.close()
|
||||
|
||||
# -- properties ----------------------------------------------------------
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
return f"http://127.0.0.1:{self._port}"
|
||||
|
||||
@property
|
||||
def token(self) -> str:
|
||||
return self._token
|
||||
|
||||
@property
|
||||
def manager(self) -> SessionManager:
|
||||
return self._manager
|
||||
|
||||
# -- workstream lifecycle ------------------------------------------------
|
||||
|
||||
def create_workstream(self, *scripts: Any, name: str = "recovery-ws") -> str:
|
||||
"""Create a ws whose scripted LLM follows ``scripts`` (positional
|
||||
:func:`fake_chat_stream` kwargs). Auto-approves tools and suppresses
|
||||
the auto-title call so the positional script stays in sync."""
|
||||
self._pending_client = _fake_client(scripts)
|
||||
ws = self._manager.create(user_id="recovery-user", name=name)
|
||||
self._prime_ws(ws)
|
||||
return ws.id
|
||||
|
||||
def open_workstream(self, ws_id: str, *scripts: Any) -> None:
|
||||
"""Rehydrate a persisted ws on THIS node (the restart path). Fresh
|
||||
UI → empty ring + storage-seeded ``_event_id``."""
|
||||
if scripts:
|
||||
self._pending_client = _fake_client(scripts)
|
||||
ws = self._manager.open(ws_id)
|
||||
if ws is None:
|
||||
raise AssertionError(f"open_workstream: ws {ws_id} not resurrectable")
|
||||
self._prime_ws(ws)
|
||||
|
||||
def _prime_ws(self, ws: Workstream) -> None:
|
||||
if isinstance(ws.ui, SessionUIBase):
|
||||
ws.ui.auto_approve = True # blanket tool auto-approval
|
||||
if ws.session is not None:
|
||||
ws.session._title_generated = True # suppress the auto-title LLM call
|
||||
|
||||
def send(self, ws_id: str, message: str = "go") -> None:
|
||||
"""POST /send — spawns the worker thread and returns immediately."""
|
||||
r = self._http.post(
|
||||
f"/v1/api/workstreams/{ws_id}/send",
|
||||
headers={"Authorization": f"Bearer {self._token}"},
|
||||
json={"message": message},
|
||||
)
|
||||
r.raise_for_status()
|
||||
|
||||
def wait_turn(self, ws_id: str, *, timeout: float = 45.0) -> None:
|
||||
"""Block until the turn's worker thread finishes (the true
|
||||
turn-complete barrier) and the ws is idle."""
|
||||
deadline = time.monotonic() + timeout
|
||||
worker: threading.Thread | None = None
|
||||
while time.monotonic() < deadline:
|
||||
ws = self._manager.get(ws_id)
|
||||
worker = ws.worker_thread if ws is not None else None
|
||||
if worker is not None:
|
||||
break
|
||||
time.sleep(0.02)
|
||||
if worker is not None:
|
||||
worker.join(timeout=max(0.5, deadline - time.monotonic()))
|
||||
if worker.is_alive():
|
||||
raise AssertionError(f"turn worker for {ws_id} did not finish in {timeout}s")
|
||||
|
||||
def get_ws(self, ws_id: str) -> Workstream | None:
|
||||
return self._manager.get(ws_id)
|
||||
|
||||
def ws_state(self, ws_id: str) -> str:
|
||||
ws = self._manager.get(ws_id)
|
||||
return ws.state.value if ws is not None else ""
|
||||
|
||||
def ring_span(self, ws_id: str) -> tuple[int | None, int]:
|
||||
"""(earliest retained ring event_id or None, latest counter) — lets a
|
||||
scenario wait for the ring to evict a specific cursor."""
|
||||
ws = self._manager.get(ws_id)
|
||||
ui = ws.ui if ws is not None else None
|
||||
if not isinstance(ui, SessionUIBase):
|
||||
return None, 0
|
||||
buf = ui._event_buffer
|
||||
earliest = buf[0][0] if buf else None
|
||||
return earliest, ui._event_id
|
||||
|
||||
def listener_poisoned(self, ws_id: str) -> bool:
|
||||
"""True once any live SSE listener on the ws has poisoned (overflow)."""
|
||||
ws = self._manager.get(ws_id)
|
||||
ui = ws.ui if ws is not None else None
|
||||
if not isinstance(ui, SessionUIBase):
|
||||
return False
|
||||
return any(getattr(q, "poisoned", False) for q in list(ui._listeners))
|
||||
|
||||
def max_event_id(self, ws_id: str) -> int | None:
|
||||
"""The storage high-water ``MAX(conversations.event_id)`` — what a
|
||||
restarted node's fresh UI seeds ``_event_id`` from."""
|
||||
result: int | None = get_storage().get_max_event_id(ws_id)
|
||||
return result
|
||||
|
||||
def fetch_history(self, ws_id: str) -> dict[str, Any]:
|
||||
r = self._http.get(
|
||||
f"/v1/api/workstreams/{ws_id}/history",
|
||||
headers={"Authorization": f"Bearer {self._token}"},
|
||||
)
|
||||
r.raise_for_status()
|
||||
result: dict[str, Any] = r.json()
|
||||
return result
|
||||
|
||||
# -- teardown ------------------------------------------------------------
|
||||
|
||||
def stop(self) -> None:
|
||||
with contextlib.suppress(Exception):
|
||||
for ws in list(self._manager.list_all()):
|
||||
with contextlib.suppress(Exception):
|
||||
self._manager.close(ws.id)
|
||||
self._server.should_exit = True
|
||||
self._thread.join(timeout=20)
|
||||
with contextlib.suppress(Exception):
|
||||
self._http.close()
|
||||
with contextlib.suppress(OSError):
|
||||
self._sock.close()
|
||||
# Restore the cluster-lane daemon factories + any patched cap defaults.
|
||||
tsrv._global_fanout_thread = _REAL_FANOUT
|
||||
tsrv._aggregate_emitter_thread = _REAL_AGGREGATE
|
||||
for meth, defaults in self._orig_defaults:
|
||||
meth.__defaults__ = defaults
|
||||
|
||||
|
||||
def _tcp_ready(port: int, timeout: float) -> bool:
|
||||
end = time.monotonic() + timeout
|
||||
while time.monotonic() < end:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
def bash_toolcall_script(
|
||||
call_id: str, command: str, *, finish_reason: str = "tool_calls"
|
||||
) -> dict[str, Any]:
|
||||
"""A scripted assistant turn issuing ONE bash tool call."""
|
||||
return dict(
|
||||
tool_calls=[{"id": call_id, "name": "bash", "arguments": json.dumps({"command": command})}],
|
||||
finish_reason=finish_reason,
|
||||
)
|
||||
|
||||
|
||||
def parallel_bash_script(commands: dict[str, str]) -> dict[str, Any]:
|
||||
"""A scripted assistant turn issuing SEVERAL bash tool calls at once
|
||||
(the parallel-pool storm), ``{call_id: command}``.
|
||||
|
||||
Each command is prefixed with a no-op ``: <call_id>;`` so the tool
|
||||
ARGUMENTS are distinct per call while the OUTPUT is unchanged (``:``
|
||||
ignores its args and prints nothing). Identical-argument parallel
|
||||
calls otherwise trip the session's repeat-tool-call guard, which
|
||||
appends a warning to the PERSISTED result only (not the live event) —
|
||||
an orthogonal divergence that would mask the recovery behavior the
|
||||
convergence assertions test.
|
||||
"""
|
||||
return dict(
|
||||
tool_calls=[
|
||||
{
|
||||
"id": cid,
|
||||
"name": "bash",
|
||||
"arguments": json.dumps({"command": f": {cid}; {cmd}"}),
|
||||
}
|
||||
for cid, cmd in commands.items()
|
||||
],
|
||||
finish_reason="tool_calls",
|
||||
)
|
||||
|
||||
|
||||
def final_text_script(content: str = "done") -> dict[str, Any]:
|
||||
"""The scripted assistant turn that ends the agent loop (no tools)."""
|
||||
return dict(content=content, finish_reason="stop")
|
||||
@@ -4,9 +4,6 @@ import asyncio
|
||||
import contextlib
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
@@ -219,98 +216,6 @@ def _seed_static_state(mgr: MCPClientManager, name: str, **overrides: Any) -> St
|
||||
return state
|
||||
|
||||
|
||||
def _run_on_loop(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 10) -> Any:
|
||||
"""Submit *coro* to *loop*, wait for the result.
|
||||
|
||||
The ONE copy shared by the MCP test files — four hand-synced copies
|
||||
had already drifted on the timeout (5s hardcoded vs a 10s default).
|
||||
The timeout is an upper bound on waiting, not a behavior assertion,
|
||||
so the most generous variant won the merge.
|
||||
"""
|
||||
fut = asyncio.run_coroutine_threadsafe(coro, loop)
|
||||
return fut.result(timeout=timeout)
|
||||
|
||||
|
||||
def _drain_background(mgr: MCPClientManager, loop: asyncio.AbstractEventLoop) -> None:
|
||||
"""Deterministically await ``mgr``'s tracked background tasks.
|
||||
|
||||
Replaces fixed sleeps for synchronizing with scheduled dead-grant
|
||||
drops / spawned refreshes: exact, and immune to slow-runner flake.
|
||||
"""
|
||||
|
||||
async def _drain() -> None:
|
||||
tasks = [t for t in list(mgr._background_tasks) if not t.done()]
|
||||
if tasks:
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
|
||||
_run_on_loop(loop, _drain())
|
||||
|
||||
|
||||
def _poll_until(predicate: Callable[[], bool], timeout: float, interval: float = 0.05) -> bool:
|
||||
"""Poll *predicate* until true or *timeout* elapses — the ONE wait loop.
|
||||
|
||||
Shared by the live MCP smoke tests' condition helpers so the
|
||||
deadline/poll pattern doesn't accrete per-file hand-synced copies.
|
||||
"""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if predicate():
|
||||
return True
|
||||
time.sleep(interval)
|
||||
return False
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
"""Grab an ephemeral localhost port for a live-server subprocess.
|
||||
|
||||
Shared by the live MCP smoke tests (flaky-server, push-refresh) so
|
||||
the socket-probe helpers stay in one place instead of drifting per
|
||||
file.
|
||||
"""
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return int(s.getsockname()[1])
|
||||
|
||||
|
||||
def _tcp_accepts(port: int) -> bool:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def _wait_tcp_ready(port: int, timeout: float) -> bool:
|
||||
"""Poll until something accepts TCP on 127.0.0.1:*port* (live tests)."""
|
||||
return _poll_until(lambda: _tcp_accepts(port), timeout)
|
||||
|
||||
|
||||
def _wait_session_live(mgr: MCPClientManager, name: str, timeout: float) -> bool:
|
||||
"""Poll until static server *name* has a live session (live tests)."""
|
||||
|
||||
def _live() -> bool:
|
||||
state = mgr._static_servers.get(name)
|
||||
return state is not None and state.session is not None
|
||||
|
||||
return _poll_until(_live, timeout)
|
||||
|
||||
|
||||
def _popen_mcp_server(script_path: Any, port: int) -> subprocess.Popen[bytes]:
|
||||
"""Start a FastMCP live-server subprocess, streams to DEVNULL.
|
||||
|
||||
The shared spawn primitive for the live MCP smoke tests
|
||||
(flaky-server flap loop, push-refresh) — the readiness wait and the
|
||||
skip-vs-raise-on-failure policy legitimately differ per test and
|
||||
stay at the call sites. ``sys.executable`` runs the same interpreter,
|
||||
so a server-side import gap surfaces as a failed TCP wait, not here.
|
||||
"""
|
||||
return subprocess.Popen(
|
||||
[sys.executable, str(script_path), str(port)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
|
||||
|
||||
def make_oidc_test_config(**overrides: Any) -> OIDCConfig:
|
||||
"""Build a test ``OIDCConfig`` with sensible defaults.
|
||||
|
||||
@@ -430,40 +335,6 @@ def mock_openai_client():
|
||||
return client
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def make_config_store():
|
||||
"""Factory for a lightweight ConfigStore double.
|
||||
|
||||
``make_config_store(**overrides)`` returns an object whose ``.get(key)``
|
||||
yields the override when present, else the registered SettingDef default —
|
||||
mirroring the real :meth:`ConfigStore.get` fail-open (a bool setting reads
|
||||
as its ``False`` default on a miss, never ``None``). Shared by the
|
||||
``server.require_project`` gate / advisory tests.
|
||||
"""
|
||||
|
||||
_unset = object()
|
||||
|
||||
def _make(**overrides: Any) -> Any:
|
||||
from turnstone.core.settings_registry import SETTINGS
|
||||
|
||||
class _ConfigStoreDouble:
|
||||
def get(self, key: str, default: Any = _unset) -> Any:
|
||||
# Mirror ConfigStore.get precedence exactly: cache (overrides)
|
||||
# first, then a caller-supplied default, then the registry
|
||||
# default, then None — so a reused caller passing an explicit
|
||||
# default for an unset key gets the same value production would.
|
||||
if key in overrides:
|
||||
return overrides[key]
|
||||
if default is not _unset:
|
||||
return default
|
||||
defn = SETTINGS.get(key)
|
||||
return defn.default if defn else None
|
||||
|
||||
return _ConfigStoreDouble()
|
||||
|
||||
return _make
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_policy_cache():
|
||||
"""Drop the in-process tool-policy cache between tests.
|
||||
|
||||
@@ -57,6 +57,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -28,5 +28,6 @@
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b"
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -49,6 +49,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -42,6 +42,7 @@
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -28,5 +28,6 @@
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b"
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -40,6 +40,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -51,6 +51,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,6 +43,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -35,6 +35,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -51,6 +51,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,6 +43,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -39,6 +39,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,10 +43,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -18,8 +18,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -34,10 +34,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -15,8 +15,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -30,10 +30,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -43,10 +43,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -18,8 +18,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -34,10 +34,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -15,8 +15,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -30,10 +30,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -39,6 +39,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true,
|
||||
"tools": [
|
||||
|
||||
@@ -21,6 +21,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true
|
||||
}
|
||||
|
||||
@@ -28,6 +28,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true,
|
||||
"tools": [
|
||||
|
||||
@@ -28,6 +28,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true,
|
||||
"tools": [
|
||||
|
||||
@@ -29,6 +29,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true,
|
||||
"tools": [
|
||||
|
||||
@@ -22,6 +22,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true
|
||||
}
|
||||
|
||||
@@ -28,6 +28,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true,
|
||||
"tools": [
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
"max_output_tokens": 4096,
|
||||
"model": "gpt-5",
|
||||
"prompt_cache_retention": "24h",
|
||||
"reasoning": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"store": false,
|
||||
"stream": true,
|
||||
"tools": [
|
||||
|
||||
+51
-1112
File diff suppressed because it is too large
Load Diff
@@ -128,72 +128,6 @@ def test_stdout_streams_to_ui_from_drain_thread():
|
||||
assert any("streamed-line" in c for c in chunks)
|
||||
|
||||
|
||||
def test_leaked_drain_stops_emitting_chunks_after_return(tmp_path):
|
||||
"""A double-``setsid`` grandchild escapes the session-group kill and
|
||||
holds the stdout pipe open past the drain join (``bash.drain_leaked``)
|
||||
— the leaked drain thread must STOP forwarding chunks to the UI once
|
||||
``_exec_bash`` returns (the ``emit_done`` gate). Without the gate,
|
||||
its lines land in later turns' panes: the UI batches per call_id,
|
||||
some providers reuse call_ids across turns, and the client grafts
|
||||
stray chunks under the completed row.
|
||||
|
||||
The grandchild's writer loop is time-bounded (~8s) so even a failed
|
||||
cleanup cannot outlive the test session, and the finally kills it so
|
||||
the drain thread hits EOF before the leaked-thread guard sweeps.
|
||||
"""
|
||||
pidfile = str(tmp_path / "leak.pid")
|
||||
chunks: list[str] = []
|
||||
|
||||
class RecordingUI(NullUI):
|
||||
def on_tool_output_chunk(self, call_id, chunk):
|
||||
chunks.append(chunk)
|
||||
|
||||
session = make_session(tool_timeout=30, ui=RecordingUI())
|
||||
# setsid detaches the grandchild from the session group (killpg
|
||||
# misses it); it inherits our stdout pipe and keeps writing.
|
||||
command = (
|
||||
f"setsid bash -c 'echo $$ > {pidfile}; "
|
||||
"for i in $(seq 1 80); do echo leak-$i; sleep 0.1; done' & echo fg-done"
|
||||
)
|
||||
leak_pid = None
|
||||
try:
|
||||
finished, result = _run_in_thread(
|
||||
lambda: session._exec_bash({"call_id": "c1", "command": command}),
|
||||
timeout=20,
|
||||
)
|
||||
assert finished, "_exec_bash hung on the escaped grandchild"
|
||||
assert result is not None
|
||||
_call_id, output = result
|
||||
assert "fg-done" in output
|
||||
# The call has RETURNED (gate set). The grandchild is still
|
||||
# writing; give its lines time to traverse the leaked drain.
|
||||
seen_at_return = len(chunks)
|
||||
time.sleep(1.0)
|
||||
# ``<= +1``: the gate documents a one-line residual race (a line
|
||||
# already past the is_set() check when the event sets). A broken
|
||||
# gate keeps forwarding ~10 lines/sec and still fails loudly.
|
||||
assert len(chunks) <= seen_at_return + 1, (
|
||||
"leaked drain thread kept forwarding chunks to the UI after "
|
||||
"the tool returned — the emit_done gate is not holding"
|
||||
)
|
||||
finally:
|
||||
deadline = time.monotonic() + 5
|
||||
while leak_pid is None and time.monotonic() < deadline:
|
||||
try:
|
||||
with open(pidfile) as f:
|
||||
leak_pid = int(f.read().strip())
|
||||
except (FileNotFoundError, ValueError):
|
||||
time.sleep(0.05)
|
||||
if leak_pid is not None:
|
||||
_kill_pid(leak_pid)
|
||||
# Let the drain thread hit EOF before the leaked-thread
|
||||
# guard sweeps the test's thread table.
|
||||
deadline = time.monotonic() + 5
|
||||
while _pid_alive(leak_pid) and time.monotonic() < deadline:
|
||||
time.sleep(0.05)
|
||||
time.sleep(0.2)
|
||||
|
||||
|
||||
def test_cancel_midbash_reports_unknown():
|
||||
"""An external ``cancel()`` during a running bash unblocks the process-bounded
|
||||
wait and reports UNKNOWN (unknown-never-none), not a clean result."""
|
||||
|
||||
@@ -186,41 +186,6 @@ class TestDisplayPath:
|
||||
assert "SUMMARY" not in contents
|
||||
assert contents == ["q", "a"] # true transcript, no injected summary
|
||||
|
||||
def test_include_compaction_projects_marker_as_system_row(self, storage_backend):
|
||||
"""The /history display path (include_compaction=True) surfaces the
|
||||
marker IN PLACE as a first-class system row — source="compaction",
|
||||
meta = the marker's stored fields — so the UI re-renders its
|
||||
compaction card after a reload. Export/search (default False)
|
||||
stay on the drop path pinned above."""
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "q")
|
||||
st.save_message(ws, "assistant", "a")
|
||||
wm = st.get_compaction_watermark(ws, 0)
|
||||
st.save_message(
|
||||
ws,
|
||||
"assistant",
|
||||
"SUMMARY",
|
||||
source="compaction",
|
||||
meta=json.dumps(
|
||||
{"watermark": wm, "before_tokens": 900, "after_tokens": 80, "trigger": "manual"}
|
||||
),
|
||||
)
|
||||
st.save_message(ws, "user", "later question")
|
||||
|
||||
msgs = st.load_messages(ws, include_compaction=True)
|
||||
assert [m.get("content") for m in msgs] == ["q", "a", "SUMMARY", "later question"]
|
||||
marker = msgs[2]
|
||||
assert marker["role"] == "system" # display row, not a fake assistant turn
|
||||
assert marker.get("_source") == "compaction"
|
||||
meta = marker.get("_source_meta")
|
||||
assert meta == {
|
||||
"watermark": wm,
|
||||
"before_tokens": 900,
|
||||
"after_tokens": 80,
|
||||
"trigger": "manual",
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# End-to-end: compaction writes the marker, resume is bounded
|
||||
|
||||
@@ -115,7 +115,7 @@ class TestSummaryTurnProvenance:
|
||||
session._generate_title()
|
||||
|
||||
uc.assert_called_once()
|
||||
prompt = uc.call_args[0][0][-1].text
|
||||
prompt = uc.call_args[0][0][-1]["content"]
|
||||
assert COMPACTION_SUMMARY_LABEL in prompt # titled FROM the real message
|
||||
|
||||
|
||||
|
||||
@@ -257,63 +257,6 @@ def test_searxng_engines_default_empty(tmp_path, monkeypatch):
|
||||
assert config_mod.get_searxng_engines() == ""
|
||||
|
||||
|
||||
def _reset_workspace_cache():
|
||||
config_mod._workspace_dir = None
|
||||
config_mod._workspace_dir_loaded = False
|
||||
|
||||
|
||||
def test_workspace_dir_from_config(tmp_path, monkeypatch):
|
||||
"""get_workspace_dir() reads from config.toml [tools] workspace_dir."""
|
||||
_reset_cache()
|
||||
_reset_workspace_cache()
|
||||
|
||||
cfg = tmp_path / "config.toml"
|
||||
cfg.write_text('[tools]\nworkspace_dir = "/srv/projects"\n')
|
||||
set_config_path(str(cfg))
|
||||
monkeypatch.delenv("TURNSTONE_WORKSPACE", raising=False)
|
||||
|
||||
assert config_mod.get_workspace_dir() == "/srv/projects"
|
||||
|
||||
|
||||
def test_workspace_dir_config_wins_over_env(tmp_path, monkeypatch):
|
||||
"""get_workspace_dir(): config.toml [tools] workspace_dir wins over env."""
|
||||
_reset_cache()
|
||||
_reset_workspace_cache()
|
||||
|
||||
cfg = tmp_path / "config.toml"
|
||||
cfg.write_text('[tools]\nworkspace_dir = "/srv/projects"\n')
|
||||
set_config_path(str(cfg))
|
||||
monkeypatch.setenv("TURNSTONE_WORKSPACE", "/ignored")
|
||||
|
||||
assert config_mod.get_workspace_dir() == "/srv/projects"
|
||||
|
||||
|
||||
def test_workspace_dir_fallback_to_env(tmp_path, monkeypatch):
|
||||
"""get_workspace_dir() falls back to $TURNSTONE_WORKSPACE."""
|
||||
_reset_cache()
|
||||
_reset_workspace_cache()
|
||||
|
||||
cfg = tmp_path / "config.toml"
|
||||
cfg.write_text("[tools]\n")
|
||||
set_config_path(str(cfg))
|
||||
monkeypatch.setenv("TURNSTONE_WORKSPACE", "/workspace")
|
||||
|
||||
assert config_mod.get_workspace_dir() == "/workspace"
|
||||
|
||||
|
||||
def test_workspace_dir_none_when_unset(tmp_path, monkeypatch):
|
||||
"""get_workspace_dir() returns None when neither config nor env is set."""
|
||||
_reset_cache()
|
||||
_reset_workspace_cache()
|
||||
|
||||
cfg = tmp_path / "config.toml"
|
||||
cfg.write_text("[tools]\n")
|
||||
set_config_path(str(cfg))
|
||||
monkeypatch.delenv("TURNSTONE_WORKSPACE", raising=False)
|
||||
|
||||
assert config_mod.get_workspace_dir() is None
|
||||
|
||||
|
||||
def test_apply_config_judge_section(tmp_path):
|
||||
"""apply_config() loads [judge] section and maps to argparse dests."""
|
||||
_reset_cache()
|
||||
|
||||
@@ -217,91 +217,3 @@ def test_whitespace_only_coord_alias_falls_through() -> None:
|
||||
)
|
||||
_invoke(factory)
|
||||
assert registry.captured_alias == "registry-default"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Coordinator MCP gate (#725) — flag × getter matrix, resolved per construction
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _capture_chatsession_kwargs(
|
||||
*,
|
||||
settings: dict[str, Any],
|
||||
mcp_client_getter: Any = None,
|
||||
getter_passed: bool = True,
|
||||
) -> Any:
|
||||
"""Run the factory through to a (patched) ChatSession and return the
|
||||
captured construction kwargs. ChatSession's own contract is covered
|
||||
elsewhere; the unit under test here is the factory's MCP gate."""
|
||||
from unittest.mock import patch
|
||||
|
||||
from tests._coord_test_helpers import _fake_registry
|
||||
|
||||
extra: dict[str, Any] = {}
|
||||
if getter_passed:
|
||||
extra["mcp_client_getter"] = mcp_client_getter
|
||||
factory = build_console_session_factory(
|
||||
registry=_fake_registry(),
|
||||
config_store=_FakeConfigStore(dict(settings)), # type: ignore[arg-type]
|
||||
node_id="console",
|
||||
coord_client_factory=lambda ws_id, uid: MagicMock(),
|
||||
**extra,
|
||||
)
|
||||
ui = MagicMock()
|
||||
ui._user_id = ""
|
||||
with patch("turnstone.console.session_factory.ChatSession") as cs:
|
||||
factory(ui, ws_id="w1")
|
||||
assert cs.call_count == 1
|
||||
return cs.call_args.kwargs
|
||||
|
||||
|
||||
def test_mcp_getter_passes_live_manager_unconditionally() -> None:
|
||||
"""Node parity: the factory passes the live console manager to every
|
||||
coordinator session (the console counterpart of the node factory's
|
||||
mcp_ref[0] read) — whether MCP tools surface is the persona's call,
|
||||
exactly as for interactive sessions."""
|
||||
manager = MagicMock()
|
||||
got = _capture_chatsession_kwargs(settings={}, mcp_client_getter=lambda: manager)
|
||||
assert got["mcp_client"] is manager
|
||||
|
||||
|
||||
def test_mcp_getter_none_manager_passes_none() -> None:
|
||||
"""Nothing configured (create_mcp_client returned None): the session
|
||||
gets None, not a crash."""
|
||||
got = _capture_chatsession_kwargs(settings={}, mcp_client_getter=lambda: None)
|
||||
assert got["mcp_client"] is None
|
||||
|
||||
|
||||
def test_mcp_no_getter_is_backward_compatible() -> None:
|
||||
got = _capture_chatsession_kwargs(settings={}, getter_passed=False)
|
||||
assert got["mcp_client"] is None
|
||||
|
||||
|
||||
def test_mcp_getter_resolved_per_construction() -> None:
|
||||
"""The getter is consulted at EVERY construction — a manager
|
||||
(re)constructed by the console ensure-helper after factory build must
|
||||
reach the next session. An instance captured at factory-build time
|
||||
fails this row."""
|
||||
from unittest.mock import patch
|
||||
|
||||
from tests._coord_test_helpers import _fake_registry
|
||||
|
||||
holder: dict[str, Any] = {"mgr": None}
|
||||
factory = build_console_session_factory(
|
||||
registry=_fake_registry(),
|
||||
config_store=_FakeConfigStore({}), # type: ignore[arg-type]
|
||||
node_id="console",
|
||||
coord_client_factory=lambda ws_id, uid: MagicMock(),
|
||||
mcp_client_getter=lambda: holder["mgr"],
|
||||
)
|
||||
ui = MagicMock()
|
||||
ui._user_id = ""
|
||||
with patch("turnstone.console.session_factory.ChatSession") as cs:
|
||||
factory(ui, ws_id="w1")
|
||||
first = cs.call_args.kwargs["mcp_client"]
|
||||
manager = MagicMock()
|
||||
holder["mgr"] = manager # the ensure-helper lazily constructed it
|
||||
factory(ui, ws_id="w2")
|
||||
second = cs.call_args.kwargs["mcp_client"]
|
||||
assert first is None
|
||||
assert second is manager
|
||||
|
||||
@@ -168,17 +168,3 @@ def test_unbounded_render_inputs_are_capped() -> None:
|
||||
assert "more preview lines not shown" in body
|
||||
assert "RAW_CAP" in body
|
||||
assert "truncated for display" in body
|
||||
|
||||
|
||||
def test_retry_note_validates_backoff_before_rendering() -> None:
|
||||
"""retry_in is a server-emitted backoff coerced with Number(); like the
|
||||
part/total pair just below it, it must be finiteness-validated (and
|
||||
non-negative) so a malformed value can't render "retrying in NaNs". The
|
||||
error text is kept regardless — it is the load-bearing half of the note."""
|
||||
body = _body()
|
||||
assert "Number.isFinite(secs) && secs >= 0" in body, (
|
||||
"retry_in must be validated (finite, non-negative) before its seconds render"
|
||||
)
|
||||
assert "Math.round(Number(evt.retry_in))" not in body, (
|
||||
"retry_in must not be Math.round(Number(...))'d without a finiteness guard"
|
||||
)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -476,7 +476,7 @@ def test_coord_spawn_metrics_increments_messages_and_resets_tool_count() -> None
|
||||
ui = ConsoleCoordinatorUI(ws_id="coord-ws", user_id="u1")
|
||||
ui._ws_messages = 5
|
||||
ui._ws_turn_tool_calls = 3
|
||||
_coord_spawn_metrics(ui)
|
||||
_coord_spawn_metrics(MagicMock(), ui)
|
||||
assert ui._ws_messages == 6
|
||||
assert ui._ws_turn_tool_calls == 0
|
||||
|
||||
@@ -489,7 +489,7 @@ def test_coord_spawn_metrics_tolerates_ui_without_counters() -> None:
|
||||
class _StubUI:
|
||||
pass
|
||||
|
||||
_coord_spawn_metrics(_StubUI()) # must not raise
|
||||
_coord_spawn_metrics(MagicMock(), _StubUI()) # must not raise
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -495,20 +495,11 @@ def test_coordinator_js_seeds_resume_cursor_only_on_initial_connect():
|
||||
Pins three invariants mirroring the app.js guards:
|
||||
1. ``refetchHistory`` takes a ``seedCursor`` flag (default false) and
|
||||
seeds ``lastEventId`` from ``hist.cursor`` only when set + non-null,
|
||||
so the clear_ui re-render caller (live stream, no reconnect — the
|
||||
one remaining seedless caller after the #882 dead-stream ruling)
|
||||
doesn't rewind the live cursor.
|
||||
2. every caller that reconnects opts in: init via
|
||||
``await refetchHistory(true)``, and the truncated resync via
|
||||
``loadHistoryThenReconnect``'s ``refetchHistory(true).finally``
|
||||
(cursor adoption is the #882 fix core — /history trims the
|
||||
in-flight turn whenever it returns a cursor).
|
||||
3. ``connectSSE`` gates ``?last_event_id=`` on ``connectCursor !=
|
||||
null`` so a cursor of 0 (a brand-new ws's first-turn boundary)
|
||||
isn't dropped — where ``connectCursor`` presents a recorded
|
||||
truncation gap over the advanced live cursor (the gap-repair
|
||||
chokepoint; see the truncated fresh-connect test in
|
||||
test_app_js.py).
|
||||
so the clear_ui / replay_truncated re-render callers (live stream,
|
||||
no reconnect) don't rewind the live cursor.
|
||||
2. the initial-connect path opts in via ``refetchHistory(true)``.
|
||||
3. ``connectSSE`` gates ``?last_event_id=`` on ``!= null`` so a cursor
|
||||
of 0 (a brand-new ws's first-turn boundary) isn't dropped.
|
||||
"""
|
||||
import re
|
||||
from pathlib import Path
|
||||
@@ -519,7 +510,7 @@ def test_coordinator_js_seeds_resume_cursor_only_on_initial_connect():
|
||||
body = coord_js.read_text(encoding="utf-8")
|
||||
assert "async function refetchHistory(seedCursor = false)" in body, (
|
||||
"refetchHistory must take a seedCursor flag (default false) so only "
|
||||
"the reconnecting callers seed the resume cursor."
|
||||
"the initial-connect caller seeds the resume cursor."
|
||||
)
|
||||
assert re.search(
|
||||
r"if\s*\(\s*seedCursor\s*&&\s*hist\.cursor\s*!=\s*null\s*\)\s*"
|
||||
@@ -529,44 +520,10 @@ def test_coordinator_js_seeds_resume_cursor_only_on_initial_connect():
|
||||
assert "await refetchHistory(true)" in body, (
|
||||
"the initial-connect path must call refetchHistory(true) to seed the cursor."
|
||||
)
|
||||
flow = re.search(r"function loadHistoryThenReconnect\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert flow is not None, "loadHistoryThenReconnect not found"
|
||||
assert "refetchHistory(true)" in flow.group(1) and ".finally(" in flow.group(1), (
|
||||
"the truncated resync (loadHistoryThenReconnect) must seed via "
|
||||
"refetchHistory(true) and reconnect in .finally."
|
||||
)
|
||||
assert re.search(
|
||||
r"if\s*\(\s*connectCursor\s*!=\s*null\s*\)\s*\{\s*url\s*\+=\s*\"\?last_event_id=\"",
|
||||
r"if\s*\(\s*lastEventId\s*!=\s*null\s*\)\s*\{\s*url\s*\+=\s*\"\?last_event_id=\"",
|
||||
body,
|
||||
), "connectSSE must gate ?last_event_id= on connectCursor != null (so cursor 0 isn't dropped)."
|
||||
|
||||
|
||||
def test_coordinator_refetch_failure_preserves_the_pane():
|
||||
"""A FAILED /history fetch must leave the message column and the
|
||||
tool-row/batch tracking untouched (#882 G3): refetchHistory's wipe +
|
||||
resets must sit AFTER the ``if (!hist) return`` guard. Pre-fix the
|
||||
wipe ran first, so a failed fetch — likeliest exactly during the
|
||||
restart windows that trigger truncated resyncs — left an EMPTY pane
|
||||
on a live stream with no retry record. Stale-but-real beats blank,
|
||||
and the maps stay valid against the untouched DOM."""
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
coord_js = Path(__file__).resolve().parent.parent / (
|
||||
"turnstone/console/static/coordinator/coordinator.js"
|
||||
)
|
||||
body = coord_js.read_text(encoding="utf-8")
|
||||
start = body.index("async function refetchHistory(seedCursor = false)")
|
||||
fn = body[start : start + 3000]
|
||||
guard = fn.index("if (!hist) return;")
|
||||
wipe = fn.index("messagesEl.replaceChildren();")
|
||||
resets = fn.index("toolRows.clear();")
|
||||
assert guard < wipe and guard < resets, (
|
||||
"refetchHistory must bail on a failed fetch BEFORE wiping the "
|
||||
"pane or clearing the tool-row maps — a failed /history must be "
|
||||
"a DOM no-op."
|
||||
)
|
||||
assert re.search(r"if \(!hist\) return;", fn), "failure guard missing"
|
||||
), "connectSSE must gate ?last_event_id= on lastEventId != null (so cursor 0 isn't dropped)."
|
||||
|
||||
|
||||
def test_coordinator_js_early_paints_pending_tool_calls():
|
||||
@@ -730,14 +687,7 @@ def test_coordinator_js_gates_send_on_cross_user_busy():
|
||||
assert "actingUserId !== me" in coord_js
|
||||
assert "composer.setSendBlocked(" in coord_js
|
||||
assert "function reconcileSendBlock()" in coord_js
|
||||
# reactive 409 fallback — the pane converts the 409 body at the fetch
|
||||
# stage; the status ARM itself lives in the shared settle helper
|
||||
# (composer_queue.settleSendResponse) with the rest of the response
|
||||
# matrix, one implementation for both panes.
|
||||
# reactive 409 fallback
|
||||
assert "r.status === 409" in coord_js
|
||||
assert 'status: "cross_user_interjection"' in coord_js
|
||||
assert "settleSendResponse(" in coord_js
|
||||
helper = (
|
||||
Path(__file__).resolve().parents[1] / "turnstone/shared_static/composer_queue.js"
|
||||
).read_text(encoding="utf-8")
|
||||
assert 'status === "cross_user_interjection"' in helper
|
||||
assert 'data.status === "cross_user_interjection"' in coord_js
|
||||
|
||||
@@ -162,15 +162,6 @@ def test_coordinator_session_uses_coordinator_tools(coord_session):
|
||||
# surfacing to a human channel without spawning a child purely
|
||||
# to ship the message. Routing is session-kind-agnostic.
|
||||
"notify",
|
||||
# ``read_resource``/``use_prompt`` joined in 1.8 (#725) — the
|
||||
# coordinator MCP surface covers tools, resources, AND prompts.
|
||||
# They sit in the BASE list unconditionally (like their
|
||||
# interactive siblings), but the WIRE strips them when no client
|
||||
# is attached or the per-user catalog counts are zero — the
|
||||
# _without_tool gate in _get_active_tools, pinned by the wire
|
||||
# matrix in test_workstream_kind.py.
|
||||
"read_resource",
|
||||
"use_prompt",
|
||||
}
|
||||
# Sub-agent tool set is zeroed on coordinator sessions.
|
||||
assert sess._task_tools == []
|
||||
|
||||
@@ -1,231 +0,0 @@
|
||||
"""Working-directory/workspace notes rendered into fs-tool descriptions.
|
||||
|
||||
Covers the pure renderer (``apply_cwd_context``), the metadata invariants the
|
||||
session wiring relies on, and the ChatSession build sites (construction, MCP
|
||||
rebuild) including the guarded ``os.getcwd()`` read and the task-agent lane.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.tools import (
|
||||
_META,
|
||||
COORDINATOR_TOOLS,
|
||||
INTERACTIVE_TOOLS,
|
||||
TASK_AGENT_TOOLS,
|
||||
TOOLS,
|
||||
apply_cwd_context,
|
||||
)
|
||||
|
||||
_FS_TOOLS = ("bash", "read_file", "write_file", "edit_file", "search", "diff_file")
|
||||
|
||||
|
||||
def _desc(tools: list[dict], name: str) -> str:
|
||||
for t in tools:
|
||||
if t["function"]["name"] == name:
|
||||
return t["function"]["description"]
|
||||
raise AssertionError(f"tool {name!r} not in list")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# apply_cwd_context (pure renderer)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestApplyCwdContext:
|
||||
def test_notes_rendered_on_fs_tools(self):
|
||||
out = apply_cwd_context(INTERACTIVE_TOOLS, "/data", "/workspace")
|
||||
assert "Commands run in /data" in _desc(out, "bash")
|
||||
assert "cd does not persist" in _desc(out, "bash")
|
||||
assert "The user's workspace directory is /workspace." in _desc(out, "bash")
|
||||
for name in ("read_file", "write_file", "edit_file", "search", "diff_file"):
|
||||
assert "Relative paths resolve against /data." in _desc(out, name)
|
||||
assert "The user's workspace directory is /workspace." in _desc(out, name)
|
||||
|
||||
def test_noteless_tools_pass_through_by_reference(self):
|
||||
out = apply_cwd_context(INTERACTIVE_TOOLS, "/data", "/workspace")
|
||||
by_name = {t["function"]["name"]: t for t in out}
|
||||
base_by_name = {t["function"]["name"]: t for t in INTERACTIVE_TOOLS}
|
||||
assert by_name["web_search"] is base_by_name["web_search"]
|
||||
# Noted tools are fresh copies.
|
||||
assert by_name["bash"] is not base_by_name["bash"]
|
||||
|
||||
def test_module_constants_never_mutated(self):
|
||||
# The fs tool dicts are SHARED across TOOLS/INTERACTIVE_TOOLS/
|
||||
# TASK_AGENT_TOOLS and aliased through merge_mcp_tools output — an
|
||||
# in-place append would corrupt every list at once.
|
||||
before = {name: _desc(TOOLS, name) for name in _FS_TOOLS}
|
||||
apply_cwd_context(INTERACTIVE_TOOLS, "/data", "/workspace")
|
||||
apply_cwd_context(TASK_AGENT_TOOLS, "/data", "/workspace")
|
||||
for name in _FS_TOOLS:
|
||||
assert _desc(TOOLS, name) == before[name]
|
||||
assert "/data" not in _desc(INTERACTIVE_TOOLS, name)
|
||||
|
||||
def test_empty_working_dir_drops_cwd_note_only(self):
|
||||
out = apply_cwd_context(INTERACTIVE_TOOLS, "", "/workspace")
|
||||
assert "Commands run in" not in _desc(out, "bash")
|
||||
assert "Relative paths resolve" not in _desc(out, "read_file")
|
||||
assert "The user's workspace directory is /workspace." in _desc(out, "bash")
|
||||
|
||||
def test_empty_workspace_drops_workspace_note_only(self):
|
||||
out = apply_cwd_context(INTERACTIVE_TOOLS, "/data", "")
|
||||
assert "Commands run in /data" in _desc(out, "bash")
|
||||
assert "workspace directory" not in _desc(out, "bash")
|
||||
|
||||
def test_both_empty_is_pass_through(self):
|
||||
out = apply_cwd_context(INTERACTIVE_TOOLS, "", "")
|
||||
assert out == INTERACTIVE_TOOLS
|
||||
assert out is not INTERACTIVE_TOOLS # still a fresh list
|
||||
|
||||
def test_mcp_style_tool_untouched(self):
|
||||
mcp_tool = {
|
||||
"type": "function",
|
||||
"function": {"name": "mcp__srv__thing", "description": "Does a thing."},
|
||||
}
|
||||
out = apply_cwd_context([mcp_tool], "/data", "/workspace")
|
||||
assert out[0] is mcp_tool
|
||||
assert out[0]["function"]["description"] == "Does a thing."
|
||||
|
||||
def test_paths_with_braces_are_literal(self):
|
||||
# str.replace substitution — a path containing brace characters must
|
||||
# land verbatim (str.format would raise or mangle here).
|
||||
out = apply_cwd_context(INTERACTIVE_TOOLS, "/data/{odd}", "")
|
||||
assert "Commands run in /data/{odd}" in _desc(out, "bash")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Metadata invariants the session wiring relies on
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestNoteMetadataInvariants:
|
||||
def test_all_fs_tools_declare_both_notes(self):
|
||||
for name in _FS_TOOLS:
|
||||
assert _META[name].get("cwd_note"), name
|
||||
assert _META[name].get("workspace_note"), name
|
||||
|
||||
def test_no_coordinator_tool_declares_notes(self):
|
||||
# The coordinator build sites skip _apply_cwd_notes on the strength
|
||||
# of this invariant.
|
||||
for t in COORDINATOR_TOOLS:
|
||||
name = t["function"]["name"]
|
||||
meta = _META.get(name) or {}
|
||||
assert not meta.get("cwd_note"), name
|
||||
assert not meta.get("workspace_note"), name
|
||||
|
||||
def test_notes_stripped_from_wire_schema(self):
|
||||
# _META_KEYS extraction: the raw JSON keys must not leak into the
|
||||
# OpenAI function dict sent to providers.
|
||||
for t in TOOLS:
|
||||
assert "cwd_note" not in t["function"]
|
||||
assert "workspace_note" not in t["function"]
|
||||
|
||||
def test_workspace_note_wording_uniform(self):
|
||||
# The workspace fact is one node-level value, so its sentence is
|
||||
# deliberately identical across the fs tools (unlike cwd_note, whose
|
||||
# prose is per-tool). Guards a one-file reword from drifting the
|
||||
# copies apart, independent of the exact wording.
|
||||
notes = {_META[name]["workspace_note"] for name in _FS_TOOLS}
|
||||
assert len(notes) == 1, notes
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ChatSession build sites
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _make_session(**kwargs):
|
||||
defaults = dict(
|
||||
client=MagicMock(),
|
||||
model="test-model",
|
||||
ui=MagicMock(),
|
||||
instructions=None,
|
||||
temperature=0.5,
|
||||
max_tokens=1024,
|
||||
tool_timeout=10,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return ChatSession(**defaults)
|
||||
|
||||
|
||||
class TestSessionCwdNotes:
|
||||
def test_fresh_session_carries_cwd_note(self, tmp_db):
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=None):
|
||||
session = _make_session()
|
||||
assert f"Commands run in {os.getcwd()}" in _desc(session._tools, "bash")
|
||||
assert f"Relative paths resolve against {os.getcwd()}." in _desc(
|
||||
session._tools, "read_file"
|
||||
)
|
||||
|
||||
def test_task_lane_carries_cwd_note(self, tmp_db):
|
||||
# Sub-agents use self._task_tools, a separate list from self._tools —
|
||||
# they run in this same process, so the same cwd applies.
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=None):
|
||||
session = _make_session()
|
||||
assert f"Commands run in {os.getcwd()}" in _desc(session._task_tools, "bash")
|
||||
|
||||
def test_getcwd_failure_degrades_to_noteless(self, tmp_db):
|
||||
# A deleted cwd (eval workdir teardown) must not break session
|
||||
# construction or an MCP background-thread rebuild.
|
||||
with (
|
||||
patch("turnstone.core.session.get_workspace_dir", return_value=None),
|
||||
patch("os.getcwd", side_effect=OSError("cwd deleted")),
|
||||
):
|
||||
session = _make_session()
|
||||
assert "Commands run in" not in _desc(session._tools, "bash")
|
||||
assert session._tools # built fine, just note-less
|
||||
|
||||
def test_workspace_rendered_when_dir_exists(self, tmp_db, tmp_path):
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=str(tmp_path)):
|
||||
session = _make_session()
|
||||
assert f"The user's workspace directory is {tmp_path}." in _desc(session._tools, "bash")
|
||||
|
||||
def test_workspace_skipped_when_dir_missing(self, tmp_db, tmp_path):
|
||||
missing = tmp_path / "does-not-exist"
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=str(missing)):
|
||||
session = _make_session()
|
||||
assert "workspace directory" not in _desc(session._tools, "bash")
|
||||
|
||||
def test_workspace_skipped_when_equal_to_cwd(self, tmp_db):
|
||||
# e.g. an operator who set working_dir: /workspace on the container —
|
||||
# one fact, not two copies of the same path.
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=os.getcwd()):
|
||||
session = _make_session()
|
||||
desc = _desc(session._tools, "bash")
|
||||
assert f"Commands run in {os.getcwd()}" in desc
|
||||
assert "workspace directory" not in desc
|
||||
|
||||
def test_constants_pristine_after_session_build(self, tmp_db):
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=None):
|
||||
_make_session()
|
||||
assert os.getcwd() not in _desc(INTERACTIVE_TOOLS, "bash")
|
||||
assert os.getcwd() not in _desc(TOOLS, "bash")
|
||||
|
||||
def test_mcp_rebuild_keeps_single_note(self, tmp_db):
|
||||
# The MCP list_changed rebuild re-derives from pristine bases — the
|
||||
# note must survive exactly once (double-append is the failure the
|
||||
# assignment-time design must never regress into).
|
||||
mock_mcp = MagicMock()
|
||||
mock_mcp.get_tools.return_value = []
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=None):
|
||||
session = _make_session(mcp_client=mock_mcp)
|
||||
session._on_mcp_tools_changed()
|
||||
session._on_mcp_tools_changed()
|
||||
assert _desc(session._tools, "bash").count(f"Commands run in {os.getcwd()}") == 1
|
||||
assert _desc(session._task_tools, "bash").count(f"Commands run in {os.getcwd()}") == 1
|
||||
|
||||
def test_mcp_drop_surface_keeps_single_note(self, tmp_db):
|
||||
# The MCP-disconnect rebuild (_drop_mcp_surface, reached via resume()
|
||||
# adopting an MCP-off persona) is the third rebuild trigger — the note
|
||||
# must survive it, exactly once, on both lanes.
|
||||
mock_mcp = MagicMock()
|
||||
mock_mcp.get_tools.return_value = []
|
||||
with patch("turnstone.core.session.get_workspace_dir", return_value=None):
|
||||
session = _make_session(mcp_client=mock_mcp)
|
||||
session._drop_mcp_surface()
|
||||
assert session._mcp_client is None
|
||||
assert _desc(session._tools, "bash").count(f"Commands run in {os.getcwd()}") == 1
|
||||
assert _desc(session._task_tools, "bash").count(f"Commands run in {os.getcwd()}") == 1
|
||||
@@ -64,75 +64,3 @@ def test_cancel_returns_promptly() -> None:
|
||||
assert time.monotonic() - start < 1.0
|
||||
stragglers = [t for t in threading.enumerate() if t.name == "dl-cancel" and not t.daemon]
|
||||
assert stragglers == [], f"non-daemon worker survived: {stragglers}"
|
||||
|
||||
|
||||
def test_on_abandon_fires_on_timeout_and_cancel_but_not_success() -> None:
|
||||
calls: list[str] = []
|
||||
|
||||
with pytest.raises(DeadlineExceededError):
|
||||
run_with_deadline(
|
||||
lambda: time.sleep(2.0),
|
||||
timeout=0.1,
|
||||
poll=0.05,
|
||||
thread_name="dl-abandon-t",
|
||||
on_abandon=lambda: calls.append("timeout"),
|
||||
)
|
||||
assert calls == ["timeout"]
|
||||
|
||||
cancel = threading.Event()
|
||||
cancel.set()
|
||||
with pytest.raises(DeadlineCancelledError):
|
||||
run_with_deadline(
|
||||
lambda: time.sleep(2.0),
|
||||
timeout=10.0,
|
||||
cancel_event=cancel,
|
||||
poll=0.05,
|
||||
thread_name="dl-abandon-c",
|
||||
on_abandon=lambda: calls.append("cancel"),
|
||||
)
|
||||
assert calls == ["timeout", "cancel"]
|
||||
|
||||
result = run_with_deadline(lambda: 7, timeout=1.0, on_abandon=lambda: calls.append("no"))
|
||||
assert result == 7
|
||||
assert calls == ["timeout", "cancel"]
|
||||
|
||||
|
||||
def test_on_abandon_errors_do_not_mask_the_deadline_error() -> None:
|
||||
def _boom() -> None:
|
||||
raise RuntimeError("abort hook broke")
|
||||
|
||||
with pytest.raises(DeadlineExceededError):
|
||||
run_with_deadline(
|
||||
lambda: time.sleep(2.0),
|
||||
timeout=0.1,
|
||||
poll=0.05,
|
||||
thread_name="dl-abandon-e",
|
||||
on_abandon=_boom,
|
||||
)
|
||||
|
||||
|
||||
class TestStreamAbortRef:
|
||||
def test_abort_closes_captured_stream(self) -> None:
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from turnstone.core.deadline import StreamAbortRef
|
||||
|
||||
ref = StreamAbortRef()
|
||||
stream = MagicMock()
|
||||
ref.append(stream)
|
||||
stream.close.assert_not_called()
|
||||
ref.abort()
|
||||
stream.close.assert_called_once()
|
||||
|
||||
def test_late_arriving_stream_closes_on_append(self) -> None:
|
||||
# The arrival race: abort fires while the worker is still inside the
|
||||
# SDK connect — the handle must close the moment it is captured.
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from turnstone.core.deadline import StreamAbortRef
|
||||
|
||||
ref = StreamAbortRef()
|
||||
ref.abort()
|
||||
stream = MagicMock()
|
||||
ref.append(stream)
|
||||
stream.close.assert_called_once()
|
||||
|
||||
@@ -1,297 +0,0 @@
|
||||
"""Unit tests for ``drain_stream`` — the #831 single non-streaming transport.
|
||||
|
||||
Every single-shot lane consumes ``create_streaming`` through this
|
||||
accumulator, so its semantics ARE the old ``create_completion`` contract:
|
||||
each case here pins a rule the per-adapter non-streaming methods used to
|
||||
implement independently (usage max-merge, tool-delta assembly, terminal
|
||||
provider_blocks, trailing-citation fold).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.providers import (
|
||||
StreamChunk,
|
||||
ToolCallDelta,
|
||||
UsageInfo,
|
||||
drain_stream,
|
||||
)
|
||||
from turnstone.core.providers._openai_common import RETRYABLE_ERROR_NAMES
|
||||
from turnstone.core.providers._protocol import IncompleteStreamError
|
||||
|
||||
|
||||
class TestContentAndReasoning:
|
||||
def test_joins_content_deltas_in_order(self):
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(content_delta="Hello, "),
|
||||
StreamChunk(content_delta="world"),
|
||||
StreamChunk(finish_reason="stop"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.content == "Hello, world"
|
||||
assert result.finish_reason == "stop"
|
||||
|
||||
def test_joins_reasoning_deltas_separately_from_content(self):
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(reasoning_delta="think "),
|
||||
StreamChunk(reasoning_delta="hard"),
|
||||
StreamChunk(content_delta="answer"),
|
||||
StreamChunk(finish_reason="stop"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.reasoning == "think hard"
|
||||
assert result.content == "answer"
|
||||
|
||||
def test_stream_without_finish_reason_raises_incomplete(self):
|
||||
# Complete-or-error: every adapter emits a finish reason on a
|
||||
# healthy stream, so its absence means the generation died
|
||||
# mid-response — partial text must never be stored as a complete
|
||||
# result (compaction summary, title). Typed and retryable.
|
||||
assert "IncompleteStreamError" in RETRYABLE_ERROR_NAMES
|
||||
with pytest.raises(IncompleteStreamError):
|
||||
drain_stream(iter([StreamChunk(content_delta="half a summar")]))
|
||||
|
||||
def test_empty_stream_raises_incomplete(self):
|
||||
with pytest.raises(IncompleteStreamError):
|
||||
drain_stream(iter([]))
|
||||
|
||||
|
||||
class TestToolCallAssembly:
|
||||
def test_merges_deltas_by_index_id_name_once_args_concat(self):
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(
|
||||
tool_call_deltas=[ToolCallDelta(index=0, id="call_1", name="read_file")]
|
||||
),
|
||||
StreamChunk(
|
||||
tool_call_deltas=[ToolCallDelta(index=0, arguments_delta='{"path": ')]
|
||||
),
|
||||
StreamChunk(
|
||||
tool_call_deltas=[ToolCallDelta(index=0, arguments_delta='"x.py"}')]
|
||||
),
|
||||
StreamChunk(finish_reason="tool_calls"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.tool_calls == [
|
||||
{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "arguments": '{"path": "x.py"}'},
|
||||
}
|
||||
]
|
||||
|
||||
def test_parallel_calls_ordered_by_index(self):
|
||||
# Interleaved argument deltas for two calls must not cross-contaminate,
|
||||
# and the assembled list is index-ordered regardless of arrival order.
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(tool_call_deltas=[ToolCallDelta(index=1, id="b", name="beta")]),
|
||||
StreamChunk(tool_call_deltas=[ToolCallDelta(index=0, id="a", name="alpha")]),
|
||||
StreamChunk(
|
||||
tool_call_deltas=[
|
||||
ToolCallDelta(index=0, arguments_delta="{}"),
|
||||
ToolCallDelta(index=1, arguments_delta='{"k": 1}'),
|
||||
]
|
||||
),
|
||||
StreamChunk(finish_reason="tool_calls"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert [tc["id"] for tc in result.tool_calls] == ["a", "b"]
|
||||
assert result.tool_calls[1]["function"]["arguments"] == '{"k": 1}'
|
||||
|
||||
def test_blank_id_preserved_for_downstream_repair(self):
|
||||
# Google compat can stream blank tool ids — the drain must hand them
|
||||
# through untouched so model_turn's pairwise blank-id repair sees them.
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(tool_call_deltas=[ToolCallDelta(index=0, name="f")]),
|
||||
StreamChunk(finish_reason="tool_calls"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.tool_calls[0]["id"] == ""
|
||||
|
||||
# Index-degenerate parallel-call de-fusion lives in the CHAT ADAPTER's
|
||||
# iterator (so the interactive loop is fixed too) — pinned in
|
||||
# test_providers.py::TestOpenAIProvider::
|
||||
# test_streaming_remaps_index_degenerate_parallel_calls. The drain
|
||||
# accumulates by index verbatim; adapters own index sanity.
|
||||
|
||||
|
||||
class TestUsageMerge:
|
||||
def test_anthropic_split_emission_max_merges(self):
|
||||
# message_start carries prompt tokens (completion 0); message_delta
|
||||
# carries completion tokens (prompt possibly absent → 0). Neither
|
||||
# first-wins nor last-wins sees both — the max-merge does.
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(
|
||||
usage=UsageInfo(
|
||||
prompt_tokens=120,
|
||||
completion_tokens=0,
|
||||
total_tokens=120,
|
||||
cache_read_tokens=100,
|
||||
)
|
||||
),
|
||||
StreamChunk(content_delta="hi"),
|
||||
StreamChunk(
|
||||
usage=UsageInfo(prompt_tokens=0, completion_tokens=42, total_tokens=42),
|
||||
finish_reason="stop",
|
||||
),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.usage.prompt_tokens == 120
|
||||
assert result.usage.completion_tokens == 42
|
||||
assert result.usage.total_tokens == 162
|
||||
assert result.usage.cache_read_tokens == 100
|
||||
|
||||
def test_single_terminal_usage_passes_through(self):
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(content_delta="x"),
|
||||
StreamChunk(finish_reason="stop"),
|
||||
StreamChunk(
|
||||
usage=UsageInfo(prompt_tokens=10, completion_tokens=5, total_tokens=15)
|
||||
),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.usage.total_tokens == 15
|
||||
|
||||
|
||||
class TestFinishAndBlocks:
|
||||
def test_finish_reason_last_non_none_wins(self):
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(finish_reason="tool_calls"),
|
||||
StreamChunk(content_delta="tail"),
|
||||
StreamChunk(finish_reason="stop"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.finish_reason == "stop"
|
||||
|
||||
def test_provider_blocks_taken_from_terminal_emission(self):
|
||||
# Every adapter attaches its full block list exactly once (on or
|
||||
# after the terminal chunk); replace-on-nonempty keeps the last set.
|
||||
blocks = [{"type": "thinking", "thinking": "t", "signature": "s"}]
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(content_delta="a"),
|
||||
StreamChunk(finish_reason="stop", provider_blocks=blocks),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.provider_blocks == blocks
|
||||
|
||||
|
||||
class TestInfoDelta:
|
||||
def test_mid_stream_status_pings_dropped(self):
|
||||
# "[Searching…]" style transient status — the non-streaming lane
|
||||
# never surfaced these, so the drain must not leak them into content.
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(info_delta="[Searching: quakes]"),
|
||||
StreamChunk(content_delta="answer"),
|
||||
StreamChunk(finish_reason="stop"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.content == "answer"
|
||||
|
||||
def test_trailing_citations_fold_matches_format_citations(self):
|
||||
# The chat/responses adapters emit format_citations("", anns).strip()
|
||||
# as a final info chunk after the finish reason. Folding it back as
|
||||
# content + "\n\n" + info must byte-match the old non-streaming
|
||||
# format_citations(content, anns) append.
|
||||
from turnstone.core.providers._openai_common import format_citations
|
||||
|
||||
class _Ann:
|
||||
type = "url_citation"
|
||||
url = "https://example.com"
|
||||
title = "Example"
|
||||
url_citation = None
|
||||
|
||||
anns = [_Ann()]
|
||||
trailing = format_citations("", anns).strip()
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(content_delta="body"),
|
||||
StreamChunk(finish_reason="stop"),
|
||||
StreamChunk(info_delta=trailing),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.content == format_citations("body", anns)
|
||||
|
||||
def test_trailing_fold_with_empty_content_matches_too(self):
|
||||
result = drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(finish_reason="tool_calls"),
|
||||
StreamChunk(info_delta="Sources:\n- x"),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert result.content == "\n\nSources:\n- x"
|
||||
|
||||
def test_finishless_stream_raises_even_with_trailing_info(self):
|
||||
# A stream that dies after a status ping must NOT return the ping
|
||||
# as content (nor the partial body as a clean result) — the
|
||||
# complete-or-error gate turns the whole stream into a retryable
|
||||
# error instead of guessing which trailing info was a citation.
|
||||
with pytest.raises(IncompleteStreamError):
|
||||
drain_stream(
|
||||
iter(
|
||||
[
|
||||
StreamChunk(content_delta="body"),
|
||||
StreamChunk(info_delta="[Searching: kubernetes CVEs]"),
|
||||
]
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
class TestErrorPropagation:
|
||||
def test_httpx_transport_error_becomes_retryable_incomplete(self):
|
||||
# Streaming moves the body read out of the SDK's wrapped request:
|
||||
# a mid-body wire failure surfaces as a raw httpx.TransportError
|
||||
# no retry predicate recognizes. The drain re-raises it (chained,
|
||||
# message preserved) as the retryable IncompleteStreamError.
|
||||
import httpx
|
||||
|
||||
def chunks():
|
||||
yield StreamChunk(content_delta="partial")
|
||||
raise httpx.RemoteProtocolError("peer closed connection")
|
||||
|
||||
with pytest.raises(IncompleteStreamError, match="RemoteProtocolError") as excinfo:
|
||||
drain_stream(chunks())
|
||||
assert isinstance(excinfo.value.__cause__, httpx.RemoteProtocolError)
|
||||
|
||||
def test_mid_stream_exception_propagates_verbatim(self):
|
||||
# Retry/deadline/fallback policy is the caller's — the drain adds
|
||||
# no exception translation, exactly like the old transport.
|
||||
def chunks():
|
||||
yield StreamChunk(content_delta="partial")
|
||||
raise RuntimeError("upstream broke")
|
||||
|
||||
with pytest.raises(RuntimeError, match="upstream broke"):
|
||||
drain_stream(chunks())
|
||||
+12
-15
@@ -200,24 +200,21 @@ class TestFlatParamLanes:
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_openai_always_reasoning_row_snaps_without_none(self) -> None:
|
||||
"""Always-reasoning rows (gpt-5.4-pro: medium/high/xhigh) declare no
|
||||
"none" level, so the knob's off position omits the param and low
|
||||
positions snap UP onto the declared floor."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.4-pro", None))
|
||||
def test_openai_o3_registry_row(self) -> None:
|
||||
"""o-series (except o1-mini) accept low/medium/high; no declared
|
||||
"none" level, so the knob's off position omits the param."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "o3", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "medium"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["medium"] == "medium"
|
||||
assert eff["max"] == "xhigh"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_openai_pro_row_wins_longest_prefix(self) -> None:
|
||||
"""gpt-5.4-pro must not prefix-fall onto the gpt-5.4 row (which
|
||||
declares "none") — the pro ladder has no off position, so the
|
||||
longest-prefix row must win or the knob would wrongly omit."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.4-pro", None))
|
||||
assert eff["none"] == "default"
|
||||
base = _as_map(effort_ladder_for_model("openai", "gpt-5.4", None))
|
||||
assert base["none"] == "none"
|
||||
def test_openai_codex_max_has_xhigh(self) -> None:
|
||||
"""gpt-5.1-codex-max must not prefix-fall onto the gpt-5.1 row
|
||||
(which lacks xhigh) — xhigh reaches the wire verbatim."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.1-codex-max", None))
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_anthropic_effort_applies_even_with_thinking_mode_none(self) -> None:
|
||||
"""output_config gates on supports_effort alone at request time."""
|
||||
|
||||
@@ -509,8 +509,8 @@ class TestExtractReasoningForHistory:
|
||||
|
||||
def test_first_block_reasoning_text_dispatches_to_openai_chat(self) -> None:
|
||||
# Phase 3 path 3: synthetic ``reasoning_text`` blocks (stamped
|
||||
# by model_turn.synth_reasoning_block for vLLM / llama.cpp /
|
||||
# Gemini-compat conversations) dispatch to
|
||||
# by ChatSession._maybe_synth_reasoning_block for vLLM /
|
||||
# llama.cpp / Gemini-compat conversations) dispatch to
|
||||
# OpenAIChatCompletionsProvider.extract_reasoning_text.
|
||||
from turnstone.core.history_decoration import extract_reasoning_for_history
|
||||
|
||||
|
||||
@@ -11,7 +11,6 @@ from __future__ import annotations
|
||||
import contextlib
|
||||
import logging
|
||||
import threading
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -40,17 +39,6 @@ class _FakeWorkstream:
|
||||
self._worker_running = False
|
||||
self._closed = False
|
||||
self.worker_thread: Any = None
|
||||
# Deferred /send entries — the wake gate yields while the order
|
||||
# barrier holds; empty/None is the default every other test
|
||||
# assumes.
|
||||
self._pending_sends: list[Any] = []
|
||||
self._pending_drain: Any = None
|
||||
|
||||
def send_barrier_active(self) -> bool:
|
||||
# Mirrors Workstream.send_barrier_active — the gate calls the
|
||||
# METHOD, so the stub must carry the same two-term pair.
|
||||
drain = self._pending_drain
|
||||
return bool(self._pending_sends) or (drain is not None and drain.is_alive())
|
||||
|
||||
|
||||
class _FakeManager:
|
||||
@@ -206,35 +194,6 @@ class TestWakeWorkstreamIfPending:
|
||||
assert wake_workstream_if_pending(ws) is False
|
||||
assert mock_send.call_count == 0
|
||||
|
||||
def test_yields_to_pending_deferred_sends(self, fake_mgr_and_ws):
|
||||
"""Deferred /send entries hold the order barrier: a wake worker
|
||||
claiming the slot would push messages already acknowledged
|
||||
"queued" behind its whole turn, so the gate yields. Re-armed
|
||||
structurally — every deferred turn's exit re-runs the gate, and
|
||||
the drain's clean exit (trigger="drain-exit") covers a list that
|
||||
emptied by pure retraction and never ran a turn."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
ws._pending_sends.append(object())
|
||||
with patch("turnstone.core.session_worker.send", return_value=True) as mock_send:
|
||||
assert wake_workstream_if_pending(ws, trigger="worker-exit") is False
|
||||
assert mock_send.call_count == 0
|
||||
# CLAIMED-entry window: list empty but the drain is alive (an
|
||||
# acked entry was popped, its dispatch in flight). The one-term
|
||||
# list check let a wake jump the acknowledged send here — the
|
||||
# barrier's drain-alive term must hold the yield.
|
||||
ws._pending_sends.clear()
|
||||
ws._pending_drain = SimpleNamespace(is_alive=lambda: True)
|
||||
with patch("turnstone.core.session_worker.send", return_value=True) as mock_send:
|
||||
assert wake_workstream_if_pending(ws, trigger="worker-exit") is False
|
||||
assert mock_send.call_count == 0
|
||||
# Barrier fully cleared (the drain retired) — the same call
|
||||
# dispatches.
|
||||
ws._pending_drain = None
|
||||
with patch("turnstone.core.session_worker.send", return_value=True) as mock_send:
|
||||
assert wake_workstream_if_pending(ws, trigger="drain-exit") is True
|
||||
assert mock_send.call_count == 1
|
||||
|
||||
def test_skips_closed_ws(self, fake_mgr_and_ws):
|
||||
"""A workstream mid-``close()`` must not get a wake spawned on
|
||||
its torn-down session, even while its ``state`` field still
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user