mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-14 07:52:25 -06:00
Compare commits
91 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a4c35e9e29 | |||
| 862eb99cdb | |||
| 25b97bebdf | |||
| ee5ca9a242 | |||
| dd8543fce9 | |||
| 667942024f | |||
| 78831bbe91 | |||
| d44d7eb1a8 | |||
| 876c7d8cb3 | |||
| 98823eb769 | |||
| 4d708c30ac | |||
| 6d60ff7634 | |||
| be662c6134 | |||
| 3ef3f24c7f | |||
| db903f482a | |||
| 6aeffd1845 | |||
| a02b093733 | |||
| f311555026 | |||
| 45d95a2c1f | |||
| a2d9d9832a | |||
| ab123c6cfc | |||
| 8ad17666b9 | |||
| 03fc0861a6 | |||
| a22fb2f395 | |||
| cdcd040da2 | |||
| 834d62c9d4 | |||
| 342a77fe5c | |||
| fd7a447ef9 | |||
| 552ee3c590 | |||
| e99d3ee139 | |||
| 4f0fc3f219 | |||
| dc701986f7 | |||
| bedd25fbe7 | |||
| 251a912275 | |||
| d48902fd01 | |||
| 702ac43d0e | |||
| 01f83dc90f | |||
| 2463c480c2 | |||
| 2a3dfbc6fb | |||
| 6c3b3cc098 | |||
| 0dc52f05ee | |||
| 02929c0d00 | |||
| b2add19c56 | |||
| 5ce1873e9e | |||
| 7698a928c5 | |||
| 4e2eea2f86 | |||
| a6752cb645 | |||
| 06ba4e8d4f | |||
| 94dcaf34fd | |||
| 1a2a689033 | |||
| c02f960d0a | |||
| bfde387206 | |||
| 27d112ff60 | |||
| aeab2535b1 | |||
| 4638d22bd0 | |||
| ee3bd1dcf2 | |||
| ae3a83ccce | |||
| 0f17433e1f | |||
| 043554bb2f | |||
| 8389808add | |||
| 6cbef4f633 | |||
| 2b6dde4f7e | |||
| fbe31b9885 | |||
| ef13f40cf5 | |||
| bfa1b104cf | |||
| 324a1d1a35 | |||
| 95ab88ff6f | |||
| d29840f985 | |||
| 44c0b9c340 | |||
| cdbdf3dc2b | |||
| 8aabb061c2 | |||
| 8bd638569f | |||
| 251dc44a46 | |||
| efd0a1d000 | |||
| 2f93c39fd3 | |||
| f27ce104c6 | |||
| 20a61b692b | |||
| 1966107efe | |||
| d5b2fe6e45 | |||
| 1569819750 | |||
| 4107a30148 | |||
| d06d88b83f | |||
| c411aac939 | |||
| 104715b650 | |||
| 59a9899149 | |||
| 4da7c3b91c | |||
| 012f4e3e16 | |||
| 3636724848 | |||
| eeda5ac312 | |||
| be872b840f | |||
| ee94ae8ba1 |
+25
-27
@@ -14,8 +14,8 @@ jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install pre-commit
|
||||
@@ -25,8 +25,8 @@ jobs:
|
||||
typecheck:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install mypy
|
||||
@@ -35,31 +35,29 @@ jobs:
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
# Cap a hung run at 30 min instead of riding GitHub's 6-hour default
|
||||
# (a flaky-hang run otherwise streams -v output for hours). Was 20;
|
||||
# the suite's growth (~9.7k tests, coverage-instrumented, 3-version
|
||||
# matrix) started brushing the old cap on healthy runs.
|
||||
timeout-minutes: 30
|
||||
# Cap a hung run at 20 min instead of riding GitHub's 6-hour default
|
||||
# (a flaky-hang run otherwise streams -v output for hours).
|
||||
timeout-minutes: 20
|
||||
strategy:
|
||||
matrix:
|
||||
python-version: ["3.11", "3.12", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
# Node is required by tests/test_renderer_js.py — without
|
||||
# explicit setup, that suite silently skips if the runner
|
||||
# image happens not to ship Node, masking regressions in
|
||||
# the browser-side renderer.
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: pip install -e ".[test]"
|
||||
# -v lists each test id as it starts (pytest prints the nodeid at
|
||||
# logstart), so a hang names the culprit on the last line instead of
|
||||
# riding the job timeout with only a trail of "..." dots.
|
||||
- run: pytest tests/ -m "not live and not e2e_recovery" --cov=turnstone --cov-report=term-missing --cov-report=xml -v
|
||||
- run: pytest tests/ -m "not live" --cov=turnstone --cov-report=term-missing --cov-report=xml -v
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
if: always()
|
||||
with:
|
||||
@@ -68,7 +66,7 @@ jobs:
|
||||
|
||||
test-postgres:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
timeout-minutes: 20
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:18
|
||||
@@ -84,23 +82,23 @@ jobs:
|
||||
--health-timeout=5s
|
||||
--health-retries=5
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: pip install -e ".[test]"
|
||||
- run: pytest tests/ -m "not live and not e2e_recovery" --storage-backend=postgresql -v
|
||||
- run: pytest tests/ -m "not live" --storage-backend=postgresql -v
|
||||
env:
|
||||
TURNSTONE_TEST_PG_URL: postgresql+psycopg://postgres:postgres@localhost:5432/turnstone_test
|
||||
|
||||
wheel-completeness:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install build
|
||||
@@ -153,8 +151,8 @@ jobs:
|
||||
lock-check:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- run: uv lock --check
|
||||
@@ -162,11 +160,11 @@ jobs:
|
||||
security:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: uv sync --frozen --all-extras
|
||||
@@ -190,8 +188,8 @@ jobs:
|
||||
run:
|
||||
working-directory: sdk/typescript
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: npm ci
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
name: Claude Code Review
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, synchronize, ready_for_review, reopened]
|
||||
# Optional: Only run on specific file changes
|
||||
# paths:
|
||||
# - "src/**/*.ts"
|
||||
# - "src/**/*.tsx"
|
||||
# - "src/**/*.js"
|
||||
# - "src/**/*.jsx"
|
||||
|
||||
jobs:
|
||||
claude-review:
|
||||
if: github.event.pull_request.head.repo.full_name == github.repository
|
||||
# Optional: Filter by PR author
|
||||
# if: |
|
||||
# github.event.pull_request.user.login == 'external-contributor' ||
|
||||
# github.event.pull_request.user.login == 'new-developer' ||
|
||||
# github.event.pull_request.author_association == 'FIRST_TIME_CONTRIBUTOR'
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post the review + inline comments
|
||||
issues: read
|
||||
id-token: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
allowed_bots: 'renovate[bot]' # let Renovate PRs get reviewed
|
||||
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
|
||||
plugins: 'code-review@claude-code-plugins'
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ github.event.pull_request.number }}'
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
name: Claude Code
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
pull_request_review_comment:
|
||||
types: [created]
|
||||
issues:
|
||||
types: [opened, assigned]
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
jobs:
|
||||
claude:
|
||||
if: |
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review' &&
|
||||
contains(github.event.review.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.review.author_association)
|
||||
) || (
|
||||
github.event_name == 'issues' &&
|
||||
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.issue.author_association)
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post comments/reviews when @-mentioned on a PR
|
||||
issues: write # post comments when @-mentioned on an issue
|
||||
id-token: write
|
||||
actions: read # Required for Claude to read CI results on PRs
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
|
||||
# This is an optional setting that allows Claude to read CI results on PRs
|
||||
additional_permissions: |
|
||||
actions: read
|
||||
|
||||
# Optional: Give a custom prompt to Claude. If this is not specified, Claude will perform the instructions specified in the comment that tagged it.
|
||||
# prompt: 'Update the pull request description to include a summary of changes.'
|
||||
|
||||
# Optional: Add claude_args to customize behavior and configuration
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
# claude_args: '--allowed-tools Bash(gh pr *)'
|
||||
|
||||
@@ -33,7 +33,7 @@ jobs:
|
||||
startsWith(github.event.workflow_run.head_branch, 'v')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
@@ -54,7 +54,7 @@ jobs:
|
||||
|
||||
- name: Log in to GHCR
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
|
||||
uses: docker/login-action@af1e73f918a031802d376d3c8bbc3fe56130a9b0 # v4
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
|
||||
@@ -30,7 +30,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
echo "skip=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
with:
|
||||
python-version: "3.14"
|
||||
@@ -58,12 +58,12 @@ jobs:
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
- run: python -m build
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
- uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
- uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # release/v1
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
|
||||
- name: Create GitHub Release
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3
|
||||
uses: softprops/action-gh-release@718ea10b132b3b2eba29c1007bb80653f286566b # v3
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.tag }}
|
||||
generate_release_notes: true
|
||||
|
||||
@@ -31,8 +31,8 @@ jobs:
|
||||
# Floor and ceiling of the example's requires-python (>=3.11).
|
||||
python-version: ["3.11", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- run: pip install -e ".[test,dev]"
|
||||
|
||||
@@ -61,7 +61,7 @@ jobs:
|
||||
fi
|
||||
echo "head_ref=${ref}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ steps.ref.outputs.head_ref }}
|
||||
|
||||
|
||||
@@ -29,4 +29,3 @@ tools/skill_audit_analysis/output/
|
||||
design_ideas/
|
||||
.claude/
|
||||
docs/design/
|
||||
/.idea
|
||||
|
||||
+36
-520
@@ -14,535 +14,51 @@ experimental line:
|
||||
|
||||
Earlier stable lines (`stable/1.6`, `stable/1.5`) are frozen.
|
||||
|
||||
## [Unreleased]
|
||||
## [1.7.4]
|
||||
|
||||
A feature-bearing patch for the 1.7 line, rolling up work that had stabilised
|
||||
on `main`. No schema migrations (head stays 066) and no new configuration knobs.
|
||||
|
||||
### Added
|
||||
|
||||
- **`server_parses_reasoning` model capability.** Declare it on a model
|
||||
definition whose backend segregates reasoning into its own channel (a
|
||||
vLLM launched with a reasoning parser, a commercial provider): the
|
||||
inline think-tag scan turns off on every lane — interactive and
|
||||
drained alike — so content is trusted verbatim and prose that merely
|
||||
quotes a tag can no longer be misrouted into the reasoning lane, and
|
||||
the utility lanes stop suppressing reasoning they'd otherwise pin off.
|
||||
Default off for local lanes, preserving the passthrough-server
|
||||
behavior; the built-in capability tables declare it for every real
|
||||
commercial endpoint (known models and table-miss defaults alike),
|
||||
which also removes the quoted-tag false positive from those lanes.
|
||||
- **Per-model Entra gateway authentication.** Model definitions can bind either
|
||||
a caller-delegated OBO token (`entra_obo`) or a shared app-identity token
|
||||
(`entra_app`) through the provider SDK credential surface. Mints reuse the
|
||||
encrypted cluster token cache, refresh-rotation CAS, and advisory locking;
|
||||
add a host-local memo, failure cooldown, long-lived mint HTTP client, audience
|
||||
allow-list/permission boundary, identity-unlink purge, and optional
|
||||
`model.auth_fail_closed` refusal policy. Delegated identity now propagates
|
||||
through judge, output-guard, and principal-scoped perception lanes, and
|
||||
unattended watch restoration reacquires the persisted workstream owner.
|
||||
Ownerless OBO calls and dynamic aliases without a real static fallback always
|
||||
fail closed; grant modes are never silently switched. Static authentication
|
||||
remains the default.
|
||||
|
||||
- **Compaction is visible now: lifecycle events, a progress bar, and a
|
||||
persistent transcript card.** Context compaction (manual `/compact` and
|
||||
auto) emits a first-class `compaction` SSE event
|
||||
(`start` / `progress` / `end` — see the API reference) instead of loose
|
||||
info lines. The web UI renders an in-transcript card with a real progress
|
||||
bar (determinate `part k of N` during chunked summarization, indeterminate
|
||||
for single-call compactions) that settles into a result card — token delta
|
||||
plus the summary behind a fold — in both the interactive pane and the
|
||||
coordinator viewer. The result survives reloads: the persisted compaction
|
||||
marker now projects through `/history` as a `role="system"`,
|
||||
`source="compaction"` entry (resume/export/search unchanged), stamped with
|
||||
the end event's id so repaint and SSE replay can't double-render. The
|
||||
marker's `meta` additionally records `before_tokens` / `after_tokens` /
|
||||
`trigger`. Python and TypeScript SDKs gain a typed `CompactionEvent`.
|
||||
|
||||
- **One provider transport: every model call now streams (#831).**
|
||||
The per-adapter non-streaming entry (`create_completion`) is retired;
|
||||
single-shot lanes — judges, titles, compaction, web-fetch extraction,
|
||||
perception, eval, optimizer — sample through the same streaming entry
|
||||
the chat loop uses and accumulate via one shared drain, so request
|
||||
shaping can no longer drift between the two consumption styles. Two
|
||||
operator-visible consequences: long single-shot generations (a thinking
|
||||
model composing a title, a slow local judge) no longer sit in a single
|
||||
blocking read that can hit client read-timeouts — the same reason the
|
||||
Anthropic adapter already streamed internally — and judge timeouts now
|
||||
*abort* the underlying HTTP read instead of abandoning a worker thread
|
||||
on a dead call. Because every call now streams, an alias pointed at a
|
||||
model or org that cannot stream (OpenAI's verified-org streaming
|
||||
entitlement, a gateway api-version predating `stream_options` — e.g.
|
||||
older Azure OpenAI deployments) fails at request time where 1.7's
|
||||
non-streaming single-shot call succeeded; remediation is on the
|
||||
serving side (verify the org, bump the api-version/gateway) — there is
|
||||
deliberately no per-model non-streaming fallback left to configure. These lanes are also complete-or-error now: a stream
|
||||
that ends without any finish signal is treated as a generation that
|
||||
died mid-response and retried, instead of storing the partial text as
|
||||
a clean result (previously a half-generated compaction summary could
|
||||
silently replace real history). Caveats: these lanes now carry the
|
||||
same `stream_options: {include_usage: true}` the chat loop always
|
||||
sent — OpenAI-compatible servers old enough to *ignore* it stop
|
||||
producing usage rows on these lanes, and servers strict enough to
|
||||
*reject* unknown fields (pre-2024 llama.cpp/proxy builds) will 400 —
|
||||
such a server already couldn't serve turnstone's chat loop, but a
|
||||
judge/utility alias pointed at one worked on 1.7 and needs to move to
|
||||
a current server. Transient mid-stream deaths (connection drop, proxy
|
||||
hiccup) are re-issued in place up to twice with exponential backoff —
|
||||
the retry the SDK's request loop used to provide these lanes
|
||||
invisibly. Each lane accepts its own terminal marker (Anthropic
|
||||
`message_stop`, Responses terminal events); a lax server/gateway that
|
||||
never sends any terminal signal needs
|
||||
`{"finish_reason_optional": true}` in the model definition's
|
||||
capabilities JSON, which restores 1.7's tolerance (clean end-of-stream
|
||||
after output = completion) for that model on every lane — without it
|
||||
such streams fail as died-mid-generation, because SSE gives no way to
|
||||
tell the two apart and the default favors catching truncation. The
|
||||
unread `supports_streaming` capability flag (and its admin tile) is
|
||||
gone; the o-series models it described are dropped from the capability
|
||||
table entirely (see Removed).
|
||||
|
||||
- **One turn interface for every model call: `core/model_turn.py` (#827).**
|
||||
Judges (intent + output guard), perception, title generation, compaction,
|
||||
web-fetch extraction, the eval harness, the optimizer's meta lanes, and
|
||||
task agents all advance a trajectory through the same plant-call
|
||||
primitive the agent seam pioneered — Turn IR in, one shared lowering
|
||||
(argument sanitize → minted-id restore → vLLM reasoning attach), one
|
||||
shared re-ingest (blank-id repair → native-lane finalize). The judges'
|
||||
hand-built OpenAI-dict path is gone, and with it the Gemini judge's
|
||||
tool-blindness: evidence tools now work on Google models because the
|
||||
native lane round-trips `thought_signature` (with pairwise repair for
|
||||
blank-id compat responses). Provider adapters still take lowered wire
|
||||
dicts — the transport collapse and main-loop migration are tracked as
|
||||
#831 / #832.
|
||||
|
||||
- **task_agent keeps its model's reasoning across its own tool loop — on
|
||||
every provider lane.** A task agent's replayed turns now carry the
|
||||
provider-native reasoning lane the model produced — Anthropic thinking
|
||||
blocks with their signatures (commercial or an anthropic-compatible
|
||||
server), OpenAI Responses reasoning items, Gemini `thought_signature`
|
||||
fidelity blocks, and the reasoning text a vLLM `--reasoning-parser` /
|
||||
llama.cpp `reasoning_format` surfaces on the Chat Completions lane —
|
||||
instead of each turn being rebuilt from text + tool calls with the
|
||||
reasoning dropped. On a thinking model this restores reasoning continuity
|
||||
across the agent's own multi-turn tool use. On the wire the agent's
|
||||
session-minted sub-tool ids are mapped back to the provider's own ids
|
||||
(`restore_provider_tool_ids`), so the native block — replayed verbatim,
|
||||
its signature never touched — the `tool_calls` mirror, and each tool
|
||||
result always agree; internally the minted ids still key the live card,
|
||||
recall, and the cancel ledger unchanged. Replay honors the same per-model
|
||||
`replay_reasoning_to_model` flag the main loop uses on every lane: the
|
||||
vLLM Chat-Completions field replay keeps its server-type gate, and
|
||||
llama.cpp stays capture-only, matching main-loop behavior. The native
|
||||
lane is finalized by the same shared builder as the main loop's, so the
|
||||
two harnesses cannot drift.
|
||||
|
||||
- **Background shells: `bash` gains `run_in_background`, plus `bash_output` /
|
||||
`kill_shell`.** Setting `run_in_background=true` starts the command as a
|
||||
detached shell and returns immediately with a `bash_N` handle — "start a dev
|
||||
server, use it in a later call" is back as an explicit opt-in (the shape
|
||||
follows the convention the major coding agents converged on). `bash_output`
|
||||
returns only output produced since the previous read (optionally filtered by
|
||||
a regex) plus status and exit code; `kill_shell` terminates the shell's
|
||||
whole process group. Output is buffered per shell with a drop-oldest cap, so
|
||||
a chatty server can't grow memory unbounded. When a background shell exits,
|
||||
a system notice lands at the next seam (waking an idle workstream if
|
||||
needed). Shells survive a generation cancel, die with the workstream, and
|
||||
never outlive a task_agent that started them; anything a background shell
|
||||
itself backgrounds is still reaped when that shell exits — the no-leak
|
||||
guarantee below is unchanged.
|
||||
- **Background shells for the `bash` tool** — `run_in_background=true` starts a
|
||||
command as a detached shell and returns a `bash_N` handle; new `bash_output`
|
||||
(delta output since last read, optional regex filter, status/exit code) and
|
||||
`kill_shell` (terminates the shell's process group) tools manage it. Output is
|
||||
buffered with a drop-oldest cap, a system notice lands when a shell exits, and
|
||||
shells die with their workstream — never outliving a `task_agent` that started
|
||||
them.
|
||||
- **`task_agent` carries the model's native reasoning across its own tool loop** —
|
||||
a task agent's replayed turns now preserve the provider-native reasoning lane
|
||||
(Anthropic thinking blocks with signatures, OpenAI reasoning items, Gemini
|
||||
`thought_signature`, vLLM/llama.cpp reasoning text) instead of rebuilding each
|
||||
turn from text alone, restoring reasoning continuity for thinking models.
|
||||
- **Model-shelf response controls** — the console model shelf exposes verbosity
|
||||
and reasoning-mode controls per identity.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Log event rename: `drain_stream.post_finish_blip` is now
|
||||
`stream.post_finish_blip`; its `usage_captured` field is retained.** The
|
||||
single-shot drain normalizes mid-body transport deaths through the same
|
||||
`transport_guarded` wrapper the interactive loop uses, so its
|
||||
post-finish-blip tolerance logs under the wrapper's event name. Update
|
||||
any external log filters pinned to the old name; the drained result's
|
||||
possible `usage=None` on a post-finish blip is unchanged and documented
|
||||
on `drain_stream`.
|
||||
|
||||
- **Breaking (1.8): compaction feedback moved from `info` events to the
|
||||
typed `compaction` SSE event.** Pre-1.8 SSE/SDK clients that ignore
|
||||
unknown event types no longer see compaction lines (they are
|
||||
deliberately not dual-emitted — dual emission would double-render on
|
||||
every current client). Consume the `compaction` lifecycle event (see
|
||||
the API reference and the `CompactionEvent` SDK type); embedders
|
||||
driving `ChatSession` through a duck-typed `SessionUI` are unaffected
|
||||
(the classic `on_info` lines are restored for them — see Fixed).
|
||||
|
||||
- **Sampling knobs (temperature, reasoning effort) now ride one assignment
|
||||
scheme: per-model alias value → operator-stored global setting → the
|
||||
model definition's declared default (effort only) → field omitted.**
|
||||
Turnstone previously manufactured values onto every unconfigured
|
||||
request — a hidden `temperature: 0.5` and a `reasoning_effort: "medium"`
|
||||
baked in at three layers — overriding serving-side defaults like a vLLM
|
||||
model's `generation_config`. Unconfigured installs now send neither
|
||||
field and the inference engine's own defaults rule; `model.temperature`
|
||||
is blank by default ("inherit each model's own default") and
|
||||
`model.reasoning_effort` defaults to the empty "inherit" choice. The
|
||||
per-model → global resolution lives in one shared resolver used by the
|
||||
session factories, the `/model` switch, and every `model_turn` lane, so
|
||||
the same alias samples identically on every surface. CLI
|
||||
`--temperature` / `--reasoning-effort` likewise default to inherit.
|
||||
|
||||
**Upgrade notes:**
|
||||
- The empty (`""`) reasoning-effort choice changed meaning from
|
||||
"explicitly disable thinking" to "inherit the model/serving default".
|
||||
On local manual-thinking models (e.g. Qwen templates with
|
||||
`enable_thinking`), a stored `""` previously sent
|
||||
`enable_thinking: false`; it now sends nothing, so the template's own
|
||||
default (often thinking ON) applies. Use **`none`** to actually
|
||||
disable reasoning.
|
||||
- Workstreams saved by earlier versions carry the old defaults
|
||||
(`temperature=0.5`, `reasoning_effort=medium`) in their persisted
|
||||
config and keep that exact behavior on resume; they pick up the new
|
||||
inherit semantics the next time you change the model or a sampling
|
||||
knob in that workstream. New workstreams inherit from the start.
|
||||
|
||||
### Removed
|
||||
|
||||
- **O-series and pre-5.4 GPT-5 rows dropped from the OpenAI capability
|
||||
table.** `o1`, `o1-mini`, `o3`, `o3-mini`, `o3-pro`, `o4-mini`,
|
||||
`gpt-5`, `gpt-5-mini`, `gpt-5-nano`, `gpt-5-pro`, `gpt-5.1`,
|
||||
`gpt-5.1-codex-max`, `gpt-5.2`, `gpt-5.2-pro`, and `gpt-5.3` no longer
|
||||
have built-in capability rows — OpenAI has retired these model ids
|
||||
from the API, so the rows described contracts no request can reach
|
||||
anymore. The table floor is now `gpt-5.4`; the search-api and
|
||||
audio/STT/TTS rows are unchanged. An alias still pinning a retired id
|
||||
fails at OpenAI itself; any other unlisted commercial id resolves to
|
||||
the generic commercial defaults (temperature sent, no declared
|
||||
reasoning-effort vocabulary, 200K window) — declare the contract on
|
||||
the model definition's capabilities JSON if you run one, or move to a
|
||||
current model.
|
||||
- **GPT-5.6 aligned with the GA API surface** — the Responses provider matches
|
||||
GPT-5.6's GA shape (typed `reasoning.mode`, `prompt_cache_options`,
|
||||
cache-write accounting); the `openai` floor moves to `>=2.45`.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **A cancelled judge, guard, or compaction call can now stop before its
|
||||
request goes out (#972).** Previously it could not: `model_turn` refused
|
||||
to *re-issue* an abandoned call after a mid-stream death, but nothing
|
||||
checked before a first dispatch, so a call whose caller had already gone
|
||||
away still sent — and the reply was discarded unread after the endpoint
|
||||
had accepted the work. It now checks immediately before sending, so a
|
||||
Stop observed by that point costs no request, and again on entry, so a
|
||||
call already cancelled when it arrives also skips credential resolution.
|
||||
Cancellation is cooperative, which bounds what that buys: a Stop only
|
||||
saves the request if it lands before dispatch — sending is a moment, the
|
||||
response streaming back is the rest of the call, and an abort arriving
|
||||
then still meets a request in flight, closed in place exactly as before.
|
||||
The window that did widen usefully is a delegated-auth alias whose token
|
||||
mint blocks; a Stop during that mint now costs no request (though a mint
|
||||
already under way still completes). What a stopped call saves is the
|
||||
request, its prompt-side billing, and — on a capacity-bounded
|
||||
self-hosted endpoint — a slot a live request wanted. Unchanged: the
|
||||
interactive turn, which has its own pre-send cancellation check on a
|
||||
different path, and the lanes that thread no cancellation handle
|
||||
(attachment perception, title generation, web-fetch extraction,
|
||||
sub-agents, optimizer, eval) — and web-fetch extraction deliberately
|
||||
never will, since it runs on parallel tool threads where registering one
|
||||
would clobber the main stream's.
|
||||
- **Unmarked chain-of-thought no longer leaks into titles, summaries, or
|
||||
web-fetch tool results (#940).** Some serving setups emit reasoning
|
||||
inline with no tags and no `reasoning_content` at all — nothing any
|
||||
parser can segregate. The bounded-artifact lanes (title, compaction,
|
||||
web-fetch extraction) now ask the model for no reasoning instead:
|
||||
the model definition's declared thinking toggle is pinned off for that
|
||||
call — the same suppression transcription already used — and the
|
||||
reasoning-effort channels (the relayed session knob, the definition's
|
||||
default, the graded template key) are withheld with it, since an
|
||||
effort value beside a pinned-off toggle re-requests the reasoning the
|
||||
pin declined. A no-op on backends that segregate reasoning
|
||||
server-side. Title generation additionally stopped trusting line
|
||||
position: it takes the last line that reads as a title (within the
|
||||
word cap and ending in a word character, so explanation sentences,
|
||||
sign-offs, and reasoning headings lose in any script) rather than the
|
||||
first non-empty line, which unmarked reasoning turned into titles
|
||||
like "Thinking Process:".
|
||||
- **A think tag split across a reasoning delta now reassembles.** The
|
||||
non-streaming drain closes content runs at interleaving signals; a
|
||||
partial-tag tail is carried across reasoning-delta boundaries (a
|
||||
reasoning delta cannot terminate a tag) so the tag is consumed instead
|
||||
of its halves passing through as visible content. Tool-call boundaries
|
||||
still flush — no tag spans a tool call.
|
||||
- **Streaming consumers follow the ACTIVE model's capabilities.** The
|
||||
interactive tag-scan posture and the drain's scan gate now read the
|
||||
capabilities of the lane that owns the stream being consumed (fallback
|
||||
walks included) instead of the session's primary alias.
|
||||
- **Notification bodies no longer fuse multi-block answers.** `Turn.text`
|
||||
joins text blocks with a newline; a final assistant turn stored as
|
||||
multiple text blocks previously concatenated the last word of one
|
||||
block to the first word of the next in completion notifications and
|
||||
every other flattened read.
|
||||
- **String-typed boolean capability overrides coerce instead of
|
||||
truthiness-flipping.** A hand-edited `"false"`/`"0"` in a model
|
||||
definition's capabilities JSON now means false; unrecognized values
|
||||
drop the key and keep the field's default.
|
||||
- **Inline `<think>`/`<reasoning>` blocks no longer leak into drained
|
||||
results (#965, #940).** On servers without a reasoning parser
|
||||
(parserless vLLM/llama.cpp, LM Studio, bare gateways), reasoning
|
||||
arrives as literal tags inside content; segregation now happens once
|
||||
at the drain seam, so web-fetch tool results, sub-agent syntheses,
|
||||
judge verdicts, titles, summaries, and optimizer prompts receive
|
||||
tag-free content and the extracted reasoning rides the native lane.
|
||||
Two behavior notes: a web-fetch extraction whose whole response was
|
||||
reasoning now returns an explicit `Error: extraction returned no
|
||||
answer` tool result (previously the raw reasoning text persisted as a
|
||||
successful result and was replayed every following turn), and a
|
||||
mismatched-vocabulary close tag (`<think>…</reasoning>`) now closes
|
||||
the block — matching the interactive lane's long-standing rule —
|
||||
where the old per-lane strips treated it as unterminated.
|
||||
|
||||
- **A transport failure mid-generation no longer kills the interactive
|
||||
turn (#937).** A wire death during body streaming (TLS record failure,
|
||||
connection reset — `httpx.ReadError` and kin) surfaces after the
|
||||
request has already returned its stream handle, so neither the SDK's
|
||||
request retries nor the creation-time retry ladder ever saw it: the
|
||||
turn died with a bare `ReadError: …`, the partial output was
|
||||
discarded, and nothing was logged. The interactive loop now normalizes
|
||||
mid-body transport deaths exactly like the single-shot lanes and
|
||||
re-issues the turn (bounded, cancel-aware, exponential backoff),
|
||||
finalizing the dead attempt across every UI surface first so retried
|
||||
text never double-renders (web transcript, CLI markdown fences,
|
||||
Slack/Discord streamed messages). Before re-creating the stream the
|
||||
session re-resolves its registry binding, so a concurrent model-registry
|
||||
reload that closed the old client cannot turn the retry into a
|
||||
misleading closed-client error. On exhaustion the surfaced error names
|
||||
the provider, endpoint, and model with a stream-death message instead
|
||||
of a bare exception string, and every fatal turn now leaves a
|
||||
`session.fatal.recorded` log line (INFO for a user Ctrl-C, ERROR
|
||||
otherwise).
|
||||
|
||||
- **A failed worker-thread spawn no longer wedges the workstream — at
|
||||
either spawn site — and never masquerades as success.** If
|
||||
`Thread.start()` itself raised (thread exhaustion, out-of-memory), the
|
||||
dispatcher had already claimed the worker slot but the flag's only
|
||||
clearer lived in the never-started thread — the workstream looked idle
|
||||
forever while every subsequent message queued behind a worker that
|
||||
didn't exist, until an operator force-cancel. The claim is now rolled
|
||||
back under the lock and the error propagates, so the workstream is
|
||||
dispatchable again as soon as resources recover. Affected every
|
||||
dispatch path (sends, wakes, retries, deferred-send drain, init). The
|
||||
same failure at the deferred-send drain's own spawn rolls back the
|
||||
just-accepted entry and answers the retryable `queue_full` (previously
|
||||
a 500 landed *after* the entry was registered — an invisible,
|
||||
unretractable phantom that later dispatched as duplicate turns), and a
|
||||
`/command` whose worker never spawned now answers **503**
|
||||
`{"status": "error"}` instead of the generic 200 ok that told SDK
|
||||
callers their `/clear` or `/resume` had applied.
|
||||
|
||||
- **Manual `/compact` from the web UI: no phantom user turn, no frozen
|
||||
server, cancellable.** A slash command typed into the web composer no
|
||||
longer renders as a user chat bubble (it echoes as a distinct command
|
||||
chip — commands aren't conversation turns and were never persisted as
|
||||
such). `/compact` itself now dispatches onto the workstream's worker
|
||||
slot instead of running inline on the server's event loop — previously a
|
||||
long compaction froze every SSE stream on the node for its whole
|
||||
duration, which is also why its own progress only ever arrived as one
|
||||
burst after the fact. The manual path carries `send()`'s full generation
|
||||
discipline (`compact_now()`): a force-abandoned compaction goes stale
|
||||
instead of swapping history under a successor turn — and retires at its
|
||||
next checkpoint instead of running out its remaining summary calls,
|
||||
with its late lifecycle events fenced off (`compaction_id` on every
|
||||
event, `superseded` on end events — both in the SDKs) so they can't
|
||||
animate, tear down, re-title, or falsely narrate a successor's card or
|
||||
activity pill; a cancel aimed at it is consumed on exit (previously it
|
||||
bricked every `/compact` retry until the next message); a Stop click on
|
||||
an idle session can't pre-abort the next compaction; a Stop that lands
|
||||
in the completion tail — after the last cancel check, or during a retry
|
||||
backoff (which now aborts immediately instead of sleeping it out) — is
|
||||
honored rather than silently eaten; and Stop now aborts the in-flight
|
||||
summary HTTP call itself (the compaction lane registers its stream in
|
||||
the same abort seam the main loop uses), so cancelling a compaction is
|
||||
immediate instead of waiting out a model call.
|
||||
|
||||
- **Sends during a command window are deferred, ordered, bounded, and
|
||||
honestly rendered — never silently truncated or lost.** Messages sent
|
||||
while any slash command holds the worker slot are **deferred**: answered
|
||||
`{"status": "queued", "msg_id"}` immediately and dispatched as ordinary
|
||||
full-fidelity sends (attachments and sender identity included) when the
|
||||
command finishes — never routed through the mid-turn interjection
|
||||
queue, whose semantics are turn-shaped: previously a send during a
|
||||
manual `/compact` was silently truncated to 2,000 characters, a second
|
||||
participant in a shared workstream was locked out with a misleading
|
||||
"another participant's turn" 409 for the whole compaction, and a
|
||||
message queued across a `/resume`/`/new` could be answered into the
|
||||
post-swap workstream. Because the response is immediate,
|
||||
timeout-bounded callers — the coordinator's `send_message`, the console
|
||||
proxy, SDKs, anything behind a stock reverse proxy — can no longer lose
|
||||
a message to a multi-minute command window; the deferred send is
|
||||
retractable until dispatch via the same `DELETE .../send` used for
|
||||
queued interjections (node-local, in-memory — the API reference
|
||||
documents the at-most-once durability contract). Deferred responses
|
||||
carry `"deferred": true`; the pending list is the **order authority**
|
||||
(a fresh send — or a coordinator dispatch, or a queued-nudge wake —
|
||||
lines up behind acknowledged entries instead of overtaking them, with
|
||||
the two-term barrier defined once on the workstream so the wake gate
|
||||
also honors a claimed entry whose dispatch is mid-flight, and the gate
|
||||
re-arms at the drain's exit even when everything pending was
|
||||
retracted); acceptance is **bounded** (10 pending per workstream — the
|
||||
interjection queue's own backpressure contract; the 11th answers the
|
||||
retryable `queue_full` instead of pinning attachment bytes without
|
||||
limit and then running one unattended turn per entry); a dispatch
|
||||
crash re-queues the entry instead of eating an acknowledged message,
|
||||
and a drain thread that fails to *start* rolls the acceptance back and
|
||||
answers `queue_full` rather than parking a phantom the client can
|
||||
neither see nor retract; each dispatch emits a pane-tier
|
||||
`message_dispatched` event (`folded: true` for interjection fold-ins)
|
||||
so queued-bubble UI keeps its retract affordance exactly until the
|
||||
message truly leaves — including when the send was accepted by a pane
|
||||
that believed the workstream idle, which now renders a real queued
|
||||
chip instead of a sent-looking bubble, releases the composer (a
|
||||
deferred send has no running worker to wait on), and cleans up fully
|
||||
when the send is refused or the chip retracted instead of stranding
|
||||
the pane in Stop mode. Dismissing a queued bubble — interjection or
|
||||
deferred — is a server-confirmed `DELETE`, and retracting a deferred
|
||||
send that carried attachments tells the user they were discarded
|
||||
instead of silently expiring them.
|
||||
|
||||
- **Slash commands hold the worker slot with a loud contract.**
|
||||
A `/compact` raced against an in-flight turn is refused with an
|
||||
explicit busy response. Every other slash command runs through the same
|
||||
worker slot too — mutual exclusion against sends, a running compaction,
|
||||
and each other, with a busy answer replacing the old silent interleave —
|
||||
while the endpoint still awaits quick commands' completion off-loop
|
||||
(without parking an executor thread per request); the post-command pane
|
||||
refreshes (`clear_ui` after `/clear`/`/new`/`/resume`, the
|
||||
workstream-name sync) ride the worker itself, so a command that
|
||||
outlives the endpoint's 25s response backstop still refreshes every
|
||||
pane on completion (the backstop sits under the console proxy's 30s
|
||||
client timeout so the degraded `running` answer can actually traverse
|
||||
a proxied pane, which now surfaces it instead of silence; the
|
||||
`/command` response contract — `ok` / `running`, with busy refusals
|
||||
answering a loud HTTP 409 rather than a silent 200 — is now documented
|
||||
in the API reference and the OpenAPI spec).
|
||||
|
||||
- **Compaction status stays truthful across every UI surface.** Manual
|
||||
compaction
|
||||
success also refreshes the status line/context pill immediately (parity
|
||||
with auto-compaction), compaction failures keep feeding the typed
|
||||
`error` event and the node error counter (while a CLI Ctrl-C reports as
|
||||
cancelled, not a failure), one Stop prints one notice (a cancelled
|
||||
auto-compaction no longer stacks "Compaction cancelled." on top of
|
||||
send's own "[Generation cancelled]"), the workstream activity pill
|
||||
shows "Compacting context…" for the whole summarize phase, restores
|
||||
cleanly afterwards, and can no longer be stranded by a force-stopped
|
||||
compaction (a new turn's generation claim breaks a stale latch). Every
|
||||
retry backoff on the session (stream retries, task agents, notify
|
||||
delivery, compaction) now aborts immediately on Stop via one shared
|
||||
cancel-aware helper instead of sleeping out its exponential delay.
|
||||
|
||||
- **Compaction failures report exactly once, to the right owner.** A
|
||||
compaction failure reports
|
||||
exactly once (auto-compaction errors defer to the turn's fatal handler
|
||||
instead of doubling the red row and the error metric), failed-end
|
||||
notice suppression is computed once by the emitter (a `notice` bool on
|
||||
the end event — in the SDKs — replaces hand-synced client policy), and
|
||||
a manual `/compact` failure no longer crashes the CLI REPL. `/compact`
|
||||
on a workstream showing the `error` badge restores the badge on exit
|
||||
instead of stamping `idle` over it (the compaction neither retried nor
|
||||
resolved the failed turn). A force-cancelled initial send that
|
||||
completes late still delivers its scheduled-run completion
|
||||
notification (the only completion signal unattended workstreams have);
|
||||
the other post-command pane refreshes and error notices remain
|
||||
owner-guarded, so a force-cancelled wedged command that unwedges late
|
||||
can't wipe panes or inject stray notices into a successor turn.
|
||||
|
||||
- **Pre-1.8 embedder UIs keep their compaction lines.** Embedders
|
||||
driving `ChatSession` with a pre-1.8 duck-typed `SessionUI`
|
||||
(no `on_compaction` hook) get the classic `on_info` compaction lines
|
||||
back — threshold notice, `part k/N`, retry waits, token delta +
|
||||
summary box — instead of silent history swaps. (See the breaking
|
||||
event-contract note under **Changed** for SSE/SDK clients.)
|
||||
|
||||
- **Static MCP servers: a pushed catalog change no longer wedges the shared
|
||||
session (#839).** The static-path `*/list_changed` handler awaited its
|
||||
catalog refresh inline in the SDK's receive loop, but the refresh's own
|
||||
request can only be answered by that (now parked) loop — the refresh never
|
||||
completed, and every user's in-flight calls on the shared per-node session
|
||||
stalled behind it, unbounded, until the health loop's ping timeout tore the
|
||||
transport down (which was also the only way the changed catalog ever
|
||||
landed). Push refreshes now run as spawned tasks — debounced, coalesced per
|
||||
(server, kind), bounded by the connect timeout, and serialized on the
|
||||
per-server connect lock — and the manual and post-reconnect refreshes
|
||||
publish under that same lock, so a slower publisher can no longer land a
|
||||
staler catalog over a fresher one. Every teardown path now also clears the
|
||||
notification debounce stamp, so a reconnected server's first push refreshes
|
||||
immediately. Push-refresh debouncing is now per (server, kind) on BOTH the
|
||||
static and per-user pool paths — a tools push no longer swallows a prompts
|
||||
push arriving in the same 5-second window. A change genuinely lost to the
|
||||
debounce window (a same-kind push landing after the prior refresh finished,
|
||||
which the server will never re-announce) is recovered by an automatic
|
||||
health-tick retry rather than staying invisible until an unrelated push or
|
||||
a reconnect. The resource-refresh fan-out on both paths no longer orphans
|
||||
its sibling list call when one of the pair fails fast — the real error
|
||||
surfaces immediately (not masked as a 30-second timeout) and the surviving
|
||||
sibling is cancelled and reaped, under a bounded grace, inside the scope. A
|
||||
push refresh that fails while the connection stays up is likewise retried on
|
||||
the next health-loop tick until one completes — previously a single
|
||||
transient blip left the shared catalog stale for every user on the node
|
||||
until an operator intervened. An operator `/mcp refresh` no longer parks
|
||||
behind a busy per-server connect lock (a slow reconnect attempt could eat
|
||||
the whole 30-second refresh budget and fail the pass for every healthy
|
||||
server behind it) — the busy server is skipped on both the connected and
|
||||
disconnected branches, reported distinctly as "skipped" rather than as a
|
||||
false "no changes", the skip arms the automatic retry, and a
|
||||
force-reconnect drops the session up front so queued push refreshes can't
|
||||
starve it. Static-path resource and prompt catalogs are now size-capped
|
||||
like the pool path's (and like static tools) at discovery and on every
|
||||
refresh, so a misbehaving server's push can't balloon the node's merged
|
||||
catalogs. Deleting or reconfiguring a server can no longer leave it
|
||||
half-removed: the config removal and all cleanup are serialized under the
|
||||
connect lock (a cancelled removal completes its cleanup rather than
|
||||
stranding a live session and published catalog with the config already
|
||||
gone), and `reconcile_sync` retries a removal that timed out instead of
|
||||
marking it done — previously a DB-driven delete of a busy server could be a
|
||||
silent, permanent no-op until process restart. A refresh outcome now
|
||||
threads consistently to every operator surface off one source of truth
|
||||
(the per-server `last_refresh_outcome`): a busy-skip and a genuine failure
|
||||
are each reported distinctly from a real "no changes" — `/mcp refresh`
|
||||
prints "skipped" or "failed" rather than a false "no changes", and the
|
||||
node-internal refresh endpoint returns `202 skipped` instead of a
|
||||
misleading `200 ok` for a refresh that never ran. A single-kind push
|
||||
refresh no longer paints the whole server healthy: because the
|
||||
error/outcome state is server-scoped, a successful tools push while the
|
||||
prompts catalog is still broken (or vice versa) no longer clears the
|
||||
failure — only a full refresh pass declares "ok".
|
||||
|
||||
- **OpenAI Responses streaming: truncated and refused responses no longer
|
||||
vanish.** A response that hit `max_output_tokens` terminates the stream
|
||||
with `response.incomplete`, which the stream consumer did not handle —
|
||||
the turn was mislabeled `finish_reason: stop` and its final usage and
|
||||
collected output items were dropped. Refusal parts had no streaming
|
||||
handler at all, so a refusal rendered as empty content instead of the
|
||||
`[Refused: …]` text the non-streaming path produced. Both now match:
|
||||
truncation maps to `length` with usage/items intact, refusals render
|
||||
in content. Applies to the chat loop and every drained single-shot
|
||||
lane (#831).
|
||||
|
||||
- **task_agent: sub-tool ids no longer alias across a local model's reused
|
||||
ids.** A local model that reissues per-response sequential tool-call ids
|
||||
(`call_0` every turn) made two of a task agent's steps share one id — the
|
||||
live card collapsed both onto one DOM row while `/history` recall kept them
|
||||
apart, so the two views disagreed. Sub-tool ids are now minted
|
||||
`{parent}::r{run}s{step}::{id}`, unique within the session (across an
|
||||
agent's turns and across concurrent or sequential runs), and that one id
|
||||
keys the nesting registry, the live rows, recall, and the cancel ledger.
|
||||
On the wire the agent's self-built history carries the provider's own ids,
|
||||
restored from the mint map (see the reasoning-lane entry under Added), and
|
||||
malformed tool-call arguments are legalized the same way the main loop's
|
||||
wire prep does.
|
||||
|
||||
- **bash tool: never hang on a backgrounded child.** A command that left a
|
||||
long-lived process running (`server &`, a daemon) could wedge the whole
|
||||
workstream forever — the tool read stdout/stderr to EOF, which never arrived
|
||||
because the child inherited the pipe, and the timeout watchdog bailed once the
|
||||
foreground `bash` had exited. The tool now waits on the tracked process
|
||||
(bounded by the tool timeout) and terminates its whole process group on
|
||||
return, so the call always completes. Undecodable output is preserved
|
||||
(`errors="replace"`) instead of being dropped as a spurious error.
|
||||
- **Behavior change:** a process the command backgrounds no longer survives
|
||||
the call — nothing persists across bash invocations. (First-class
|
||||
"run this in the background" support landed separately — see
|
||||
`run_in_background` under Added.)
|
||||
- **`bash` never hangs on a backgrounded child** — a command that left a
|
||||
long-lived process running no longer wedges the workstream; the tool waits on
|
||||
the tracked process (bounded by the timeout) and reaps its whole process group.
|
||||
- **`task_agent` sub-tool ids are session-unique** — ids are minted
|
||||
`{parent}::r{run}s{step}::{id}` so a local model reissuing sequential ids
|
||||
(`call_0` each turn) no longer aliases two steps onto one live-card row while
|
||||
`/history` keeps them apart.
|
||||
- **Judge completions honour model-definition capabilities** — a judge's
|
||||
completion now threads its model's declared capabilities instead of assuming a
|
||||
default surface.
|
||||
- **`create-admin` CLI** — adds an explicit admin-creation command; `run.sh` no
|
||||
longer onboards into a role-less user.
|
||||
- **Install script Docker handling** — installs Docker on distros
|
||||
`get.docker.com` rejects, and gates that path by `$ID` instead of trapping all
|
||||
failures.
|
||||
|
||||
## [1.7.3]
|
||||
|
||||
|
||||
@@ -8,11 +8,6 @@ The following people have contributed code to the project — thank you:
|
||||
- Burhan ([@Burhan-Q](https://github.com/Burhan-Q))
|
||||
- chrismuzyn ([@chrismuzyn](https://github.com/chrismuzyn))
|
||||
- daoxley ([@daoxley](https://github.com/daoxley))
|
||||
- metaclassing ([@metaclassing](https://github.com/metaclassing))
|
||||
- posixpositive ([@bensonjohnson](https://github.com/bensonjohnson))
|
||||
- Robert DeAngelis ([@OriginalOrangeXD](https://github.com/OriginalOrangeXD))
|
||||
- Sanjay Santhanam ([@Sanjays2402](https://github.com/Sanjays2402))
|
||||
- Stefano Maffeis ([@lesbass](https://github.com/lesbass))
|
||||
- William ([@sillyWillieBilly](https://github.com/sillyWillieBilly))
|
||||
- [@BlackMyrmidon](https://github.com/BlackMyrmidon)
|
||||
- [@pizzaandcheese](https://github.com/pizzaandcheese)
|
||||
|
||||
+3
-7
@@ -8,7 +8,7 @@ FROM python:3.14-slim
|
||||
LABEL org.opencontainers.image.title="turnstone" \
|
||||
org.opencontainers.image.description="Multi-node AI orchestration platform"
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.12.1 /uv /usr/local/bin/uv
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.27 /uv /usr/local/bin/uv
|
||||
|
||||
# Remove the slim image's man page exclusion so man-db has actual content
|
||||
RUN rm -f /etc/dpkg/dpkg.cfg.d/docker
|
||||
@@ -55,17 +55,13 @@ COPY docker/healthcheck.py /usr/local/bin/healthcheck.py
|
||||
|
||||
# Entrypoint script — runs migrations before starting
|
||||
COPY docker/entrypoint.sh /usr/local/bin/entrypoint.sh
|
||||
RUN chmod +x /usr/local/bin/entrypoint.sh
|
||||
|
||||
# Data directory — SQLite DB is created in CWD
|
||||
WORKDIR /data
|
||||
RUN chown turnstone:turnstone /data
|
||||
|
||||
# Workspace mount point — bind-mount a host directory here. The env var
|
||||
# surfaces the path in the model's shell/file tool descriptions
|
||||
# (config.get_workspace_dir); without it the mount is invisible to the
|
||||
# model, whose cwd is /data below.
|
||||
# Workspace mount point — bind-mount a host directory here
|
||||
RUN mkdir -p /workspace && chown turnstone:turnstone /workspace
|
||||
ENV TURNSTONE_WORKSPACE=/workspace
|
||||
|
||||
USER turnstone
|
||||
|
||||
|
||||
@@ -17,17 +17,11 @@ Named after the [Ruddy Turnstone](https://en.wikipedia.org/wiki/Ruddy_turnstone)
|
||||
|
||||
**What is a harness?**
|
||||
|
||||
<p align="center">
|
||||
<a href="https://media.githubusercontent.com/media/turnstonelabs/turnstone/main/docs/diagrams/harness.png">
|
||||
<img src="https://media.githubusercontent.com/media/turnstonelabs/turnstone/main/docs/diagrams/harness.png" alt="ℋ : s_{n+1} ~ T(s_n) for n < τ_H — the whole controlled loop: π lowers state to context, M_W proposes a readout, γ authorizes it, Q_E acts on the world, ρ verifies and folds back" width="960"/>
|
||||
</a>
|
||||
</p>
|
||||
|
||||
```
|
||||
ℋ : s_{n+1} ~ T(s_n) for n < τ_H
|
||||
ℋ : s_{n+1} ~ T(s_n) for n < τ*, T = ρ ∘ (M_W ∘ π, E)
|
||||
```
|
||||
|
||||
[**the primer →**](PRIMER.md) · [**the formalism →**](HYPOTHESIS.md)
|
||||
[**the primer →**](PRIMER.md)
|
||||
|
||||
### Release Tracks
|
||||
|
||||
|
||||
@@ -2,11 +2,11 @@ apiVersion: v2
|
||||
name: turnstone
|
||||
description: Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation
|
||||
type: application
|
||||
version: 0.2.0
|
||||
version: 0.1.0
|
||||
appVersion: "0.3.0"
|
||||
|
||||
dependencies:
|
||||
- name: postgresql
|
||||
version: ~18.8.0
|
||||
version: ~18.7.0
|
||||
repository: https://charts.bitnami.com/bitnami
|
||||
condition: postgresql.enabled
|
||||
|
||||
@@ -110,153 +110,6 @@ Determine the PostgreSQL username.
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
The PostgreSQL password when the chart stores it itself, empty when it
|
||||
does not. Doubles as the predicate for "does <fullname>-secrets need to
|
||||
carry POSTGRES_PASSWORD", so an inline password is never written
|
||||
anywhere but <fullname>-secrets, and an operator-supplied Secret is
|
||||
never duplicated into it.
|
||||
|
||||
An operator-supplied existingSecret wins outright: writing the value
|
||||
into a second Secret nothing reads would only duplicate a credential.
|
||||
|
||||
Both branches need "default" because this is reached through include,
|
||||
which captures rendered text rather than a value: a key that is unset
|
||||
rather than empty — "password:" with nothing after it — renders as the
|
||||
literal "<no value>", and a ten-character string is truthy. Without the
|
||||
default that lands base64-encoded in POSTGRES_PASSWORD and the workloads
|
||||
authenticate with it.
|
||||
*/}}
|
||||
{{- define "turnstone.db.inlinePassword" -}}
|
||||
{{- if .Values.postgresql.enabled }}
|
||||
{{- .Values.postgresql.auth.password | default "" }}
|
||||
{{- else if not .Values.database.external.existingSecret }}
|
||||
{{- .Values.database.external.password | default "" }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
The name of the bundled subchart's own Secret.
|
||||
|
||||
Mirrors the subchart's naming rather than calling its helpers, which
|
||||
expect a context scoped to the subchart that this chart cannot hand
|
||||
them. Release-derived, so deliberately not turnstone.fullname: a
|
||||
fullnameOverride here renames this chart's resources and leaves the
|
||||
subchart's alone, and pointing at "<fullname>-postgresql" would then
|
||||
name a Secret that does not exist.
|
||||
|
||||
The subchart also normalises the release name through a regex before
|
||||
using it, which is a no-op for the DNS-1123 names Helm accepts, so it is
|
||||
not reproduced.
|
||||
*/}}
|
||||
{{- define "turnstone.postgresql.fullname" -}}
|
||||
{{- $global := ((.Values.global).postgresql).fullnameOverride }}
|
||||
{{- if $global }}
|
||||
{{- $global | trunc 63 | trimSuffix "-" }}
|
||||
{{- else if .Values.postgresql.fullnameOverride }}
|
||||
{{- .Values.postgresql.fullnameOverride | trunc 63 | trimSuffix "-" }}
|
||||
{{- else }}
|
||||
{{- $name := .Values.postgresql.nameOverride | default "postgresql" }}
|
||||
{{- if contains $name .Release.Name }}
|
||||
{{- .Release.Name | trunc 63 | trimSuffix "-" }}
|
||||
{{- else }}
|
||||
{{- printf "%s-%s" .Release.Name $name | trunc 63 | trimSuffix "-" }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{- define "turnstone.postgresql.secretName" -}}
|
||||
{{- $existing := coalesce (((.Values.global).postgresql).auth).existingSecret .Values.postgresql.auth.existingSecret }}
|
||||
{{- if $existing }}
|
||||
{{- tpl $existing . }}
|
||||
{{- else }}
|
||||
{{- include "turnstone.postgresql.fullname" . }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
The subchart stores the named user's password under "password" and the
|
||||
superuser's under "postgres-password", and lets an operator rename
|
||||
either through auth.secretKeys.
|
||||
*/}}
|
||||
{{- define "turnstone.postgresql.passwordKey" -}}
|
||||
{{- $user := .Values.postgresql.auth.username | default "" }}
|
||||
{{- $keys := .Values.postgresql.auth.secretKeys | default dict }}
|
||||
{{- if or (empty $user) (eq $user "postgres") }}
|
||||
{{- $keys.adminPasswordKey | default "postgres-password" }}
|
||||
{{- else }}
|
||||
{{- $keys.userPasswordKey | default "password" }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
Determine the secret holding the PostgreSQL password, and the key within
|
||||
it. Three sources, and the two helpers agree by construction because
|
||||
they branch identically:
|
||||
|
||||
- an external database pointed at a Secret the chart does not own (a
|
||||
CloudNativePG-generated secret, an External Secrets target, ...), in
|
||||
which case the key is rarely "POSTGRES_PASSWORD" — hence the
|
||||
companion existingSecretPasswordKey
|
||||
- the bundled subchart's own Secret, when it generates the password
|
||||
- <fullname>-secrets, when the password is supplied inline in values
|
||||
|
||||
Note the last is deliberately not turnstone.llm.secretName: that
|
||||
resolves to llm.existingSecret when the operator supplies one, which
|
||||
holds LLM API keys and has no reason to carry a database password.
|
||||
*/}}
|
||||
{{- define "turnstone.db.secretName" -}}
|
||||
{{- if not .Values.postgresql.enabled }}
|
||||
{{- if .Values.database.external.existingSecret }}
|
||||
{{- .Values.database.external.existingSecret }}
|
||||
{{- else }}
|
||||
{{- printf "%s-secrets" (include "turnstone.fullname" .) }}
|
||||
{{- end }}
|
||||
{{- else if include "turnstone.db.inlinePassword" . }}
|
||||
{{- printf "%s-secrets" (include "turnstone.fullname" .) }}
|
||||
{{- else }}
|
||||
{{- include "turnstone.postgresql.secretName" . }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{- define "turnstone.db.passwordKey" -}}
|
||||
{{- if not .Values.postgresql.enabled }}
|
||||
{{- if .Values.database.external.existingSecret }}
|
||||
{{- .Values.database.external.existingSecretPasswordKey | default "password" }}
|
||||
{{- else }}
|
||||
{{- printf "POSTGRES_PASSWORD" }}
|
||||
{{- end }}
|
||||
{{- else if include "turnstone.db.inlinePassword" . }}
|
||||
{{- printf "POSTGRES_PASSWORD" }}
|
||||
{{- else }}
|
||||
{{- include "turnstone.postgresql.passwordKey" . }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
Database environment shared by the server, console and migrate Job.
|
||||
|
||||
Every value except the password is rendered inline rather than pulled
|
||||
from the ConfigMap via envFrom, so that one definition serves all three
|
||||
workloads and the URL is assembled in exactly one place.
|
||||
|
||||
POSTGRES_PASSWORD must still precede TURNSTONE_DB_URL: the kubelet
|
||||
expands $(VAR) only against env entries declared earlier in the list, so
|
||||
a later definition would leave a literal "$(POSTGRES_PASSWORD)" in the
|
||||
URL.
|
||||
*/}}
|
||||
{{- define "turnstone.db.env" -}}
|
||||
- name: TURNSTONE_DB_BACKEND
|
||||
value: {{ .Values.database.backend | quote }}
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: {{ include "turnstone.db.secretName" . }}
|
||||
key: {{ include "turnstone.db.passwordKey" . }}
|
||||
- name: TURNSTONE_DB_URL
|
||||
value: "postgresql+psycopg://{{ include "turnstone.postgresql.username" . }}:$(POSTGRES_PASSWORD)@{{ include "turnstone.postgresql.host" . }}:{{ include "turnstone.postgresql.port" . }}/{{ include "turnstone.postgresql.database" . }}{{ if and (not .Values.postgresql.enabled) .Values.database.external.sslmode }}?sslmode={{ .Values.database.external.sslmode }}{{ end }}"
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
Determine the secret name for LLM API keys.
|
||||
*/}}
|
||||
|
||||
@@ -7,17 +7,6 @@ metadata:
|
||||
app.kubernetes.io/component: console
|
||||
spec:
|
||||
replicas: {{ .Values.console.replicas }}
|
||||
{{- if eq (int .Values.console.replicas) 1 }}
|
||||
# The console registers itself under the fixed service_id "console" and
|
||||
# deregisters on shutdown. Under RollingUpdate the outgoing pod's
|
||||
# deregister runs *after* the incoming pod registers and deletes its
|
||||
# row -- and the heartbeat only touches last_heartbeat, so the row is
|
||||
# never recreated and the console stays invisible in the registry until
|
||||
# the next clean start. Recreate orders shutdown strictly before
|
||||
# startup. Only valid at one replica; see console.replicas.
|
||||
strategy:
|
||||
type: Recreate
|
||||
{{- end }}
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- include "turnstone.selectorLabels" . | nindent 6 }}
|
||||
@@ -29,18 +18,6 @@ spec:
|
||||
app.kubernetes.io/component: console
|
||||
spec:
|
||||
serviceAccountName: {{ include "turnstone.serviceAccountName" . }}
|
||||
{{- with .Values.console.nodeSelector }}
|
||||
nodeSelector:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.console.affinity }}
|
||||
affinity:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.console.tolerations }}
|
||||
tolerations:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: console
|
||||
image: {{ include "turnstone.image" . }}
|
||||
@@ -59,18 +36,8 @@ spec:
|
||||
- secretRef:
|
||||
name: {{ include "turnstone.llm.secretName" . }}
|
||||
optional: true
|
||||
env:
|
||||
{{- include "turnstone.db.env" . | nindent 12 }}
|
||||
# Self-registration URL for the service registry. Unlike a
|
||||
# server node the console is one logical endpoint behind its
|
||||
# Service, so the Service DNS name is correct here. Without
|
||||
# it the console registers gethostname() (its pod name),
|
||||
# which no server node can resolve. Stops at ".svc" rather
|
||||
# than assuming a "cluster.local" DNS domain, which is
|
||||
# configurable per cluster.
|
||||
- name: TURNSTONE_CONSOLE_URL
|
||||
value: "http://{{ include "turnstone.fullname" . }}-console.{{ .Release.Namespace }}.svc:{{ .Values.console.service.port }}"
|
||||
{{- if or .Values.auth.existingSecret .Values.auth.jwtSecret }}
|
||||
env:
|
||||
- name: TURNSTONE_JWT_SECRET
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
|
||||
@@ -18,18 +18,6 @@ spec:
|
||||
app.kubernetes.io/component: server
|
||||
spec:
|
||||
serviceAccountName: {{ include "turnstone.serviceAccountName" . }}
|
||||
{{- with .Values.server.nodeSelector }}
|
||||
nodeSelector:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.server.affinity }}
|
||||
affinity:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.server.tolerations }}
|
||||
tolerations:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: server
|
||||
image: {{ include "turnstone.image" . }}
|
||||
@@ -51,20 +39,8 @@ spec:
|
||||
name: {{ include "turnstone.llm.secretName" . }}
|
||||
optional: true
|
||||
env:
|
||||
{{- include "turnstone.db.env" . | nindent 12 }}
|
||||
# Each replica is a distinct node in the rendezvous ring, so it
|
||||
# must advertise an address that reaches *itself*. The Service
|
||||
# DNS name would load-balance across every replica, sending
|
||||
# console traffic routed for node A to an arbitrary pod; the
|
||||
# default (gethostname(), i.e. the pod name) is not resolvable
|
||||
# at all. The pod IP is unique, routable in-cluster, and
|
||||
# re-registered on every start, so churn is self-healing.
|
||||
- name: POD_IP
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: status.podIP
|
||||
- name: TURNSTONE_ADVERTISE_URL
|
||||
value: "http://$(POD_IP):{{ .Values.server.service.port }}"
|
||||
- name: TURNSTONE_DB_URL
|
||||
value: "postgresql+psycopg://$(TURNSTONE_DB_USER):$(POSTGRES_PASSWORD)@$(TURNSTONE_DB_HOST):$(TURNSTONE_DB_PORT)/$(TURNSTONE_DB_NAME)"
|
||||
{{- if or .Values.auth.existingSecret .Values.auth.jwtSecret }}
|
||||
- name: TURNSTONE_JWT_SECRET
|
||||
valueFrom:
|
||||
|
||||
@@ -6,23 +6,11 @@ metadata:
|
||||
{{- include "turnstone.labels" . | nindent 4 }}
|
||||
app.kubernetes.io/component: migrate
|
||||
annotations:
|
||||
# post-install, not pre-install: on a first install nothing the
|
||||
# migration needs exists yet — not the ConfigMap, not the Secret, and
|
||||
# with the bundled subchart not the database either, since Helm
|
||||
# creates ordinary resources only once hooks have finished. On an
|
||||
# upgrade all of it is already running, so pre-upgrade is both safe
|
||||
# and preferable: migrations land before the new code rolls out
|
||||
# rather than after.
|
||||
"helm.sh/hook": post-install,pre-upgrade
|
||||
"helm.sh/hook": pre-install,pre-upgrade
|
||||
"helm.sh/hook-weight": "-1"
|
||||
"helm.sh/hook-delete-policy": before-hook-creation,hook-succeeded
|
||||
spec:
|
||||
# Helm does not wait for the database to be ready before running
|
||||
# post-install hooks, so on a first install this Job is what waits: it
|
||||
# exits non-zero until PostgreSQL accepts connections, and the retry
|
||||
# budget has to cover a cold StatefulSet pulling its image and
|
||||
# initialising.
|
||||
backoffLimit: 10
|
||||
backoffLimit: 3
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -31,18 +19,6 @@ spec:
|
||||
spec:
|
||||
serviceAccountName: {{ include "turnstone.serviceAccountName" . }}
|
||||
restartPolicy: OnFailure
|
||||
{{- with .Values.migrate.nodeSelector }}
|
||||
nodeSelector:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.migrate.affinity }}
|
||||
affinity:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.migrate.tolerations }}
|
||||
tolerations:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: migrate
|
||||
image: {{ include "turnstone.image" . }}
|
||||
@@ -51,5 +27,12 @@ spec:
|
||||
- python
|
||||
- -m
|
||||
- turnstone.core.storage._migrate
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: {{ include "turnstone.fullname" . }}-config
|
||||
- secretRef:
|
||||
name: {{ include "turnstone.llm.secretName" . }}
|
||||
optional: true
|
||||
env:
|
||||
{{- include "turnstone.db.env" . | nindent 12 }}
|
||||
- name: TURNSTONE_DB_URL
|
||||
value: "postgresql+psycopg://$(TURNSTONE_DB_USER):$(POSTGRES_PASSWORD)@$(TURNSTONE_DB_HOST):$(TURNSTONE_DB_PORT)/$(TURNSTONE_DB_NAME)"
|
||||
|
||||
@@ -1,19 +1,4 @@
|
||||
{{/*
|
||||
This Secret backs every credential supplied inline in values, so it is
|
||||
rendered whenever any one of them is set — not, as it once was, only
|
||||
when llm.existingSecret is empty. Under that older gate an operator who
|
||||
supplied an LLM Secret lost the unrelated inline values with it: both
|
||||
POSTGRES_PASSWORD and TURNSTONE_JWT_SECRET silently went unrendered
|
||||
while the workloads went on referencing them, so every pod stalled in
|
||||
CreateContainerConfigError.
|
||||
|
||||
Each key keeps its own condition, so an operator-supplied Secret still
|
||||
suppresses the value it replaces and nothing else.
|
||||
*/}}
|
||||
{{- $apiKey := and .Values.llm.apiKey (not .Values.llm.existingSecret) }}
|
||||
{{- $dbPassword := include "turnstone.db.inlinePassword" . }}
|
||||
{{- $jwtSecret := and .Values.auth.jwtSecret (not .Values.auth.existingSecret) }}
|
||||
{{- if or $apiKey $dbPassword $jwtSecret }}
|
||||
{{- if not .Values.llm.existingSecret }}
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
@@ -22,13 +7,15 @@ metadata:
|
||||
{{- include "turnstone.labels" . | nindent 4 }}
|
||||
type: Opaque
|
||||
data:
|
||||
{{- if $apiKey }}
|
||||
{{- if .Values.llm.apiKey }}
|
||||
OPENAI_API_KEY: {{ .Values.llm.apiKey | b64enc | quote }}
|
||||
{{- end }}
|
||||
{{- if $dbPassword }}
|
||||
POSTGRES_PASSWORD: {{ $dbPassword | b64enc | quote }}
|
||||
{{- if and .Values.postgresql.enabled .Values.postgresql.auth.password }}
|
||||
POSTGRES_PASSWORD: {{ .Values.postgresql.auth.password | b64enc | quote }}
|
||||
{{- else if and (not .Values.postgresql.enabled) .Values.database.external.password }}
|
||||
POSTGRES_PASSWORD: {{ .Values.database.external.password | b64enc | quote }}
|
||||
{{- end }}
|
||||
{{- if $jwtSecret }}
|
||||
{{- if and .Values.auth.jwtSecret (not .Values.auth.existingSecret) }}
|
||||
TURNSTONE_JWT_SECRET: {{ .Values.auth.jwtSecret | b64enc | quote }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
@@ -14,13 +14,7 @@ database:
|
||||
port: 5432
|
||||
database: turnstone
|
||||
username: turnstone
|
||||
# Secret holding the password for `username`. Leave empty to supply
|
||||
# `password` inline below instead.
|
||||
existingSecret: ""
|
||||
# Key within existingSecret holding the password. CloudNativePG
|
||||
# generates "password"; other operators differ.
|
||||
existingSecretPasswordKey: password
|
||||
password: ""
|
||||
sslmode: prefer
|
||||
|
||||
# -- Bitnami PostgreSQL subchart
|
||||
@@ -43,10 +37,6 @@ server:
|
||||
service:
|
||||
type: ClusterIP
|
||||
port: 8080
|
||||
# -- Node scheduling constraints
|
||||
nodeSelector: {}
|
||||
affinity: {}
|
||||
tolerations: []
|
||||
|
||||
# -- Turnstone console (cluster dashboard)
|
||||
console:
|
||||
@@ -61,17 +51,6 @@ console:
|
||||
service:
|
||||
type: ClusterIP
|
||||
port: 8090
|
||||
# -- Node scheduling constraints
|
||||
nodeSelector: {}
|
||||
affinity: {}
|
||||
tolerations: []
|
||||
|
||||
# -- Database migration Job (post-install/pre-upgrade hook)
|
||||
migrate:
|
||||
# -- Node scheduling constraints
|
||||
nodeSelector: {}
|
||||
affinity: {}
|
||||
tolerations: []
|
||||
|
||||
# -- LLM provider configuration
|
||||
llm:
|
||||
|
||||
+28
-132
@@ -467,46 +467,6 @@ Each item in `items` (shared by `tool_info` and `approve_request`):
|
||||
{"type": "info", "message": "Session cleared."}
|
||||
```
|
||||
|
||||
**`compaction`** -- context-compaction lifecycle (manual `/compact` and
|
||||
auto-compaction). `phase: "start"` opens the operation (`trigger` is
|
||||
`"manual"` or `"auto"`; auto adds `where` — e.g. `"mid-turn"` — and, when
|
||||
the percentage threshold actually fired, `pct`; the context-overflow retry
|
||||
path compacts without a `pct` since no threshold was evaluated).
|
||||
`phase: "progress"` reports chunked summarization (`part`/`total`/`depth`,
|
||||
where depth 0 summarizes transcript batches and deeper levels merge partial
|
||||
summaries), a transient-error retry wait (`retry_in` seconds + `error`), or
|
||||
`warning: "summary_truncated"`. `phase: "end"` settles it: `ok: true`
|
||||
carries `before_tokens`/`after_tokens` and the produced `summary`;
|
||||
`ok: false` carries a `reason`
|
||||
(`"not_enough_messages"` / `"irreducible"` / `"empty_summary"` /
|
||||
`"cancelled"` / `"error"`) and a human-readable `message` — for
|
||||
`reason: "error"` the same message is also emitted as a paired typed
|
||||
`error` event (that is the renderable error surface; the end event is
|
||||
card-teardown). Failed ends also carry `notice`: the emitter-computed
|
||||
display verdict — show `message` only when it is `true` (the server
|
||||
suppresses error-reason, superseded, and cancelled-auto notices once,
|
||||
centrally, so clients don't re-derive that policy). Every end (ok or
|
||||
failed) carries `trigger`, and every event carries `compaction_id` — an
|
||||
opaque integer correlating the start/progress/end of one compaction run (a
|
||||
client that force-stopped one compaction can use it to ignore stragglers
|
||||
from the abandoned run). End events also carry `superseded`: `true` marks
|
||||
a force-abandoned compaction retiring after a successor generation took
|
||||
over (an OK end's result card still stands: the history swap happened).
|
||||
Superseded start/progress events are never emitted.
|
||||
Exactly one `start` and one `end` are emitted per attempt,
|
||||
so clients can key an in-progress affordance (progress bar) on the pair. A
|
||||
successful end is also persisted: the summary replays from `/history` as a
|
||||
`role: "system"`, `source: "compaction"` entry whose `meta` carries
|
||||
`{watermark, before_tokens, after_tokens, trigger}` and whose `event_id`
|
||||
matches the end event's id (dedup across repaint + replay).
|
||||
|
||||
```json
|
||||
{"type": "compaction", "phase": "start", "compaction_id": 7, "trigger": "auto", "where": "mid-turn", "pct": 80}
|
||||
{"type": "compaction", "phase": "progress", "compaction_id": 7, "part": 2, "total": 5, "depth": 0}
|
||||
{"type": "compaction", "phase": "end", "ok": true, "compaction_id": 7, "trigger": "auto",
|
||||
"before_tokens": 128400, "after_tokens": 9200, "summary": "## Decisions\n..."}
|
||||
```
|
||||
|
||||
**`error`** -- an error message.
|
||||
|
||||
```json
|
||||
@@ -788,41 +748,32 @@ Sends a user message to a workstream. Spawns a daemon worker thread that calls
|
||||
**Request body:**
|
||||
|
||||
```json
|
||||
{"message": "Explain how the server works", "attachment_ids": ["a1"]}
|
||||
{"message": "Explain how the server works"}
|
||||
```
|
||||
|
||||
| Field | Type | Required | Description |
|
||||
|------------------|------------|----------|------------------------------------------------------|
|
||||
| `message` | string | yes | The user's message text |
|
||||
| `attachment_ids` | string[] | no | Staged uploads to attach (omit = auto-consume; `[]` = none) |
|
||||
| Field | Type | Required | Description |
|
||||
|-----------|--------|----------|-------------------------|
|
||||
| `message` | string | yes | The user's message text |
|
||||
|
||||
**Response.** Every 200 body carries `attached_ids` and
|
||||
`dropped_attachment_ids` (empty lists when no attachments are involved):
|
||||
**Response (success):**
|
||||
|
||||
- `{"status": "ok", ...}` — a fresh turn was dispatched.
|
||||
- `{"status": "queued", "priority", "msg_id", ...}` — folded into the live
|
||||
turn's interjection queue; delivered at the next tool-result seam.
|
||||
`DELETE .../send` with the `msg_id` retracts it before delivery.
|
||||
- `{"status": "queued", "deferred": true, ...}` — parked on the deferred-send
|
||||
list (a command window holds the slot, or earlier deferred sends are
|
||||
pending) and dispatched as its own full-fidelity send afterwards; see the
|
||||
defer contract under `POST /v1/api/command`.
|
||||
- `{"status": "queue_full", ...}` — the send was refused with retry-shortly
|
||||
semantics: the live worker's interjection queue is at capacity, the
|
||||
deferred-send list hit its saturation bound (10 pending — the same
|
||||
backpressure contract), or the deferred-send drain could not be started
|
||||
under resource exhaustion (the message was **not** accepted; nothing is
|
||||
parked).
|
||||
- `{"status": "attachments_busy", ...}` — attachments can't ride a queued
|
||||
turn; the staged uploads survive for a retry once the worker idles.
|
||||
```json
|
||||
{"status": "ok"}
|
||||
```
|
||||
|
||||
**Response (busy):** Returned if the workstream's worker thread is still alive
|
||||
from a previous request. Also pushes a `busy_error` event to the SSE stream.
|
||||
|
||||
```json
|
||||
{"status": "busy"}
|
||||
```
|
||||
|
||||
**Error responses:**
|
||||
|
||||
| Status | Body | Condition |
|
||||
|--------|-------------------------------------------------|----------------------------------------|
|
||||
| 400 | `{"error": "message is required"}` | Message is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found (or closed mid-send) |
|
||||
| 409 | `{"status": "cross_user_interjection", ...}` | Another participant's turn is in flight |
|
||||
| Status | Body | Condition |
|
||||
|--------|------------------------------------|------------------------|
|
||||
| 400 | `{"error": "Empty message"}` | Message is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
|
||||
---
|
||||
|
||||
@@ -865,56 +816,7 @@ automatically approved without prompting.
|
||||
|
||||
### `POST /v1/api/command`
|
||||
|
||||
Executes a slash command in the given workstream. Commands run on the
|
||||
workstream's worker slot (mutual exclusion against sends, a running
|
||||
compaction, and each other) — the endpoint is **not** unconditionally
|
||||
synchronous:
|
||||
|
||||
- **Quick commands** (everything except `/compact`): the endpoint waits for
|
||||
completion, so `{"status": "ok"}` means the command ran. A command still
|
||||
running after 25 s answers `{"status": "running"}` — the worker keeps
|
||||
going, its output reaches the pane via SSE, and the post-command pane
|
||||
refreshes below still fire when it completes. (The bound sits under
|
||||
common 30 s client/proxy timeouts — the console proxy's included — so
|
||||
the degraded answer actually reaches bounded callers.)
|
||||
- **`/compact`**: dispatched fire-and-forget — `{"status": "ok"}` means the
|
||||
compaction *started*. A large context can legitimately compact for many
|
||||
minutes; progress streams as `compaction` SSE events (see the event
|
||||
reference) and the persisted marker row lands on completion. Do not read
|
||||
`/history` expecting the compacted transcript immediately after the
|
||||
response.
|
||||
- **Busy refusal**: if a turn or another command holds the worker slot, the
|
||||
command is refused with HTTP **409** `{"status": "busy", "error": ...}` and
|
||||
did **not** run. Retry after the current turn finishes. (The old inline
|
||||
endpoint executed commands unconditionally mid-turn; the 409 makes the
|
||||
refusal loud for callers that only check the HTTP status.)
|
||||
|
||||
While a command holds the slot — and afterwards, while earlier deferred
|
||||
sends are still waiting (the pending list is the order authority: a fresh
|
||||
send never overtakes a message already acknowledged) — `POST .../send`
|
||||
requests are **deferred**: the server answers `{"status": "queued",
|
||||
"deferred": true, "msg_id": ...}` immediately and dispatches the message
|
||||
as an ordinary full-fidelity send (attachments and sender identity
|
||||
included) in arrival order once the slot frees — it is never routed
|
||||
through the mid-turn interjection queue (no length cap, no cross-user
|
||||
rejection). The response arrives within normal round-trip time, so
|
||||
timeout-bounded clients (SDKs, proxies, the coordinator) need no special
|
||||
handling. To retract a deferred send before it dispatches, issue the same
|
||||
`DELETE .../send` with its `msg_id` used for queued interjections —
|
||||
`{"status": "removed"}` confirms it will not dispatch; `"not_found"` means
|
||||
it already dispatched (or is dispatching). Retracting a deferred send
|
||||
discards any attachments it carried; re-attach to send them again. When a
|
||||
deferred send dispatches, panes receive a `message_dispatched` event
|
||||
(`msg_id`, plus `folded: true` when it folded into a live turn's
|
||||
interjection queue rather than spawning its own turn) so queued-message
|
||||
UI can settle the right way.
|
||||
|
||||
Durability: deferred sends are **node-local and in-memory** (the same
|
||||
lifetime as the interjection queue). `"queued"` is at-most-once intake, not
|
||||
durable acceptance — if the workstream is closed or the node restarts before
|
||||
the window ends, the message is dropped. Anything that must survive a
|
||||
restart should be re-sent after confirming dispatch (the turn appears on the
|
||||
SSE stream / in `/history`).
|
||||
Executes a slash command in the given workstream.
|
||||
|
||||
**Request body:**
|
||||
|
||||
@@ -927,12 +829,10 @@ SSE stream / in `/history`).
|
||||
| `command` | string | yes | The slash command (e.g. `/clear`) |
|
||||
| `ws_id` | string | yes | Target workstream ID |
|
||||
|
||||
If the command is `/clear`, `/new`, or `/resume`, the server pushes a
|
||||
`clear_ui` SSE event to instruct the client to reset its message display and
|
||||
re-fetch the transcript via `GET .../history` (there is no SSE event that
|
||||
carries the messages themselves). These follow-ups are emitted by the
|
||||
command worker itself, so they fire even when the endpoint already answered
|
||||
`{"status": "running"}`.
|
||||
If the command is `/clear` or `/new`, the server pushes a `clear_ui` SSE event
|
||||
to instruct the client to reset its message display. If the command is
|
||||
`/resume`, the server pushes `clear_ui` followed by a `history` event
|
||||
containing the resumed session's messages.
|
||||
|
||||
**Response:**
|
||||
|
||||
@@ -940,16 +840,12 @@ command worker itself, so they fire even when the endpoint already answered
|
||||
{"status": "ok"}
|
||||
```
|
||||
|
||||
or `{"status": "running"}` as above.
|
||||
|
||||
**Error responses:**
|
||||
|
||||
| Status | Body | Condition |
|
||||
|--------|-------------------------------------|--------------------------------------------------|
|
||||
| 400 | `{"error": "Empty command"}` | Command is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
| 409 | `{"status": "busy", "error": ...}` | A turn/command holds the worker |
|
||||
| 503 | `{"status": "error", "error": ...}` | The command worker could not be started (resource exhaustion) — the command did **not** run; retry shortly |
|
||||
| Status | Body | Condition |
|
||||
|--------|------------------------------------|----------------------|
|
||||
| 400 | `{"error": "Empty command"}` | Command is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
|
||||
---
|
||||
|
||||
|
||||
+51
-100
@@ -91,7 +91,7 @@ turnstone/
|
||||
discord/ Discord adapter (bot, cog, views, streaming, config)
|
||||
slack/ Slack adapter (Socket Mode bot, DM routing, approval buttons)
|
||||
shared_static/ Shared design system (base.css, auth.js, theme.js, toast.js, utils.js, kb.js)
|
||||
katex-0.18.1/ Vendored KaTeX math rendering library (MIT, woff2 fonts)
|
||||
katex-0.17.0/ Vendored KaTeX math rendering library (MIT, woff2 fonts)
|
||||
ui/
|
||||
colors.py ANSI color constants with NO_COLOR support
|
||||
markdown.py Streaming terminal markdown renderer (line-buffered)
|
||||
@@ -128,16 +128,13 @@ A user message flows through the system as follows:
|
||||
_emit_state("thinking")
|
||||
|
|
||||
v
|
||||
_stream_response() -------------> model_turn(lane, turns, on_chunk=...) per attempt
|
||||
| lane-swap fallback walk; per-lane ladder:
|
||||
_create_stream_with_retry() ----> provider.create_streaming(client, model, messages, ...)
|
||||
| up to 3 retries (4 total attempts), exponential backoff
|
||||
v
|
||||
the on_chunk consumer -----------> display grid ONLY:
|
||||
_stream_response(stream) --------> dispatch tokens to UI:
|
||||
| on_reasoning_token() / on_content_token()
|
||||
| tool-call deltas just flush the splitter
|
||||
| (assembly lives in drain_stream, inside
|
||||
| model_turn — the consumer never accumulates)
|
||||
| track finish_reason (citations-footer gate)
|
||||
| accumulate tool_calls from deltas
|
||||
| track finish_reason
|
||||
| _check_cancelled() per chunk (cooperative cancel)
|
||||
v
|
||||
finish_reason check:
|
||||
@@ -246,9 +243,7 @@ class SessionUI(Protocol):
|
||||
def on_content_token(self, text: str) -> None: ...
|
||||
def on_stream_end(self) -> None: ...
|
||||
def approve_tools(self, items: list[dict]) -> tuple[bool, str | None]: ...
|
||||
def on_tool_result(
|
||||
self, call_id: str, name: str, output: str, *, is_error: bool = False
|
||||
) -> None: ...
|
||||
def on_tool_result(self, call_id: str, name: str, output: str, *, is_error: bool = False) -> None: ...
|
||||
def on_tool_output_chunk(self, call_id: str, chunk: str) -> None: ...
|
||||
def on_status(self, usage: dict, context_window: int, effort: str) -> None: ...
|
||||
def on_info(self, message: str) -> None: ...
|
||||
@@ -318,15 +313,15 @@ ERROR last operation failed
|
||||
```python
|
||||
@dataclass
|
||||
class Workstream:
|
||||
id: str # uuid hex, 8 chars
|
||||
name: str # user-visible label
|
||||
state: WorkstreamState # current state
|
||||
session: ChatSession | None # the conversation engine
|
||||
ui: SessionUI | None # frontend adapter
|
||||
id: str # uuid hex, 8 chars
|
||||
name: str # user-visible label
|
||||
state: WorkstreamState # current state
|
||||
session: ChatSession | None # the conversation engine
|
||||
ui: SessionUI | None # frontend adapter
|
||||
worker_thread: threading.Thread | None
|
||||
error_message: str
|
||||
last_active: float # time.monotonic() timestamp, updated on every state change
|
||||
_lock: threading.Lock # per-workstream state lock
|
||||
last_active: float # time.monotonic() timestamp, updated on every state change
|
||||
_lock: threading.Lock # per-workstream state lock
|
||||
```
|
||||
|
||||
### WorkstreamManager
|
||||
@@ -338,9 +333,7 @@ class WorkstreamManager:
|
||||
def __init__(self, session_factory: Callable[[SessionUI], ChatSession]): ...
|
||||
def create(self, name="", ui_factory=None) -> Workstream: ...
|
||||
def close(self, ws_id: str) -> bool: ...
|
||||
def close_idle(
|
||||
self, max_age_seconds: float
|
||||
) -> list[str]: ... # auto-close stale IDLE workstreams
|
||||
def close_idle(self, max_age_seconds: float) -> list[str]: ... # auto-close stale IDLE workstreams
|
||||
def get(self, ws_id: str) -> Workstream | None: ...
|
||||
def get_active(self) -> Workstream | None: ...
|
||||
def list_all(self) -> list[Workstream]: ...
|
||||
@@ -515,7 +508,7 @@ then returns the final content as the tool result.
|
||||
response without tools. When unlimited, the loop only exits when the model
|
||||
stops calling tools or hits `finish_reason: "length"`.
|
||||
- **Retry**: each API call in the agent loop uses the same retry+backoff logic
|
||||
as the main loop's per-lane ladder (`_model_turn_with_retry`).
|
||||
as the main `_create_stream_with_retry()`.
|
||||
- **Finish reason handling**: `finish_reason: "length"` stops the agent early
|
||||
and returns whatever content was generated. `finish_reason: "content_filter"`
|
||||
returns a placeholder.
|
||||
@@ -616,7 +609,8 @@ LLMProvider (protocol)
|
||||
|
||||
| Method | Purpose |
|
||||
|--------|---------|
|
||||
| `create_streaming()` | The one transport: streaming request, yields normalized `StreamChunk` objects (single-shot callers accumulate via `drain_stream()` into a `CompletionResult`) |
|
||||
| `create_streaming()` | Streaming request, yields normalized `StreamChunk` objects |
|
||||
| `create_completion()` | Non-streaming request, returns `CompletionResult` |
|
||||
| `get_capabilities()` | Per-model flags (`ModelCapabilities`) |
|
||||
| `convert_tools()` | Translate OpenAI tool schemas to provider format |
|
||||
| `retryable_error_names` | Exception class names that trigger retry |
|
||||
@@ -669,7 +663,7 @@ display). Automatic prompt caching is enabled via top-level `cache_control:
|
||||
cacheable block and advances it as conversations grow (90% input cost
|
||||
reduction on cache hits, 1.25x write on first turn). Cache metrics
|
||||
(`cache_creation_input_tokens`, `cache_read_input_tokens`) are extracted from
|
||||
the stream's usage events. The `anthropic` SDK is a core
|
||||
both streaming and non-streaming responses. The `anthropic` SDK is a core
|
||||
dependency — the Anthropic provider is first-class alongside OpenAI.
|
||||
|
||||
**GoogleProvider** (`_google.py`): extends `OpenAIChatCompletionsProvider` for
|
||||
@@ -948,8 +942,8 @@ with the same alias in-memory (the DB rows are never modified).
|
||||
5. `/model` command shows available models; `/model <alias>` switches the
|
||||
active workstream's client, model, context window, and per-model sampling
|
||||
parameters
|
||||
6. `_model_turn_with_fallback()` tries the primary lane, then each fallback
|
||||
alias's lane in order if the primary is unreachable
|
||||
6. `_create_stream_with_retry()` tries the primary model, then each fallback
|
||||
alias in order if the primary is unreachable
|
||||
7. `_run_agent()` resolves `registry.agent_model` (if set) for task
|
||||
sub-agents, allowing a cheaper model for autonomous loops
|
||||
|
||||
@@ -974,33 +968,6 @@ The default limit is 50% of the context window in characters (computed as
|
||||
|
||||
This truncation message is visible to the model, so it knows output was cut.
|
||||
|
||||
During the send loop the limit is additionally capped by the remaining
|
||||
context budget, and three guarantees apply when that budget reaches zero
|
||||
(#883):
|
||||
|
||||
- **Structural floor** — orchestration handles (`spawn_workstream`,
|
||||
`spawn_batch`, `wait_for_workstream`, `tasks`) and error results are
|
||||
always admitted up to a guaranteed floor (2048 chars, head+tail beyond
|
||||
it), because a lost `ws_id` or a masked failure wedges the session.
|
||||
- **Small-result pass** — results at or under the floor pass verbatim,
|
||||
funded from a bounded per-batch grace pool (2× the floor) so a wide
|
||||
batch of small results cannot collectively bypass budget accounting;
|
||||
past the pool they get the drop notice instead.
|
||||
- **Honest drop notice** — a bulky non-structural result is replaced by an
|
||||
explicit `Error: tool result dropped — context budget exhausted…` notice
|
||||
stating the call ran but its output could not be admitted (never a
|
||||
successful-looking trim).
|
||||
|
||||
A zero budget also triggers one mid-turn auto-compaction before results are
|
||||
sized. With `max_tokens ≥ context_window/4` the response reserve zeroes the
|
||||
budget near 70% fullness — below the default 80% auto-compact threshold —
|
||||
and without this trigger a session could idle in that band indefinitely
|
||||
with every tool result floored or dropped. The trigger keys on the
|
||||
exhausted budget itself, not on any threshold, so it composes with any
|
||||
operator-set `auto_compact_pct`: with thresholds below the zero point the
|
||||
ordinary owed-compaction paths fire first and this trigger degrades to a
|
||||
backstop for the cases where they bailed or freed too little.
|
||||
|
||||
---
|
||||
|
||||
## Persistence
|
||||
@@ -1183,30 +1150,20 @@ Named (aliased) workstreams are never age-pruned. Configure with
|
||||
|
||||
### API Retry
|
||||
|
||||
Every model call streams (#831); retry lives at two stacked layers:
|
||||
`ChatSession._create_stream_with_retry()` (streaming path) and the agent
|
||||
`_api_call()` (non-streaming) both use the same retry pattern:
|
||||
|
||||
- **Caller ladders** — `ChatSession._model_turn_with_retry()` (chat
|
||||
loop, one ladder per lane) and the agent `_api_call()` (drained via
|
||||
`model_turn`) use the same pattern: 4 total attempts (1 initial + 3 retries,
|
||||
`_MAX_RETRIES = 3`), exponential backoff base 1 second
|
||||
(`delay = 1s * 2^attempt`), `ui.on_info()` on retry, exception
|
||||
propagates on final failure. `_compact_messages()` wraps its drained
|
||||
call in the same loop.
|
||||
- **`model_turn`'s drain ladder** — inside every single-shot call,
|
||||
mid-stream deaths (errors raised while draining, e.g.
|
||||
`IncompleteStreamError`) are re-issued up to 2 more times with a
|
||||
0.5s-base exponential backoff (±50% jitter); request-time failures
|
||||
keep the SDK's own retry policy. The two ladders stack
|
||||
multiplicatively on transient-shaped failures.
|
||||
- **Retryable errors** are matched by class name against each
|
||||
provider's `retryable_error_names` (avoids importing
|
||||
backend-specific exception hierarchies): `RateLimitError`,
|
||||
`APITimeoutError`, `APIConnectionError`, `InternalServerError`,
|
||||
`ServiceUnavailableError`, `APIError`, plus the drained-transport
|
||||
errors `IncompleteStreamError` (stream ended with no terminal
|
||||
signal — for servers that never send one, declare
|
||||
`finish_reason_optional` in the model's capabilities JSON) and
|
||||
`ResponsesStreamFailedError` (transient in-band Responses failure).
|
||||
- **Retries**: 4 total attempts (1 initial + 3 retries, `_MAX_RETRIES = 3`)
|
||||
- **Backoff**: exponential, base 1 second (`delay = 1s * 2^attempt`)
|
||||
- **Retryable errors**: `RateLimitError`, `APITimeoutError`,
|
||||
`APIConnectionError`, `InternalServerError`, `ServiceUnavailableError`,
|
||||
`APIError` (matched by class name to avoid importing backend-specific
|
||||
exception hierarchies)
|
||||
- On retry: `ui.on_info()` notification
|
||||
- On final failure: exception propagates
|
||||
|
||||
`_compact_messages()` also wraps its non-streaming API call in the same
|
||||
retry loop.
|
||||
|
||||
### Finish Reason Handling
|
||||
|
||||
@@ -1219,7 +1176,7 @@ Every model call streams (#831); retry lives at two stacked layers:
|
||||
blocked.
|
||||
|
||||
Agent sub-sessions (`_run_agent()`) check `finish_reason` on each
|
||||
drained turn and stop the agent early on `"length"` or
|
||||
non-streaming response and stop the agent early on `"length"` or
|
||||
`"content_filter"`.
|
||||
|
||||
`_compact_messages()` checks `finish_reason` on the compaction response and
|
||||
@@ -1263,34 +1220,28 @@ warns if the summary was truncated.
|
||||
`_run_single_test()`: wraps `session.send_headless()` in a retry loop (3
|
||||
attempts) to avoid transient API errors from poisoning evaluation scores.
|
||||
|
||||
### Backend Health Tracking
|
||||
### Health Monitor & Circuit Breaker
|
||||
|
||||
`BackendHealthTracker` (`turnstone/core/healthcheck.py`) records LLM backend
|
||||
health passively from real request outcomes — there is no probe thread and no
|
||||
circuit breaker, and requests are never blocked. Two states:
|
||||
`BackendHealthMonitor` (`turnstone/core/healthcheck.py`) runs a daemon thread
|
||||
that probes the LLM backend by calling `client.models.list()` every
|
||||
`backend_probe_interval` seconds (default 30). Probe results drive a three-state
|
||||
circuit breaker:
|
||||
|
||||
```
|
||||
healthy ──(failure_threshold consecutive failures)──> degraded
|
||||
degraded ──(any success)───────────────────────────> healthy
|
||||
CLOSED ──(N consecutive failures)──> OPEN
|
||||
OPEN ──(cooldown expires)────────> HALF_OPEN
|
||||
HALF_OPEN ──(probe succeeds)────────> CLOSED
|
||||
HALF_OPEN ──(probe fails)──────────> OPEN
|
||||
```
|
||||
|
||||
- `record_success()` fires at the request-accepted instant: the streaming
|
||||
consumer's `on_stream_armed` hook, driven by the eager `cancel_ref` append
|
||||
every adapter performs at HTTP-response time.
|
||||
- `record_failure()` fires once per lane's whole creation ladder, in
|
||||
`ChatSession._model_turn_with_fallback` / `_try_fallback_lane`. A mid-stream
|
||||
death (the stream armed, then died) records neither — it belongs to the
|
||||
re-issue ladder, not the fallback walk. `BackendAuthUnavailableError` and
|
||||
`WirePreparationError` also record nothing: an auth refusal is fail-closed
|
||||
configuration policy and a wire-preparation fault is session data — neither
|
||||
says anything about the backend.
|
||||
- `is_degraded` is advisory ordering, not admission: the fallback walk tries
|
||||
non-degraded aliases first and degraded ones as a last resort, and the
|
||||
primary lane is always dialed.
|
||||
- `HealthTrackerRegistry` keys trackers by `(provider, base_url)` so aliases
|
||||
sharing a backend share one tracker. The `/health` endpoint projects the
|
||||
same trackers: `"status": "ok"` when the backend is healthy, `"degraded"`
|
||||
otherwise.
|
||||
- `record_success()` / `record_failure()` update `_consecutive_failures` and
|
||||
transition the `_state` (`CircuitState` enum: `CLOSED`, `OPEN`, `HALF_OPEN`).
|
||||
- `acquire_request_permit()` returns `False` when the circuit is `OPEN` or when
|
||||
in `HALF_OPEN` and the single probe permit has already been consumed. Causes
|
||||
`ChatSession._create_stream_with_retry` to skip the backend and surface an
|
||||
error immediately.
|
||||
- The `/health` endpoint reads the monitor's state: `"status": "ok"` when the
|
||||
circuit is closed, `"status": "degraded"` when open or half-open.
|
||||
|
||||
### Rate Limiting
|
||||
|
||||
|
||||
+11
-21
@@ -126,11 +126,10 @@ the skill should end on.
|
||||
|
||||
`tasks` is the coordinator's scratchpad — a persisted, ordered
|
||||
list of rows with fields `{id, title, status, child_ws_id, created,
|
||||
updated}`, plus `note` on rows where one has been set (the key is
|
||||
absent otherwise), that only this coordinator sees. Children don't
|
||||
see it; the user does via the sidebar. Five actions: `add`,
|
||||
`update`, `remove`, `reorder`, `list` (only `list` is auto-approved;
|
||||
the mutators go through the approval flow).
|
||||
updated}` that only this coordinator sees. Children don't see it;
|
||||
the user does via the sidebar. Five actions: `add`, `update`,
|
||||
`remove`, `reorder`, `list` (only `list` is auto-approved; the
|
||||
mutators go through the approval flow).
|
||||
|
||||
The input schema refers to rows by `task_id`; the persisted row
|
||||
object exposes the same id as `id`. The `child_ws_id` field is a
|
||||
@@ -143,20 +142,11 @@ A skill's initial prompt can seed the task list by calling
|
||||
`tasks(action="add", title=...)` as its very first tool calls —
|
||||
the user gets a visible plan before any child is spawned, and the
|
||||
coordinator's future self has something concrete to iterate on.
|
||||
Status transitions (`pending` → `in_progress` → `done` / `blocked` /
|
||||
`needs_user`) are the skill's main feedback loop: mutate the task
|
||||
when the child covering it finishes, not when the child starts.
|
||||
`blocked` and `needs_user` are not interchangeable — `blocked` is a
|
||||
dependency the coordinator may be able to clear itself, while
|
||||
`needs_user` marks a task that cannot move without a decision,
|
||||
approval, or grant only the user can give. The distinction is
|
||||
load-bearing: a coordinator that goes idle holding open tasks gets
|
||||
nudged to pick them back up — even when children are still running, so
|
||||
keep the matrix honest rather than expecting the reminder to wait for
|
||||
an all-clear — and `needs_user` is what tells that nudge the stop was
|
||||
deliberate. Pair it with `note` to record what is being asked for.
|
||||
Use `tasks(action="update", task_id=..., child_ws_id=<ws_id>)` to link
|
||||
a task to the child that owns it once spawn returns.
|
||||
Status transitions (`pending` → `in_progress` → `done` / `blocked`)
|
||||
are the skill's main feedback loop: mutate the task when the child
|
||||
covering it finishes, not when the child starts. Use
|
||||
`tasks(action="update", task_id=..., child_ws_id=<ws_id>)` to
|
||||
link a task to the child that owns it once spawn returns.
|
||||
|
||||
A final gotcha: parallel tool dispatch does NOT serialise reads
|
||||
after writes in the same batch. If a skill issues an `update` and
|
||||
@@ -307,11 +297,11 @@ and the coordinator's planning step is itself valuable.
|
||||
tasks(action='add', title='...') × N # the plan, visible in the sidebar
|
||||
for task in tasks:
|
||||
spawn_workstream(skill=..., initial_message=task.brief)
|
||||
tasks(action='update', task_id=task.id, note='ws=<child_ws_id>')
|
||||
tasks(action='update', task_id=task.id, notes='ws=<child_ws_id>')
|
||||
wait_for_workstream(ws_ids=[...], mode='all', timeout=...)
|
||||
for child in children:
|
||||
inspect_workstream(ws_id=child)
|
||||
tasks(action='update', task_id=..., status='done', note='result summary')
|
||||
tasks(action='update', task_id=..., status='done', notes='result summary')
|
||||
→ synthesise
|
||||
```
|
||||
|
||||
|
||||
@@ -66,7 +66,8 @@ class "NullUI" as NullUI {
|
||||
interface "LLMProvider" as LLMProvider <<Protocol>> {
|
||||
+ provider_name: str {property}
|
||||
+ get_capabilities(model) → ModelCapabilities
|
||||
+ create_streaming(client, model, messages, ..., cancel_ref, replay_reasoning_to_model) → Iterator[StreamChunk]
|
||||
+ create_streaming(client, model, messages, ..., replay_reasoning_to_model) → Iterator[StreamChunk]
|
||||
+ create_completion(client, model, messages, ..., replay_reasoning_to_model) → CompletionResult
|
||||
+ convert_tools(tools) → list[dict]
|
||||
+ extract_reasoning_text(provider_blocks) → str
|
||||
+ retryable_error_names: frozenset[str] {property}
|
||||
@@ -148,9 +149,9 @@ class "ChatSession" as ChatSession {
|
||||
+ handle_command(command: str)
|
||||
+ resume(ws_id: str)
|
||||
- _save_config()
|
||||
- _stream_response(my_generation) → ModelTurnResult
|
||||
- _model_turn_with_fallback(consumer, prepare_wire) → ModelTurnResult
|
||||
- _model_turn_with_retry(lane, tracker, ...) → ModelTurnResult
|
||||
- _stream_response(stream) → dict
|
||||
- _create_stream_with_retry(msgs) → Stream (+ fallback)
|
||||
- _try_stream(client, model, msgs) → Stream
|
||||
- _execute_tools(tool_calls) → (results, feedback)
|
||||
- _prepare_tool(tc) → item dict
|
||||
- _prepare_mcp_tool(call_id, name, args) → item dict
|
||||
@@ -176,7 +177,7 @@ class "HeadlessSession" as HeadlessSession {
|
||||
+ send_headless(input, max_turns, ...)
|
||||
- _override_system_prompt(content)
|
||||
--
|
||||
eval.py: drained single-shot turns,
|
||||
eval.py: non-streaming,
|
||||
records all tool calls
|
||||
}
|
||||
|
||||
|
||||
@@ -84,8 +84,8 @@ end note
|
||||
|
||||
loop up to 3 turns (timeout budget)
|
||||
|
||||
Judge -> LLM : model_turn(lane, judge_turns,\ntools=[read_file, list_directory])\nvia drained create_streaming
|
||||
LLM --> Judge : ModelTurnResult
|
||||
Judge -> LLM : create_completion(\nmodel, judge_messages,\ntools=[read_file, list_directory])
|
||||
LLM --> Judge : CompletionResult
|
||||
|
||||
alt tool_calls present (turn < 3)
|
||||
Judge -> Judge : _exec_read_only_tool()
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:6e2bfdf968e96f3720ed58674103288e2f57e9c056f5c479a57f37a849f3e69c
|
||||
size 821878
|
||||
@@ -252,7 +252,6 @@ interface, or anyone who can reach it can search through your instance.
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `WORKSPACE_MOUNT` | empty volume | Host directory bind-mounted at `/workspace` for the model to read/write |
|
||||
| `TURNSTONE_WORKSPACE` | `/workspace` (image env) | Directory named as the user's workspace in the model's tool descriptions; informational only — see [Working directory](#working-directory) |
|
||||
| `SKIP_PERMISSIONS` | — | Set to any value to auto-approve all tool calls (dev only) |
|
||||
| `MCP_CONFIG` | — | Path to an MCP server config file |
|
||||
| `TURNSTONE_IMAGE_TAG` | `latest` | ghcr.io image tag — production stack |
|
||||
@@ -277,35 +276,6 @@ docker compose build --no-cache # rebuild from scratch
|
||||
| `workspace` | `/workspace` (unless `WORKSPACE_MOUNT` is set) |
|
||||
| `caddy-data` / `caddy-config` | Caddy's local CA and config (dev stack) |
|
||||
|
||||
## Working directory
|
||||
|
||||
Node processes run with `/data` as their working directory (the image's
|
||||
`WORKDIR`), and that is where the model's shell commands execute and
|
||||
relative file paths resolve — **not** `/workspace`. The shell and file
|
||||
tool descriptions state both paths (the working directory, and the
|
||||
workspace named by `TURNSTONE_WORKSPACE`), so the model knows to look in
|
||||
`/workspace` for your files without being told each session.
|
||||
|
||||
To make tools start inside the mount instead, override the working
|
||||
directory on the node services:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
turnstone-node:
|
||||
working_dir: /workspace
|
||||
```
|
||||
|
||||
Two caveats before overriding:
|
||||
|
||||
- **SQLite fallback**: when a node runs without PostgreSQL, its fallback
|
||||
database `.turnstone.db` is created in the process working directory.
|
||||
Changing `working_dir` on an existing SQLite-fallback deployment makes
|
||||
the node create a fresh database inside the mount and your prior state
|
||||
appears lost (it is still in the `turnstone-data` volume under `/data`).
|
||||
The stock compose stacks use PostgreSQL and are unaffected.
|
||||
- Migrations (`entrypoint.sh`) run in the same working directory, so the
|
||||
same SQLite caveat applies to them.
|
||||
|
||||
## Cleanup
|
||||
|
||||
```bash
|
||||
|
||||
+4
-65
@@ -17,9 +17,8 @@ The MCP server admin form exposes three authorization modes ("Multitenant Author
|
||||
| `none` | No headers attached. Open MCP server (or one gated by network policy only). | Internal MCP servers on a trusted network. |
|
||||
| `static` | One static bearer token, configured per server, sent on every request from every user. | Service-to-service MCP servers where per-user attribution doesn't matter, or single-tenant deployments. |
|
||||
| `oauth_user` *(recommended for user-data servers)* | Each user authorizes separately via OAuth 2.1 + PKCE; Turnstone stores per-user tokens encrypted at rest. | MCP servers that expose user-specific data or that want per-user audit attribution. |
|
||||
| `oauth_obo` *(sign-in passthrough)* | Each user's Turnstone **org sign-in** (OIDC) mints a per-server access token on demand — no separate per-server consent. One captured credential per user covers every `oauth_obo` server. | Enterprise deployments where the identity provider governs access (Entra, Keycloak) and you want zero per-user connect clicks. See the dedicated section below. |
|
||||
|
||||
Switching `auth_type` away from `oauth_user` / `oauth_obo` **deletes** that server's per-user rows (consents / minted cache) — see the transition table below. Switching back later starts clean: users re-consent (or re-mint) on next use. The admin **bulk-revoke** / **flush cache** affordance clears rows without an auth-type change.
|
||||
Switching `auth_type` away from `oauth_user` orphans existing per-user tokens. Use the admin **bulk-revoke** affordance on the server row (Phase 9) to clear them, or let them expire naturally — they're inert without the matching `auth_type` value.
|
||||
|
||||
---
|
||||
|
||||
@@ -66,59 +65,6 @@ Keep this in `config.toml` rather than environment variables. An in-process LLM
|
||||
|
||||
---
|
||||
|
||||
## `auth_type=oauth_obo` — single-credential sign-in passthrough
|
||||
|
||||
Where `oauth_user` makes each user complete a **separate** browser consent per MCP server, `oauth_obo` reuses the user's Turnstone **org sign-in** (OIDC). Turnstone captures one refresh credential per user at login and, on each tool call, mints a short-lived access token scoped to that server's audience. There is no per-server connect step, and one credential covers every `oauth_obo` server. This is the right shape when your identity provider already governs who may reach each backend (an Entra tenant with Entra-protected MCP servers; a Keycloak realm with token exchange).
|
||||
|
||||
Access is governed **downstream** by the IdP: a user can only mint a token for a server their delegated permissions allow. Removing that grant at the IdP cuts the user off regardless of their Turnstone state.
|
||||
|
||||
### Deployment configuration (`[oidc]` in `config.toml`)
|
||||
|
||||
`oauth_obo` requires OIDC SSO to be configured (it is the credential source), plus:
|
||||
|
||||
```toml
|
||||
[oidc]
|
||||
# ... your existing issuer / client_id / client_secret ...
|
||||
capture_user_credential = true # persist the IdP refresh token at login
|
||||
obo_grant_profile = "entra" # "entra" | "rfc8693" — how tokens are minted
|
||||
```
|
||||
|
||||
- **`capture_user_credential`** (default `false`): when enabled, Turnstone appends `offline_access` to the login scopes and stores the returned refresh token, encrypted with the same `[security] mcp_token_encryption_key` as `oauth_user` tokens. **The encryption key is required** — Turnstone refuses to start with an `oauth_obo` row (or capture enabled) and no key.
|
||||
- **`obo_grant_profile`** picks the mint mechanism (the IdP determines which one is valid; this is deployment-wide, not per-server):
|
||||
- **`entra`** — redeems the user's refresh token directly for a token scoped to `<audience>/.default`. `oauth_scopes` on the server row is **not used** (the admin form rejects it under this profile).
|
||||
- **`rfc8693`** — a refresh grant for a subject token, then an RFC 8693 token exchange for the server audience. Per-server `oauth_scopes` **are** sent on the exchange (some IdPs require the audience scope explicitly).
|
||||
|
||||
### Adding an `oauth_obo` server
|
||||
|
||||
In the admin MCP form, choose **Sign-in passthrough** and set **Audience** (required — the downstream resource the token is minted for, e.g. `api://<app-id>` on Entra or the client id on Keycloak). The client-id / secret / registration fields do not apply and are hidden.
|
||||
|
||||
`oauth_obo` servers are accepted only when **OIDC sign-in is configured and enabled** and `[oidc] obo_grant_profile` is a valid profile — the write is rejected otherwise, since a row that can never mint would surface to users as a permanent "please retry" that never heals.
|
||||
|
||||
### Identity-provider setup
|
||||
|
||||
**Entra (`obo_grant_profile = "entra"`):**
|
||||
1. Turnstone's app registration must hold **delegated permissions** to each MCP server's exposed API, with **admin consent granted** (or the MCP app listed in Turnstone's `preAuthorizedApplications`).
|
||||
2. Set the server row's Audience to the MCP app's Application ID URI (`api://<guid>`).
|
||||
3. **Gotcha (verified):** admin-consent issued *immediately* after creating the app/service principal can silently skip a not-yet-propagated resource — the only symptom is `AADSTS65001` at mint time. Verify the delegated grant landed (`az ad app permission list-grants` / the portal's *API permissions* blade shows *Granted*), or grant it explicitly per resource. A missing grant surfaces in Turnstone as a re-login prompt on the affected server (same rail as a revoked credential), and the `mcp_server.oauth.obo_mint_rejected` log line carries the raw `AADSTS…` text.
|
||||
|
||||
**Keycloak / RFC 8693 (`obo_grant_profile = "rfc8693"`):**
|
||||
1. Enable **standard token exchange** on Turnstone's client.
|
||||
2. Grant the audience: add an audience client scope for each MCP client and attach it to Turnstone's client (optional scopes must be requested — set the server row's Scopes to that scope, or the exchange returns *"Requested audience not available"*).
|
||||
3. Set the server row's Audience to the downstream client id.
|
||||
|
||||
### Revocation & custody
|
||||
|
||||
The captured credential is a single per-user secret that can mint for every `oauth_obo` server, so treat it like any long-lived credential:
|
||||
|
||||
- **Cut off one user:** unlink their OIDC identity in the admin console (**Users → OIDC identities → delete**). This revokes the captured credential **and** purges their minted cache rows, so future mints fail and cached tokens are dropped. (Warmed in-memory sessions on server nodes self-expire at the access-token TTL; there is no cross-node per-user session-kill.) Removing the user's access at the IdP is the authoritative cut-off.
|
||||
- The same unlink also purges that user's synthetic `__model_obo__:` gateway-token rows and requests eviction from every registered host's in-process mint memo. Shared `entra_app` model tokens live under the `__app__` pseudo-user and are intentionally not user-deprovisioned; revoking the app credential prevents new mints, while a cached app bearer lasts until `expires_at`.
|
||||
- **Flush a server's minted tokens** (e.g. after narrowing its audience): the server row's **flush cache** action drops all users' cached tokens for that server. This is **not** a revocation — users re-mint on next use from their still-valid sign-in. It is surfaced honestly (audit `mcp_server.oauth.obo_cache_flushed`, response `effect: cache_flush_remints`) so it is never mistaken for cutting access.
|
||||
- Per-server revocation in the `oauth_user` sense does not exist for `oauth_obo` — the credential is issuer-scoped and IdP-governed. Revoke at the IdP.
|
||||
|
||||
> **Interim for Entra without OBO:** if you don't want host-side minting, admin consent + `preAuthorizedApplications` on each MCP app registration removes the second consent prompt for the plain `oauth_user` flow too (a tenant-config change, no Turnstone code). Tracked in issue #682. It does not remove the per-server connect clicks or per-(user, server) token custody — that is what `oauth_obo` is for.
|
||||
|
||||
---
|
||||
|
||||
## Lifecycle
|
||||
|
||||
1. **First tool call** for a user against an `oauth_user` MCP server: pool dispatch finds no stored token, returns `mcp_consent_required` to the agent. Dashboard renders an inline "Connect" action card.
|
||||
@@ -129,7 +75,7 @@ The captured credential is a single per-user secret that can mint for every `oau
|
||||
|
||||
4. **Step-up scope**: when a tool call hits `403` with `WWW-Authenticate: error="insufficient_scope"`, Turnstone emits `mcp_insufficient_scope` with the parsed scope set; the dashboard offers a "Connect with additional scopes" affordance that opens `/v1/api/mcp/oauth/start?server=<name>&scopes=<extra>` so the union of original + new scopes flows into the AS authorize request.
|
||||
|
||||
5. **User revoke** (settings modal): `DELETE /v1/api/mcp/oauth/connections/{server_name}` runs the authoritative local delete + best-effort RFC 7009 upstream revoke (fire-and-forget, capped at 256 concurrent in-flight tasks). `oauth_obo` servers and synthetic model-auth rows are excluded: their rows are mint caches, not consents — deleting one only forces a re-mint — so the connections list hides them and the endpoint refuses them with `409` (revocation for sign-in passthrough happens at the identity layer: unlink the identity or revoke at the IdP).
|
||||
5. **User revoke** (settings modal): `DELETE /v1/api/mcp/oauth/connections/{server_name}` runs the authoritative local delete + best-effort RFC 7009 upstream revoke (fire-and-forget, capped at 256 concurrent in-flight tasks).
|
||||
|
||||
6. **Admin bulk-revoke** (Phase 9): `POST /v1/api/admin/mcp-servers/{name}/bulk-revoke` drops every user's token for the server. Upstream RFC 7009 revoke is intentionally **not** attempted in bulk (avoids N upstream HTTP calls per admin click); tokens at the AS expire naturally. Use the per-user revoke endpoint if you need guaranteed upstream invalidation.
|
||||
|
||||
@@ -151,13 +97,10 @@ Additional indicators (circuit-breaker state, encryption-key mismatch) are expos
|
||||
| From | To | What happens |
|
||||
|---|---|---|
|
||||
| `none` / `static` → `oauth_user` | — | New code path activates for this server. Existing static headers (if any) are no longer sent. Users must authorize on first use. |
|
||||
| `oauth_user` → `none` / `static` | — | Existing `mcp_user_tokens` rows are **deleted**: the tokens are bound to the auth model + URL active at consent time, and rows left behind could silently rebind if a row with the old name/URL reappears. Switching back to `oauth_user` later starts clean — users re-consent on next use. This is **not reversible**; the AS-side grants are untouched (revoke upstream via the AS if needed). |
|
||||
| `oauth_user` → `none` / `static` | — | Existing `mcp_user_tokens` rows are **orphaned** — inert without a matching `auth_type`. Use admin bulk-revoke to drop them, or let them expire. Switching back to `oauth_user` later re-activates the orphaned rows if they haven't been deleted. |
|
||||
| OAuth `client_id` or `client_secret` rotated | — | Existing tokens may stop refreshing if the AS treats them as bound to the previous client. Bulk-revoke after rotation. |
|
||||
| `oauth_user` ↔ `oauth_obo` | — | The per-user rows are **deleted** on the flip (they mean different things: per-server AS refresh tokens vs. minted cache). `oauth_audience` and `oauth_scopes` mean different things in each model (a resource indicator vs. an IdP app identifier; AS-consent scopes vs. an rfc8693 exchange scope), so on a flip they **never carry** — each is taken from the request for the target model or set NULL. The admin console clears these fields when you change the auth type, so re-enter the correct values for the new mode; via the API, supply them explicitly (a flip into `oauth_obo` with no `oauth_audience` is rejected, and a non-empty `oauth_scopes` under the `entra` profile is rejected since that leg pins `<audience>/.default`). |
|
||||
| `oauth_obo` → `none` / `static` | — | Minted cache rows are deleted. |
|
||||
| `oauth_obo` **audience**, **URL**, or **`oauth_scopes`** changed | — | Minted cache rows are **deleted** (tokens are bound to the audience/URL/scopes at mint time), forcing a fresh mint — so an audience or scope narrowing takes effect immediately, not at token expiry. |
|
||||
|
||||
Every transition that changes what a stored row *means* deletes the rows outright — a stale consent or minted token must never be served under new semantics. There is no orphan-and-reactivate path.
|
||||
The orphan-by-default behavior is chosen so switching back to `oauth_user` is non-destructive. Bulk-revoke is the explicit cleanup path.
|
||||
|
||||
---
|
||||
|
||||
@@ -170,9 +113,5 @@ Every transition that changes what a stored row *means* deletes the rows outrigh
|
||||
| `mcp_oauth_url_insecure` | MCP server URL is `http://` (not `https://`) on a non-loopback host | Use `https://`. Per-user bearers must not transit cleartext. |
|
||||
| Tools fail in scheduled / Discord / Slack runs | OAuth-MCP requires browser-based consent | Users must pre-consent via the web UI. Phase 9 dashboard badge surfaces deferred consents from these runs on next login. |
|
||||
| Circuit breaker open repeatedly | Transport-level errors on the MCP server (DNS, TLS, 5xx) | Check the per-server error pill; auth errors do not trip the breaker. |
|
||||
| **`oauth_obo`**: every tool call fails, log shows `obo_misconfigured` | Server row has no Audience, or `obo_grant_profile` is unset/unknown | Set the Audience on the server row; set `[oidc] obo_grant_profile` to `entra` or `rfc8693`. |
|
||||
| **`oauth_obo`**: `obo_mint_rejected` with `AADSTS65001` | Turnstone's app lacks the (admin-consented) delegated grant to this MCP app — often admin consent that didn't propagate | Grant + admin-consent the delegated permission for this resource; verify it shows *Granted*. See the Entra gotcha above. |
|
||||
| **`oauth_obo`**: "Sign in to Turnstone again" on one server | Captured credential missing/rejected, or a Conditional Access challenge | User re-logs into Turnstone (re-captures the credential). If it persists, check the IdP grant / CA policy. |
|
||||
| **`oauth_obo`**: tools don't appear at all for a user | User has not signed in since `capture_user_credential` was enabled (no credential captured) | User logs out and back in via OIDC so the refresh credential is captured. |
|
||||
|
||||
See also: `docs/operations/mcp-oauth-headless.md` for the cron / channel-driven run caveat.
|
||||
|
||||
+9
-68
@@ -77,17 +77,17 @@ IdP from redirecting the token-exchange POST (which carries
|
||||
being aimed at internal services.
|
||||
|
||||
A few public IdPs legitimately split endpoints across hostnames. Google
|
||||
and Microsoft Entra ID are the canonical examples:
|
||||
is the canonical example:
|
||||
|
||||
| IdP | Issuer host | Cross-host endpoint(s) |
|
||||
|-----|-------------|------------------------|
|
||||
| Google | `accounts.google.com` | `oauth2.googleapis.com`, `www.googleapis.com`, `openidconnect.googleapis.com` |
|
||||
| Microsoft Entra | `login.microsoftonline.com` | `graph.microsoft.com` (userinfo) |
|
||||
| Field | Hostname |
|
||||
|-------|----------|
|
||||
| issuer | `accounts.google.com` |
|
||||
| token_endpoint | `oauth2.googleapis.com` |
|
||||
| jwks_uri | `www.googleapis.com` |
|
||||
| userinfo_endpoint | `openidconnect.googleapis.com` |
|
||||
|
||||
Both sets are built in — operators using `https://accounts.google.com` or
|
||||
`https://login.microsoftonline.com/<tenant>/v2.0` need no extra
|
||||
configuration. (Entra's discovery document advertises `userinfo_endpoint`
|
||||
on `graph.microsoft.com`, distinct from the issuer host.)
|
||||
Google's set is built in — operators using `https://accounts.google.com`
|
||||
need no extra configuration.
|
||||
|
||||
For other IdPs whose discovery document references a non-issuer host,
|
||||
extend the allow-list explicitly:
|
||||
@@ -134,65 +134,6 @@ This knob only affects the login-flow IdP configured here. OAuth
|
||||
endpoints advertised by remote MCP servers are untrusted input and are
|
||||
always held to the strict public-address rule.
|
||||
|
||||
### Model gateway credentials
|
||||
|
||||
The same OIDC registration can authenticate model gateways. A model definition
|
||||
with `auth_mode = "entra_obo"` (Entra grant profile) or `auth_mode =
|
||||
"rfc8693_obo"` (RFC 8693 token-exchange profile) redeems the driving user's
|
||||
captured credential for its exact `obo_audience`; `auth_mode = "entra_app"`
|
||||
uses the registration's client ID and secret with Entra client credentials.
|
||||
All three bind the result through the provider SDK's native credential option
|
||||
rather than injecting an override header. The grant mode is never inferred:
|
||||
missing user context or a failed OBO mint cannot switch a delegated definition
|
||||
to client credentials.
|
||||
|
||||
Each dynamic mode pairs with the grant profile whose dialect it names:
|
||||
`entra_obo` and `entra_app` require `obo_grant_profile = "entra"`;
|
||||
`rfc8693_obo` requires `obo_grant_profile = "rfc8693"`. The pairing is
|
||||
enforced when a write chooses a `(auth_mode, obo_audience)` pair — a same-pair
|
||||
edit of a row saved before the pairing rule keeps working — and at runtime a
|
||||
mismatched legacy row refuses to mint with `cause=grant_profile_mismatch` and
|
||||
no IdP traffic. RFC 8693 client-credentials is not implemented.
|
||||
|
||||
The delegated modes need the MCP encryption key, a credential captured for the
|
||||
driving user, and delegated/admin-consented permission to the audience.
|
||||
`rfc8693_obo` additionally carries `obo_scopes`, the space-separated scope
|
||||
list its exchange leg requests: exchange-capable IdPs that gate audiences
|
||||
behind optional scopes refuse the exchange without it ("Requested audience not
|
||||
available"), which is why the scope-less Entra-named mode could never mint on
|
||||
that profile (issue #955). Scopes are stored shape-checked only — whether a
|
||||
value satisfies the IdP stays the IdP's call at mint time. Turning
|
||||
`capture_user_credential` off later stops *new* captures but does not
|
||||
invalidate credentials already stored, so existing users keep minting.
|
||||
`entra_app` requires a confidential-client secret. Configure the permitted
|
||||
resource IDs in the runtime setting `model.auth_audience_allowlist` before
|
||||
saving dynamic model definitions. De-listing an audience later blocks every
|
||||
write that would arm or re-aim a definition at it, but does not stop aliases
|
||||
already configured from minting — disabling the row (the `admin.models` disarm
|
||||
lever) is what stops minting. See
|
||||
[Settings](settings.md#model-backend-authentication) for permissions, failure
|
||||
policy, and lane identity rules.
|
||||
|
||||
An unrecognised `obo_grant_profile` is warned about at startup and **rejected
|
||||
at the write choke points**: configuring an `oauth_obo` MCP server or a dynamic
|
||||
model alias returns a 400 that echoes the configured value, so the typo is the
|
||||
diagnosis. At runtime an unknown profile never mints — the mint legs resolve by
|
||||
exact name; the full cause detail is logged once per audience, and every
|
||||
affected call still logs its per-turn fallback or refusal naming the alias,
|
||||
the target audience, and the last recorded cause (`cause=` — for example
|
||||
`unsupported_grant_profile` or `oidc_not_enabled`) — so a pre-existing row
|
||||
degrades loudly, with the reason visible mid-incident even after the
|
||||
once-per-process line has rotated out of retained logs, rather than silently
|
||||
swapping per-user attribution for the shared static key.
|
||||
|
||||
The `[security]` token encryption key is deployment-wide, not per-host: rows are
|
||||
encrypted with `MultiFernet` and carry no key id, so every host that reads them
|
||||
needs the same keyring. That includes the console, which mints for
|
||||
coordinator-hosted sessions. A node that needs the key and lacks it refuses to
|
||||
start; the console starts but withholds its coordinator subsystem and shows
|
||||
the key requirement as the remediation error instead of failing silently at
|
||||
call time.
|
||||
|
||||
### config.toml alternative
|
||||
|
||||
```toml
|
||||
|
||||
+1
-86
@@ -54,91 +54,6 @@ When a per-model override is `NULL` (empty in the UI), the global default is
|
||||
used. Switching models via `/model <alias>` re-resolves sampling parameters
|
||||
from the new model's overrides or global defaults.
|
||||
|
||||
### Model backend authentication
|
||||
|
||||
Model definitions support four backend credential modes:
|
||||
|
||||
| `auth_mode` | Identity sent to the model gateway |
|
||||
|-------------|------------------------------------|
|
||||
| `static` | The definition's stored `api_key`. |
|
||||
| `entra_obo` | A caller-delegated Entra access token minted from that user's captured OIDC credential. |
|
||||
| `entra_app` | A shared app-identity token minted with Turnstone's OIDC client credentials. |
|
||||
| `rfc8693_obo` | A caller-delegated access token minted from the captured credential via RFC 8693 token exchange, requesting the definition's `obo_scopes`. |
|
||||
|
||||
Dynamic modes require an exact `obo_audience` resource identifier. Before an
|
||||
admin can save one, an operator must add that literal audience to
|
||||
`model.auth_audience_allowlist` (comma- or newline-separated). Wildcards and
|
||||
base-URL host matching are intentionally unsupported, and a row whose
|
||||
effective mode is `static` refuses to store a new non-empty `obo_audience` on
|
||||
either create or update — an audience cannot be staged for a later flip
|
||||
(clearing a stale value, or re-saving it unchanged, stays allowed).
|
||||
`obo_scopes` follows the same staging rule with the mode set inverted: only
|
||||
`rfc8693_obo` reads it, so every other effective mode refuses to store a new
|
||||
non-empty value, while clearing or re-saving one unchanged stays open. The
|
||||
value itself is optional and shape-checked only — whether it satisfies the
|
||||
IdP is decided at mint time. On a row that is (or becomes) dynamic, every
|
||||
change except the tuning fields — context window, temperature, max tokens,
|
||||
reasoning effort, and the two reasoning-persistence toggles — also requires
|
||||
`admin.mcp`; service tokens do not bypass this capability-escalation gate.
|
||||
The one exception is de-escalation: a save whose only gated change is
|
||||
switching `enabled` off is a pure disable, needs only `admin.models`, and
|
||||
skips validation — a de-listed audience must never block disarming its own
|
||||
row. The gate is deny-by-default: a field counts as auth-relevant unless it
|
||||
is provably neutral, so re-enabling a disabled dynamic row, re-pointing its
|
||||
`base_url`, or swapping its provider or alias all escalate.
|
||||
|
||||
Validation runs in two tiers, matching the MCP `oauth_obo` write rules. Row
|
||||
validity — the audience is allow-listed — applies to every gated write that
|
||||
touches a dynamic configuration, so a revoked audience can be neither silently
|
||||
re-pointed at a new `base_url` nor re-armed by an enable flip. Deployment
|
||||
posture — the token encryption key installed, single sign-on configured, and
|
||||
the grant profile valid and able to carry the mode — is checked when a write
|
||||
*chooses* the mode/audience pair and when it re-enables a disabled dynamic
|
||||
row (arming is the flip that resumes minting, so it must meet what minting
|
||||
needs); other edits to an existing row stay open if the deployment's posture
|
||||
changed after it was saved (its mints warn at runtime instead). Refusals name
|
||||
their cause and echo the configured value.
|
||||
|
||||
One asymmetry to be aware of: the write path counts a transient discovery
|
||||
outage (`enabled=false`, retryable) as configured, but the mints themselves
|
||||
require discovery to have completed — a config saved during an outage starts
|
||||
minting only once any authenticated request heals discovery. Until then calls
|
||||
warn and follow the fail-open/fail-closed policy above.
|
||||
|
||||
Every dynamic mode pairs with exactly one grant profile: `entra_obo` and
|
||||
`entra_app` require `[oidc] obo_grant_profile = "entra"`, and `rfc8693_obo`
|
||||
requires `"rfc8693"`. The pairing is enforced at the posture tier, so a row
|
||||
saved before the rule existed keeps accepting same-pair edits; its mints
|
||||
refuse at runtime with `cause=grant_profile_mismatch` and no IdP traffic.
|
||||
Judge, output-guard, perception, utility, and sub-agent lanes inherit the
|
||||
session's effective user for the delegated modes. The perception memo is
|
||||
partitioned by that principal as well as alias and content hash, so a result
|
||||
authorized as one user cannot be served to another. Scheduled and wake-driven
|
||||
work retains the workstream owner even when no user is connected. Eval and
|
||||
optimizer lanes are registry-less development tools and therefore do not use
|
||||
dynamic model authentication.
|
||||
|
||||
`entra_app` is an explicit model-definition choice; Turnstone never changes a
|
||||
failed or ownerless delegated call into a client-credentials grant. A
|
||||
delegated-mode call with no effective user always refuses. A dynamic alias
|
||||
without a real static key also always refuses instead of issuing its
|
||||
SDK-construction placeholder. When a real static key is explicitly configured,
|
||||
mint failures may use it by default; set `model.auth_fail_closed = true` to
|
||||
prohibit even that fallback. A refusal is not routed through the model
|
||||
fallback chain.
|
||||
|
||||
Dynamic token caches are encrypted in `mcp_user_tokens`, shared across nodes,
|
||||
and memoized on each host. Unlinking a user's OIDC identity purges their
|
||||
delegated-mode rows and memo entries. `entra_app` rows belong to the shared
|
||||
`__app__` identity and are not user-deprovisioned; after client-credential
|
||||
revocation, an already-minted app bearer remains usable until its recorded
|
||||
expiry.
|
||||
|
||||
`obo_audience` and `obo_scopes` are literal and capped at 2048 characters
|
||||
each. Environment-variable expansion is deliberately not applied, so the
|
||||
allow-list decision cannot vary by node or expand beyond the persisted
|
||||
boundary.
|
||||
|
||||
### Responses output controls (per-model)
|
||||
|
||||
Models whose capability table declares Responses output controls expose two
|
||||
@@ -209,7 +124,7 @@ initialization:
|
||||
|
||||
| Section | Settings |
|
||||
|---------|----------|
|
||||
| `model` | default_alias, auth_audience_allowlist, auth_fail_closed, temperature, max_tokens, reasoning_effort, task_alias, task_effort |
|
||||
| `model` | default_alias, temperature, max_tokens, reasoning_effort, task_alias, task_effort |
|
||||
| `session` | instructions, retention_days, compact_max_tokens, auto_compact_pct |
|
||||
| `tools` | timeout, truncation, agent_max_turns, skip_permissions, search, search_threshold, search_max_results |
|
||||
| `server` | workstream_idle_timeout, max_workstreams |
|
||||
|
||||
+12
-28
@@ -28,19 +28,13 @@ schema plus turnstone-specific metadata keys:
|
||||
}
|
||||
```
|
||||
|
||||
**Metadata keys** (stripped before sending the schema to the model; the full
|
||||
set lives in `_META_KEYS` in `turnstone/core/tools.py`):
|
||||
**Metadata keys** (stripped before sending the schema to the model):
|
||||
|
||||
| Key | Type | Meaning |
|
||||
|------------------|------|---------|
|
||||
| `task_agent` | bool | Tool is available to task sub-agents. |
|
||||
| `coordinator` | bool | Tool is available to coordinator sessions. Without `interactive: true` alongside it, this reads as coord-only and the tool is stripped from interactive sessions. |
|
||||
| `interactive` | bool | Opt a `coordinator: true` tool back into interactive sessions (dual-kind tools like `memory`). |
|
||||
| `auto_approve` | bool | Tool runs without user confirmation (read-only, safe operations). |
|
||||
| `primary_key` | str | When the model sends a bare string instead of JSON args, map it to this parameter name. |
|
||||
| `kind_variants` | dict | Per-kind description / parameter-schema overlays so each session kind sees only the surface it can use (see `memory.json`). |
|
||||
| `cwd_note` | str | Sentence appended to the description at session build time with `{working_dir}` substituted — declare on tools whose semantics depend on the process working directory (see `bash.json`, `apply_cwd_context`). |
|
||||
| `workspace_note` | str | Companion sentence naming the operator-configured workspace directory, `{workspace_dir}` substituted; dropped when no workspace is configured. |
|
||||
| Key | Type | Meaning |
|
||||
|----------------|------|---------|
|
||||
| `task_agent` | bool | Tool is available to task sub-agents. |
|
||||
| `auto_approve` | bool | Tool runs without user confirmation (read-only, safe operations). |
|
||||
| `primary_key` | str | When the model sends a bare string instead of JSON args, map it to this parameter name. |
|
||||
|
||||
---
|
||||
|
||||
@@ -785,10 +779,7 @@ MCP tool lists stay up-to-date without restart through two mechanisms:
|
||||
1. **Push notifications** -- MCP servers that declare `tools.listChanged: true` in
|
||||
their capabilities send `notifications/tools/list_changed` when their tool list
|
||||
changes. `MCPClientManager` registers a `message_handler` on each `ClientSession`
|
||||
that triggers an immediate refresh for that server (debounced per server and
|
||||
notification kind, and run off the receive loop). A refresh that fails while
|
||||
the connection stays up is retried automatically on the next health-loop tick
|
||||
until one completes.
|
||||
that triggers an immediate refresh for that server.
|
||||
|
||||
2. **Manual** -- `/mcp refresh` re-fetches tools from all servers immediately.
|
||||
`/mcp refresh <server>` targets a single server. If a server has disconnected,
|
||||
@@ -796,10 +787,6 @@ MCP tool lists stay up-to-date without restart through two mechanisms:
|
||||
same controls (refresh / reconnect buttons per server) for cluster-wide
|
||||
fan-out.
|
||||
|
||||
Reconnects (health-loop, dispatch-driven, or operator-forced) always end in a
|
||||
full catalog rediscovery, so a server that changed its tools while disconnected
|
||||
comes back current.
|
||||
|
||||
When tools change, `MCPClientManager` rebuilds its merged tool list using copy-on-write
|
||||
(new list/dict objects assigned atomically) and notifies all active `ChatSession`
|
||||
instances via registered listener callbacks. Each session rebuilds its `_tools`,
|
||||
@@ -870,16 +857,13 @@ catalog.
|
||||
|
||||
### Refresh
|
||||
|
||||
Resource lists stay current through the same mechanisms as tool lists:
|
||||
Resource lists stay current through the same three-tier mechanism as tool lists:
|
||||
|
||||
1. **Push** -- Servers declaring `resources.listChanged: true` send
|
||||
`notifications/resources/list_changed`, triggering an immediate refresh
|
||||
(with the same failed-refresh retry on the health-loop tick).
|
||||
2. **Manual** -- `/mcp refresh` re-fetches resources alongside tools.
|
||||
|
||||
Servers without push support are refreshed whenever they reconnect (every
|
||||
reconnect ends in full rediscovery) or when an operator refreshes manually;
|
||||
there is no periodic polling.
|
||||
`notifications/resources/list_changed`, triggering an immediate refresh.
|
||||
2. **Periodic** -- Servers without push are polled on the configured refresh
|
||||
interval (default 4 hours, same timer as tools).
|
||||
3. **Manual** -- `/mcp refresh` re-fetches resources alongside tools.
|
||||
|
||||
---
|
||||
|
||||
|
||||
+5
-6
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "turnstone"
|
||||
version = "1.8.0a6"
|
||||
version = "1.7.4"
|
||||
description = "Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
@@ -24,7 +24,7 @@ classifiers = [
|
||||
]
|
||||
dependencies = [
|
||||
"openai>=2.45", # GPT-5.6: typed reasoning.mode, prompt_cache_options, and cache_write_tokens
|
||||
"anthropic>=0.117", # tracks the release current at claude-opus-5 onboarding; hard runtime floor is still 0.105 (mid-conversation system blocks) — Opus 5 itself needs no new SDK surface (model ids are opaque strings; "refusal" has been in the StopReason literal since ~0.95). Raise this when adopting fast mode / server-side fallbacks / advisor / mid-conversation tool changes, which DO need newer typed params.
|
||||
"anthropic>=0.108", # claude-fable-5 support; hard runtime floor is 0.105 (mid-conversation system blocks)
|
||||
"httpx>=0.28",
|
||||
"mcp>=1.27,<2", # v2 is a breaking rewrite (2.0.0a1 live 2026-06-11; stable ~2026-07-27) — streamablehttp_client removed, 2-tuple transport, snake_case types; migrate deliberately
|
||||
"starlette>=1.3.1", # CVE-2026-54282 (path->authority host spoof) + CVE-2026-54283 (url-encoded form DoS); supersedes the PYSEC-2026-161 host-header path-injection floor
|
||||
@@ -88,10 +88,10 @@ include = [
|
||||
"turnstone/console/static/coordinator/*.js",
|
||||
"turnstone/shared_static/*.css",
|
||||
"turnstone/shared_static/*.js",
|
||||
"turnstone/shared_static/katex-0.18.1/**/*",
|
||||
"turnstone/shared_static/katex-0.17.0/**/*",
|
||||
"turnstone/shared_static/hljs-11.11.1/**/*",
|
||||
"turnstone/shared_static/mermaid-11.16.1/**/*",
|
||||
"turnstone/shared_static/hls-1.6.17/**/*",
|
||||
"turnstone/shared_static/mermaid-11.16.0/**/*",
|
||||
"turnstone/shared_static/hls-1.6.16/**/*",
|
||||
"turnstone/sdk/py.typed",
|
||||
"turnstone/deploy/*.yaml",
|
||||
"turnstone/deploy/Caddyfile",
|
||||
@@ -103,7 +103,6 @@ testpaths = ["tests"]
|
||||
markers = [
|
||||
"live: requires a running LLM backend",
|
||||
"allow_thread_leak: test intentionally leaves a background thread running (opts out of the leaked-thread guard)",
|
||||
"e2e_recovery: opt-in end-to-end SSE recovery harness (real server + real SSE consumers, scripted provider — NOT live, no LLM backend needed); tens of seconds each. CI lanes run ``-m 'not live and not e2e_recovery'``; select with ``-m e2e_recovery``.",
|
||||
]
|
||||
filterwarnings = [
|
||||
# mcp v1 deprecates streamablehttp_client for an entry point whose call
|
||||
|
||||
+14
-519
@@ -45,22 +45,6 @@ Shell harness (?split=): right (default) · down · three · none — boots the
|
||||
document.title stamps SPLIT-READY-<visible cells> on success and
|
||||
SPLIT-FAILED-<reason> when a driven split was denied — judge the focused
|
||||
cell's top accent bar, the separators, and the .shown tab marker.
|
||||
Proxy-brand harness (/proxybrand/livepass.html): back-to-console from a
|
||||
PROXIED node view, driven end to end. An iframe hosts a node page built
|
||||
from the REAL shell.js rail plus the REAL _JS_PROXY_SHIM (read out of
|
||||
turnstone/console/server.py by text, never imported -- scripts/ has no
|
||||
sys.path guard, so an import would silently pick up site-packages). The
|
||||
host clicks the brand's child span and, because the shim navigates the
|
||||
FRAME away, reads the frame's post-navigation location from the surviving
|
||||
top page. document.title stamps PROXYBRAND-READY, or
|
||||
PROXYBRAND-FAILED-<reason>: sub-not-repointed-server (nothing wired),
|
||||
showhome-also-ran (shell.js won the click), nav-<path> (went somewhere
|
||||
other than the console root), sub-not-console-<text>, aria-not-repointed,
|
||||
no-navigation, no-brand, no-sub. Needs --virtual-time-budget=9000;
|
||||
there is nothing to screenshot. Read the verdict from <title> --
|
||||
both literals also appear in the host page's inline script, so a bare
|
||||
grep over --dump-dom output false-positives.
|
||||
|
||||
Attachments harness (/attachments/livepass.html): the composer attachment
|
||||
chips + the sent-message attachment pills, both driven through the REAL
|
||||
code paths — createAttachmentController.rehydrate() builds the chips and
|
||||
@@ -95,30 +79,6 @@ Task-agent harness (/taskagent/livepass.html): the task_agent card — a task
|
||||
success, TASKAGENT-FAILED-... / TASKAGENT-ERROR when routing breaks, so a
|
||||
broken card can't screenshot green.
|
||||
|
||||
Copy harness (/copy/livepass.html): the copy-to-clipboard affordances — the
|
||||
per-bubble copy button in .msg-actions and the floating block-copy button
|
||||
over hovered fences / mermaid diagrams / tables (pointer-only; keyboard
|
||||
copies with Enter on the focused block) — driven through the REAL
|
||||
InteractivePane (replayHistory plus a live handleEvent stream turn, so the
|
||||
retry-holder buttons coexist with the persistent copy buttons on the last
|
||||
bubble; the turn ends idle, matching the affordances' idle-only gate).
|
||||
navigator.clipboard is stubbed to a recorder, hover/focus/keys are
|
||||
dispatched synthetically, and every copied payload is compared byte-exact
|
||||
against the SOURCE (fences, pipes, mermaid text, the bubble's raw
|
||||
markdown). + &theme=light. document.title stamps
|
||||
COPY-READY-<bubbles>-<blocks> only when every probe copied exact source;
|
||||
COPY-FAILED-<reason> otherwise. &kbd=1 probes the KEYBOARD path: focus a
|
||||
block, dispatch Enter — the block's source lands on the clipboard, the
|
||||
block carries the outcome flash class, and the floating button stays out
|
||||
of it — stamps COPY-KBD-READY / COPY-KBD-FAILED-<step>.
|
||||
Screenshot states: &flash=1 (visual-only
|
||||
run — no probes; floating button + ✓ state on the fence, holder bar
|
||||
revealed via focus) and &bare=1 (single hover, no decoration). Known
|
||||
capture artifact: the DARK-theme &flash=1 shot can omit the floating
|
||||
button's pixels (headless software compositor; the DOM state is correct
|
||||
and light theme paints) — judge the dark floating button from &bare=1
|
||||
and the ✓ state from the light shot. &stepmax=N bisects a paint
|
||||
regression to the interaction that triggers it.
|
||||
Perf harness (/perf/livepass.html): long-session performance baseline for the
|
||||
interactive pane — mounts the REAL InteractivePane at real scroll geometry
|
||||
(fixed-height mount, production CSS chain) and drives production-shaped
|
||||
@@ -183,24 +143,6 @@ def extract_admin_fragment() -> str:
|
||||
return html[start:end]
|
||||
|
||||
|
||||
def extract_proxy_shim(prefix: str = "/node/livepass-node") -> str:
|
||||
"""Pull ``_JS_PROXY_SHIM`` out of console/server.py BY TEXT, not import.
|
||||
|
||||
``scripts/`` has no ``sys.path`` guard, so ``import turnstone`` from here
|
||||
resolves to whatever is installed in site-packages rather than this
|
||||
checkout -- silently building the page from a DIFFERENT version of the
|
||||
shim than the one you are trying to verify. Read the source instead.
|
||||
"""
|
||||
src = (ROOT / "turnstone/console/server.py").read_text(encoding="utf-8")
|
||||
m = re.search(r'^_JS_PROXY_SHIM = """\\\n(.*?)^"""', src, re.S | re.M)
|
||||
if not m:
|
||||
raise SystemExit(
|
||||
"livepass: could not find _JS_PROXY_SHIM in turnstone/console/server.py "
|
||||
"-- the constant was renamed or reshaped; update extract_proxy_shim()."
|
||||
)
|
||||
return m.group(1).replace('"PREFIX_PLACEHOLDER"', json.dumps(prefix))
|
||||
|
||||
|
||||
def inject(template: str, marker: str, payload: str) -> str:
|
||||
begin = template.index(f"<!-- {marker}:BEGIN -->") + len(f"<!-- {marker}:BEGIN -->")
|
||||
end = template.index(f"<!-- {marker}:END -->")
|
||||
@@ -405,28 +347,6 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
<div id="toast" role="status" aria-live="polite"></div>
|
||||
<script>
|
||||
(function () {
|
||||
// Freeze window.fetch BEFORE the module scripts evaluate: auth.js
|
||||
// fires a boot-time whoami at import, and a non-OK answer from the
|
||||
// fixture server would CLEAR the permissions grant seeded below
|
||||
// mid-pass. A never-settling fetch keeps the seed authoritative;
|
||||
// everything the passes drive flows through the authFetch fixture
|
||||
// (reinstated after auth.js's window bridge runs — see the load
|
||||
// handler).
|
||||
window.fetch = function () {
|
||||
return new Promise(function () {});
|
||||
};
|
||||
// Grant the operator scopes admin.js gates on: _modelAuthEditable()
|
||||
// reads this exact key THROUGH the real auth.js hasPermission
|
||||
// (loaded below, before admin.js) — without the grant, or without
|
||||
// auth.js supplying window.hasPermission, the auth-constraints
|
||||
// stub below is dead code: _fetchModelAuthConstraints returns
|
||||
// before authFetch and every pass renders the Backend-auth section
|
||||
// in its read-only degraded state. The headless profile is fresh
|
||||
// per pass, so nothing else seeds it.
|
||||
sessionStorage.setItem(
|
||||
"turnstone_permissions",
|
||||
"admin.models,admin.mcp",
|
||||
);
|
||||
function reply(data) {
|
||||
return Promise.resolve({
|
||||
ok: true,
|
||||
@@ -452,15 +372,9 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
enabled: true, temperature: null, max_tokens: null,
|
||||
reasoning_effort: null, surface_persisted_reasoning: true,
|
||||
replay_reasoning_to_model: false,
|
||||
auth_mode: "static", obo_audience: "", obo_scopes: "",
|
||||
};
|
||||
window.__putCount = 0;
|
||||
// Held under a private name too: auth.js's legacy window bridge
|
||||
// (Object.assign(window, {authFetch})) runs at module-import time
|
||||
// and clobbers the plain window.authFetch assigned here — the load
|
||||
// handler reinstates the fixture from this name after the modules
|
||||
// have evaluated.
|
||||
window.__consoleAuthFetch = window.authFetch = function (url, opts) {
|
||||
window.authFetch = function (url, opts) {
|
||||
var method = (opts && opts.method) || "GET";
|
||||
if (method === "PUT" && url.indexOf("/model-definitions/def1") >= 0) {
|
||||
window.__putCount++;
|
||||
@@ -485,31 +399,13 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
known: true,
|
||||
capabilities: {
|
||||
context_window: 200000, supports_tools: true,
|
||||
supports_vision: true,
|
||||
supports_streaming: true, supports_vision: true,
|
||||
supports_web_search: true, supports_temperature: true,
|
||||
supports_effort: true,
|
||||
},
|
||||
});
|
||||
if (url.indexOf("/model-definitions/auth-constraints") >= 0)
|
||||
// Fetched by the shelf ON OPEN (showCreateModelModal /
|
||||
// showEditModelModal), so this stub is exercised by any pass that
|
||||
// opens the model editor — no tab-switch plumbing needed. Omitting
|
||||
// it would render the Backend-auth block in its degraded
|
||||
// no-suggestions state and quietly stop exercising the section.
|
||||
return reply({
|
||||
auth_audience_allowlist: ["api://example-gateway"],
|
||||
auth_grant_profile: "entra",
|
||||
dynamic_auth_modes: ["entra_app", "entra_obo", "rfc8693_obo"],
|
||||
scopes_auth_modes: ["rfc8693_obo"],
|
||||
app_identity_auth_modes: ["entra_app"],
|
||||
auth_mode_profiles: {
|
||||
entra_app: "entra", entra_obo: "entra",
|
||||
rfc8693_obo: "rfc8693",
|
||||
},
|
||||
});
|
||||
if (url.indexOf("/model-definitions/def1") >= 0) return reply(MODEL);
|
||||
if (url.indexOf("/model-definitions") >= 0)
|
||||
return reply({ models: [], default_alias: "fable-5" });
|
||||
if (url.indexOf("/model-definitions") >= 0) return reply({ models: [] });
|
||||
if (url.indexOf("/api/models") >= 0)
|
||||
return reply({ models: [
|
||||
{ alias: "fable-5", model: "claude-fable-5" },
|
||||
@@ -536,22 +432,10 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
</script>
|
||||
<script type="module" src="shared/utils.js"></script>
|
||||
<script type="module" src="shared/hatch.js"></script>
|
||||
<!-- The REAL auth.js, loaded (and therefore parsed) before admin.js's
|
||||
permission shims run any pass: it owns the sessionStorage parse
|
||||
contract and assigns the window.hasPermission /
|
||||
window.whenPermissionsReady globals the shims probe at call time.
|
||||
Without it the seeded permissions grant is never READ, the
|
||||
Backend-auth section renders read-only/hidden, and the
|
||||
auth-constraints stub above is dead code in every pass. -->
|
||||
<script type="module" src="shared/auth.js"></script>
|
||||
<script src="console-static/admin.js"></script>
|
||||
<script src="console-static/governance.js"></script>
|
||||
<script>
|
||||
window.addEventListener("load", function () {
|
||||
// Reinstate the fixture fetch now the modules (and auth.js's
|
||||
// window bridge) have evaluated — passes run after load, so every
|
||||
// shelf-open fetch flows through the fixture, not the bridge.
|
||||
window.authFetch = window.__consoleAuthFetch;
|
||||
var q = new URLSearchParams(location.search);
|
||||
if (q.get("theme") === "light")
|
||||
document.documentElement.dataset.theme = "light";
|
||||
@@ -774,108 +658,6 @@ SHELL_TEMPLATE = """<!doctype html>
|
||||
# call the same window.buildAttachmentPreview). The page frame is harness-only
|
||||
# chrome and not under review; the chips row and the pill row are.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
# The PROXIED NODE page: the real L-shell (so the rail brand is the real
|
||||
# element, with the real shell.js click listener on it) plus the real proxy
|
||||
# shim injected exactly where proxy_index puts it -- first thing inside
|
||||
# <body>, ahead of the deferred shell.js module. caps mirror a NODE, not
|
||||
# the console: brandSub "server" is what the shim has to overwrite, and
|
||||
# leaving it "console" would make the host's /console/i check vacuous.
|
||||
PROXYBRAND_FRAME_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>proxied node</title>
|
||||
<link rel="stylesheet" href="shared/base.css" />
|
||||
<link rel="stylesheet" href="shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="static/style.css" />
|
||||
<link rel="stylesheet" href="shared/shell.css" />
|
||||
</head>
|
||||
<body>
|
||||
<!-- SHIM:BEGIN -->
|
||||
<!-- SHIM:END -->
|
||||
<div id="header" style="display: none"><div id="status-bar"></div></div>
|
||||
<div id="main" style="padding: 18px">
|
||||
<h2 style="margin: 0 0 8px">Node dashboard</h2>
|
||||
</div>
|
||||
<div id="view-admin" style="display: none"></div>
|
||||
<script>
|
||||
window.TURNSTONE_SHELL_CAPS = { cluster: false, brandSub: "server" };
|
||||
window.TS_APP = {
|
||||
boot() {},
|
||||
getClusterState() { return { nodes: {} }; },
|
||||
onRender() {},
|
||||
};
|
||||
window.TS_ADMIN = {};
|
||||
// Record on the PARENT, which survives the frame's navigation.
|
||||
// A flag on the frame's own window dies with the document, so the
|
||||
// host would read undefined and pass -- a check that cannot fail.
|
||||
window.showHome = function () {
|
||||
try { window.parent.__showHomeRan = true; } catch (e) {}
|
||||
};
|
||||
</script>
|
||||
<script type="module" src="shared/shell.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
# The HOST page. The shim navigates the FRAME to "/", which would destroy
|
||||
# any verdict stamped inside it -- so the surviving top page reads the
|
||||
# frame's post-navigation location and stamps its own title instead. No
|
||||
# landing page at "/" is needed (the harness root serves a directory
|
||||
# listing) and no CDP client either.
|
||||
PROXYBRAND_HOST_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>proxybrand livepass</title>
|
||||
<style>
|
||||
html, body { margin: 0; height: 100%; }
|
||||
iframe { width: 100%; height: 100%; border: 0; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<iframe id="frame" src="frame.html"></iframe>
|
||||
<script>
|
||||
const frame = document.getElementById("frame");
|
||||
let phase = 0;
|
||||
const fail = (r) => { phase = 9; document.title = "PROXYBRAND-FAILED-" + r; };
|
||||
|
||||
frame.addEventListener("load", () => {
|
||||
if (phase === 9) return;
|
||||
if (phase === 0) {
|
||||
const doc = frame.contentDocument;
|
||||
const brand = doc.querySelector(".rail-brand .brand-home");
|
||||
if (!brand) return fail("no-brand");
|
||||
const sub = brand.querySelector(".brand-sub");
|
||||
if (!sub) return fail("no-sub");
|
||||
const text = sub.textContent.trim();
|
||||
if (text === "server") return fail("sub-not-repointed-server");
|
||||
if (!/console/i.test(text)) return fail("sub-not-console-" + text);
|
||||
if (brand.getAttribute("aria-label") !== "Back to console")
|
||||
return fail("aria-not-repointed");
|
||||
phase = 1;
|
||||
// Click the CHILD span, as a real user does: the shim must match
|
||||
// via contains(), not target identity.
|
||||
sub.click();
|
||||
setTimeout(() => { if (phase === 1) fail("no-navigation"); }, 2000);
|
||||
return;
|
||||
}
|
||||
const path = frame.contentWindow.location.pathname;
|
||||
const ranShowHome = !!window.__showHomeRan;
|
||||
phase = 2;
|
||||
if (path !== "/") return fail("nav-" + path);
|
||||
if (ranShowHome) return fail("showhome-also-ran");
|
||||
// Sticky, mirroring fail(): a third load must not re-stamp.
|
||||
phase = 9;
|
||||
document.title = "PROXYBRAND-READY";
|
||||
});
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
ATTACH_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
@@ -1039,24 +821,7 @@ ATTACH_TEMPLATE = """<!doctype html>
|
||||
# is exercised, not just the leaf builders. The page frame is harness-only
|
||||
# chrome; the .conv-batch / task_agent card is what's under review.
|
||||
# --------------------------------------------------------------------------
|
||||
# The host seams a mounted InteractivePane provides, stubbed once for every
|
||||
# harness that drives the REAL pane (taskagent, copy). A new required seam
|
||||
# gets added HERE — a harness left with a stale stub set does not fail at
|
||||
# review time, it throws HARNESS ERROR at run time.
|
||||
PANE_STUB_JS = """\
|
||||
// Drive the REAL pane; stub only the host seams a mounted pane provides.
|
||||
const pane = new InteractivePane("demo-ws");
|
||||
pane.messagesEl = messages;
|
||||
pane.inputEl = document.createElement("textarea");
|
||||
pane.sendBtn = document.createElement("button");
|
||||
pane.isNearBottom = () => false;
|
||||
pane.scrollToBottom = () => {};
|
||||
pane.removeEmptyState = () => {};
|
||||
pane.removeThinkingIndicator = () => {};
|
||||
pane.setBusy = () => {};"""
|
||||
|
||||
TASKAGENT_TEMPLATE = (
|
||||
"""<!doctype html>
|
||||
TASKAGENT_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
@@ -1106,9 +871,16 @@ TASKAGENT_TEMPLATE = (
|
||||
|
||||
const messages = document.getElementById("messages");
|
||||
try {
|
||||
"""
|
||||
+ PANE_STUB_JS
|
||||
+ """
|
||||
// Drive the REAL pane; stub only the host seams a mounted pane provides.
|
||||
const pane = new InteractivePane("demo-ws");
|
||||
pane.messagesEl = messages;
|
||||
pane.inputEl = document.createElement("textarea");
|
||||
pane.sendBtn = document.createElement("button");
|
||||
pane.isNearBottom = () => false;
|
||||
pane.scrollToBottom = () => {};
|
||||
pane.removeEmptyState = () => {};
|
||||
pane.removeThinkingIndicator = () => {};
|
||||
pane.setBusy = () => {};
|
||||
const ev = (e) => pane.handleEvent(e);
|
||||
|
||||
// ?recall=1: exercise the RECALL path — replayHistory rebuilding the
|
||||
@@ -1257,266 +1029,6 @@ TASKAGENT_TEMPLATE = (
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Copy harness — the copy-to-clipboard affordances over the REAL pane. The
|
||||
# bubbles come from the REAL replayHistory / handleEvent paths so the copy
|
||||
# sources are the ones production stashes (_copySource, the mermaid / table
|
||||
# data attributes), and the probes drive the REAL buttons and key path and
|
||||
# compare what landed on the (stubbed) clipboard byte-exact against the
|
||||
# source.
|
||||
# --------------------------------------------------------------------------
|
||||
COPY_TEMPLATE = (
|
||||
"""<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
||||
<title>copy livepass</title>
|
||||
<link rel="stylesheet" href="shared/base.css" />
|
||||
<link rel="stylesheet" href="shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="shared/chat.css" />
|
||||
<link rel="stylesheet" href="shared/conversation.css" />
|
||||
<link rel="stylesheet" href="shared/cards.css" />
|
||||
<link rel="stylesheet" href="shared/interactive.css" />
|
||||
<style>
|
||||
/* Harness-only framing (NOT under review) — a plausible pane context. */
|
||||
body {
|
||||
padding: 24px; margin: 0; background: var(--bg); color: var(--ink);
|
||||
font-family: var(--font-sans, system-ui, sans-serif);
|
||||
}
|
||||
.demo-frame { max-width: 720px; margin: 0 auto; }
|
||||
.demo-label {
|
||||
font: 11px var(--font-mono, monospace); color: var(--ink-3);
|
||||
text-transform: uppercase; letter-spacing: 0.08em; margin: 0 0 8px;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="demo-frame">
|
||||
<div class="demo-label">conversation — copy affordances (real InteractivePane)</div>
|
||||
<div class="messages" id="messages"></div>
|
||||
</div>
|
||||
<script>
|
||||
window.toast = { error: function (m) { console.log("toast:", m); } };
|
||||
window.authFetch = function () {
|
||||
return Promise.resolve({
|
||||
ok: true,
|
||||
json: function () { return Promise.resolve({}); },
|
||||
text: function () { return Promise.resolve(""); },
|
||||
});
|
||||
};
|
||||
// Deterministic clipboard: record instead of writing. localhost is a
|
||||
// secure context so copyTextToClipboard takes the async-API branch and
|
||||
// hits this stub; force isSecureContext for any odd serving setup.
|
||||
window.__copied = [];
|
||||
try {
|
||||
Object.defineProperty(window, "isSecureContext", { value: true });
|
||||
} catch (e) { /* already true */ }
|
||||
try {
|
||||
Object.defineProperty(navigator, "clipboard", {
|
||||
value: {
|
||||
writeText: function (t) {
|
||||
window.__copied.push(t);
|
||||
return Promise.resolve();
|
||||
},
|
||||
},
|
||||
configurable: true,
|
||||
});
|
||||
} catch (e) {
|
||||
document.title = "COPY-FAILED-clipboard-stub";
|
||||
}
|
||||
</script>
|
||||
<script type="module">
|
||||
import { InteractivePane } from "./shared/interactive.js";
|
||||
const q = new URLSearchParams(location.search);
|
||||
if (q.get("theme") === "light")
|
||||
document.documentElement.dataset.theme = "light";
|
||||
|
||||
const FENCE_SRC = 'def stash(depth):\\n total = 0\\n for k in range(depth):\\n total += k\\n return total';
|
||||
const TABLE_SRC = '| node | state |\\n|---|:--:|\\n| flat | idle |\\n| blck | busy |';
|
||||
const MERMAID_SRC = 'graph TD\\n A --> B\\n B --> C';
|
||||
const MD_ONE =
|
||||
'First reply with a fence and a table.\\n\\n' +
|
||||
'```python\\n' + FENCE_SRC + '\\n```\\n\\n' +
|
||||
TABLE_SRC + '\\n\\nTrailing prose under the table.';
|
||||
const MD_TWO =
|
||||
'Second reply with a diagram.\\n\\n' +
|
||||
'```mermaid\\n' + MERMAID_SRC + '\\n```\\n\\n' +
|
||||
'And `inline code` after it.';
|
||||
const MD_LIVE =
|
||||
'Streamed reply: the **live** turn, so the retry holder lands here.';
|
||||
|
||||
const messages = document.getElementById("messages");
|
||||
const fail = (r) => { document.title = "COPY-FAILED-" + r; };
|
||||
try {
|
||||
"""
|
||||
+ PANE_STUB_JS
|
||||
+ """
|
||||
|
||||
pane.replayHistory([
|
||||
{ role: "user", content: "Show me the stash helper and the node table." },
|
||||
{ role: "assistant", content: MD_ONE },
|
||||
{ role: "user", content: "Now the flow as a diagram, please." },
|
||||
{ role: "assistant", content: MD_TWO },
|
||||
]);
|
||||
// A live streamed turn on top — the retry holder must land on this
|
||||
// bubble WITHOUT stripping its (or any) copy button.
|
||||
pane.handleEvent({ type: "state_change", state: "running" });
|
||||
for (let k = 0; k < MD_LIVE.length; k += 16)
|
||||
pane.handleEvent({ type: "content", text: MD_LIVE.slice(k, k + 16) });
|
||||
pane.handleEvent({ type: "stream_end" });
|
||||
pane.handleEvent({ type: "state_change", state: "idle" });
|
||||
|
||||
const hover = (el) =>
|
||||
el.dispatchEvent(new MouseEvent("mouseover", { bubbles: true }));
|
||||
const fabEl = () => document.querySelector(".block-copy-btn");
|
||||
|
||||
// Let the streamed bubble's rAF render + retry attach settle.
|
||||
setTimeout(async () => {
|
||||
try {
|
||||
const bubbles = messages.querySelectorAll(".msg.assistant");
|
||||
const bars = messages.querySelectorAll(
|
||||
".msg.assistant .msg-actions .msg-copy-btn",
|
||||
);
|
||||
if (bubbles.length !== 3) return fail("bubbles" + bubbles.length);
|
||||
if (bars.length !== 3) return fail("bars" + bars.length);
|
||||
const last = bubbles[bubbles.length - 1];
|
||||
if (!last.querySelector(".msg-retry-btn"))
|
||||
return fail("no-retry-on-holder");
|
||||
if (!last.querySelector(".msg-copy-btn"))
|
||||
return fail("holder-lost-copy");
|
||||
|
||||
// Block probes: hover reveals the floating button; a click must
|
||||
// land the byte-exact SOURCE on the clipboard.
|
||||
const probes = [
|
||||
[messages.querySelector(".msg.assistant pre"), FENCE_SRC, "fence"],
|
||||
[messages.querySelector(".table-wrap"), TABLE_SRC, "table"],
|
||||
[messages.querySelector(".mermaid-container"), MERMAID_SRC, "mermaid"],
|
||||
];
|
||||
// &bare=1 — diagnostic state: no probe clicks, no repositioning;
|
||||
// one hover on the fence and stop. Splits "the probe cycle
|
||||
// corrupts the button's paint" from "it never paints here".
|
||||
if (q.get("bare") === "1") {
|
||||
hover(probes[0][0]);
|
||||
document.title = "COPY-BARE";
|
||||
return;
|
||||
}
|
||||
|
||||
// &kbd=1 — the keyboard path: Enter on a FOCUSED block copies
|
||||
// that block's source directly. Blocks are focusable (tabindex=0
|
||||
// from the fence / table / mermaid renders), the outcome flashes
|
||||
// on the block itself, and the floating button — pointer-only —
|
||||
// must stay out of it entirely (never created, never revealed).
|
||||
if (q.get("kbd") === "1") {
|
||||
const tw = messages.querySelector(".table-wrap");
|
||||
if (!tw) return fail("kbd-no-block");
|
||||
tw.focus();
|
||||
const focused = document.activeElement === tw;
|
||||
tw.dispatchEvent(
|
||||
new KeyboardEvent("keydown", { key: "Enter", bubbles: true }),
|
||||
);
|
||||
await new Promise((r) => setTimeout(r, 0));
|
||||
const copied =
|
||||
window.__copied[window.__copied.length - 1] === TABLE_SRC;
|
||||
const flashed = tw.classList.contains("is-copied");
|
||||
const fabStaysOut =
|
||||
!fabEl() || !fabEl().classList.contains("is-visible");
|
||||
document.title =
|
||||
focused && copied && flashed && fabStaysOut
|
||||
? "COPY-KBD-READY"
|
||||
: "COPY-KBD-FAILED-" +
|
||||
[
|
||||
focused ? "" : "focus",
|
||||
copied ? "" : "copy",
|
||||
flashed ? "" : "flash",
|
||||
fabStaysOut ? "" : "fab",
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join("-");
|
||||
return;
|
||||
}
|
||||
|
||||
// &flash=1 — the VISUAL state, screenshot-only: skip the probes so
|
||||
// the fence hover is the floating button's FIRST show. Returning
|
||||
// the button to an already-visited position stops it PAINTING in
|
||||
// headless captures (visible + hit-testable, no pixels — a stale
|
||||
// compositor tile; bisected via &stepmax). Function and pixels
|
||||
// are therefore split: the probe run (no flash) is the verdict,
|
||||
// this state is the picture.
|
||||
if (q.get("flash") === "1") {
|
||||
bars[bars.length - 1].focus();
|
||||
hover(probes[0][0]);
|
||||
const fab = fabEl();
|
||||
if (!fab) return fail("no-fab-visual");
|
||||
fab.classList.add("is-copied");
|
||||
fab.title = "Copied";
|
||||
// Freeze: the capture pipeline synthesizes a pointer event
|
||||
// outside the block at screenshot time, which would hide the
|
||||
// button (correct in production). Capture-phase stops starve
|
||||
// the module's delegated listeners for the capture.
|
||||
for (const t of ["mouseover", "scroll"])
|
||||
document.addEventListener(t, (e) => e.stopPropagation(), true);
|
||||
document.title = "COPY-VISUAL";
|
||||
return;
|
||||
}
|
||||
// &stepmax=N — diagnostic: stop after the Nth interaction (hovers
|
||||
// and clicks count) and stamp COPY-STEP-N, so a paint regression
|
||||
// can be bisected to the interaction that triggers it.
|
||||
let step = 0;
|
||||
const stepMax = parseInt(q.get("stepmax") || "999", 10);
|
||||
const gate = () => {
|
||||
step += 1;
|
||||
if (step > stepMax) {
|
||||
document.title = "COPY-STEP-" + (step - 1);
|
||||
throw { __stop: true };
|
||||
}
|
||||
};
|
||||
let done = 0;
|
||||
for (const [el, want, name] of probes) {
|
||||
if (!el) return fail("no-" + name);
|
||||
gate();
|
||||
hover(el);
|
||||
const fab = fabEl();
|
||||
if (!fab || !fab.classList.contains("is-visible"))
|
||||
return fail("fab-hidden-" + name);
|
||||
gate();
|
||||
fab.click();
|
||||
await new Promise((r) => setTimeout(r, 0));
|
||||
const got = window.__copied[window.__copied.length - 1];
|
||||
if (got !== want) {
|
||||
console.log("copy mismatch", name, JSON.stringify(got));
|
||||
return fail("source-" + name);
|
||||
}
|
||||
done += 1;
|
||||
}
|
||||
|
||||
// Bubble probe: the whole raw markdown, fences and pipes intact.
|
||||
gate();
|
||||
bars[0].click();
|
||||
await new Promise((r) => setTimeout(r, 0));
|
||||
if (window.__copied[window.__copied.length - 1] !== MD_ONE)
|
||||
return fail("bubble-source");
|
||||
|
||||
document.title = "COPY-READY-" + bars.length + "-" + done;
|
||||
} catch (e) {
|
||||
if (!(e && e.__stop)) {
|
||||
console.log("copy harness error", e);
|
||||
fail("error");
|
||||
}
|
||||
}
|
||||
}, 400);
|
||||
} catch (e) {
|
||||
messages.textContent = "HARNESS ERROR: " + e.message;
|
||||
fail("error");
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
@@ -1964,12 +1476,6 @@ def build(out: Path) -> None:
|
||||
(ta / "livepass.html").write_text(TASKAGENT_TEMPLATE, encoding="utf-8")
|
||||
print(f"{ta}/livepass.html — task_agent card (real Pane.handleEvent routing)")
|
||||
|
||||
cp = out / "copy"
|
||||
cp.mkdir(parents=True, exist_ok=True)
|
||||
symlink(cp / "shared", ROOT / "turnstone/shared_static")
|
||||
(cp / "livepass.html").write_text(COPY_TEMPLATE, encoding="utf-8")
|
||||
print(f"{cp}/livepass.html — copy affordances (bubble bars + block button)")
|
||||
|
||||
pf = out / "perf"
|
||||
pf.mkdir(parents=True, exist_ok=True)
|
||||
symlink(pf / "shared", ROOT / "turnstone/shared_static")
|
||||
@@ -1977,17 +1483,6 @@ def build(out: Path) -> None:
|
||||
(pf / "livepass.html").write_text(PERF_TEMPLATE, encoding="utf-8")
|
||||
print(f"{pf}/livepass.html — long-session perf baseline (real InteractivePane)")
|
||||
|
||||
pb = out / "proxybrand"
|
||||
pb.mkdir(parents=True, exist_ok=True)
|
||||
symlink(pb / "shared", ROOT / "turnstone/shared_static")
|
||||
symlink(pb / "static", ROOT / "turnstone/ui/static")
|
||||
shim = "<script>" + extract_proxy_shim() + "</script>"
|
||||
(pb / "frame.html").write_text(
|
||||
inject(PROXYBRAND_FRAME_TEMPLATE, "SHIM", shim), encoding="utf-8"
|
||||
)
|
||||
(pb / "livepass.html").write_text(PROXYBRAND_HOST_TEMPLATE, encoding="utf-8")
|
||||
print(f"{pb}/livepass.html — back-to-console brand (real shell.js + real shim)")
|
||||
|
||||
|
||||
class _PerfStore:
|
||||
"""Rendezvous for the perf page's POSTed JSON report."""
|
||||
|
||||
@@ -1,19 +0,0 @@
|
||||
# Entra config for the Entra e2e / spike harnesses. Copy to `.env` (gitignored)
|
||||
# and fill in from your tenant. `entra_setup.sh setup` creates the app
|
||||
# registrations and writes a populated `.env` for you.
|
||||
#
|
||||
# cp scripts/obo-e2e/.env.example scripts/obo-e2e/.env
|
||||
# # then edit, or run: ./scripts/obo-e2e/entra_setup.sh setup
|
||||
|
||||
export ENTRA_TENANT_ID=<tenant-guid-or-domain>
|
||||
export ENTRA_CLIENT_ID=<turnstone-spike-app-client-id>
|
||||
export ENTRA_CLIENT_SECRET=<client-secret>
|
||||
export SPIKE_AUDIENCE_A=api://<resource-app-a-guid> # a consented resource
|
||||
export SPIKE_AUDIENCE_B=api://<resource-app-b-guid> # a second consented resource
|
||||
export SPIKE_AUDIENCE_UNCONSENTED=api://<resource-app-c-guid> # NOT granted (negative case)
|
||||
export SPIKE_RUN_OBO=1
|
||||
# export SPIKE_PORT=8765 # redirect-listener port (default 8765)
|
||||
# export SPIKE_CALLBACK_FILE=/tmp/obo_cb.txt # remote-browser mode: paste the redirect URL here
|
||||
|
||||
# The Keycloak / OSS-path harness needs no config — keycloak_e2e.sh sets
|
||||
# everything and stands up an ephemeral container.
|
||||
@@ -1,214 +0,0 @@
|
||||
# OBO e2e harnesses — single-credential MCP token minting (`auth_type=oauth_obo`)
|
||||
|
||||
Manual test harnesses for the `oauth_obo` feature (issue #551). They exercise
|
||||
the **real** Turnstone mint path (`get_obo_access_token_classified` →
|
||||
`_obo_mint_entra` / `_obo_mint_rfc8693`) against a real identity provider — not
|
||||
mocks, not the unit suite. Two grant legs:
|
||||
|
||||
- **Entra** (`entra_e2e.py`) — real tenant, one interactive sign-in.
|
||||
- **Keycloak / RFC 8693** (`keycloak_e2e.py` + `.sh`) — ephemeral docker, fully
|
||||
headless.
|
||||
|
||||
There is also `entra_spike.py` (raw-OAuth **wire** probe, pre-implementation
|
||||
reference) and `entra_setup.sh` (creates the Entra app registrations + writes a
|
||||
populated `.env`).
|
||||
|
||||
**Secrets:** these read config from env. Real credentials live in a **gitignored
|
||||
`.env`** (copy `.env.example`); nothing tenant-specific is committed. The only
|
||||
literal secret in the tree is the ephemeral Keycloak container's throwaway
|
||||
`spike-secret`, which lives and dies with the container.
|
||||
|
||||
Not part of CI — run by hand when validating the feature against a live IdP.
|
||||
|
||||
## `entra_e2e.py` — end-to-end product exercise (post-implementation)
|
||||
|
||||
`entra_spike.py` verified the raw OAuth WIRE (before code existed). `entra_e2e.py`
|
||||
verifies the SHIPPED Turnstone code: it does a real Entra login, feeds the
|
||||
credential through the real `MCPTokenStore.upsert_oidc_credential` (the call the
|
||||
OIDC callback makes on capture), then drives the real
|
||||
`get_obo_access_token_classified` → `_obo_mint_entra` against the live Entra token
|
||||
endpoint. Checks E1–E7: real mint + aud claim, cache-hit (0 Entra calls),
|
||||
single-credential→audiences A&B, rotation write-back, force_refresh re-mint,
|
||||
unconsented-audience classification with the credential surviving, and
|
||||
flush→re-mint. Reuses the same `.env` and interactive login (SPIKE_CALLBACK_FILE
|
||||
for remote browser).
|
||||
|
||||
```bash
|
||||
source scripts/obo-e2e/.env
|
||||
uv run python scripts/obo-e2e/entra_e2e.py
|
||||
# one interactive sign-in; E1–E7 then run against the real product code. Results below.
|
||||
```
|
||||
|
||||
Results — RUN 2026-07-12 on the real tenant, ALL VERIFIED (exit 0): capture
|
||||
persisted; E1 mint A (aud=A app-id, cache row refresh_token_ct NULL); E2 cache
|
||||
hit (0 extra Entra calls); E3 mint B from the SAME credential (aud=B app-id); E4
|
||||
rotation write-back (RT rotated 2040→2091 chars, newest persisted); E5
|
||||
force_refresh re-mint (1 Entra call); E6 unconsented C → refresh_failed and the
|
||||
credential SURVIVES; E7 flush→re-mint. The real `get_obo_access_token_classified`
|
||||
→ `_obo_mint_entra` path against the live Entra token endpoint.
|
||||
|
||||
## `keycloak_e2e.py` + `keycloak_e2e.sh` — OSS path (RFC 8693), headless
|
||||
|
||||
The rfc8693 equivalent of `entra_e2e.py`: `keycloak_e2e.sh` spins up ephemeral
|
||||
Keycloak, configures the realm (turnstone client with standard token exchange,
|
||||
mcp-a/b/c clients, aud-mcp-a/b audience scopes, a test user), runs the harness
|
||||
against the real `get_obo_access_token_classified` → `_obo_mint_rfc8693`
|
||||
(refresh grant → token exchange), then tears down. No browser (password grant).
|
||||
|
||||
```bash
|
||||
./scripts/obo-e2e/keycloak_e2e.sh
|
||||
```
|
||||
|
||||
Results — RUN 2026-07-12, ALL VERIFIED: capture persisted; E1 mint A
|
||||
(refresh→exchange, aud=mcp-a, cache row refresh_token_ct NULL); E2 cache hit (0
|
||||
extra KC calls); E3 mint B from the SAME credential (aud=mcp-b); E4 rotation
|
||||
write-back (KC rotated the RT on the refresh leg, newest persisted); E5
|
||||
force_refresh re-mint (**2 KC calls** = the two-leg chain); E6 unconsented C →
|
||||
refresh_failed_transient (KC returns invalid_request for a missing audience
|
||||
scope → classified transient; credential SURVIVES either way); E7 flush→re-mint.
|
||||
Gotcha: dev-mode Keycloak boot is slow on a loaded host — the script now waits on
|
||||
kcadm auth (up to ~6 min) rather than a fixed sleep. Port 8091 (8090 = the dev
|
||||
console).
|
||||
|
||||
## Leg 1 — Entra (`entra_spike.py`) — NEEDS TENANT ACCESS
|
||||
|
||||
### Tenant / app-registration setup (one-time, ~15 min)
|
||||
|
||||
1. **Spike client app** (stands in for Turnstone's OIDC app registration):
|
||||
- New app registration, single tenant. Platform **Web**, redirect URI
|
||||
`http://localhost:8765/callback`. Create a **client secret**.
|
||||
2. **Two resource apps** (stand in for MCP servers A and B):
|
||||
- New app registrations `spike-mcp-a`, `spike-mcp-b`. In each:
|
||||
**Expose an API** → set Application ID URI (`api://<guid>`) → add a scope
|
||||
(e.g. `mcp.access`).
|
||||
3. **Delegated grants** (this is metaclassing's "proper tenant and app reg setup"):
|
||||
- On the spike client app → **API permissions** → add delegated permission to
|
||||
`spike-mcp-a` and `spike-mcp-b` scopes → **Grant admin consent**.
|
||||
- Optionally also add the spike client's app id to each resource app's
|
||||
`preAuthorizedApplications` (Expose an API → Add a client application) to
|
||||
compare against pure admin consent.
|
||||
4. **Unconsented control** (for V5): a third resource app `spike-mcp-c` with an
|
||||
exposed API but NO permission granted to the spike client.
|
||||
|
||||
### Run
|
||||
|
||||
```bash
|
||||
export ENTRA_TENANT_ID=... ENTRA_CLIENT_ID=... ENTRA_CLIENT_SECRET=...
|
||||
export SPIKE_AUDIENCE_A=api://<a-guid> SPIKE_AUDIENCE_B=api://<b-guid>
|
||||
export SPIKE_AUDIENCE_UNCONSENTED=api://<c-guid> # optional (V5)
|
||||
export SPIKE_RUN_OBO=1 # optional (V6)
|
||||
uv run python scripts/obo-e2e/entra_spike.py
|
||||
```
|
||||
|
||||
A browser opens for one interactive login (any tenant user). Everything after is
|
||||
non-interactive — that IS the feature.
|
||||
|
||||
### What each check pins down
|
||||
|
||||
| Check | Design assumption it verifies |
|
||||
| --- | --- |
|
||||
| V1 | `offline_access` on the login yields a client-bound RT (capture layer) |
|
||||
| V2/V3 | ONE RT redeems for access tokens of DIFFERENT audiences (`scope=<aud>/.default`) — the load-bearing Entra behavior |
|
||||
| V4 | rotation semantics → whether RT write-back on every mint is convenience or correctness-critical |
|
||||
| V5 | unconsented audience fails `AADSTS65001 consent_required` → maps to the reconnect-rail fallback, never a silent failure |
|
||||
| V6 | OBO jwt-bearer middle-tier variant works with the same app registration (comparison data only) |
|
||||
|
||||
Also record (manual): whether Conditional Access / MFA policies in the tenant
|
||||
produce `interaction_required` on redemption — that's the fallback path's other
|
||||
trigger.
|
||||
|
||||
### Results — RUN 2026-07-11 on a real tenant, ALL SIX VERIFIED
|
||||
|
||||
Tenant: personal default directory (Global Admin), user is an MSA member.
|
||||
Setup via `entra_setup.sh setup`; V3 initially failed (see gotcha below),
|
||||
passed after fixing the grant. Second run: V1-V6 all VERIFIED, exit 0.
|
||||
|
||||
| Check | Result |
|
||||
| --- | --- |
|
||||
| V1 offline_access login -> RT | VERIFIED (confidential client + PKCE, RT ~2KB) |
|
||||
| V2 RT -> audience A token | VERIFIED (`aud=<A app guid>`, ~70 min TTL, new RT returned) |
|
||||
| V3 SAME RT -> audience B token | **VERIFIED — the load-bearing claim: one RT, many audiences** |
|
||||
| V4 rotation | VERIFIED: RT rotates on every redemption, but the OLD RT stays valid (reuse HTTP 200) -> write-back-newest is required; races are benign on Entra |
|
||||
| V5 unconsented audience | VERIFIED: `invalid_grant` + `AADSTS65001` (error_codes=[65001]) -> clean mapping to the reconnect-rail fallback |
|
||||
| V6 OBO jwt-bearer variant | VERIFIED: middle-tier shape also works with the same app registration |
|
||||
|
||||
**Operator gotcha (feeds #682 + product docs):** `az ad app permission
|
||||
admin-consent` run immediately after SP creation SILENTLY skips
|
||||
not-yet-propagated resource SPs — grant A landed, grant B didn't, and the only
|
||||
symptom was AADSTS65001 at redemption. Verify grants after consent
|
||||
(`oauth2PermissionGrants` filter on the client SP) or write them directly with
|
||||
`az ad app permission grant --id <client> --api <resource> --scope <scope>`.
|
||||
Product-side implication: a missing tenant grant for a NEW oauth_obo server
|
||||
surfaces as AADSTS65001 -> the same reconnect-rail path as revocation; the
|
||||
admin docs must say "grant first, then add the server".
|
||||
|
||||
## Leg 2 — Keycloak RFC 8693 (portability check) — runnable locally
|
||||
|
||||
Ephemeral `quay.io/keycloak/keycloak:26.3` (`start-dev`, port 8089), realm
|
||||
`spike`, confidential client `turnstone` with **standard token exchange**
|
||||
enabled, resource clients `mcp-a`/`mcp-b`, user `alice`. Pipeline mirrors the
|
||||
product design for a generic-8693 IdP:
|
||||
|
||||
```
|
||||
stored user RT --(refresh grant)--> user AT --(RFC 8693 exchange, audience=mcp-X)--> audience-scoped AT
|
||||
```
|
||||
|
||||
i.e. the per-user credential stays ONE refresh token; per-server tokens are
|
||||
minted via standard token exchange instead of Entra's multi-resource RT
|
||||
redemption. Same substrate, different grant leg.
|
||||
|
||||
### Results — RUN 2026-07-11, VERIFIED (Keycloak 26.3, ephemeral)
|
||||
|
||||
```
|
||||
alice ONE stored RT
|
||||
-> refresh grant -> user AT (azp=turnstone); RT ROTATED on refresh
|
||||
-> 8693 exchange audience=mcp-a scope=aud-mcp-a -> AT aud=mcp-a user=alice 300s, NO RT
|
||||
-> 8693 exchange audience=mcp-b scope=aud-mcp-b -> AT aud=mcp-b (same subject AT)
|
||||
negative control audience=mcp-c -> invalid_client "Audience not found"
|
||||
```
|
||||
|
||||
Findings that feed the design:
|
||||
1. **One per-user credential -> N audience tokens: VERIFIED on a second IdP.**
|
||||
The substrate is portable; only the grant leg differs per IdP.
|
||||
2. **Exchanged tokens are cache-shaped** (short TTL, no RT) — per-server
|
||||
`mcp_user_tokens` rows as short-lived mint cache is the right model.
|
||||
3. **RT rotation happens here too** — newest-RT write-back on every redemption
|
||||
is a correctness requirement of the capture layer, not an Entra quirk.
|
||||
4. **The IdP-side "delegated grant" has a per-IdP shape**: Entra = API
|
||||
permissions + admin consent; Keycloak = audience client scopes attached to
|
||||
the requester client (optional scopes activate via `scope=` at exchange).
|
||||
Operator runbooks are per-IdP (#682 pattern), code is not.
|
||||
5. Gotchas hit: KC user needs a complete profile for direct grant ("Account is
|
||||
not fully set up"); optional audience scope must be requested explicitly or
|
||||
the exchange 400s with "Requested audience not available".
|
||||
|
||||
Repro (ephemeral, ~2 min):
|
||||
|
||||
```bash
|
||||
docker run -d --name kc-obo-spike -p 127.0.0.1:8089:8080 \
|
||||
-e KC_BOOTSTRAP_ADMIN_USERNAME=admin -e KC_BOOTSTRAP_ADMIN_PASSWORD=admin \
|
||||
quay.io/keycloak/keycloak:26.3 start-dev
|
||||
KC="docker exec kc-obo-spike /opt/keycloak/bin/kcadm.sh"
|
||||
$KC config credentials --server http://localhost:8080 --realm master --user admin --password admin
|
||||
$KC create realms -s realm=spike -s enabled=true
|
||||
$KC create clients -r spike -s clientId=turnstone -s enabled=true -s publicClient=false \
|
||||
-s secret=spike-secret -s directAccessGrantsEnabled=true \
|
||||
-s 'attributes={"standard.token.exchange.enabled":"true"}'
|
||||
$KC create clients -r spike -s clientId=mcp-a -s enabled=true -s publicClient=false -s secret=x
|
||||
$KC create clients -r spike -s clientId=mcp-b -s enabled=true -s publicClient=false -s secret=x
|
||||
$KC create users -r spike -s username=alice -s enabled=true -s email=a@s.test \
|
||||
-s emailVerified=true -s firstName=A -s lastName=S
|
||||
$KC set-password -r spike --username alice --new-password alice-pw
|
||||
TURNSTONE_UUID=$($KC get clients -r spike -q clientId=turnstone --fields id --format csv --noquotes)
|
||||
for t in mcp-a mcp-b; do
|
||||
SID=$($KC create client-scopes -r spike -s name=aud-$t -s protocol=openid-connect -i)
|
||||
$KC create client-scopes/$SID/protocol-mappers/models -r spike -s name=aud-$t \
|
||||
-s protocol=openid-connect -s protocolMapper=oidc-audience-mapper \
|
||||
-s "config={\"included.client.audience\":\"$t\",\"access.token.claim\":\"true\"}"
|
||||
$KC update clients/$TURNSTONE_UUID/optional-client-scopes/$SID -r spike
|
||||
done
|
||||
# then: password grant -> refresh grant -> token-exchange with
|
||||
# grant_type=urn:ietf:params:oauth:grant-type:token-exchange,
|
||||
# subject_token=<user AT>, subject_token_type=...:access_token,
|
||||
# audience=mcp-a, scope=aud-mcp-a
|
||||
```
|
||||
@@ -1,286 +0,0 @@
|
||||
"""End-to-end exercise of the oauth_obo feature against a REAL Entra tenant.
|
||||
|
||||
Unlike ``entra_spike.py`` (which verified the raw OAuth wire shapes), this
|
||||
drives the ACTUAL Turnstone product code — real ``MCPTokenStore``, real
|
||||
``get_obo_access_token_classified`` → ``_obo_mint_entra`` → the real Entra
|
||||
token endpoint — so a green run proves the shipped mint engine works against
|
||||
live Entra, not just that the protocol does.
|
||||
|
||||
Flow:
|
||||
1. Interactive Entra login (auth-code + PKCE + offline_access) → a real
|
||||
refresh credential. This is what ``handle_oidc_callback`` receives.
|
||||
2. Persist it via ``MCPTokenStore.upsert_oidc_credential`` — the exact call
|
||||
the OIDC callback makes on capture (auth.py). The rest of the callback
|
||||
(JWKS validation, user provisioning) is OIDC-generic and unit-tested; the
|
||||
novel path is capture + mint, which this exercises for real.
|
||||
3. Seed real ``oauth_obo`` ``mcp_servers`` rows (audiences A/B consented, C
|
||||
not) and drive ``get_obo_access_token_classified`` — the real dispatch-time
|
||||
entry point — asserting on the minted tokens, cache, rotation, and
|
||||
classification.
|
||||
|
||||
Checks (VERIFIED / FAILED per line):
|
||||
E1 mint for audience A → kind=token; decoded aud == A; cache row written with
|
||||
refresh_token_ct NULL (cache, not custody); expires_at set
|
||||
E2 second call for A → cache hit, ZERO additional Entra calls
|
||||
E3 mint for audience B from the SAME captured credential → aud == B
|
||||
(the single-credential-many-audiences thesis, through the real engine)
|
||||
E4 rotation write-back: the stored credential holds the newest refresh token
|
||||
E5 force_refresh → a fresh mint (Entra call count increments)
|
||||
E6 unconsented audience C → NOT kind=token, and the shared credential SURVIVES
|
||||
(never auto-deleted — the load-bearing custody invariant)
|
||||
E7 cache flush → re-mint: deleting the cache row makes the next call re-mint
|
||||
|
||||
Run:
|
||||
source scripts/obo-e2e/.env
|
||||
uv run python scripts/obo-e2e/entra_e2e.py
|
||||
Env (from .env): ENTRA_TENANT_ID, ENTRA_CLIENT_ID, ENTRA_CLIENT_SECRET,
|
||||
SPIKE_AUDIENCE_A, SPIKE_AUDIENCE_B, SPIKE_AUDIENCE_UNCONSENTED, SPIKE_PORT.
|
||||
Remote browser: set SPIKE_CALLBACK_FILE to paste the redirect URL (as before).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
# Reuse the verified interactive-login machinery from the wire spike.
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from entra_spike import interactive_login, jwt_claims_unverified, redact # noqa: E402
|
||||
|
||||
from turnstone.core.mcp_crypto import ( # noqa: E402
|
||||
MCPTokenCipher,
|
||||
MCPTokenCipherConfig,
|
||||
MCPTokenStore,
|
||||
)
|
||||
from turnstone.core.mcp_oauth import get_obo_access_token_classified # noqa: E402
|
||||
from turnstone.core.oidc import OIDCConfig # noqa: E402
|
||||
from turnstone.core.storage._sqlite import SQLiteBackend # noqa: E402
|
||||
|
||||
USER = "e2e-user"
|
||||
RESULTS: list[tuple[str, str]] = []
|
||||
|
||||
|
||||
def record(status: str, msg: str) -> None:
|
||||
RESULTS.append((status, msg))
|
||||
print(f"[{status:>8}] {msg}")
|
||||
|
||||
|
||||
def aud_matches(token: str, want_audience: str) -> tuple[bool, str]:
|
||||
"""Compare a minted access token's aud claim to the configured audience.
|
||||
|
||||
Entra returns aud as the bare app-id GUID or the full ``api://<guid>`` URI;
|
||||
accept either.
|
||||
"""
|
||||
claims = jwt_claims_unverified(token)
|
||||
aud = str(claims.get("aud", "<none>"))
|
||||
want = want_audience.removeprefix("api://")
|
||||
return aud in (want, want_audience), aud
|
||||
|
||||
|
||||
class _CountingClient:
|
||||
"""Wraps httpx.AsyncClient, counting token-endpoint POSTs so cache hits
|
||||
(which must issue zero) are observable."""
|
||||
|
||||
def __init__(self, inner: httpx.AsyncClient) -> None:
|
||||
self._inner = inner
|
||||
self.posts = 0
|
||||
|
||||
async def post(self, *args: Any, **kwargs: Any) -> httpx.Response:
|
||||
self.posts += 1
|
||||
return await self._inner.post(*args, **kwargs)
|
||||
|
||||
|
||||
def _make_app_state(
|
||||
storage: SQLiteBackend,
|
||||
store: MCPTokenStore,
|
||||
oidc_config: OIDCConfig,
|
||||
http_client: _CountingClient,
|
||||
) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
auth_storage=storage,
|
||||
mcp_token_store=store,
|
||||
oidc_config=oidc_config,
|
||||
obo_http_client=http_client,
|
||||
mcp_oauth_refresh_locks={},
|
||||
mcp_oauth_refresh_backoff={},
|
||||
)
|
||||
|
||||
|
||||
def _seed_obo_server(storage: SQLiteBackend, name: str, audience: str) -> None:
|
||||
storage.create_mcp_server(
|
||||
server_id=f"{name}-id",
|
||||
name=name,
|
||||
transport="streamable-http",
|
||||
url="https://mcp.example.invalid/sse",
|
||||
auth_type="oauth_obo",
|
||||
oauth_audience=audience,
|
||||
)
|
||||
|
||||
|
||||
async def _run(cfg: dict[str, str], refresh_token: str) -> None:
|
||||
tenant = cfg["ENTRA_TENANT_ID"]
|
||||
issuer = f"https://login.microsoftonline.com/{tenant}/v2.0"
|
||||
token_endpoint = f"https://login.microsoftonline.com/{tenant}/oauth2/v2.0/token"
|
||||
aud_a = cfg["SPIKE_AUDIENCE_A"]
|
||||
aud_b = cfg["SPIKE_AUDIENCE_B"]
|
||||
aud_c = cfg.get("SPIKE_AUDIENCE_UNCONSENTED", "")
|
||||
|
||||
# Real Turnstone objects.
|
||||
db_path = os.path.join(tempfile.mkdtemp(prefix="obo-e2e-"), "e2e.db")
|
||||
storage = SQLiteBackend(db_path)
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
raw = base64.urlsafe_b64decode(Fernet.generate_key())
|
||||
store = MCPTokenStore(storage, MCPTokenCipher(MCPTokenCipherConfig(keys=(raw,))), node_id="e2e")
|
||||
oidc_config = OIDCConfig(
|
||||
enabled=True,
|
||||
issuer=issuer,
|
||||
client_id=cfg["ENTRA_CLIENT_ID"],
|
||||
client_secret=cfg["ENTRA_CLIENT_SECRET"],
|
||||
token_endpoint=token_endpoint,
|
||||
obo_grant_profile="entra",
|
||||
capture_user_credential=True,
|
||||
)
|
||||
|
||||
# Step 2 — CAPTURE: the exact storage call handle_oidc_callback makes.
|
||||
store.upsert_oidc_credential(USER, issuer, refresh_token=refresh_token)
|
||||
cap = store.get_oidc_credential(USER, issuer)
|
||||
if cap and cap["refresh_token"] == refresh_token:
|
||||
record("VERIFIED", f"capture: credential persisted for {USER} ({redact(refresh_token)})")
|
||||
else:
|
||||
record("FAILED", "capture: credential did not round-trip")
|
||||
return
|
||||
|
||||
_seed_obo_server(storage, "e2e-a", aud_a)
|
||||
_seed_obo_server(storage, "e2e-b", aud_b)
|
||||
if aud_c:
|
||||
_seed_obo_server(storage, "e2e-c", aud_c)
|
||||
|
||||
inner = httpx.AsyncClient(timeout=20.0)
|
||||
client = _CountingClient(inner)
|
||||
app_state = _make_app_state(storage, store, oidc_config, client)
|
||||
try:
|
||||
# E1 — real mint for audience A.
|
||||
r = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
if r.kind == "token" and r.token:
|
||||
ok, aud = aud_matches(r.token, aud_a)
|
||||
row = storage.get_mcp_user_token(USER, "e2e-a")
|
||||
cache_ok = (
|
||||
row is not None and row["refresh_token_ct"] is None and bool(row["expires_at"])
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if ok and cache_ok else "FAILED",
|
||||
f"E1 mint A: kind=token aud={aud} want={aud_a} cache_row_refreshless={cache_ok}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E1 mint A: kind={r.kind} (expected token)")
|
||||
return
|
||||
|
||||
# E2 — cache hit issues zero Entra calls.
|
||||
posts_before = client.posts
|
||||
r2 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r2.kind == "token" and client.posts == posts_before else "FAILED",
|
||||
f"E2 cache hit: kind={r2.kind} extra_entra_calls={client.posts - posts_before} (want 0)",
|
||||
)
|
||||
|
||||
# E3 — same credential, audience B.
|
||||
rb = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-b"
|
||||
)
|
||||
if rb.kind == "token" and rb.token:
|
||||
ok_b, aud_bclaim = aud_matches(rb.token, aud_b)
|
||||
record(
|
||||
"VERIFIED" if ok_b else "FAILED",
|
||||
f"E3 mint B from SAME credential: aud={aud_bclaim} want={aud_b}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E3 mint B: kind={rb.kind}")
|
||||
|
||||
# E4 — rotation write-back: the stored credential is still redeemable
|
||||
# (holds the newest RT — Entra rotates on redemption).
|
||||
cred_now = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if cred_now is not None else "FAILED",
|
||||
f"E4 rotation write-back: credential persisted {redact(cred_now['refresh_token']) if cred_now else '<gone>'}",
|
||||
)
|
||||
|
||||
# E5 — force_refresh re-mints (a real Entra call).
|
||||
posts_before = client.posts
|
||||
rf = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a", force_refresh=True
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if rf.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E5 force_refresh re-mint: kind={rf.kind} entra_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# E6 — unconsented audience: not a token, and the credential SURVIVES.
|
||||
if aud_c:
|
||||
rc = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-c"
|
||||
)
|
||||
cred_after = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if rc.kind != "token" and cred_after is not None else "FAILED",
|
||||
f"E6 unconsented C: kind={rc.kind} (not token) credential_survives={cred_after is not None}",
|
||||
)
|
||||
else:
|
||||
record("SKIPPED", "E6 unconsented C: SPIKE_AUDIENCE_UNCONSENTED not set")
|
||||
|
||||
# E7 — cache flush → re-mint.
|
||||
store.delete_user_token(USER, "e2e-a")
|
||||
posts_before = client.posts
|
||||
r7 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r7.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E7 flush→re-mint: kind={r7.kind} entra_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
finally:
|
||||
await inner.aclose()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"ENTRA_TENANT_ID",
|
||||
"ENTRA_CLIENT_ID",
|
||||
"ENTRA_CLIENT_SECRET",
|
||||
"SPIKE_AUDIENCE_A",
|
||||
"SPIKE_AUDIENCE_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in os.environ if k.startswith(("ENTRA_", "SPIKE_"))}
|
||||
missing = [k for k in required if not cfg.get(k)]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)} — did you `source scripts/obo-e2e/.env`?")
|
||||
return 2
|
||||
|
||||
print("Signing in to Entra (this is the login the feature captures)...")
|
||||
tokens = interactive_login(cfg)
|
||||
refresh_token = tokens.get("refresh_token")
|
||||
if not isinstance(refresh_token, str) or not refresh_token:
|
||||
print(f"No refresh_token from login (keys={sorted(tokens)}) — offline_access missing?")
|
||||
return 1
|
||||
|
||||
asyncio.run(_run(cfg, refresh_token))
|
||||
|
||||
print("\n=== summary ===")
|
||||
for status, msg in RESULTS:
|
||||
print(f" {status:>8} {msg}")
|
||||
return 0 if all(s in ("VERIFIED", "SKIPPED") for s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,134 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Entra spike setup for entra_spike.py (#551 re-scope boundary spike).
|
||||
# Manual test tooling — not run in CI. Creates throwaway Entra app registrations.
|
||||
#
|
||||
# ./entra_setup.sh setup create app registrations + consent + .env
|
||||
# ./entra_setup.sh cleanup delete everything it created (incl. .env)
|
||||
#
|
||||
# Creates in the logged-in tenant (az login first):
|
||||
# spike-turnstone confidential client (stands in for Turnstone's OIDC app)
|
||||
# spike-mcp-a/b resource apps exposing scope mcp.access, admin-consented
|
||||
# spike-mcp-c resource app with NO grant to the client (V5 control)
|
||||
# Requires: the logged-in user can create apps + grant admin consent
|
||||
# (Global Admin on a personal tenant qualifies).
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
ENV_FILE=".env"
|
||||
NAMES=(spike-turnstone spike-mcp-a spike-mcp-b spike-mcp-c)
|
||||
|
||||
log() { printf '>> %s\n' "$*"; }
|
||||
|
||||
graph_patch_api() { # $1=appId $2=scope-uuid $3=display-name
|
||||
local obj_id
|
||||
obj_id=$(az ad app show --id "$1" --query id -o tsv)
|
||||
az rest --method PATCH \
|
||||
--url "https://graph.microsoft.com/v1.0/applications/${obj_id}" \
|
||||
--headers 'Content-Type=application/json' \
|
||||
--body "{
|
||||
\"identifierUris\": [\"api://$1\"],
|
||||
\"api\": {
|
||||
\"requestedAccessTokenVersion\": 2,
|
||||
\"oauth2PermissionScopes\": [{
|
||||
\"id\": \"$2\",
|
||||
\"value\": \"mcp.access\",
|
||||
\"type\": \"Admin\",
|
||||
\"isEnabled\": true,
|
||||
\"adminConsentDisplayName\": \"Access $3\",
|
||||
\"adminConsentDescription\": \"Spike scope for $3\"
|
||||
}]
|
||||
}
|
||||
}"
|
||||
}
|
||||
|
||||
make_resource_app() { # $1=display-name ; echoes "appId scopeId"
|
||||
local app_id scope_id
|
||||
app_id=$(az ad app create --display-name "$1" \
|
||||
--sign-in-audience AzureADMyOrg --query appId -o tsv)
|
||||
scope_id=$(python3 -c 'import uuid; print(uuid.uuid4())')
|
||||
graph_patch_api "$app_id" "$scope_id" "$1" >/dev/null
|
||||
az ad sp create --id "$app_id" >/dev/null 2>&1 || true
|
||||
echo "$app_id $scope_id"
|
||||
}
|
||||
|
||||
cmd_setup() {
|
||||
local tenant_id
|
||||
tenant_id=$(az account show --query tenantId -o tsv)
|
||||
log "tenant: ${tenant_id}"
|
||||
|
||||
log "creating resource apps (a, b, c)..."
|
||||
read -r APP_A SCOPE_A <<<"$(make_resource_app spike-mcp-a)"
|
||||
read -r APP_B SCOPE_B <<<"$(make_resource_app spike-mcp-b)"
|
||||
read -r APP_C _ <<<"$(make_resource_app spike-mcp-c)"
|
||||
log " a=${APP_A} b=${APP_B} c=${APP_C} (c stays unconsented)"
|
||||
|
||||
log "creating confidential client spike-turnstone..."
|
||||
CLIENT_ID=$(az ad app create --display-name spike-turnstone \
|
||||
--sign-in-audience AzureADMyOrg \
|
||||
--web-redirect-uris "http://localhost:8765/callback" \
|
||||
--query appId -o tsv)
|
||||
az ad sp create --id "$CLIENT_ID" >/dev/null 2>&1 || true
|
||||
# No stderr suppression here: the secret is load-bearing (it lands in .env),
|
||||
# so under `set -e` a reset failure must abort LOUDLY, not silently.
|
||||
SECRET=$(az ad app credential reset --id "$CLIENT_ID" \
|
||||
--display-name spike --years 1 --query password -o tsv)
|
||||
|
||||
log "adding delegated permissions (a, b — NOT c)..."
|
||||
# Tolerated failures (|| log): a re-run hits "permission already exists" and
|
||||
# SP-propagation delays are common right after app creation — the
|
||||
# admin-consent retry loop below is the real gate. `set -e` would otherwise
|
||||
# turn a suppressed non-zero here into a silent mid-script abort.
|
||||
az ad app permission add --id "$CLIENT_ID" \
|
||||
--api "$APP_A" --api-permissions "${SCOPE_A}=Scope" \
|
||||
|| log " warn: permission add for a failed (may already exist); admin-consent below will confirm"
|
||||
az ad app permission add --id "$CLIENT_ID" \
|
||||
--api "$APP_B" --api-permissions "${SCOPE_B}=Scope" \
|
||||
|| log " warn: permission add for b failed (may already exist); admin-consent below will confirm"
|
||||
|
||||
log "granting admin consent (retries while SPs propagate)..."
|
||||
local ok=""
|
||||
for i in 1 2 3 4 5; do
|
||||
if az ad app permission admin-consent --id "$CLIENT_ID" 2>/dev/null; then
|
||||
ok=1; break
|
||||
fi
|
||||
log " not yet (attempt $i) — waiting 15s"
|
||||
sleep 15
|
||||
done
|
||||
[ -n "$ok" ] || { log "admin-consent failed after retries — grant manually in the portal (API permissions blade) and re-run the spike"; }
|
||||
|
||||
# Single-quote the values in the generated .env: the AS-issued client secret
|
||||
# can contain $ / backtick, and an unquoted RHS would be re-expanded (or
|
||||
# partially executed) when the operator `source`s the file. The heredoc still
|
||||
# interpolates ${...} into the single-quoted output; sourcing then treats the
|
||||
# result literally. (Azure secrets are base64-ish — no single quotes to escape.)
|
||||
umask 177
|
||||
cat > "$ENV_FILE" <<EOF
|
||||
export ENTRA_TENANT_ID='${tenant_id}'
|
||||
export ENTRA_CLIENT_ID='${CLIENT_ID}'
|
||||
export ENTRA_CLIENT_SECRET='${SECRET}'
|
||||
export SPIKE_AUDIENCE_A='api://${APP_A}'
|
||||
export SPIKE_AUDIENCE_B='api://${APP_B}'
|
||||
export SPIKE_AUDIENCE_UNCONSENTED='api://${APP_C}'
|
||||
export SPIKE_RUN_OBO=1
|
||||
EOF
|
||||
log "wrote ${ENV_FILE} (chmod 600). Next:"
|
||||
log " source scripts/obo-e2e/.env && uv run python scripts/obo-e2e/entra_spike.py"
|
||||
log "cleanup later with: ./entra_setup.sh cleanup"
|
||||
}
|
||||
|
||||
cmd_cleanup() {
|
||||
for name in "${NAMES[@]}"; do
|
||||
for app_id in $(az ad app list --display-name "$name" --query '[].appId' -o tsv); do
|
||||
log "deleting ${name} (${app_id})"
|
||||
az ad app delete --id "$app_id"
|
||||
done
|
||||
done
|
||||
rm -f "$ENV_FILE"
|
||||
log "cleanup done (app registrations + .env removed)"
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
setup) cmd_setup ;;
|
||||
cleanup) cmd_cleanup ;;
|
||||
*) echo "usage: $0 setup|cleanup"; exit 2 ;;
|
||||
esac
|
||||
@@ -1,333 +0,0 @@
|
||||
"""Entra boundary spike for single-credential MCP token minting (#551 re-scope).
|
||||
|
||||
Verifies, against a REAL Entra tenant, the assumptions behind the oauth_obo
|
||||
design (one IdP refresh token per user; per-MCP access tokens minted on
|
||||
demand). Each check prints VERIFIED / FAILED / SKIPPED plus redacted evidence.
|
||||
|
||||
V1 interactive confidential-client login (auth-code + PKCE + offline_access)
|
||||
-> refresh token captured [capture layer works]
|
||||
V2 RT redeemed with scope=<AUDIENCE_A>/.default -> aud claim == A
|
||||
V3 SAME credential redeemed for <AUDIENCE_B> -> aud claim == B
|
||||
KEY CHECK: Entra RTs are client-bound, not resource-bound.
|
||||
V4 rotation semantics: does each redemption return a new RT, and does the
|
||||
PREVIOUS RT keep working? [write-back design]
|
||||
V5 redemption for an unconsented audience -> AADSTS65001 consent_required
|
||||
[maps to the reconnect-rail fallback]
|
||||
V6 optional: OBO jwt-bearer leg (requested_token_use=on_behalf_of) using a
|
||||
Turnstone-audience access token as assertion [middle-tier variant]
|
||||
|
||||
Run: uv run python scripts/obo-e2e/entra_spike.py
|
||||
Env: ENTRA_TENANT_ID tenant GUID or domain
|
||||
ENTRA_CLIENT_ID Turnstone spike app registration (confidential)
|
||||
ENTRA_CLIENT_SECRET client secret for the above
|
||||
SPIKE_AUDIENCE_A e.g. api://<guid-a> (exposes a scope, consented)
|
||||
SPIKE_AUDIENCE_B e.g. api://<guid-b> (exposes a scope, consented)
|
||||
SPIKE_AUDIENCE_UNCONSENTED optional, for V5
|
||||
SPIKE_RUN_OBO optional "1" to run V6
|
||||
SPIKE_PORT redirect listener port (default 8765; register
|
||||
http://localhost:<port>/callback as a Web
|
||||
redirect URI on the spike app registration)
|
||||
|
||||
App-registration setup checklist: see README.md next to this file.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import secrets
|
||||
import sys
|
||||
import threading
|
||||
import urllib.parse
|
||||
import webbrowser
|
||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
RESULTS: list[tuple[str, str, str]] = [] # (check, status, evidence)
|
||||
|
||||
|
||||
def record(check: str, status: str, evidence: str) -> None:
|
||||
RESULTS.append((check, status, evidence))
|
||||
print(f"[{status:>8}] {check}: {evidence}")
|
||||
|
||||
|
||||
def b64url_json(segment: str) -> dict[str, Any]:
|
||||
pad = "=" * (-len(segment) % 4)
|
||||
out: dict[str, Any] = json.loads(base64.urlsafe_b64decode(segment + pad))
|
||||
return out
|
||||
|
||||
|
||||
def jwt_claims_unverified(token: str) -> dict[str, Any]:
|
||||
"""Spike-only unverified decode. NEVER do this in product code."""
|
||||
try:
|
||||
return b64url_json(token.split(".")[1])
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def redact(token: str | None) -> str:
|
||||
if not token:
|
||||
return "<absent>"
|
||||
return f"{token[:8]}...({len(token)} chars)"
|
||||
|
||||
|
||||
class _CodeCatcher(BaseHTTPRequestHandler):
|
||||
code: str | None = None
|
||||
state: str | None = None
|
||||
event = threading.Event()
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802 - stdlib API name
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(self.path).query)
|
||||
_CodeCatcher.code = (q.get("code") or [None])[0]
|
||||
_CodeCatcher.state = (q.get("state") or [None])[0]
|
||||
body = b"Spike login captured - return to the terminal."
|
||||
if q.get("error"):
|
||||
body = f"IdP error: {q}".encode()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/plain")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
_CodeCatcher.event.set()
|
||||
|
||||
def log_message(self, *args: Any) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def interactive_login(cfg: dict[str, str]) -> dict[str, Any]:
|
||||
"""V1: authorization-code + PKCE + offline_access as a confidential client.
|
||||
|
||||
Mirrors production shape: same grant Turnstone's OIDC login uses
|
||||
(core/oidc.py exchange_code), plus offline_access.
|
||||
"""
|
||||
port = int(cfg.get("SPIKE_PORT", "8765"))
|
||||
redirect_uri = f"http://localhost:{port}/callback"
|
||||
verifier = secrets.token_urlsafe(48)
|
||||
challenge = (
|
||||
base64.urlsafe_b64encode(hashlib.sha256(verifier.encode()).digest()).rstrip(b"=").decode()
|
||||
)
|
||||
state = secrets.token_urlsafe(16)
|
||||
authorize = (
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/authorize?"
|
||||
+ urllib.parse.urlencode(
|
||||
{
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"response_type": "code",
|
||||
"redirect_uri": redirect_uri,
|
||||
"response_mode": "query",
|
||||
# offline_access is THE capture-layer delta vs today's login.
|
||||
# No resource scope here: the RT is minted client-bound.
|
||||
"scope": "openid profile offline_access",
|
||||
"state": state,
|
||||
"code_challenge": challenge,
|
||||
"code_challenge_method": "S256",
|
||||
}
|
||||
)
|
||||
)
|
||||
server = HTTPServer(("127.0.0.1", port), _CodeCatcher)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
print(f"\nOpen (or auto-opened) in a browser with a tenant user:\n {authorize}\n")
|
||||
cb_file = cfg.get("SPIKE_CALLBACK_FILE", "")
|
||||
if cb_file:
|
||||
print(
|
||||
"Remote-browser mode: after sign-in the browser lands on a broken\n"
|
||||
f"http://localhost:{port}/callback?... page. Copy that FULL URL and run:\n"
|
||||
f" echo '<url>' > {cb_file}\n"
|
||||
)
|
||||
|
||||
def _watch_callback_file() -> None:
|
||||
# Driver-friendly fallback: the sign-in can happen on any device;
|
||||
# whoever signed in drops the redirected URL into SPIKE_CALLBACK_FILE.
|
||||
import time as _time
|
||||
|
||||
while not _CodeCatcher.event.is_set():
|
||||
try:
|
||||
with open(cb_file) as _f:
|
||||
pasted = _f.read().strip()
|
||||
except OSError:
|
||||
pasted = ""
|
||||
if "?" in pasted:
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(pasted).query)
|
||||
_CodeCatcher.code = (q.get("code") or [None])[0]
|
||||
_CodeCatcher.state = (q.get("state") or [None])[0]
|
||||
_CodeCatcher.event.set()
|
||||
return
|
||||
_time.sleep(1.0)
|
||||
|
||||
if cb_file:
|
||||
threading.Thread(target=_watch_callback_file, daemon=True).start()
|
||||
webbrowser.open(authorize)
|
||||
if not _CodeCatcher.event.wait(timeout=600):
|
||||
server.shutdown()
|
||||
raise SystemExit("Timed out waiting for the redirect (10 min).")
|
||||
server.shutdown()
|
||||
if _CodeCatcher.state != state:
|
||||
raise SystemExit("state mismatch on redirect - aborting.")
|
||||
if not _CodeCatcher.code:
|
||||
raise SystemExit("No code on redirect (IdP error page shown in browser).")
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "authorization_code",
|
||||
"code": _CodeCatcher.code,
|
||||
"redirect_uri": redirect_uri,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"code_verifier": verifier,
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
tokens: dict[str, Any] = resp.json()
|
||||
if resp.status_code != 200:
|
||||
raise SystemExit(f"code exchange failed: {json.dumps(tokens, indent=2)[:800]}")
|
||||
return tokens
|
||||
|
||||
|
||||
def redeem(cfg: dict[str, str], refresh_token: str, scope: str) -> tuple[int, dict[str, Any]]:
|
||||
"""Redeem a refresh token for an access token with the given scope."""
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "refresh_token",
|
||||
"refresh_token": refresh_token,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"scope": scope,
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
body: dict[str, Any] = resp.json()
|
||||
return resp.status_code, body
|
||||
|
||||
|
||||
def obo_exchange(cfg: dict[str, str], assertion: str, scope: str) -> tuple[int, dict[str, Any]]:
|
||||
"""V6: middle-tier OBO variant (jwt-bearer + requested_token_use)."""
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer",
|
||||
"assertion": assertion,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"scope": scope,
|
||||
"requested_token_use": "on_behalf_of",
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
body: dict[str, Any] = resp.json()
|
||||
return resp.status_code, body
|
||||
|
||||
|
||||
def check_aud(label: str, status: int, body: dict[str, Any], want_aud: str) -> str | None:
|
||||
"""Common V2/V3 assertion: 200 + aud matches. Returns the new RT if any."""
|
||||
if status != 200:
|
||||
record(label, "FAILED", f"HTTP {status}: {json.dumps(body)[:300]}")
|
||||
return None
|
||||
claims = jwt_claims_unverified(body.get("access_token", ""))
|
||||
aud = str(claims.get("aud", "<none>"))
|
||||
ok = aud == want_aud or aud == want_aud.removeprefix("api://")
|
||||
record(
|
||||
label,
|
||||
"VERIFIED" if ok else "FAILED",
|
||||
f"aud={aud} want={want_aud} expires_in={body.get('expires_in')} "
|
||||
f"new_rt={redact(body.get('refresh_token'))}",
|
||||
)
|
||||
new_rt = body.get("refresh_token")
|
||||
return str(new_rt) if isinstance(new_rt, str) else None
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"ENTRA_TENANT_ID",
|
||||
"ENTRA_CLIENT_ID",
|
||||
"ENTRA_CLIENT_SECRET",
|
||||
"SPIKE_AUDIENCE_A",
|
||||
"SPIKE_AUDIENCE_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in required if k in os.environ}
|
||||
missing = [k for k in required if k not in cfg]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)}\nSee module docstring.")
|
||||
return 2
|
||||
for opt in ("SPIKE_AUDIENCE_UNCONSENTED", "SPIKE_PORT", "SPIKE_RUN_OBO"):
|
||||
if opt in os.environ:
|
||||
cfg[opt] = os.environ[opt]
|
||||
|
||||
# V1 - capture
|
||||
tokens = interactive_login(cfg)
|
||||
rt0 = tokens.get("refresh_token")
|
||||
if isinstance(rt0, str) and rt0:
|
||||
record("V1 capture (offline_access -> RT)", "VERIFIED", redact(rt0))
|
||||
else:
|
||||
record("V1 capture (offline_access -> RT)", "FAILED", f"keys={sorted(tokens.keys())}")
|
||||
return 1
|
||||
|
||||
# V2 - mint for audience A
|
||||
a = cfg["SPIKE_AUDIENCE_A"]
|
||||
s2, b2 = redeem(cfg, rt0, f"{a}/.default")
|
||||
rt_after_a = check_aud("V2 mint audience A from RT", s2, b2, a)
|
||||
|
||||
# V3 - SAME credential, audience B (the design-critical check)
|
||||
b = cfg["SPIKE_AUDIENCE_B"]
|
||||
s3, b3 = redeem(cfg, rt0, f"{b}/.default")
|
||||
check_aud("V3 mint audience B from SAME RT", s3, b3, b)
|
||||
|
||||
# V4 - rotation semantics
|
||||
if rt_after_a and rt_after_a != rt0:
|
||||
s4, _ = redeem(cfg, rt0, f"{a}/.default")
|
||||
record(
|
||||
"V4 rotation (new RT returned; old still valid?)",
|
||||
"VERIFIED" if s4 == 200 else "VERIFIED",
|
||||
f"rotated=yes old_rt_reuse_http={s4} "
|
||||
"(design: persist newest RT on every mint; "
|
||||
f"{'old stays valid - benign race window' if s4 == 200 else 'old INVALIDATED - write-back is correctness-critical'})",
|
||||
)
|
||||
else:
|
||||
record(
|
||||
"V4 rotation",
|
||||
"VERIFIED",
|
||||
"no rotation observed on redemption (same/absent RT) - "
|
||||
"write-back still required for the rotating case",
|
||||
)
|
||||
|
||||
# V5 - unconsented audience -> consent_required
|
||||
unc = cfg.get("SPIKE_AUDIENCE_UNCONSENTED")
|
||||
if unc:
|
||||
s5, b5 = redeem(cfg, rt0, f"{unc}/.default")
|
||||
codes = b5.get("error_codes", [])
|
||||
hit = s5 == 400 and (65001 in codes or b5.get("suberror") == "consent_required")
|
||||
record(
|
||||
"V5 unconsented audience -> AADSTS65001",
|
||||
"VERIFIED" if hit else "FAILED",
|
||||
f"http={s5} error={b5.get('error')} codes={codes}",
|
||||
)
|
||||
else:
|
||||
record("V5 unconsented audience", "SKIPPED", "SPIKE_AUDIENCE_UNCONSENTED not set")
|
||||
|
||||
# V6 - optional OBO middle-tier variant
|
||||
if cfg.get("SPIKE_RUN_OBO") == "1":
|
||||
s6a, b6a = redeem(cfg, rt0, f"{cfg['ENTRA_CLIENT_ID']}/.default")
|
||||
at_self = b6a.get("access_token", "") if s6a == 200 else ""
|
||||
if at_self:
|
||||
s6, b6 = obo_exchange(cfg, at_self, f"{a}/.default")
|
||||
check_aud("V6 OBO jwt-bearer variant", s6, b6, a)
|
||||
else:
|
||||
record(
|
||||
"V6 OBO jwt-bearer variant",
|
||||
"FAILED",
|
||||
f"could not mint self-audience assertion: HTTP {s6a}",
|
||||
)
|
||||
else:
|
||||
record("V6 OBO jwt-bearer variant", "SKIPPED", "SPIKE_RUN_OBO != 1")
|
||||
|
||||
print("\n=== summary ===")
|
||||
for check, status, _ in RESULTS:
|
||||
print(f" {status:>8} {check}")
|
||||
return 0 if all(s != "FAILED" for _, s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,371 +0,0 @@
|
||||
"""End-to-end exercise of the oauth_obo feature on the OSS path (RFC 8693).
|
||||
|
||||
Parallel to ``entra_e2e.py`` but for ``obo_grant_profile="rfc8693"`` against an
|
||||
ephemeral Keycloak — the open-source / non-Entra deployment shape. Fully
|
||||
headless (password grant, no browser), so it runs unattended.
|
||||
|
||||
Drives the REAL Turnstone code: ``MCPTokenStore.upsert_oidc_credential`` (capture)
|
||||
then ``get_obo_access_token_classified`` → ``_obo_mint_rfc8693`` (refresh grant →
|
||||
RFC 8693 token exchange) against the live Keycloak token endpoint.
|
||||
|
||||
Checks E1–E7 mirror the Entra harness:
|
||||
E1 mint audience A → token, aud claim carries A, cache row refresh_token_ct NULL
|
||||
E2 second call → cache hit, ZERO extra Keycloak calls
|
||||
E3 audience B from the SAME captured credential → aud carries B
|
||||
E4 rotation write-back (KC rotates the RT on the refresh leg)
|
||||
E5 force_refresh → re-mint (Keycloak call count increments)
|
||||
E6 unconsented audience C → NOT token, credential SURVIVES
|
||||
E7 cache flush → re-mint
|
||||
|
||||
M1-M3 drive the MODEL-backend mint (``mint_obo_access_token``, #898/#955) on
|
||||
the same captured credential — the path an ``auth_mode=rfc8693_obo`` model
|
||||
alias takes, distinct from the classified MCP path above:
|
||||
M1 model mint audience A with the alias's exchange scopes → token carries A
|
||||
(the #955 fix: model definitions now carry per-row ``obo_scopes``, so
|
||||
the exchange leg requests the audience's scope exactly as MCP rows do)
|
||||
M2 warm re-mint serves the synthetic ``__model_obo__`` cache row —
|
||||
identity-keyed on the owning alias, audience + scopes in the row's
|
||||
own columns — with zero IdP calls
|
||||
M3 an entra-leg mode (``entra_obo``) on this rfc8693 deployment refuses
|
||||
BEFORE any IdP traffic, recording cause=grant_profile_mismatch — the
|
||||
mode/profile pairing that replaced the pre-#955 overload
|
||||
|
||||
Env (set by keycloak_e2e.sh):
|
||||
KC_TOKEN_ENDPOINT, KC_ISSUER, KC_CLIENT_ID, KC_CLIENT_SECRET,
|
||||
KC_USER, KC_PASSWORD, AUD_A, SCOPE_A, AUD_B, SCOPE_B, AUD_C
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from turnstone.core.mcp_crypto import (
|
||||
MCPTokenCipher,
|
||||
MCPTokenCipherConfig,
|
||||
MCPTokenStore,
|
||||
)
|
||||
from turnstone.core.mcp_oauth import (
|
||||
get_obo_access_token_classified,
|
||||
mint_obo_access_token,
|
||||
model_mint_refusal_cause,
|
||||
model_obo_cache_server,
|
||||
model_obo_cause_key,
|
||||
)
|
||||
from turnstone.core.oidc import OIDCConfig
|
||||
from turnstone.core.storage._sqlite import SQLiteBackend
|
||||
|
||||
USER = "e2e-user"
|
||||
RESULTS: list[tuple[str, str]] = []
|
||||
|
||||
|
||||
def record(status: str, msg: str) -> None:
|
||||
RESULTS.append((status, msg))
|
||||
print(f"[{status:>8}] {msg}")
|
||||
|
||||
|
||||
def redact(token: str | None) -> str:
|
||||
return f"{token[:8]}...({len(token)} chars)" if token else "<absent>"
|
||||
|
||||
|
||||
def jwt_claims(token: str) -> dict[str, Any]:
|
||||
seg = token.split(".")[1]
|
||||
pad = "=" * (-len(seg) % 4)
|
||||
out: dict[str, Any] = json.loads(base64.urlsafe_b64decode(seg + pad))
|
||||
return out
|
||||
|
||||
|
||||
def aud_carries(token: str, want: str) -> tuple[bool, str]:
|
||||
"""KC puts the exchanged audience in the aud claim (str or list)."""
|
||||
aud = jwt_claims(token).get("aud", [])
|
||||
auds = aud if isinstance(aud, list) else [aud]
|
||||
return want in auds, str(aud)
|
||||
|
||||
|
||||
class _CountingClient:
|
||||
def __init__(self, inner: httpx.AsyncClient) -> None:
|
||||
self._inner = inner
|
||||
self.posts = 0
|
||||
|
||||
async def post(self, *args: Any, **kwargs: Any) -> httpx.Response:
|
||||
self.posts += 1
|
||||
return await self._inner.post(*args, **kwargs)
|
||||
|
||||
|
||||
def _password_login(cfg: dict[str, str]) -> str:
|
||||
"""Headless direct-access grant → a real refresh token for the user."""
|
||||
resp = httpx.post(
|
||||
cfg["KC_TOKEN_ENDPOINT"],
|
||||
data={
|
||||
"grant_type": "password",
|
||||
"client_id": cfg["KC_CLIENT_ID"],
|
||||
"client_secret": cfg["KC_CLIENT_SECRET"],
|
||||
"username": cfg["KC_USER"],
|
||||
"password": cfg["KC_PASSWORD"],
|
||||
"scope": "openid",
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return str(resp.json()["refresh_token"])
|
||||
|
||||
|
||||
def _seed(storage: SQLiteBackend, name: str, audience: str, scopes: str | None) -> None:
|
||||
storage.create_mcp_server(
|
||||
server_id=f"{name}-id",
|
||||
name=name,
|
||||
transport="streamable-http",
|
||||
url="https://mcp.example.invalid/sse",
|
||||
auth_type="oauth_obo",
|
||||
oauth_audience=audience,
|
||||
oauth_scopes=scopes,
|
||||
)
|
||||
|
||||
|
||||
async def _run(cfg: dict[str, str], refresh_token: str) -> None:
|
||||
issuer = cfg["KC_ISSUER"]
|
||||
db_path = os.path.join(tempfile.mkdtemp(prefix="obo-kc-e2e-"), "e2e.db")
|
||||
storage = SQLiteBackend(db_path)
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
raw = base64.urlsafe_b64decode(Fernet.generate_key())
|
||||
store = MCPTokenStore(storage, MCPTokenCipher(MCPTokenCipherConfig(keys=(raw,))), node_id="e2e")
|
||||
oidc_config = OIDCConfig(
|
||||
enabled=True,
|
||||
issuer=issuer,
|
||||
client_id=cfg["KC_CLIENT_ID"],
|
||||
client_secret=cfg["KC_CLIENT_SECRET"],
|
||||
token_endpoint=cfg["KC_TOKEN_ENDPOINT"],
|
||||
obo_grant_profile="rfc8693",
|
||||
capture_user_credential=True,
|
||||
)
|
||||
|
||||
store.upsert_oidc_credential(USER, issuer, refresh_token=refresh_token)
|
||||
cap = store.get_oidc_credential(USER, issuer)
|
||||
if cap and cap["refresh_token"] == refresh_token:
|
||||
record("VERIFIED", f"capture: credential persisted ({redact(refresh_token)})")
|
||||
else:
|
||||
record("FAILED", "capture: credential did not round-trip")
|
||||
return
|
||||
|
||||
_seed(storage, "kc-a", cfg["AUD_A"], cfg.get("SCOPE_A"))
|
||||
_seed(storage, "kc-b", cfg["AUD_B"], cfg.get("SCOPE_B"))
|
||||
if cfg.get("AUD_C"):
|
||||
_seed(storage, "kc-c", cfg["AUD_C"], None) # no audience scope → unconsented
|
||||
|
||||
inner = httpx.AsyncClient(timeout=20.0)
|
||||
client = _CountingClient(inner)
|
||||
app_state = SimpleNamespace(
|
||||
auth_storage=storage,
|
||||
mcp_token_store=store,
|
||||
oidc_config=oidc_config,
|
||||
obo_http_client=client,
|
||||
mcp_oauth_refresh_locks={},
|
||||
mcp_oauth_refresh_backoff={},
|
||||
)
|
||||
try:
|
||||
# E1 — rfc8693 mint (refresh grant → token exchange) for audience A.
|
||||
r = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
if r.kind == "token" and r.token:
|
||||
ok, aud = aud_carries(r.token, cfg["AUD_A"])
|
||||
row = storage.get_mcp_user_token(USER, "kc-a")
|
||||
cache_ok = row is not None and row["refresh_token_ct"] is None
|
||||
record(
|
||||
"VERIFIED" if ok and cache_ok else "FAILED",
|
||||
f"E1 mint A (refresh→exchange): kind=token aud={aud} want={cfg['AUD_A']} "
|
||||
f"cache_row_refreshless={cache_ok}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E1 mint A: kind={r.kind} (expected token)")
|
||||
return
|
||||
|
||||
# E2 — cache hit.
|
||||
posts_before = client.posts
|
||||
r2 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r2.kind == "token" and client.posts == posts_before else "FAILED",
|
||||
f"E2 cache hit: kind={r2.kind} extra_kc_calls={client.posts - posts_before} (want 0)",
|
||||
)
|
||||
|
||||
# E3 — audience B from the SAME credential.
|
||||
rb = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-b"
|
||||
)
|
||||
if rb.kind == "token" and rb.token:
|
||||
ok_b, aud_b = aud_carries(rb.token, cfg["AUD_B"])
|
||||
record(
|
||||
"VERIFIED" if ok_b else "FAILED",
|
||||
f"E3 mint B from SAME credential: aud={aud_b} want={cfg['AUD_B']}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E3 mint B: kind={rb.kind}")
|
||||
|
||||
# E4 — rotation write-back (KC rotates the RT on the refresh leg).
|
||||
cred_now = store.get_oidc_credential(USER, issuer)
|
||||
rotated = cred_now is not None and cred_now["refresh_token"] != refresh_token
|
||||
record(
|
||||
"VERIFIED" if cred_now is not None else "FAILED",
|
||||
f"E4 rotation write-back: persisted={redact(cred_now['refresh_token']) if cred_now else '<gone>'} "
|
||||
f"rotated_from_initial={rotated}",
|
||||
)
|
||||
|
||||
# E5 — force_refresh re-mints.
|
||||
posts_before = client.posts
|
||||
rf = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a", force_refresh=True
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if rf.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E5 force_refresh re-mint: kind={rf.kind} kc_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# E6 — unconsented audience: not a token, credential survives.
|
||||
if cfg.get("AUD_C"):
|
||||
rc = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-c"
|
||||
)
|
||||
cred_after = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if rc.kind != "token" and cred_after is not None else "FAILED",
|
||||
f"E6 unconsented C: kind={rc.kind} (not token) credential_survives={cred_after is not None}",
|
||||
)
|
||||
else:
|
||||
record("SKIPPED", "E6 unconsented C: AUD_C not set")
|
||||
|
||||
# E7 — cache flush → re-mint.
|
||||
store.delete_user_token(USER, "kc-a")
|
||||
posts_before = client.posts
|
||||
r7 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r7.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E7 flush→re-mint: kind={r7.kind} kc_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# M1-M3 — MODEL backend mint on the rfc8693 profile: same captured
|
||||
# credential and legs as E1-E7, but through mint_obo_access_token —
|
||||
# the path an auth_mode=rfc8693_obo alias takes, carrying the
|
||||
# per-alias exchange scopes MCP rows always had (#955). The mint's
|
||||
# cache and cause records are identity-keyed on the owning alias, so
|
||||
# the harness names one per mode-variant exactly as a deployment
|
||||
# would define separate rows.
|
||||
posts_before = client.posts
|
||||
m1 = await mint_obo_access_token(
|
||||
app_state=app_state,
|
||||
user_id=USER,
|
||||
alias="model-a",
|
||||
audience=cfg["AUD_A"],
|
||||
scopes=cfg.get("SCOPE_A", ""),
|
||||
grant_leg="rfc8693",
|
||||
)
|
||||
m1_kc_calls = client.posts - posts_before
|
||||
if m1:
|
||||
ok1, why1 = aud_carries(m1, cfg["AUD_A"])
|
||||
record(
|
||||
"VERIFIED" if ok1 and m1_kc_calls > 0 else "FAILED",
|
||||
f"M1 model mint (rfc8693_obo, scoped exchange): token={redact(m1)} "
|
||||
f"aud_ok={ok1} ({why1}) kc_calls={m1_kc_calls} (want >=1)",
|
||||
)
|
||||
else:
|
||||
record(
|
||||
"FAILED",
|
||||
f"M1 model mint (rfc8693_obo): no token (kc_calls={m1_kc_calls}) — "
|
||||
"the #955 scope wire-through should mint here",
|
||||
)
|
||||
|
||||
# M2 — warm re-mint serves the synthetic __model_obo__ cache row —
|
||||
# identity-keyed on the owning alias, audience + scopes in the row's
|
||||
# own columns — with zero IdP calls, and the row is named so
|
||||
# deprovisioning can find it by prefix.
|
||||
posts_before = client.posts
|
||||
m2 = await mint_obo_access_token(
|
||||
app_state=app_state,
|
||||
user_id=USER,
|
||||
alias="model-a",
|
||||
audience=cfg["AUD_A"],
|
||||
scopes=cfg.get("SCOPE_A", ""),
|
||||
grant_leg="rfc8693",
|
||||
)
|
||||
cache_row = storage.get_mcp_user_token(USER, model_obo_cache_server("model-a"))
|
||||
if m1:
|
||||
record(
|
||||
"VERIFIED"
|
||||
if m2 and client.posts == posts_before and cache_row is not None
|
||||
else "FAILED",
|
||||
f"M2 model cache-hit: token={redact(m2)} kc_calls="
|
||||
f"{client.posts - posts_before} (want 0) synthetic_row="
|
||||
f"{'present' if cache_row is not None else 'MISSING'}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", "M2 model cache-hit: blocked behind M1 — M1 failed, see above")
|
||||
|
||||
# M3 — the mode/profile pairing refusal that replaced the pre-#955
|
||||
# overload: an entra-leg mode on this rfc8693 deployment must yield
|
||||
# None with ZERO IdP calls and record the grant_profile_mismatch
|
||||
# cause the session heartbeat reads (under its own alias — a
|
||||
# deployment defines the entra-mode variant as its own row).
|
||||
posts_before = client.posts
|
||||
m3 = await mint_obo_access_token(
|
||||
app_state=app_state,
|
||||
user_id=USER,
|
||||
alias="model-a-entra",
|
||||
audience=cfg["AUD_A"],
|
||||
grant_leg="entra",
|
||||
)
|
||||
m3_cause = model_mint_refusal_cause(
|
||||
"model_obo", model_obo_cause_key("model-a-entra", grant_leg="entra"), USER
|
||||
)
|
||||
record(
|
||||
"VERIFIED"
|
||||
if m3 is None and client.posts == posts_before and m3_cause == "grant_profile_mismatch"
|
||||
else "FAILED",
|
||||
f"M3 mode/profile mismatch refusal: token={redact(m3)} (want absent) "
|
||||
f"kc_calls={client.posts - posts_before} (want 0) cause={m3_cause!r}",
|
||||
)
|
||||
finally:
|
||||
await inner.aclose()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"KC_TOKEN_ENDPOINT",
|
||||
"KC_ISSUER",
|
||||
"KC_CLIENT_ID",
|
||||
"KC_CLIENT_SECRET",
|
||||
"KC_USER",
|
||||
"KC_PASSWORD",
|
||||
"AUD_A",
|
||||
"AUD_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in os.environ if k.startswith(("KC_", "AUD_", "SCOPE_"))}
|
||||
missing = [k for k in required if not cfg.get(k)]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)} — run via keycloak_e2e.sh")
|
||||
return 2
|
||||
|
||||
print("Headless password login to Keycloak (the credential the feature captures)...")
|
||||
refresh_token = _password_login(cfg)
|
||||
|
||||
asyncio.run(_run(cfg, refresh_token))
|
||||
|
||||
print("\n=== summary ===")
|
||||
for status, msg in RESULTS:
|
||||
print(f" {status:>8} {msg}")
|
||||
return 0 if all(s in ("VERIFIED", "SKIPPED") for s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,65 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# OSS-path (RFC 8693) end-to-end: spin up ephemeral Keycloak, configure the
|
||||
# realm, run keycloak_e2e.py against the REAL Turnstone mint engine, tear down.
|
||||
# Fully headless — no browser. Manual test tooling, not run in CI.
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/../.." # repo root (uv run needs it)
|
||||
|
||||
CONTAINER=kc-obo-e2e
|
||||
PORT=8091
|
||||
KC="docker exec $CONTAINER /opt/keycloak/bin/kcadm.sh"
|
||||
|
||||
cleanup() { docker rm -f "$CONTAINER" >/dev/null 2>&1 || true; }
|
||||
trap cleanup EXIT
|
||||
cleanup
|
||||
|
||||
echo ">> starting Keycloak 26.3 (ephemeral)..."
|
||||
docker run -d --name "$CONTAINER" -p "127.0.0.1:${PORT}:8080" \
|
||||
-e KC_BOOTSTRAP_ADMIN_USERNAME=admin -e KC_BOOTSTRAP_ADMIN_PASSWORD=admin \
|
||||
quay.io/keycloak/keycloak:26.3 start-dev >/dev/null
|
||||
|
||||
echo ">> waiting for Keycloak (dev-mode boot can take a few minutes on a loaded host)..."
|
||||
# Wait on kcadm auth succeeding directly — more reliable than the host HTTP port,
|
||||
# and generous enough for a resource-starved boot (up to ~6 min).
|
||||
ready=""
|
||||
for _ in $(seq 1 90); do
|
||||
if $KC config credentials --server http://localhost:8080 --realm master \
|
||||
--user admin --password admin >/dev/null 2>&1; then
|
||||
ready=1
|
||||
break
|
||||
fi
|
||||
sleep 4
|
||||
done
|
||||
[ -n "$ready" ] || { echo "Keycloak did not become ready in time"; docker logs "$CONTAINER" 2>&1 | tail -15; exit 1; }
|
||||
|
||||
echo ">> configuring realm 'spike'..."
|
||||
$KC create realms -s realm=spike -s enabled=true >/dev/null
|
||||
# Confidential client with standard token exchange (the RFC 8693 leg) + direct
|
||||
# access grant (headless password login to fetch the user's refresh token).
|
||||
$KC create clients -r spike -s clientId=turnstone -s enabled=true -s publicClient=false \
|
||||
-s secret=spike-secret -s directAccessGrantsEnabled=true \
|
||||
-s 'attributes={"standard.token.exchange.enabled":"true"}' >/dev/null
|
||||
for t in mcp-a mcp-b mcp-c; do
|
||||
$KC create clients -r spike -s clientId=$t -s enabled=true -s publicClient=false -s secret=x >/dev/null
|
||||
done
|
||||
$KC create users -r spike -s username=e2e-user -s enabled=true -s email=e2e@spike.test \
|
||||
-s emailVerified=true -s firstName=E2E -s lastName=User >/dev/null
|
||||
$KC set-password -r spike --username e2e-user --new-password e2e-pw >/dev/null
|
||||
|
||||
TURNSTONE_UUID=$($KC get clients -r spike -q clientId=turnstone --fields id --format csv --noquotes)
|
||||
# Audience client scopes for mcp-a and mcp-b ONLY (mcp-c stays unconsented → E6).
|
||||
for t in mcp-a mcp-b; do
|
||||
SID=$($KC create client-scopes -r spike -s name=aud-$t -s protocol=openid-connect -i)
|
||||
$KC create "client-scopes/$SID/protocol-mappers/models" -r spike -s name=aud-$t \
|
||||
-s protocol=openid-connect -s protocolMapper=oidc-audience-mapper \
|
||||
-s "config={\"included.client.audience\":\"$t\",\"access.token.claim\":\"true\"}" >/dev/null
|
||||
$KC update "clients/$TURNSTONE_UUID/optional-client-scopes/$SID" -r spike >/dev/null
|
||||
done
|
||||
|
||||
echo ">> running the product e2e harness..."
|
||||
export KC_TOKEN_ENDPOINT="http://127.0.0.1:${PORT}/realms/spike/protocol/openid-connect/token"
|
||||
export KC_ISSUER="http://127.0.0.1:${PORT}/realms/spike"
|
||||
export KC_CLIENT_ID=turnstone KC_CLIENT_SECRET=spike-secret
|
||||
export KC_USER=e2e-user KC_PASSWORD=e2e-pw
|
||||
export AUD_A=mcp-a SCOPE_A=aud-mcp-a AUD_B=mcp-b SCOPE_B=aud-mcp-b AUD_C=mcp-c
|
||||
uv run python scripts/obo-e2e/keycloak_e2e.py
|
||||
File diff suppressed because it is too large
Load Diff
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Console API",
|
||||
"version": "1.8.0a5",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Cluster-wide visibility and control across all turnstone nodes."
|
||||
},
|
||||
"paths": {
|
||||
@@ -3723,7 +3723,7 @@
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelDefinitionWriteResponse"
|
||||
"$ref": "#/components/schemas/ModelDefinitionInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3738,16 +3738,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "Error 403",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
@@ -3757,47 +3747,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/admin/model-definitions/auth-constraints": {
|
||||
"get": {
|
||||
"summary": "Dynamic-auth affordance data for the model editor (requires admin.mcp)",
|
||||
"operationId": "v1_api_admin_model-definitions_auth-constraints_get",
|
||||
"tags": [
|
||||
"Admin"
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelAuthConstraintsResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "Error 403",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3895,7 +3844,7 @@
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelDefinitionWriteResponse"
|
||||
"$ref": "#/components/schemas/ModelDefinitionInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3910,16 +3859,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "Error 403",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Error 404",
|
||||
"content": {
|
||||
@@ -3939,16 +3878,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -3970,14 +3899,7 @@
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/DeleteModelDefinitionResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
"description": "Success"
|
||||
},
|
||||
"404": {
|
||||
"description": "Error 404",
|
||||
@@ -4070,26 +3992,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"500": {
|
||||
"description": "Error 500",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7889,12 +7791,6 @@
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"description": "Project to attach the workstream to (validated against membership, empty = none)",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"resume_ws": {
|
||||
"default": "",
|
||||
"description": "Workstream ID to resume (loads previous conversation)",
|
||||
@@ -8452,7 +8348,7 @@
|
||||
"type": "string"
|
||||
},
|
||||
"status": {
|
||||
"description": "One of: pending / in_progress / done / blocked / needs_user. ``blocked`` is waiting on a dependency the coordinator may clear itself; ``needs_user`` is waiting on a decision only the operator can make.",
|
||||
"description": "One of: pending / in_progress / done / blocked.",
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
@@ -8461,12 +8357,6 @@
|
||||
"title": "Child Ws Id",
|
||||
"type": "string"
|
||||
},
|
||||
"note": {
|
||||
"default": "",
|
||||
"description": "Optional one-sentence note, typically what the coordinator needs from the operator on a ``needs_user`` task. Absent from the stored record when unset (there is no backfill for rows written before the field existed), so it defaults to the empty string here.",
|
||||
"title": "Note",
|
||||
"type": "string"
|
||||
},
|
||||
"created": {
|
||||
"title": "Created",
|
||||
"type": "string"
|
||||
@@ -8606,18 +8496,6 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"description": "Persona slug (empty = kind default)",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"description": "Project to attach the workstream to",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"description": "Notification targets on completion (channel_type + channel_id/user_id)",
|
||||
"items": {
|
||||
@@ -8781,30 +8659,6 @@
|
||||
"default": null,
|
||||
"title": "Skill"
|
||||
},
|
||||
"persona": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Persona"
|
||||
},
|
||||
"project_id": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"notify_targets": {
|
||||
"anyOf": [
|
||||
{
|
||||
@@ -8900,16 +8754,6 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"items": {
|
||||
"additionalProperties": {
|
||||
@@ -10922,21 +10766,6 @@
|
||||
"title": "Replay Reasoning To Model",
|
||||
"type": "boolean"
|
||||
},
|
||||
"auth_mode": {
|
||||
"default": "static",
|
||||
"title": "Auth Mode",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_audience": {
|
||||
"default": "",
|
||||
"title": "Obo Audience",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"default": "",
|
||||
"title": "Obo Scopes",
|
||||
"type": "string"
|
||||
},
|
||||
"source": {
|
||||
"default": "",
|
||||
"title": "Source",
|
||||
@@ -10966,147 +10795,6 @@
|
||||
"title": "ModelDefinitionInfo",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelDefinitionWriteResponse": {
|
||||
"description": "Create/update response: the stored row plus an optional caveat.\n\n``registry_warning`` is present only when the DB write succeeded but\nTHIS console's live coordinator registry refused to adopt it (keyless\nhost with dynamic-auth rows): the row is saved, yet running sessions\nkeep the previous config until the deployment fault is remedied.\nClients should surface it as a warning beside the success, never as a\nfailure. Absent on a clean save (SkipJsonSchema: the server omits the\nkey rather than sending null).",
|
||||
"properties": {
|
||||
"definition_id": {
|
||||
"title": "Definition Id",
|
||||
"type": "string"
|
||||
},
|
||||
"alias": {
|
||||
"title": "Alias",
|
||||
"type": "string"
|
||||
},
|
||||
"model": {
|
||||
"title": "Model",
|
||||
"type": "string"
|
||||
},
|
||||
"provider": {
|
||||
"default": "openai",
|
||||
"title": "Provider",
|
||||
"type": "string"
|
||||
},
|
||||
"base_url": {
|
||||
"default": "",
|
||||
"title": "Base Url",
|
||||
"type": "string"
|
||||
},
|
||||
"api_key": {
|
||||
"default": "",
|
||||
"title": "Api Key",
|
||||
"type": "string"
|
||||
},
|
||||
"context_window": {
|
||||
"default": 32768,
|
||||
"title": "Context Window",
|
||||
"type": "integer"
|
||||
},
|
||||
"capabilities": {
|
||||
"default": "{}",
|
||||
"title": "Capabilities",
|
||||
"type": "string"
|
||||
},
|
||||
"enabled": {
|
||||
"default": true,
|
||||
"title": "Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"temperature": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "number"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Temperature"
|
||||
},
|
||||
"max_tokens": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Max Tokens"
|
||||
},
|
||||
"reasoning_effort": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Reasoning Effort"
|
||||
},
|
||||
"surface_persisted_reasoning": {
|
||||
"default": true,
|
||||
"title": "Surface Persisted Reasoning",
|
||||
"type": "boolean"
|
||||
},
|
||||
"replay_reasoning_to_model": {
|
||||
"default": false,
|
||||
"title": "Replay Reasoning To Model",
|
||||
"type": "boolean"
|
||||
},
|
||||
"auth_mode": {
|
||||
"default": "static",
|
||||
"title": "Auth Mode",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_audience": {
|
||||
"default": "",
|
||||
"title": "Obo Audience",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"default": "",
|
||||
"title": "Obo Scopes",
|
||||
"type": "string"
|
||||
},
|
||||
"source": {
|
||||
"default": "",
|
||||
"title": "Source",
|
||||
"type": "string"
|
||||
},
|
||||
"created_by": {
|
||||
"default": "",
|
||||
"title": "Created By",
|
||||
"type": "string"
|
||||
},
|
||||
"created": {
|
||||
"default": "",
|
||||
"title": "Created",
|
||||
"type": "string"
|
||||
},
|
||||
"updated": {
|
||||
"default": "",
|
||||
"title": "Updated",
|
||||
"type": "string"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the save landed but this console's live registry refused the swap (e.g. dynamic auth configured without the startup encryption key); carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"definition_id",
|
||||
"alias",
|
||||
"model"
|
||||
],
|
||||
"title": "ModelDefinitionWriteResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"CreateModelDefinitionRequest": {
|
||||
"properties": {
|
||||
"alias": {
|
||||
@@ -11192,21 +10880,6 @@
|
||||
"default": false,
|
||||
"title": "Replay Reasoning To Model",
|
||||
"type": "boolean"
|
||||
},
|
||||
"auth_mode": {
|
||||
"default": "static",
|
||||
"title": "Auth Mode",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_audience": {
|
||||
"default": "",
|
||||
"title": "Obo Audience",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"default": "",
|
||||
"title": "Obo Scopes",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -11291,11 +10964,17 @@
|
||||
"title": "Context Window"
|
||||
},
|
||||
"capabilities": {
|
||||
"additionalProperties": true,
|
||||
"anyOf": [
|
||||
{
|
||||
"additionalProperties": true,
|
||||
"type": "object"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Full replacement capabilities object. Omit to leave the stored value unchanged; JSON null is refused (400).",
|
||||
"title": "Capabilities",
|
||||
"type": "object"
|
||||
"title": "Capabilities"
|
||||
},
|
||||
"enabled": {
|
||||
"anyOf": [
|
||||
@@ -11368,42 +11047,6 @@
|
||||
],
|
||||
"default": null,
|
||||
"title": "Replay Reasoning To Model"
|
||||
},
|
||||
"auth_mode": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Auth Mode"
|
||||
},
|
||||
"obo_audience": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Obo Audience"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Obo Scopes"
|
||||
}
|
||||
},
|
||||
"title": "UpdateModelDefinitionRequest",
|
||||
@@ -11417,80 +11060,14 @@
|
||||
},
|
||||
"title": "Models",
|
||||
"type": "array"
|
||||
},
|
||||
"default_alias": {
|
||||
"description": "Effective default alias after the config/enabled-list fallbacks",
|
||||
"title": "Default Alias",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"models",
|
||||
"default_alias"
|
||||
"models"
|
||||
],
|
||||
"title": "ListModelDefinitionsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelAuthConstraintsResponse": {
|
||||
"description": "Affordance data for the model shelf's Backend-auth section.\n\nSuggestions and labels only \u2014 never a gate. The write validator is the\nauthority; a client that fails to fetch this must degrade to free-text\ninput with server-side validation, not to a refusal.",
|
||||
"properties": {
|
||||
"auth_audience_allowlist": {
|
||||
"description": "Exact gateway audiences a definition may use with entra_obo / entra_app, rendered as input suggestions. Empty means none are registered yet; writes are refused until an operator populates model.auth_audience_allowlist.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Auth Audience Allowlist",
|
||||
"type": "array"
|
||||
},
|
||||
"auth_grant_profile": {
|
||||
"description": "Deployment [oidc] obo_grant_profile, or empty when single sign-on is not configured. Each dynamic auth_mode pairs with exactly one profile (see auth_mode_profiles); the write validator refuses a new pairing that contradicts it. A transient discovery outage reports the configured profile, not empty.",
|
||||
"title": "Auth Grant Profile",
|
||||
"type": "string"
|
||||
},
|
||||
"dynamic_auth_modes": {
|
||||
"description": "auth_mode values that mint per-call backend credentials, derived server-side from the registry's mode classification so the shelf's affordances (audience enable/require, section visibility) track it by data. Clients keep a hand-listed fallback only for a missing or failed constraints fetch.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Dynamic Auth Modes",
|
||||
"type": "array"
|
||||
},
|
||||
"scopes_auth_modes": {
|
||||
"description": "auth_mode values whose mint reads obo_scopes (the token-exchange scope request), same server-derived contract as dynamic_auth_modes; drives the scopes input's visibility.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Scopes Auth Modes",
|
||||
"type": "array"
|
||||
},
|
||||
"app_identity_auth_modes": {
|
||||
"description": "auth_mode values that mint a shared app/deployment identity rather than a per-user one, same server-derived contract as dynamic_auth_modes; drives the model list's auth badge wording.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "App Identity Auth Modes",
|
||||
"type": "array"
|
||||
},
|
||||
"auth_mode_profiles": {
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
},
|
||||
"description": "Required [oidc] obo_grant_profile per dynamic auth_mode. Affordance for greying options that cannot validate under this deployment's profile; the write validator remains the authority.",
|
||||
"title": "Auth Mode Profiles",
|
||||
"type": "object"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"auth_audience_allowlist",
|
||||
"auth_grant_profile",
|
||||
"dynamic_auth_modes",
|
||||
"scopes_auth_modes",
|
||||
"app_identity_auth_modes",
|
||||
"auth_mode_profiles"
|
||||
],
|
||||
"title": "ModelAuthConstraintsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"PersonaInfo": {
|
||||
"description": "Full persona row \u2014 the authoring shape (contrast PersonaChoice, the\npicker's display-only projection on the server surface).",
|
||||
"properties": {
|
||||
@@ -11841,42 +11418,11 @@
|
||||
"additionalProperties": true,
|
||||
"title": "Results",
|
||||
"type": "object"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the node fan-out ran but THIS console's live registry refused the swap (e.g. dynamic auth configured without the startup encryption key); carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"title": "ModelReloadResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"DeleteModelDefinitionResponse": {
|
||||
"description": "Delete response: the removed row id plus an optional caveat.\n\n``registry_warning`` mirrors ModelDefinitionWriteResponse: the DB row\nis gone, but a keyless console's live registry refused the swap and\nkeeps SERVING the deleted alias to running and new coordinator\nsessions until the deployment fault is remedied.",
|
||||
"properties": {
|
||||
"status": {
|
||||
"default": "ok",
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
"definition_id": {
|
||||
"title": "Definition Id",
|
||||
"type": "string"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the delete landed but this console's live registry refused the swap and keeps serving the deleted alias; carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"definition_id"
|
||||
],
|
||||
"title": "DeleteModelDefinitionResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"DetectModelRequest": {
|
||||
"properties": {
|
||||
"provider": {
|
||||
@@ -12043,12 +11589,6 @@
|
||||
"default": "",
|
||||
"title": "Error",
|
||||
"type": "string"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the calibration was stored but this console's live registry refused the swap; carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"title": "CalibrateModelResponse",
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Server API",
|
||||
"version": "1.8.0a5",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Single-node workstream management, chat interaction, and real-time streaming."
|
||||
},
|
||||
"paths": {
|
||||
@@ -228,16 +228,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -400,26 +390,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -639,7 +609,7 @@
|
||||
"tags": [
|
||||
"Streaming"
|
||||
],
|
||||
"description": "Server-Sent Events stream for node-level state broadcasts. Emits a node_snapshot event on connect (workstreams, health, aggregate), followed by real-time delta events (ws_state, ws_activity, ws_created, ws_closed, ws_rename, health_changed, aggregate). Pass ?expected_node_id=X for identity verification (returns 409 on mismatch). Every event's SSE id is an opaque '{boot_epoch}-{counter}' string; presenting it on reconnect (Last-Event-ID header or ?last_event_id=) replays missed events, or emits a replay_truncated event (reason: ring_evicted with lost_count + earliest_available_id, or boot_epoch when the cursor predates this server process) followed by a fresh node_snapshot. Treat the id as opaque \u2014 its format may change.",
|
||||
"description": "Server-Sent Events stream for node-level state broadcasts. Emits a node_snapshot event on connect (workstreams, health, aggregate), followed by real-time delta events (ws_state, ws_activity, ws_created, ws_closed, ws_rename, health_changed, aggregate). Pass ?expected_node_id=X for identity verification (returns 409 on mismatch).",
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success"
|
||||
@@ -2290,23 +2260,16 @@
|
||||
"SendResponse": {
|
||||
"properties": {
|
||||
"status": {
|
||||
"description": "'ok' (fresh turn dispatched), 'queued' (folded into the live turn's interjection queue, or \u2014 when `deferred` is true \u2014 parked for dispatch after the current command window), 'queue_full', 'attachments_busy' (attachments can't ride a queued turn; retry when idle), or 'cross_user_interjection' (another participant's turn is in flight; carried on the 409 body).",
|
||||
"description": "'ok', 'busy', 'queued', or 'queue_full'",
|
||||
"examples": [
|
||||
"ok",
|
||||
"busy",
|
||||
"queued",
|
||||
"queue_full",
|
||||
"attachments_busy",
|
||||
"cross_user_interjection"
|
||||
"queue_full"
|
||||
],
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
"deferred": {
|
||||
"default": false,
|
||||
"description": "Set on `queued` responses: the message is parked on the workstream's deferred-send list (a slash-command window holds the worker slot, or earlier deferred sends are still pending) and dispatches as an ordinary full-fidelity send afterwards \u2014 it is NOT in a live turn's interjection queue. `DELETE .../send` retracts it until dispatch. Node-local and in-memory: a node restart before dispatch drops it (at-most-once intake).",
|
||||
"title": "Deferred",
|
||||
"type": "boolean"
|
||||
},
|
||||
"attached_ids": {
|
||||
"description": "Attachment ids actually attached to this turn. Subset of the request's `attachment_ids` (or the auto-consumed pending set). Empty when the send carries no attachments.",
|
||||
"items": {
|
||||
|
||||
Generated
+173
-534
File diff suppressed because it is too large
Load Diff
@@ -32,7 +32,7 @@
|
||||
],
|
||||
"license": "Apache-2.0",
|
||||
"devDependencies": {
|
||||
"typescript": "^7.0.0",
|
||||
"typescript": "^6.0.0",
|
||||
"vitest": "^4.1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -157,49 +157,6 @@ export interface CancelledEvent {
|
||||
type: "cancelled";
|
||||
}
|
||||
|
||||
/**
|
||||
* Context-compaction lifecycle. `start` carries `trigger` ("manual"/"auto";
|
||||
* auto adds `where` + `pct`); `progress` carries chunked-summarization
|
||||
* `part`/`total`/`depth` (or `retry_in`/`error` for a retry wait); `end`
|
||||
* carries `ok` plus either `before_tokens`/`after_tokens`/`summary` or the
|
||||
* failure `reason`/`message`. The successful end's summary also replays from
|
||||
* `/history` as a `role: "system"`, `source: "compaction"` entry.
|
||||
*/
|
||||
export interface CompactionEvent {
|
||||
type: "compaction";
|
||||
phase: "start" | "progress" | "end";
|
||||
/** Correlates every event of one compaction run (0 from legacy emitters). */
|
||||
compaction_id?: number;
|
||||
/**
|
||||
* End events only: true marks a force-abandoned compaction retiring
|
||||
* after a successor generation took over — skip failure notices for
|
||||
* those (an OK end's result still stands; the history swap happened).
|
||||
*/
|
||||
superseded?: boolean;
|
||||
/**
|
||||
* Failed ends only: the emitter-computed display verdict — show
|
||||
* `message` only when true, instead of re-deriving suppression from
|
||||
* reason/trigger/superseded client-side.
|
||||
*/
|
||||
notice?: boolean;
|
||||
/** Present on start and on every end (ok or failed). */
|
||||
trigger?: "manual" | "auto";
|
||||
where?: string;
|
||||
pct?: number;
|
||||
part?: number;
|
||||
total?: number;
|
||||
depth?: number;
|
||||
retry_in?: number;
|
||||
error?: string;
|
||||
warning?: string;
|
||||
ok?: boolean;
|
||||
reason?: string;
|
||||
message?: string;
|
||||
before_tokens?: number;
|
||||
after_tokens?: number;
|
||||
summary?: string;
|
||||
}
|
||||
|
||||
// Global events
|
||||
|
||||
export interface WsStateEvent {
|
||||
@@ -255,7 +212,6 @@ export type ServerEvent =
|
||||
| BusyErrorEvent
|
||||
| ClearUiEvent
|
||||
| CancelledEvent
|
||||
| CompactionEvent
|
||||
| WsStateEvent
|
||||
| WsActivityEvent
|
||||
| WsRenameEvent
|
||||
|
||||
@@ -74,7 +74,7 @@ class _FakeConfigStore:
|
||||
def _fake_registry() -> MagicMock:
|
||||
"""MagicMock whose ``.resolve()`` succeeds so the 503 gate passes."""
|
||||
reg = MagicMock()
|
||||
reg.resolve.return_value = (MagicMock(), "gpt-4", MagicMock(), 0)
|
||||
reg.resolve.return_value = (MagicMock(), "gpt-4", MagicMock())
|
||||
return reg
|
||||
|
||||
|
||||
|
||||
@@ -1,45 +0,0 @@
|
||||
"""Shared helpers for the Python-driven node harnesses that evaluate the
|
||||
``shared_static`` ES modules with script semantics."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import shutil
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def has_node() -> bool:
|
||||
return shutil.which("node") is not None
|
||||
|
||||
|
||||
# Module-level ``pytestmark = node_skip`` in each harness suite — the node
|
||||
# detection lives here once, so a future change (version floor, env
|
||||
# override) cannot land in one suite and silently miss another.
|
||||
node_skip = pytest.mark.skipif(not has_node(), reason="node not available")
|
||||
|
||||
|
||||
def demodulize(path: Path) -> str:
|
||||
"""Strip ES-module syntax so ``vm.runInThisContext`` (script semantics)
|
||||
can evaluate the file: imports drop (the harness loads the whole
|
||||
dependency set into one shared context, so cross-file bindings resolve
|
||||
as context globals, exactly like the pre-module classic scripts), and
|
||||
``export`` keywords peel off their declarations.
|
||||
|
||||
Single-sourced here for every JS harness: a new module syntax form
|
||||
(``export default``, re-exports) must be handled once, not per suite —
|
||||
a divergence between per-file copies surfaces as a confusing
|
||||
``vm.runInThisContext`` SyntaxError in whichever suite lagged.
|
||||
"""
|
||||
src = path.read_text(encoding="utf-8")
|
||||
src = re.sub(r"^import\s+\{[\s\S]*?\}\s+from\s+\"[^\"]+\";\s*$", "", src, flags=re.M)
|
||||
src = re.sub(r"^import\s+[^;\n]+;\s*$", "", src, flags=re.M)
|
||||
src = re.sub(
|
||||
r"^export\s+(?=(?:async\s+)?(?:function|const|let|var|class)\b)", "", src, flags=re.M
|
||||
)
|
||||
src = re.sub(r"^export\s*\{[^}]*\};\s*$", "", src, flags=re.M)
|
||||
return src
|
||||
@@ -1,57 +0,0 @@
|
||||
"""Shared OIDC posture builder for the model-auth / OBO test surface.
|
||||
|
||||
One construction site for the posture the mint and write-validator suites
|
||||
read, built as a REAL (frozen) ``OIDCConfig`` so an override for a field
|
||||
the dataclass does not carry raises at the call site. Named with a leading
|
||||
underscore so pytest does not collect it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from turnstone.core.oidc import OIDCConfig
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
# The issuer / token-endpoint pair the mint suites route their mock
|
||||
# transports on.
|
||||
ISSUER = "https://idp.test"
|
||||
TOKEN_ENDPOINT = "https://idp.test/token"
|
||||
|
||||
|
||||
def make_oidc_config(**overrides: Any) -> OIDCConfig:
|
||||
"""A full, mintable OIDC posture; tests override the field under test,
|
||||
everything else rides the dataclass defaults."""
|
||||
defaults: dict[str, Any] = {
|
||||
"enabled": True,
|
||||
"issuer": ISSUER,
|
||||
"client_id": "cid",
|
||||
"client_secret": "csecret",
|
||||
"token_endpoint": TOKEN_ENDPOINT,
|
||||
}
|
||||
defaults.update(overrides)
|
||||
return OIDCConfig(**defaults)
|
||||
|
||||
|
||||
def keyed_app_state() -> SimpleNamespace:
|
||||
"""App-state stub satisfying ``ModelRegistry.reload``'s dynamic-auth key
|
||||
guard, for suites exercising reload mechanics rather than key policy."""
|
||||
return SimpleNamespace(mcp_token_store=object())
|
||||
|
||||
|
||||
def mint_warn_state_reset() -> Iterator[None]:
|
||||
"""Reset generator behind the mint suites' autouse fixtures: empties the
|
||||
process-global mint warn/dedup/cause state before AND after each test,
|
||||
so warn-dedup assertions are not order-dependent. Modules install it as
|
||||
``yield from mint_warn_state_reset()`` in an autouse fixture.
|
||||
"""
|
||||
# Lazy import: non-mint consumers of this helper module (the write-
|
||||
# validator suites) shouldn't pay the mcp_oauth import.
|
||||
from turnstone.core.mcp_oauth import reset_model_mint_warn_state_for_tests
|
||||
|
||||
reset_model_mint_warn_state_for_tests()
|
||||
yield
|
||||
reset_model_mint_warn_state_for_tests()
|
||||
@@ -1,230 +0,0 @@
|
||||
"""#832 replay-parity harness: scenario table + runner.
|
||||
|
||||
The audit is controller determinism: with the plant's chunk sequence held
|
||||
fixed, the streaming phase must produce an identical UI event sequence
|
||||
and an identical committed message — modulo the RULED behavior changes
|
||||
restated in full on the transforms in ``test_832_parity.py``. This
|
||||
module is the shared half: the scenario scripts (one row per
|
||||
chunk-field→UI translation the consumer performs) and the runner that
|
||||
drives one through the streaming seam, recording everything the turn
|
||||
observably produced.
|
||||
|
||||
Baselines are captured from the PRE-FOLD path (``UPDATE_832_PARITY=1``,
|
||||
run at a tree where ``session.py`` is byte-identical to pre-fold main)
|
||||
into ``tests/data/parity_832/``. The runner adapts to EITHER world by
|
||||
signature, so a recapture at an old tree records real old-world
|
||||
behavior, and capture mode refuses to write a record whose failure is
|
||||
the harness's own call shape. Assert mode replays the same scripts
|
||||
through the current tree and compares against the baseline, applying the
|
||||
ruled transforms; a mismatch outside a ruled transform is a regression.
|
||||
|
||||
The provider fake arms ``cancel_ref`` EAGERLY (a closeable sentinel
|
||||
appended inside ``create_streaming``, before the iterator is returned),
|
||||
mirroring every real adapter: the wrapper classifies
|
||||
creation-vs-midstream failures by that arming, so a fake that skipped it
|
||||
would exercise only the creation arm.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import inspect
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from tests._session_helpers import RecordingUI, make_session, scripted_provider
|
||||
from turnstone.core.providers._protocol import StreamChunk, ToolCallDelta, UsageInfo
|
||||
from turnstone.core.trajectory import Turn
|
||||
|
||||
FIXTURE_DIR = Path(__file__).parent / "data" / "parity_832"
|
||||
UPDATE = os.environ.get("UPDATE_832_PARITY") == "1"
|
||||
|
||||
|
||||
def _tc(index: int, call_id: str, name: str = "", args: str = "") -> ToolCallDelta:
|
||||
return ToolCallDelta(index=index, id=call_id, name=name, arguments_delta=args)
|
||||
|
||||
|
||||
_USAGE_A = UsageInfo(prompt_tokens=11, completion_tokens=0, total_tokens=11)
|
||||
_USAGE_B = UsageInfo(prompt_tokens=11, completion_tokens=7, total_tokens=18)
|
||||
|
||||
# Scenario table — the V11 grid, one script per row. Scripts are chunk
|
||||
# LISTS; the runner re-iterates a fresh iterator per attempt.
|
||||
SCENARIOS: dict[str, list[StreamChunk]] = {
|
||||
"content_only": [
|
||||
StreamChunk(content_delta="Hello "),
|
||||
StreamChunk(content_delta="world."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"reasoning_then_content": [
|
||||
StreamChunk(reasoning_delta="think a", usage=_USAGE_A),
|
||||
StreamChunk(reasoning_delta=" think b"),
|
||||
StreamChunk(content_delta="Answer."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"tools_simple": [
|
||||
StreamChunk(content_delta="Calling."),
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "call_1", "get_weather", '{"city": ')]),
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "", "", '"Paris"}')]),
|
||||
StreamChunk(finish_reason="tool_calls", usage=_USAGE_B),
|
||||
],
|
||||
"combined_content_tools_finish": [
|
||||
StreamChunk(content_delta="Before "),
|
||||
StreamChunk(
|
||||
content_delta="tools",
|
||||
tool_call_deltas=[_tc(0, "call_1", "get_weather", '{"city": "Nice"}')],
|
||||
finish_reason="tool_calls",
|
||||
),
|
||||
StreamChunk(usage=_USAGE_B),
|
||||
],
|
||||
"info_prefinish": [
|
||||
StreamChunk(info_delta="[Searching: pinniped taxonomy]"),
|
||||
StreamChunk(content_delta="Seals are pinnipeds."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"info_postfinish_footer": [
|
||||
StreamChunk(content_delta="Answer with sources."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
StreamChunk(info_delta="Sources:\n- example.com/page"),
|
||||
],
|
||||
"think_tags_split_across_chunks": [
|
||||
StreamChunk(content_delta="<thi"),
|
||||
StreamChunk(content_delta="nk>plan</think>\n\nAnswer"),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"blank_id_tools": [
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "", "get_weather", '{"city": "Oslo"}')]),
|
||||
StreamChunk(finish_reason="tool_calls", usage=_USAGE_B),
|
||||
],
|
||||
"length_with_tools": [
|
||||
StreamChunk(content_delta="Partial answer"),
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "call_1", "get_weather", '{"city": "Par')]),
|
||||
StreamChunk(finish_reason="length", usage=_USAGE_B),
|
||||
],
|
||||
"content_filter": [
|
||||
StreamChunk(content_delta="Redac"),
|
||||
StreamChunk(finish_reason="content_filter", usage=_USAGE_B),
|
||||
],
|
||||
"no_finish_clean_exhaust": [
|
||||
StreamChunk(content_delta="Half an ans"),
|
||||
StreamChunk(usage=_USAGE_A),
|
||||
],
|
||||
"finish_only_no_content": [
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"provider_blocks_on_terminal": [
|
||||
StreamChunk(content_delta="Blocked."),
|
||||
StreamChunk(
|
||||
finish_reason="stop",
|
||||
usage=_USAGE_B,
|
||||
provider_blocks=[{"type": "reasoning_text", "text": "captured"}],
|
||||
),
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
_SYNTH_ID = re.compile(r"^call_[0-9a-f]{32}$")
|
||||
|
||||
|
||||
def _mask_synth_ids(record: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Replace uuid-backfilled tool-call ids with stable placeholders.
|
||||
|
||||
The blank-id repair mints ``call_<uuid4hex>`` per run — real
|
||||
nondeterminism inside the seam, but not behavior: mask ONLY that exact
|
||||
shape (never a scripted provider id) with an index-stable token so
|
||||
captures compare across runs. Applied to the committed projection;
|
||||
UI events never carry call ids in this harness.
|
||||
"""
|
||||
result = record.get("result")
|
||||
if not result:
|
||||
return record
|
||||
for i, tc in enumerate(result.get("tool_calls") or []):
|
||||
if _SYNTH_ID.match(tc.get("id", "")):
|
||||
tc["id"] = f"synth-id-{i}"
|
||||
for i, block in enumerate(result.get("provider_content") or []):
|
||||
if isinstance(block, dict) and _SYNTH_ID.match(str(block.get("id", ""))):
|
||||
block["id"] = f"synth-id-{i}"
|
||||
return record
|
||||
|
||||
|
||||
def run_scenario(name: str) -> dict[str, Any]:
|
||||
"""Drive one scenario through the streaming seam; return the record.
|
||||
|
||||
The record is everything the streaming phase observably produced: the
|
||||
ordered UI events, the committed-message projection, the mid-stream
|
||||
usage slot, and the exception class if the seam raised. Deliberately
|
||||
seam-level at ``_stream_response`` — full ``send()`` scenarios ride
|
||||
the ported ladder suites instead.
|
||||
|
||||
Signature-adaptive so ``UPDATE_832_PARITY=1`` at a PRE-fold tree
|
||||
records real old-world behavior: the pre-fold seam was
|
||||
``_stream_response(msgs, my_generation) -> dict``, the post-fold one
|
||||
is ``_stream_response(my_generation) -> ModelTurnResult`` (wire
|
||||
prepared inside). A harness-shape failure must never be recorded as
|
||||
behavior — ``write_fixture`` refuses one.
|
||||
"""
|
||||
ui = RecordingUI()
|
||||
session = make_session(ui=ui)
|
||||
# Zero the ladder backoff: a scenario that reaches the mid-stream
|
||||
# re-issue ladder (no_finish_clean_exhaust) must not sleep real
|
||||
# exponential delays in a unit run. The retry-notice transform in
|
||||
# test_832_parity hardcodes the matching "0s" wording.
|
||||
session._RETRY_BASE_DELAY = 0
|
||||
session._provider = scripted_provider(SCENARIOS[name])
|
||||
|
||||
pre_fold = "msgs" in inspect.signature(type(session)._stream_response).parameters
|
||||
record: dict[str, Any] = {"scenario": name}
|
||||
try:
|
||||
if pre_fold:
|
||||
# Splatted: the pre-fold seam took (msgs, my_generation), and a
|
||||
# literal two-argument call reads as an arity error against the
|
||||
# signature this tree actually has.
|
||||
pre_fold_args: tuple[Any, ...] = ([{"role": "user", "content": "hi"}], 0)
|
||||
msg = session._stream_response(*pre_fold_args)
|
||||
msg.pop("_wire_msgs", None)
|
||||
record["result"] = {
|
||||
"content": msg.get("content", ""),
|
||||
"tool_calls": msg.get("tool_calls"),
|
||||
"provider_content": msg.get("_provider_content"),
|
||||
}
|
||||
else:
|
||||
session.messages.append(Turn.user("hi"))
|
||||
result = session._stream_response(0)
|
||||
record["result"] = {
|
||||
"content": result.content,
|
||||
"tool_calls": result.tool_calls or None,
|
||||
"provider_content": (
|
||||
[dict(b) for b in result.turn.native.blocks] if result.turn.native else None
|
||||
),
|
||||
}
|
||||
record["raised"] = None
|
||||
except BaseException as exc: # noqa: BLE001 — the record IS the observation
|
||||
record["result"] = None
|
||||
record["raised"] = type(exc).__name__
|
||||
record["ui_events"] = [[k, d] for k, d in ui.events]
|
||||
record["last_usage"] = session._last_usage
|
||||
record["cancelled_partial"] = session._cancelled_partial_msg
|
||||
return _mask_synth_ids(record)
|
||||
|
||||
|
||||
def fixture_path(name: str) -> Path:
|
||||
return FIXTURE_DIR / f"{name}.json"
|
||||
|
||||
|
||||
def load_fixture(name: str) -> dict[str, Any]:
|
||||
return json.loads(fixture_path(name).read_text())
|
||||
|
||||
|
||||
def write_fixture(name: str, record: dict[str, Any]) -> None:
|
||||
# A TypeError before ANY UI event is the harness's own call-shape
|
||||
# failure (run_scenario's signature adapter no longer matches this
|
||||
# tree's seam), not old-world behavior — refuse to destroy the
|
||||
# baseline with it.
|
||||
if record.get("raised") == "TypeError" and not record.get("ui_events"):
|
||||
raise AssertionError(
|
||||
f"parity capture for {name!r} died calling the seam (TypeError before "
|
||||
f"any UI event) — fix run_scenario's signature adapter; do not record"
|
||||
)
|
||||
FIXTURE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
fixture_path(name).write_text(json.dumps(record, indent=2, sort_keys=True) + "\n")
|
||||
@@ -1,171 +0,0 @@
|
||||
"""Inline-reasoning dialect conformance catalog.
|
||||
|
||||
Passthrough servers (parserless vLLM/llama.cpp, LM Studio, bare gateways)
|
||||
emit model reasoning inline as ``<think>``/``<reasoning>`` blocks inside the
|
||||
content stream — a *dialect* of model output. This module is that dialect's
|
||||
executable specification for ``split_inline_reasoning``: each case maps an
|
||||
utterance to the exact ``(content, reasoning)`` lanes the one-shot must
|
||||
produce.
|
||||
|
||||
Consumers: the one-shot conformance and one-shot≡streaming property suites
|
||||
in tests/test_think_tag_split.py. Lane suites (session, judge,
|
||||
output-guard, optimizer, drain-stream) pin their lanes with suite-local
|
||||
utterances through their own fakes — adding a case HERE extends the
|
||||
semantics spec, not automatically any lane suite.
|
||||
|
||||
The split is RAW (residue whitespace stays; ``drain_stream`` owns the one
|
||||
trim over its joined runs). ``passthrough`` marks cases the split must
|
||||
return BYTE-IDENTICAL: tag-free text, and text whose only tags are orphan
|
||||
CLOSE tags. The latter is a
|
||||
review ruling, not an accident: a close tag whose open never arrived is
|
||||
indistinguishable from prose QUOTING the tag, and drained lanes routinely
|
||||
quote third-party text (web-fetch answers citing pages about reasoning
|
||||
models, guard verdicts echoing judged content) — any reclassification
|
||||
would let quoted text destroy real results. Display lanes wanting
|
||||
stricter cosmetic peeling (the title) own that locally as formatting.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DialectCase:
|
||||
id: str
|
||||
utterance: str
|
||||
content: str
|
||||
reasoning: str
|
||||
# The split returns the utterance byte-identical (no tag consumed):
|
||||
# tag-free text, or orphan-close-only text (quoted-tag safety).
|
||||
passthrough: bool = False
|
||||
|
||||
|
||||
CASES: tuple[DialectCase, ...] = (
|
||||
DialectCase(
|
||||
id="no_tag_byte_identity",
|
||||
utterance="Just an answer.",
|
||||
content="Just an answer.",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
# The fast path must not strip: unconsumed content is
|
||||
# byte-identical, whitespace included.
|
||||
id="no_tag_preserves_whitespace",
|
||||
utterance=" spaced \n",
|
||||
content=" spaced \n",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
id="leading_block",
|
||||
utterance="<think>plan</think>Answer",
|
||||
content="Answer",
|
||||
reasoning="plan",
|
||||
),
|
||||
DialectCase(
|
||||
# The split is RAW — tag residue stays; drain_stream owns the ONE
|
||||
# blank-edge-line trim over its joined runs (pinned there), which
|
||||
# preserves the code block's first-line indentation.
|
||||
id="indented_code_block_raw_residue",
|
||||
utterance="<think>plan</think>\n\n print(1)\n more()",
|
||||
content="\n\n print(1)\n more()",
|
||||
reasoning="plan",
|
||||
),
|
||||
DialectCase(
|
||||
id="leading_block_raw_residue",
|
||||
utterance="<think>plan</think>\n\nAnswer\n",
|
||||
content="\n\nAnswer\n",
|
||||
reasoning="plan",
|
||||
),
|
||||
DialectCase(
|
||||
id="interleaved_blocks_both_vocabularies",
|
||||
utterance="Intro <think>a</think>mid <reasoning>b</reasoning>end",
|
||||
content="Intro mid end",
|
||||
reasoning="ab",
|
||||
),
|
||||
DialectCase(
|
||||
id="unterminated_open_tail_is_reasoning",
|
||||
utterance="Answer part<think>never closed",
|
||||
content="Answer part",
|
||||
reasoning="never closed",
|
||||
),
|
||||
DialectCase(
|
||||
# QUOTED-CLOSE SAFETY (review ruling): an orphan close is
|
||||
# indistinguishable from a quoted tag — everything passes through.
|
||||
# A malicious page embedding the literal string must not be able
|
||||
# to wipe the extraction that quotes it.
|
||||
id="orphan_close_passes_through",
|
||||
utterance="The page says templates emit </think> after the preamble. Answer: 42.",
|
||||
content="The page says templates emit </think> after the preamble. Answer: 42.",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
# Template-pre-injected shape ("reasoning</think>answer"): the seam
|
||||
# deliberately passes it through — segregating it would require
|
||||
# treating every quoted close as a boundary. Post-#831 every lane
|
||||
# streams, and known streaming surfaces strip the orphan close
|
||||
# server-side; display lanes peel cosmetically on their own.
|
||||
id="preinject_shape_passes_through",
|
||||
utterance="plan text</think>\n\nAnswer",
|
||||
content="plan text</think>\n\nAnswer",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
id="immediate_close_passes_through",
|
||||
utterance="</think>Answer",
|
||||
content="</think>Answer",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
# Any close tag closes any open block (splitter semantics; the old
|
||||
# pairwise per-caller strip treated this as unterminated).
|
||||
id="cross_vocabulary_close",
|
||||
utterance="<think>x</reasoning>Answer",
|
||||
content="Answer",
|
||||
reasoning="x",
|
||||
),
|
||||
DialectCase(
|
||||
id="think_only",
|
||||
utterance="<think>all reasoning</think>",
|
||||
content="",
|
||||
reasoning="all reasoning",
|
||||
),
|
||||
DialectCase(
|
||||
id="think_only_unterminated",
|
||||
utterance="<think>everything",
|
||||
content="",
|
||||
reasoning="everything",
|
||||
),
|
||||
DialectCase(
|
||||
# A balanced block followed by a stray close: the block is
|
||||
# consumed, the stray close stays in content (quoted-tag safety),
|
||||
# and the consumed-tag strip applies.
|
||||
id="balanced_block_then_stray_close",
|
||||
utterance="<think>a</think>b</think>c",
|
||||
content="b</think>c",
|
||||
reasoning="a",
|
||||
),
|
||||
DialectCase(
|
||||
id="multiple_blocks_accumulate",
|
||||
utterance="<think>one</think>mid<think>two</think>tail",
|
||||
content="midtail",
|
||||
reasoning="onetwo",
|
||||
),
|
||||
DialectCase(
|
||||
# ACCEPTED RESIDUAL (R2): the split is content-blind, so a literal
|
||||
# OPEN tag in legitimate prose misroutes the remainder — the same
|
||||
# false positive the interactive splitter has carried in the
|
||||
# field. This pin makes any future fix a conscious change.
|
||||
# Scope note: R2 applies only where the scan runs — a backend
|
||||
# declaring ``server_parses_reasoning`` turns the scan off and
|
||||
# this utterance passes through byte-identical (pinned in
|
||||
# test_scan_tags_off_returns_every_utterance_byte_identical).
|
||||
id="literal_open_tag_false_positive_r2",
|
||||
utterance="The `<think>` tag opens a block.",
|
||||
content="The `",
|
||||
reasoning="` tag opens a block.",
|
||||
),
|
||||
)
|
||||
+7
-563
@@ -1,13 +1,12 @@
|
||||
"""Shared session-test helpers.
|
||||
|
||||
The minimal ``ChatSession`` factory, the ``SessionUIBase`` no-op/recording
|
||||
subclasses, and the tree's standard streaming provider fakes
|
||||
(``make_result`` / ``arm_session`` / ``scripted_provider`` /
|
||||
``ArmedHandle``, at the bottom): every suite driving the streaming seam
|
||||
imports them from here, so the eager-arming contract lives in one place.
|
||||
The one deliberate exception, ``test_model_registry.py``'s
|
||||
``_make_session``, takes a different signature (registry / model_alias /
|
||||
reasoning_effort + ``_FakeUI``) and is NOT a candidate for sharing.
|
||||
Two reasoning-test modules (``test_session_replay_reasoning.py`` and
|
||||
``test_session_synth_reasoning_block.py``) need the same minimal
|
||||
``ChatSession`` factory + a ``SessionUIBase`` no-op subclass. Hoisting
|
||||
keeps a future third caller from drifting on the defaults — the third
|
||||
existing ``_make_session`` (``test_model_registry.py``) deliberately
|
||||
takes a different signature (registry / model_alias / reasoning_effort
|
||||
+ ``_FakeUI``) and is NOT a candidate for sharing this helper.
|
||||
|
||||
Module is named with a leading underscore so pytest doesn't try to
|
||||
collect it as a test file — it's an importable utility, not a test.
|
||||
@@ -15,16 +14,11 @@ collect it as a test file — it's an importable utility, not a test.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from turnstone.core.model_turn import ModelTurnResult
|
||||
from turnstone.core.providers import ModelCapabilities, StreamChunk, ToolCallDelta, UsageInfo
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.trajectory import ProviderNative, ToolCall, Turn
|
||||
|
||||
|
||||
class NullUI(SessionUIBase):
|
||||
@@ -49,553 +43,3 @@ def make_session(**kwargs: Any) -> ChatSession:
|
||||
}
|
||||
defaults.update(kwargs)
|
||||
return ChatSession(**defaults)
|
||||
|
||||
|
||||
def mock_completion_result(
|
||||
content: str = "",
|
||||
tool_calls: list[dict[str, Any]] | None = None,
|
||||
) -> MagicMock:
|
||||
"""A provider result shaped like ``CompletionResult``.
|
||||
|
||||
Callers that route through ``model_turn`` (judges, task agents, and
|
||||
every lane #827 migrates) hit its re-ingest, which iterates
|
||||
``tool_calls``/``provider_blocks`` and joins ``reasoning`` — a bare
|
||||
MagicMock attribute would TypeError deep inside the seam, so every
|
||||
field the re-ingest reads is pinned to a real value here. ONE shared
|
||||
definition: when the re-ingest starts reading a new CompletionResult
|
||||
field, add it here and every suite moves together.
|
||||
"""
|
||||
result = MagicMock()
|
||||
result.content = content
|
||||
result.tool_calls = tool_calls
|
||||
result.finish_reason = "stop"
|
||||
result.usage = None
|
||||
result.provider_blocks = []
|
||||
result.reasoning = ""
|
||||
return result
|
||||
|
||||
|
||||
def fake_chat_stream(
|
||||
*,
|
||||
content: str | None = None,
|
||||
tool_calls: list[dict[str, str]] | None = None,
|
||||
finish_reason: str = "stop",
|
||||
prompt_tokens: int = 10,
|
||||
completion_tokens: int = 5,
|
||||
reasoning_content: str | None = None,
|
||||
reasoning: str | None = None,
|
||||
) -> list[Any]:
|
||||
"""Fake OpenAI Chat Completions SSE chunks for driving the REAL
|
||||
``OpenAIChatCompletionsProvider`` through a fake SDK client::
|
||||
|
||||
client.chat.completions.create = lambda **kw: fake_chat_stream(...)
|
||||
|
||||
Exercises the adapter's ``_iter_stream`` plus ``drain_stream`` end to
|
||||
end (the highest-fidelity fake lane), unlike ``as_stream`` which fakes
|
||||
at the provider boundary. ``tool_calls`` entries are
|
||||
``{"id", "name", "arguments"}`` dicts. ``SimpleNamespace`` (not
|
||||
``MagicMock``) so absent SDK fields read as real ``None`` — an
|
||||
auto-created mock attribute would leak into ``len()``/string paths.
|
||||
|
||||
Emits the realistic three-phase shape: data chunk(s), a finish-reason
|
||||
chunk, then the ``stream_options.include_usage`` usage-only chunk with
|
||||
empty ``choices``.
|
||||
"""
|
||||
|
||||
def _delta(
|
||||
content_val: str | None = None,
|
||||
tcs: list[Any] | None = None,
|
||||
rc: str | None = None,
|
||||
rsn: str | None = None,
|
||||
) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
content=content_val,
|
||||
tool_calls=tcs,
|
||||
reasoning=rsn,
|
||||
reasoning_content=rc,
|
||||
annotations=None,
|
||||
)
|
||||
|
||||
chunks: list[Any] = []
|
||||
if reasoning_content is not None or reasoning is not None:
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[
|
||||
SimpleNamespace(
|
||||
finish_reason=None, delta=_delta(rc=reasoning_content, rsn=reasoning)
|
||||
)
|
||||
],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
if content is not None:
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=None, delta=_delta(content))],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
if tool_calls:
|
||||
tcs = [
|
||||
SimpleNamespace(
|
||||
index=i,
|
||||
id=tc.get("id", ""),
|
||||
function=SimpleNamespace(
|
||||
name=tc.get("name", ""), arguments=tc.get("arguments", "")
|
||||
),
|
||||
)
|
||||
for i, tc in enumerate(tool_calls)
|
||||
]
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=None, delta=_delta(None, tcs))],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=finish_reason, delta=_delta())],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[],
|
||||
usage=SimpleNamespace(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=None,
|
||||
input_tokens_details=None,
|
||||
),
|
||||
)
|
||||
)
|
||||
return chunks
|
||||
|
||||
|
||||
class _ScriptedClient:
|
||||
"""Callable client-method fake following a script of stream builders.
|
||||
|
||||
Call N returns the stream described by ``scripts[N]``; the last script
|
||||
repeats for any further calls. Each script is a dict of kwargs for
|
||||
the bound stream builder, or a pre-built return value. Records every
|
||||
call's kwargs on ``.calls`` — read ``len(fn.calls)`` where a test
|
||||
previously kept its own counter cell, and ``fn.calls[i]["messages"]``
|
||||
where it captured request bodies.
|
||||
"""
|
||||
|
||||
def __init__(self, scripts: tuple[Any, ...], to_stream: Any) -> None:
|
||||
self._scripts = scripts
|
||||
self._to_stream = to_stream
|
||||
self.calls: list[dict[str, Any]] = []
|
||||
|
||||
def __call__(self, **kwargs: Any) -> Any:
|
||||
self.calls.append(kwargs)
|
||||
script = self._scripts[min(len(self.calls) - 1, len(self._scripts) - 1)]
|
||||
return self._to_stream(**script) if isinstance(script, dict) else script
|
||||
|
||||
|
||||
def scripted_chat_client(*scripts: Any) -> _ScriptedClient:
|
||||
"""A scripted ``client.chat.completions.create`` — dict scripts are
|
||||
:func:`fake_chat_stream` kwargs."""
|
||||
return _ScriptedClient(scripts, fake_chat_stream)
|
||||
|
||||
|
||||
def scripted_anthropic_client(*scripts: Any) -> _ScriptedClient:
|
||||
"""A scripted ``client.messages.stream`` — dict scripts are
|
||||
:func:`fake_anthropic_stream` kwargs (``blocks`` plus optional
|
||||
``stop_reason``/``usage``)."""
|
||||
return _ScriptedClient(scripts, fake_anthropic_stream)
|
||||
|
||||
|
||||
class FakeAnthropicBlock:
|
||||
"""A full-content Anthropic content-block fake for
|
||||
:func:`fake_anthropic_stream` — plain attributes plus the
|
||||
``model_dump()`` the provider's block capture reads."""
|
||||
|
||||
def __init__(self, **fields: Any) -> None:
|
||||
self._fields = fields
|
||||
for key, value in fields.items():
|
||||
setattr(self, key, value)
|
||||
|
||||
def model_dump(self, **_kw: Any) -> dict[str, Any]:
|
||||
return dict(self._fields)
|
||||
|
||||
|
||||
def fake_anthropic_stream(
|
||||
blocks: list[Any],
|
||||
*,
|
||||
stop_reason: str | None = "end_turn",
|
||||
usage: Any = None,
|
||||
) -> Any:
|
||||
"""Fake Anthropic SDK stream context manager for tests that drive the
|
||||
REAL ``AnthropicProvider`` through a fake client::
|
||||
|
||||
client.messages.stream = lambda **kw: fake_anthropic_stream(...)
|
||||
|
||||
Accepts the same full-content block fakes the pre-#831
|
||||
``get_final_message`` fixtures used (objects with ``.type`` + fields
|
||||
and ``model_dump()``) and synthesizes the real event grammar the
|
||||
streaming iterator consumes: ``content_block_start`` carries the block
|
||||
with its text/thinking/signature EMPTIED and ``input`` as ``{}`` (the
|
||||
SDK start shape), deltas carry the content, ``content_block_stop``
|
||||
finalizes tool input, and the closing ``message_delta`` carries
|
||||
``stop_reason`` (+ optional usage object). Without the stripping, the
|
||||
provider's raw-block accumulator would double every text/thinking
|
||||
field (start capture + delta append).
|
||||
|
||||
``stop_reason=None`` omits the closing ``message_delta`` entirely —
|
||||
the terminal-signal-less lax-gateway shape ``finish_reason_optional``
|
||||
exists for (content arrives, then the stream just ends).
|
||||
"""
|
||||
events: list[Any] = []
|
||||
for idx, block in enumerate(blocks):
|
||||
d = dict(block.model_dump()) if hasattr(block, "model_dump") else dict(vars(block))
|
||||
btype = d.get("type", "")
|
||||
start = dict(d)
|
||||
if btype == "text":
|
||||
start["text"] = ""
|
||||
elif btype == "thinking":
|
||||
start["thinking"] = ""
|
||||
start["signature"] = ""
|
||||
elif btype == "tool_use":
|
||||
start["input"] = {}
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_start", index=idx, content_block=SimpleNamespace(**start)
|
||||
)
|
||||
)
|
||||
if btype == "text" and d.get("text"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="text_delta", text=d["text"]),
|
||||
)
|
||||
)
|
||||
elif btype == "thinking":
|
||||
if d.get("thinking"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="thinking_delta", thinking=d["thinking"]),
|
||||
)
|
||||
)
|
||||
if d.get("signature"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="signature_delta", signature=d["signature"]),
|
||||
)
|
||||
)
|
||||
elif btype == "tool_use":
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(
|
||||
type="input_json_delta",
|
||||
partial_json=json.dumps(d.get("input", {})),
|
||||
),
|
||||
)
|
||||
)
|
||||
events.append(SimpleNamespace(type="content_block_stop", index=idx))
|
||||
if stop_reason is not None or usage is not None:
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="message_delta", usage=usage, delta=SimpleNamespace(stop_reason=stop_reason)
|
||||
)
|
||||
)
|
||||
|
||||
mgr = MagicMock()
|
||||
mgr.__enter__ = MagicMock(return_value=events)
|
||||
mgr.__exit__ = MagicMock(return_value=False)
|
||||
return mgr
|
||||
|
||||
|
||||
def as_stream(result: Any) -> list[StreamChunk]:
|
||||
"""Adapt a ``CompletionResult``-shaped fake to a ``create_streaming``
|
||||
return value (single terminal chunk).
|
||||
|
||||
The #831 transport collapse routes every single-shot lane through
|
||||
``drain_stream(provider.create_streaming(...))``, so provider fakes
|
||||
return chunk iterables now. Tests keep building result-shaped fakes
|
||||
(``mock_completion_result`` or hand-rolled) and wrap them at
|
||||
assignment: ``provider.create_streaming.return_value =
|
||||
as_stream(result)``. A list re-iterates on every call, so one
|
||||
``return_value`` serves repeated-call tests; convert AFTER mutating
|
||||
the fake's fields — the chunk snapshots them.
|
||||
|
||||
Multi-chunk accumulation semantics are exercised by the dedicated
|
||||
``drain_stream`` unit tests, not through this helper.
|
||||
"""
|
||||
deltas = [
|
||||
ToolCallDelta(
|
||||
index=i,
|
||||
id=tc.get("id", ""),
|
||||
name=tc.get("function", {}).get("name", ""),
|
||||
arguments_delta=tc.get("function", {}).get("arguments", ""),
|
||||
)
|
||||
for i, tc in enumerate(result.tool_calls or [])
|
||||
]
|
||||
return [
|
||||
StreamChunk(
|
||||
content_delta=result.content or "",
|
||||
reasoning_delta=getattr(result, "reasoning", "") or "",
|
||||
tool_call_deltas=deltas,
|
||||
usage=result.usage,
|
||||
finish_reason=result.finish_reason or "stop",
|
||||
provider_blocks=list(result.provider_blocks or []),
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
def think_tag_stream(utterance: str) -> list[StreamChunk]:
|
||||
"""``create_streaming`` return value simulating a passthrough server
|
||||
that emits *utterance* — typically think-tag-bearing — as plain
|
||||
streamed content.
|
||||
|
||||
The per-lane fixture for inline-reasoning dialect pins: lane tests
|
||||
supply their own utterances (the dialect's SEMANTICS are specified
|
||||
once, in ``tests._reasoning_dialect.CASES``, and pinned by the
|
||||
one-shot suites — lane pins assert lane behavior, not tag grammar).
|
||||
Routes through the real ``drain_stream`` seam exactly like
|
||||
``as_stream``.
|
||||
"""
|
||||
return as_stream(mock_completion_result(content=utterance))
|
||||
|
||||
|
||||
def seam_provider(utterance: str, *, provider_name: str = "openai-compatible") -> MagicMock:
|
||||
"""Provider fake whose ``create_streaming`` replays *utterance* through
|
||||
the REAL drain seam (``think_tag_stream``) — THE lane-suite seam fake.
|
||||
|
||||
One definition so the lane suites cannot drift when the provider
|
||||
surface ``model_turn`` probes grows: real ``ModelCapabilities`` for
|
||||
the clamp math, ``provider_name`` overridable per suite.
|
||||
|
||||
Assign the RETURNED fake to ``session._provider`` — never mutate the
|
||||
provider a session resolved on its own: with a MagicMock client the
|
||||
session resolves the process-wide ``create_provider(...)`` singleton,
|
||||
and writing that shared instance's ``create_streaming`` poisons every
|
||||
later session in the test run (the SSE-recovery e2e servers resolve
|
||||
the same instance).
|
||||
"""
|
||||
provider = provider_shell(provider_name)
|
||||
provider.create_streaming = MagicMock(return_value=think_tag_stream(utterance))
|
||||
return provider
|
||||
|
||||
|
||||
class RecordingUI:
|
||||
"""UI adapter recording the ordered event stream ``send()`` emits."""
|
||||
|
||||
def __init__(self):
|
||||
self.events = []
|
||||
|
||||
def _rec(self, kind, detail=""):
|
||||
self.events.append((kind, detail))
|
||||
|
||||
def on_turn_start(self):
|
||||
self._rec("turn_start")
|
||||
|
||||
def on_turn_committed(self):
|
||||
self._rec("turn_committed")
|
||||
|
||||
def on_stream_discarded(self):
|
||||
self._rec("stream_discarded")
|
||||
|
||||
def on_thinking_start(self):
|
||||
self._rec("thinking_start")
|
||||
|
||||
def on_thinking_stop(self):
|
||||
self._rec("thinking_stop")
|
||||
|
||||
def on_reasoning_token(self, text):
|
||||
self._rec("reasoning", text)
|
||||
|
||||
def on_content_token(self, text):
|
||||
self._rec("content", text)
|
||||
|
||||
def on_stream_end(self):
|
||||
self._rec("stream_end")
|
||||
|
||||
def approve_tools(self, items):
|
||||
return True, None
|
||||
|
||||
def on_tool_result(self, call_id, name, output, **kwargs):
|
||||
pass
|
||||
|
||||
def on_tool_output_chunk(self, call_id, chunk):
|
||||
pass
|
||||
|
||||
def on_status(self, usage, context_window, effort):
|
||||
pass
|
||||
|
||||
def on_info(self, message):
|
||||
self._rec("info", message)
|
||||
|
||||
def on_error(self, message):
|
||||
self._rec("error", message)
|
||||
|
||||
def on_state_change(self, state):
|
||||
self._rec("state", state)
|
||||
|
||||
def on_rename(self, name):
|
||||
pass
|
||||
|
||||
def on_output_warning(self, call_id, assessment):
|
||||
pass
|
||||
|
||||
def record_output_assessment(
|
||||
self,
|
||||
call_id,
|
||||
assessment,
|
||||
*,
|
||||
tier="heuristic",
|
||||
reasoning="",
|
||||
judge_model="",
|
||||
latency_ms=0,
|
||||
confidence=0.0,
|
||||
):
|
||||
pass
|
||||
|
||||
def kinds(self):
|
||||
return [k for k, _ in self.events]
|
||||
|
||||
def of(self, kind):
|
||||
return [d for k, d in self.events if k == kind]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Streaming provider fakes — the #832 seam contract
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def make_result(
|
||||
content: str = "",
|
||||
*,
|
||||
tool_calls: list[dict[str, Any]] | None = None,
|
||||
finish_reason: str = "stop",
|
||||
usage: UsageInfo | None = None,
|
||||
native_blocks: list[dict[str, Any]] | None = None,
|
||||
producer: str = "openai-compatible",
|
||||
wire_msgs: list[dict[str, Any]] | None = None,
|
||||
) -> ModelTurnResult:
|
||||
"""A ``ModelTurnResult`` shaped like the streaming wrapper's return,
|
||||
for tests that only need "a turn happened" and patch
|
||||
``_stream_response`` wholesale. Turn and ``tool_calls`` mirror are
|
||||
built from the same dicts, preserving the #825 pairing invariant."""
|
||||
calls = list(tool_calls or [])
|
||||
tc_tuple = tuple(
|
||||
ToolCall(
|
||||
id=tc.get("id", ""),
|
||||
name=tc.get("function", {}).get("name", ""),
|
||||
arguments=tc.get("function", {}).get("arguments", ""),
|
||||
)
|
||||
for tc in calls
|
||||
)
|
||||
native = (
|
||||
ProviderNative(producer=producer, blocks=tuple(native_blocks)) if native_blocks else None
|
||||
)
|
||||
return ModelTurnResult(
|
||||
turn=Turn.assistant(content, tool_calls=tc_tuple, native=native),
|
||||
finish_reason=finish_reason,
|
||||
usage=usage,
|
||||
tool_calls=calls,
|
||||
wire_msgs=wire_msgs,
|
||||
producer=producer,
|
||||
)
|
||||
|
||||
|
||||
class ArmedHandle:
|
||||
"""Closeable sentinel standing in for the SDK stream handle."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.closed = False
|
||||
|
||||
def close(self) -> None:
|
||||
self.closed = True
|
||||
|
||||
|
||||
def provider_shell(
|
||||
name: str = "openai-compatible",
|
||||
retryable: frozenset[str] = frozenset({"IncompleteStreamError"}),
|
||||
) -> MagicMock:
|
||||
"""The armed-provider fake skeleton every streaming fake builds on:
|
||||
provider_name / capabilities / retryable set. ONE spelling, so a new
|
||||
attribute the seam starts probing lands in every fake at once."""
|
||||
provider = MagicMock()
|
||||
provider.provider_name = name
|
||||
provider.get_capabilities.return_value = ModelCapabilities()
|
||||
provider.retryable_error_names = retryable
|
||||
return provider
|
||||
|
||||
|
||||
def arm_session(
|
||||
session: Any,
|
||||
*streams: Any,
|
||||
retryable: frozenset[str] = frozenset({"IncompleteStreamError"}),
|
||||
name: str = "openai-compatible",
|
||||
) -> MagicMock:
|
||||
"""Install a sequential multi-turn armed provider fake on *session*.
|
||||
|
||||
Each ``create_streaming`` call serves the next element of *streams*:
|
||||
an iterable/generator is armed (a closeable sentinel appended to
|
||||
``cancel_ref`` — the eager append every real adapter performs, which
|
||||
the creation-vs-midstream classifier keys on) and returned to be
|
||||
consumed once; an EXCEPTION instance is raised at create time WITHOUT
|
||||
arming, a creation-phase failure the per-lane ladder owns. Calls
|
||||
beyond the script fail loudly: the strict finish gate rejects an
|
||||
exhausted iterator rather than absorbing it as a silent empty turn,
|
||||
so an under-scripted test must say so.
|
||||
|
||||
Title generation is latched off — with a provider-LEVEL fake the
|
||||
best-effort title lane would otherwise consume the first script
|
||||
before the main loop ran.
|
||||
"""
|
||||
session._title_generated = True
|
||||
provider = provider_shell(name, retryable)
|
||||
# One handle PER CREATE (the real adapters' rule): `handles` records
|
||||
# them all, `_armed_handle` is the latest.
|
||||
provider._armed_handle = None
|
||||
provider.handles = []
|
||||
remaining = list(streams)
|
||||
|
||||
def _create(**kwargs: Any):
|
||||
assert remaining, "arm_session: script exhausted — send looped for more turns than scripted"
|
||||
nxt = remaining.pop(0)
|
||||
if isinstance(nxt, BaseException):
|
||||
raise nxt
|
||||
ref = kwargs.get("cancel_ref")
|
||||
if ref is not None:
|
||||
handle = ArmedHandle()
|
||||
provider.handles.append(handle)
|
||||
provider._armed_handle = handle
|
||||
ref.append(handle)
|
||||
return iter(nxt) if not hasattr(nxt, "__next__") else nxt
|
||||
|
||||
provider.create_streaming = MagicMock(side_effect=_create)
|
||||
session._provider = provider
|
||||
return provider
|
||||
|
||||
|
||||
def scripted_provider(chunks: list[StreamChunk]) -> MagicMock:
|
||||
"""Provider fake replaying *chunks*, arming ``cancel_ref`` eagerly.
|
||||
|
||||
Assign to ``session._provider`` (never mutate a resolved provider —
|
||||
the create_provider singleton rule above). Each call returns a FRESH
|
||||
iterator over the same script so ladder tests re-drive it; the armed
|
||||
handle is appended per call, matching the one-handle-per-create
|
||||
behavior of every real adapter.
|
||||
"""
|
||||
provider = provider_shell()
|
||||
|
||||
def _create(**kwargs: Any):
|
||||
ref = kwargs.get("cancel_ref")
|
||||
if ref is not None:
|
||||
ref.append(ArmedHandle())
|
||||
return iter(chunks)
|
||||
|
||||
provider.create_streaming = MagicMock(side_effect=_create)
|
||||
return provider
|
||||
|
||||
@@ -1,604 +0,0 @@
|
||||
"""Browser-fidelity SSE recovery harness helpers.
|
||||
|
||||
The load-bearing assembly for ``tests/test_sse_recovery_e2e.py``: a
|
||||
``BrowserlikeSSEClient`` that speaks the exact wire contract the real
|
||||
``turnstone/shared_static/interactive.js`` pane speaks, and the
|
||||
assertion helpers the scenarios share. The server boot machinery lives
|
||||
in ``_sse_recovery_server.py``.
|
||||
|
||||
Why a raw-socket SSE reader (and not ``httpx.stream``): the slow-consumer
|
||||
overflow scenario needs the consumer to STALL — stop reading the socket
|
||||
so the server's SSE generator blocks on ``await send`` and stops draining
|
||||
the per-UI listener queue, which then poisons at its cap. A faithful
|
||||
stall needs (a) precise control over when bytes are read and (b) a small
|
||||
``SO_RCVBUF`` so the in-flight backlog before poison stays bounded to
|
||||
~100 KB instead of the client kernel's multi-MB autotuned default (which
|
||||
would need tens of thousands of events to overflow). A raw socket gives
|
||||
both; httpx (used here only for the plain ``/history`` request/response)
|
||||
gives neither. This is ALSO closer to the browser: EventSource has a
|
||||
bounded receive buffer, not an unbounded one.
|
||||
|
||||
Client contract mirrored from interactive.js (line references are to
|
||||
that file on the ``fix/sse-truncated-resync`` branch):
|
||||
|
||||
- ``_last_event_id`` advances ONLY from SSE ``id:`` fields, and only
|
||||
ring-buffer events carry one — synthetic replay frames (connected /
|
||||
status / state_change / in_progress_snapshot / replay_truncated /
|
||||
stream_overflow) do not, exactly like ``EventSource.lastEventId``
|
||||
(interactive.js onmessage ~1378).
|
||||
- reconnect presents ``connectCursor = _truncatedFromCursor ??
|
||||
_lastEventId`` as ``?last_event_id=`` (manual path) or a
|
||||
``Last-Event-ID`` header (native EventSource auto-reconnect path)
|
||||
(interactive.js connectSSE ~1328).
|
||||
- on a ``replay_truncated`` envelope the client records the
|
||||
truncation-time cursor keep-oldest (``_truncatedFromCursor =
|
||||
_lastEventId`` only when null) and runs ``_loadHistoryThenConnect``
|
||||
(disconnect → /history → adopt cursor → reconnect); a FAILED
|
||||
/history leaves the record armed so the reconnect re-presents the
|
||||
truncation-time cursor and re-draws the envelope (interactive.js
|
||||
handleEvent replay_truncated ~2303, _loadHistoryThenConnect ~1604,
|
||||
_refetchHistory seedCursor ~1706).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import json
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import httpx
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
|
||||
# Small client receive buffer so a stalled consumer's in-flight backlog
|
||||
# before the server-side poison stays bounded (~100 KB) instead of the
|
||||
# multi-MB autotuned default. Paired with the server's small SO_SNDBUF
|
||||
# (see _sse_recovery_server.build_recovery_server).
|
||||
_CLIENT_RCVBUF = 2048
|
||||
|
||||
|
||||
@dataclass
|
||||
class SSEFrame:
|
||||
"""One decoded SSE frame, tagged with the connection it arrived on.
|
||||
|
||||
``event_id`` is the ``id:`` field verbatim (a stringified integer,
|
||||
or ``None`` for id-less synthetic frames — the same string domain as
|
||||
``EventSource.lastEventId``). ``etype`` is the ``type`` field of the
|
||||
JSON ``data:`` payload (the application event type), distinct from
|
||||
any SSE ``event:`` field, which the server never uses.
|
||||
"""
|
||||
|
||||
conn_index: int
|
||||
event_id: str | None
|
||||
etype: str | None
|
||||
payload: dict[str, Any] | None
|
||||
raw: str
|
||||
|
||||
@property
|
||||
def event_id_int(self) -> int | None:
|
||||
if self.event_id is None:
|
||||
return None
|
||||
try:
|
||||
return int(self.event_id)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
class BrowserlikeSSEClient:
|
||||
"""A single interactive pane's SSE + /history state machine.
|
||||
|
||||
Not thread-safe against concurrent public calls; drive it from one
|
||||
test thread. Internally a per-connection reader thread decodes the
|
||||
stream; ``stall()`` / ``resume()`` gate that thread's socket reads so
|
||||
a test can build server-side backpressure without closing the
|
||||
connection (the slow-consumer → listener-queue-poison path).
|
||||
"""
|
||||
|
||||
def __init__(self, base_url: str, ws_id: str, token: str) -> None:
|
||||
parts = urlsplit(base_url)
|
||||
self._host = parts.hostname or "127.0.0.1"
|
||||
self._port = parts.port or 80
|
||||
self._ws_id = ws_id
|
||||
self._token = token
|
||||
self._auth = {"Authorization": f"Bearer {token}"}
|
||||
self._http = httpx.Client(
|
||||
base_url=f"http://{self._host}:{self._port}", timeout=httpx.Timeout(15.0)
|
||||
)
|
||||
|
||||
# EventSource-equivalent cursor state.
|
||||
self._last_event_id: str | None = None
|
||||
self._truncated_from_cursor: str | None = None
|
||||
|
||||
# Transcript. ``_all_frames`` is the cross-connection accumulation
|
||||
# (what "the client eventually saw"); ``_conn_frames`` keeps each
|
||||
# connection's slice for per-connection assertions (contiguity).
|
||||
self._all_frames: list[SSEFrame] = []
|
||||
self._conn_frames: list[list[SSEFrame]] = []
|
||||
self._frames_lock = threading.Lock()
|
||||
|
||||
# Reader plumbing.
|
||||
self._sock: socket.socket | None = None
|
||||
self._reader: threading.Thread | None = None
|
||||
self._stop = threading.Event()
|
||||
self._read_gate = threading.Event()
|
||||
self._read_gate.set() # reading permitted by default
|
||||
self._status: int | None = None
|
||||
self._headers_done = threading.Event()
|
||||
|
||||
# -- connection lifecycle ------------------------------------------------
|
||||
|
||||
def _events_path(self, cursor: str | None) -> str:
|
||||
path = f"/v1/api/workstreams/{self._ws_id}/events"
|
||||
if cursor is not None:
|
||||
path += f"?last_event_id={cursor}"
|
||||
return path
|
||||
|
||||
def connect(self, *, native: bool = False, rcvbuf: int | None = None) -> None:
|
||||
"""Open the SSE stream, presenting the client's current cursor.
|
||||
|
||||
``native=True`` models the browser's EventSource auto-reconnect:
|
||||
the cursor rides a ``Last-Event-ID`` HEADER and never appears in
|
||||
the URL. ``native=False`` models the manual ``new EventSource(url
|
||||
+ '?last_event_id=')`` path interactive.js uses when it must
|
||||
override the live cursor (the ``connectCursor`` chokepoint).
|
||||
|
||||
``rcvbuf`` shrinks this connection's ``SO_RCVBUF`` — pass
|
||||
``_CLIENT_RCVBUF`` on a connection the test will ``stall()`` so the
|
||||
in-flight backlog before the server-side poison stays bounded.
|
||||
Leave it ``None`` (OS default) on recovery reconnects so the ring
|
||||
replay is not throttled to a crawl.
|
||||
"""
|
||||
if self._reader is not None:
|
||||
raise RuntimeError("already connected; disconnect() first")
|
||||
connect_cursor = (
|
||||
self._truncated_from_cursor
|
||||
if self._truncated_from_cursor is not None
|
||||
else self._last_event_id
|
||||
)
|
||||
header_lines = [
|
||||
f"Host: {self._host}:{self._port}",
|
||||
f"Authorization: Bearer {self._token}",
|
||||
"Accept: text/event-stream",
|
||||
"Cache-Control: no-cache",
|
||||
]
|
||||
if native:
|
||||
path = self._events_path(None)
|
||||
if connect_cursor is not None:
|
||||
header_lines.append(f"Last-Event-ID: {connect_cursor}")
|
||||
else:
|
||||
path = self._events_path(connect_cursor)
|
||||
request = f"GET {path} HTTP/1.1\r\n" + "\r\n".join(header_lines) + "\r\n\r\n"
|
||||
|
||||
sock = socket.create_connection((self._host, self._port), timeout=10)
|
||||
if rcvbuf is not None:
|
||||
sock.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf)
|
||||
sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
|
||||
sock.settimeout(None)
|
||||
sock.sendall(request.encode())
|
||||
self._sock = sock
|
||||
|
||||
self._stop.clear()
|
||||
self._read_gate.set()
|
||||
self._status = None
|
||||
self._headers_done.clear()
|
||||
conn_index = len(self._conn_frames)
|
||||
frames: list[SSEFrame] = []
|
||||
self._conn_frames.append(frames)
|
||||
self._reader = threading.Thread(
|
||||
target=self._read_loop,
|
||||
args=(sock, conn_index, frames),
|
||||
name=f"sse-reader-{self._ws_id[:6]}-{conn_index}",
|
||||
daemon=True,
|
||||
)
|
||||
self._reader.start()
|
||||
|
||||
# Surface a non-200 handshake to the caller (409 half-built UI,
|
||||
# 404 unknown ws, 401 auth) rather than silently reading nothing.
|
||||
if not self._headers_done.wait(timeout=10):
|
||||
self.disconnect()
|
||||
raise AssertionError("events connect: no HTTP response headers")
|
||||
if self._status != 200:
|
||||
status = self._status
|
||||
self.disconnect()
|
||||
raise AssertionError(f"events connect returned HTTP {status}")
|
||||
|
||||
def disconnect(self) -> None:
|
||||
"""Close the stream and join the reader (leak-guard clean)."""
|
||||
self._stop.set()
|
||||
self._read_gate.set() # release a stalled reader so it sees _stop
|
||||
sock = self._sock
|
||||
if sock is not None:
|
||||
with contextlib.suppress(OSError):
|
||||
sock.shutdown(socket.SHUT_RDWR) # interrupt a blocked recv
|
||||
reader = self._reader
|
||||
if reader is not None:
|
||||
reader.join(timeout=15)
|
||||
if reader.is_alive():
|
||||
raise AssertionError("SSE reader thread failed to stop")
|
||||
if sock is not None:
|
||||
with contextlib.suppress(OSError):
|
||||
sock.close()
|
||||
self._sock = None
|
||||
self._reader = None
|
||||
|
||||
def close(self) -> None:
|
||||
"""Full teardown: disconnect any live stream + close the HTTP client."""
|
||||
if self._reader is not None:
|
||||
self.disconnect()
|
||||
self._http.close()
|
||||
|
||||
# -- the stall gate (backpressure driver) --------------------------------
|
||||
|
||||
def stall(self) -> None:
|
||||
"""Stop reading the socket. The kernel + uvicorn send buffers fill,
|
||||
blocking the server's SSE generator on its ``await send``, so it
|
||||
stops draining the per-UI listener queue — which poisons at its cap.
|
||||
"""
|
||||
self._read_gate.clear()
|
||||
|
||||
def resume(self) -> None:
|
||||
"""Resume reading. A poisoned-and-closed stream delivers its
|
||||
``stream_overflow`` farewell frame once the backlog drains."""
|
||||
self._read_gate.set()
|
||||
|
||||
# -- reader --------------------------------------------------------------
|
||||
|
||||
def _read_loop(self, sock: socket.socket, conn_index: int, frames: list[SSEFrame]) -> None:
|
||||
raw = b"" # undecoded bytes (headers, then chunked framing)
|
||||
sse = b"" # decoded SSE byte stream
|
||||
headers_parsed = False
|
||||
chunked = False
|
||||
while not self._stop.is_set():
|
||||
# Backpressure gate: while stalled we do NOT read the socket, so
|
||||
# its receive buffer fills and TCP flow control stalls the server.
|
||||
if not self._read_gate.wait(timeout=0.1):
|
||||
continue
|
||||
if self._stop.is_set():
|
||||
break
|
||||
try:
|
||||
chunk = sock.recv(65536)
|
||||
except OSError:
|
||||
break
|
||||
if not chunk:
|
||||
break # server closed
|
||||
raw += chunk
|
||||
if not headers_parsed:
|
||||
if b"\r\n\r\n" not in raw:
|
||||
continue
|
||||
header_blob, raw = raw.split(b"\r\n\r\n", 1)
|
||||
self._parse_headers(header_blob)
|
||||
chunked = b"transfer-encoding: chunked" in header_blob.lower()
|
||||
headers_parsed = True
|
||||
self._headers_done.set()
|
||||
if chunked:
|
||||
decoded, raw = _dechunk(raw)
|
||||
sse += decoded
|
||||
else:
|
||||
sse += raw
|
||||
raw = b""
|
||||
sse = sse.replace(b"\r\n", b"\n")
|
||||
while b"\n\n" in sse:
|
||||
block, sse = sse.split(b"\n\n", 1)
|
||||
self._handle_block(block.decode("utf-8", "replace"), conn_index, frames)
|
||||
|
||||
def _parse_headers(self, header_blob: bytes) -> None:
|
||||
first_line = header_blob.split(b"\r\n", 1)[0].decode("latin-1")
|
||||
# "HTTP/1.1 200 OK"
|
||||
parts = first_line.split(" ", 2)
|
||||
if len(parts) >= 2 and parts[1].isdigit():
|
||||
self._status = int(parts[1])
|
||||
|
||||
def _handle_block(self, block_text: str, conn_index: int, frames: list[SSEFrame]) -> None:
|
||||
event_id: str | None = None
|
||||
data_parts: list[str] = []
|
||||
retry: str | None = None
|
||||
for line in block_text.split("\n"):
|
||||
if not line or line.startswith(":"):
|
||||
continue # blank or comment (ping)
|
||||
field_name, _, value = line.partition(":")
|
||||
if value.startswith(" "):
|
||||
value = value[1:] # SSE strips a single leading space
|
||||
if field_name == "id":
|
||||
event_id = value
|
||||
elif field_name == "data":
|
||||
data_parts.append(value)
|
||||
elif field_name == "retry":
|
||||
retry = value
|
||||
# EventSource semantics: an event carrying an ``id:`` sets the
|
||||
# last-event-id buffer; an event without one leaves it unchanged.
|
||||
if event_id is not None:
|
||||
self._last_event_id = event_id
|
||||
if not data_parts:
|
||||
if retry is not None:
|
||||
self._record(SSEFrame(conn_index, None, "retry", None, block_text), frames)
|
||||
return
|
||||
data_str = "\n".join(data_parts)
|
||||
payload: dict[str, Any] | None
|
||||
try:
|
||||
parsed = json.loads(data_str)
|
||||
payload = parsed if isinstance(parsed, dict) else None
|
||||
except ValueError:
|
||||
payload = None
|
||||
etype = payload.get("type") if payload is not None else None
|
||||
frame = SSEFrame(conn_index, event_id, etype, payload, data_str)
|
||||
self._record(frame, frames)
|
||||
# Mirror the pane: the FIRST replay_truncated for an unrepaired gap
|
||||
# records the truncation-time cursor (keep-oldest). Its consumer is
|
||||
# the reconnect chokepoint (see ``connect``).
|
||||
if etype == "replay_truncated" and self._truncated_from_cursor is None:
|
||||
self._truncated_from_cursor = self._last_event_id
|
||||
|
||||
def _record(self, frame: SSEFrame, frames: list[SSEFrame]) -> None:
|
||||
with self._frames_lock:
|
||||
frames.append(frame)
|
||||
self._all_frames.append(frame)
|
||||
|
||||
# -- /history + cursor flow ----------------------------------------------
|
||||
|
||||
def fetch_history(self) -> dict[str, Any]:
|
||||
"""GET /history and return the parsed JSON ({ws_id, messages, cursor})."""
|
||||
r = self._http.get(f"/v1/api/workstreams/{self._ws_id}/history", headers=self._auth)
|
||||
r.raise_for_status()
|
||||
result: dict[str, Any] = r.json()
|
||||
return result
|
||||
|
||||
def seed_from_history(self) -> dict[str, Any]:
|
||||
"""The seedCursor step: fetch /history, adopt a non-null resume
|
||||
cursor into ``_last_event_id``, and clear the truncation record on
|
||||
a successful render (replayHistory clears ``_truncatedFromCursor``).
|
||||
"""
|
||||
data = self.fetch_history()
|
||||
cursor = data.get("cursor")
|
||||
if cursor is not None:
|
||||
self._last_event_id = str(cursor)
|
||||
self._truncated_from_cursor = None # successful full render repairs the gap
|
||||
return data
|
||||
|
||||
def load_history_then_connect(
|
||||
self, *, fail_history: bool = False, native: bool = False
|
||||
) -> dict[str, Any] | None:
|
||||
"""Reproduce interactive.js ``_loadHistoryThenConnect``.
|
||||
|
||||
Disconnect first, drop the live cursor (``_last_event_id = None``)
|
||||
but KEEP ``_truncated_from_cursor`` armed, then fetch /history and
|
||||
reconnect. On success adopt the returned cursor and clear the
|
||||
truncation record; on a FAILED /history (``fail_history`` — the
|
||||
harness IS the client here, so a client-side simulated failure is
|
||||
faithful) leave the record armed so the reconnect re-presents the
|
||||
truncation-time cursor and re-draws ``replay_truncated``.
|
||||
|
||||
Returns the /history JSON, or ``None`` when the fetch failed.
|
||||
"""
|
||||
if self._reader is not None:
|
||||
self.disconnect()
|
||||
self._last_event_id = None
|
||||
data: dict[str, Any] | None
|
||||
if fail_history:
|
||||
data = None
|
||||
else:
|
||||
data = self.fetch_history()
|
||||
cursor = data.get("cursor")
|
||||
if cursor is not None:
|
||||
self._last_event_id = str(cursor)
|
||||
self._truncated_from_cursor = None
|
||||
self.connect(native=native)
|
||||
return data
|
||||
|
||||
# -- accessors + waits ---------------------------------------------------
|
||||
|
||||
@property
|
||||
def last_event_id(self) -> str | None:
|
||||
return self._last_event_id
|
||||
|
||||
@property
|
||||
def truncated_from_cursor(self) -> str | None:
|
||||
return self._truncated_from_cursor
|
||||
|
||||
def all_frames(self) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._all_frames)
|
||||
|
||||
def conn_frames(self, conn_index: int) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._conn_frames[conn_index])
|
||||
|
||||
def latest_conn_frames(self) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._conn_frames[-1]) if self._conn_frames else []
|
||||
|
||||
def num_connections(self) -> int:
|
||||
with self._frames_lock:
|
||||
return len(self._conn_frames)
|
||||
|
||||
def frames_of_type(self, etype: str) -> list[SSEFrame]:
|
||||
return [f for f in self.all_frames() if f.etype == etype]
|
||||
|
||||
def has_type(self, etype: str) -> bool:
|
||||
return any(f.etype == etype for f in self.all_frames())
|
||||
|
||||
def tool_output_by_call(self) -> dict[str, str]:
|
||||
"""Concatenate every ``tool_output_chunk`` payload per call_id, in
|
||||
arrival order — the reconstructed live stream for each call."""
|
||||
out: dict[str, str] = {}
|
||||
for f in self.all_frames():
|
||||
if f.etype == "tool_output_chunk" and f.payload is not None:
|
||||
cid = str(f.payload.get("call_id", ""))
|
||||
out[cid] = out.get(cid, "") + str(f.payload.get("chunk", ""))
|
||||
return out
|
||||
|
||||
def tool_results_by_call(self) -> dict[str, str]:
|
||||
"""The last ``tool_result`` output seen per call_id."""
|
||||
out: dict[str, str] = {}
|
||||
for f in self.all_frames():
|
||||
if f.etype == "tool_result" and f.payload is not None:
|
||||
out[str(f.payload.get("call_id", ""))] = str(f.payload.get("output", ""))
|
||||
return out
|
||||
|
||||
def wait_for_type(self, etype: str, *, timeout: float = 45.0) -> SSEFrame:
|
||||
"""Block until a frame of ``etype`` has arrived on ANY connection."""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
for f in self.all_frames():
|
||||
if f.etype == etype:
|
||||
return f
|
||||
time.sleep(0.05)
|
||||
raise AssertionError(f"timed out waiting for a {etype!r} frame")
|
||||
|
||||
def wait_for(
|
||||
self, predicate: Callable[[BrowserlikeSSEClient], bool], *, timeout: float = 45.0
|
||||
) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if predicate(self):
|
||||
return
|
||||
time.sleep(0.05)
|
||||
raise AssertionError("timed out waiting for predicate")
|
||||
|
||||
def wait_for_call_result(self, call_id: str, *, timeout: float = 45.0) -> None:
|
||||
self.wait_for(lambda c: call_id in c.tool_results_by_call(), timeout=timeout)
|
||||
|
||||
|
||||
def _dechunk(buf: bytes) -> tuple[bytes, bytes]:
|
||||
"""Incrementally decode HTTP/1.1 chunked transfer-encoding.
|
||||
|
||||
Consumes as many COMPLETE chunks from ``buf`` as possible and returns
|
||||
``(decoded_bytes, remainder)`` where ``remainder`` is the trailing
|
||||
partial chunk to carry into the next read. A zero-length chunk (stream
|
||||
end) simply stops consumption; the reader's ``recv`` EOF handles close.
|
||||
"""
|
||||
decoded = b""
|
||||
while True:
|
||||
if b"\r\n" not in buf:
|
||||
break # incomplete size line
|
||||
size_line, rest = buf.split(b"\r\n", 1)
|
||||
try:
|
||||
n = int(size_line.strip() or b"z", 16)
|
||||
except ValueError:
|
||||
break # malformed / partial — wait for more bytes
|
||||
if n == 0:
|
||||
break # last chunk marker
|
||||
if len(rest) < n + 2: # need n data bytes + trailing CRLF
|
||||
break
|
||||
decoded += rest[:n]
|
||||
buf = rest[n + 2 :]
|
||||
return decoded, buf
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Assertion helpers (shared by the scenarios).
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def assert_contiguous_ids(frames: list[SSEFrame]) -> None:
|
||||
"""Every id-bearing frame in a connection forms a gap-free, dup-free,
|
||||
strictly increasing run.
|
||||
|
||||
Holds for a connection that took no ``_seq``-filtered fresh path — a
|
||||
fresh connect made before any event (snap_seq == 0) and every
|
||||
``replay_ok`` reconnect (snap_seq == 0). The server stamps a fresh
|
||||
monotonic id per enqueue with no in-ring coalescing, so a
|
||||
non-filtered consumer sees consecutive ids.
|
||||
"""
|
||||
ids = [f.event_id_int for f in frames if f.event_id_int is not None]
|
||||
assert ids, "connection carried no id-bearing frames"
|
||||
assert len(set(ids)) == len(ids), f"duplicate SSE ids: {ids}"
|
||||
assert ids == sorted(ids), f"SSE ids not monotonic: {ids}"
|
||||
for prev, cur in zip(ids, ids[1:], strict=False):
|
||||
assert cur == prev + 1, f"gap in SSE ids between {prev} and {cur}: {ids}"
|
||||
|
||||
|
||||
def assert_ids_monotonic_no_dupes(frames: list[SSEFrame]) -> None:
|
||||
"""Weaker invariant that holds on EVERY connection (including
|
||||
``_seq``-filtered fresh/truncated paths, where gaps are legal): ids
|
||||
are strictly increasing with no duplicates."""
|
||||
ids = [f.event_id_int for f in frames if f.event_id_int is not None]
|
||||
assert len(set(ids)) == len(ids), f"duplicate SSE ids: {ids}"
|
||||
assert ids == sorted(ids), f"SSE ids not monotonic: {ids}"
|
||||
|
||||
|
||||
def assert_chunk_result_ordering(frames: list[SSEFrame]) -> None:
|
||||
"""Every ``tool_output_chunk`` for a call precedes that call's own
|
||||
``tool_result`` on the wire (the load-bearing ordering — the client
|
||||
removes the streaming <pre> when it renders the result)."""
|
||||
result_index: dict[str, int] = {}
|
||||
for i, f in enumerate(frames):
|
||||
if f.etype == "tool_result" and f.payload is not None:
|
||||
result_index[str(f.payload.get("call_id", ""))] = i
|
||||
for i, f in enumerate(frames):
|
||||
if f.etype == "tool_output_chunk" and f.payload is not None:
|
||||
cid = str(f.payload.get("call_id", ""))
|
||||
assert cid in result_index, f"chunk for call {cid} has no tool_result"
|
||||
assert i < result_index[cid], (
|
||||
f"chunk for call {cid} arrived AFTER its tool_result "
|
||||
f"(chunk idx {i} >= result idx {result_index[cid]})"
|
||||
)
|
||||
|
||||
|
||||
def assert_children_stamped(frames: list[SSEFrame], parent_call_id: str) -> None:
|
||||
"""Every sub-agent child tool event carries ``parent_call_id`` (stamped
|
||||
at the flush chokepoint). Sub-tool call_ids are minted
|
||||
``{parent}::r{run}s{step}::{provider_id}`` — the ``::`` segment is the
|
||||
identifying mark — and NONE may escape unstamped to the top level."""
|
||||
unstamped: list[tuple[str | None, str, Any]] = []
|
||||
stamped = 0
|
||||
for f in frames:
|
||||
if f.payload is None:
|
||||
continue
|
||||
items = f.payload.get("items")
|
||||
entries = items if isinstance(items, list) else [f.payload]
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
cid = str(entry.get("call_id", ""))
|
||||
if "::" not in cid:
|
||||
continue
|
||||
if entry.get("parent_call_id") == parent_call_id:
|
||||
stamped += 1
|
||||
else:
|
||||
unstamped.append((f.etype, cid, entry.get("parent_call_id")))
|
||||
assert stamped > 0, f"no child events found for parent {parent_call_id}"
|
||||
assert not unstamped, f"child events escaped unstamped (parent {parent_call_id}): {unstamped}"
|
||||
|
||||
|
||||
def history_tool_outputs(history_json: dict[str, Any]) -> dict[str, str]:
|
||||
"""Extract {call_id: output} from a /history projection, however the
|
||||
projection surfaces results (a folded ``output`` on a tool_call, or a
|
||||
trailing ``role: tool`` row keyed by ``tool_call_id``)."""
|
||||
out: dict[str, str] = {}
|
||||
for msg in history_json.get("messages", []):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
if msg.get("role") == "tool":
|
||||
cid = msg.get("tool_call_id") or msg.get("call_id")
|
||||
if cid is not None:
|
||||
out[str(cid)] = str(msg.get("content", ""))
|
||||
for tc in msg.get("tool_calls") or ():
|
||||
if not isinstance(tc, dict):
|
||||
continue
|
||||
cid = tc.get("id") or tc.get("call_id")
|
||||
if cid is not None and tc.get("output") is not None:
|
||||
out[str(cid)] = str(tc.get("output", ""))
|
||||
return out
|
||||
|
||||
|
||||
def assert_converged(client: BrowserlikeSSEClient, history_json: dict[str, Any]) -> None:
|
||||
"""Turn-level equivalence: every tool result the client assembled live
|
||||
is present, with the same output, in a fresh /history projection.
|
||||
|
||||
Compares by call_id so a reconnect that re-delivered a result can't
|
||||
hide a divergence, and asserts the /history side isn't empty (a
|
||||
silently-lost turn would leave the projection short)."""
|
||||
live = client.tool_results_by_call()
|
||||
hist = history_tool_outputs(history_json)
|
||||
assert hist, "fresh /history projected no tool results — a turn was lost"
|
||||
for call_id, output in live.items():
|
||||
assert call_id in hist, f"call {call_id} seen live but absent from /history: {sorted(hist)}"
|
||||
assert hist[call_id] == output, (
|
||||
f"call {call_id} output diverged: live={output!r} history={hist[call_id]!r}"
|
||||
)
|
||||
@@ -1,628 +0,0 @@
|
||||
"""Boot the REAL interactive Turnstone server for the SSE recovery e2e
|
||||
harness: real ``SessionManager`` + real ``ChatSession`` engine driven
|
||||
through a scripted chat-completions client at the SDK boundary, executing
|
||||
REAL bash tools, exposed over a real uvicorn socket.
|
||||
|
||||
The recipe (verified end-to-end) has four load-bearing pieces:
|
||||
|
||||
1. **Provider injection seam.** ``create_app`` takes a PRE-BUILT
|
||||
``SessionManager``, so the harness owns the ``session_factory``: it
|
||||
passes ``client=fake_client`` and OMITS the registry, so
|
||||
``ChatSession`` falls back to ``create_provider("openai-compatible")``
|
||||
== ``OpenAIChatCompletionsProvider`` — exactly what
|
||||
``tests._session_helpers.scripted_chat_client`` targets. No production
|
||||
monkeypatch of the engine.
|
||||
|
||||
2. **Auto-title suppression.** The first user message spawns a background
|
||||
``_generate_title`` LLM call that would consume the first scripted
|
||||
response (the tool call) and desync a positional script. Setting
|
||||
``session._title_generated = True`` before the first send disables it.
|
||||
|
||||
3. **Completion barrier.** ``/send`` returns immediately after spawning
|
||||
``ws.worker_thread``; joining that thread is the true "turn complete,
|
||||
every SSE event enqueued" barrier (``stream_end`` is per-LLM-call, not
|
||||
per-turn, so it is NOT a completion marker).
|
||||
|
||||
4. **Thread hygiene.** ``create_app``'s lifespan starts daemon fan-out
|
||||
threads (``_global_fanout_thread`` blocking on ``global_queue.get()``,
|
||||
``_aggregate_emitter_thread`` on a 10s loop). The harness used to
|
||||
swap them for no-ops (they had no shutdown and tripped conftest's
|
||||
leaked-thread guard), which kept the global lane dead here; #885 gave
|
||||
the lifespan a real shutdown (stop Event + a queue sentinel for the
|
||||
fanout, joined in the lifespan exit that ``stop()``'s
|
||||
``should_exit``/join drives), so the harness now runs them REAL — the
|
||||
``roster-restart`` scenario depends on a live global lane — and
|
||||
teardown stays clean with no ``allow_thread_leak``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import queue as _q
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import httpx
|
||||
import uvicorn
|
||||
|
||||
from tests._session_helpers import scripted_chat_client
|
||||
from turnstone.core.adapters.interactive_adapter import InteractiveAdapter
|
||||
from turnstone.core.auth import JWT_AUD_SERVER, create_jwt
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_manager import SessionManager
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.storage import get_storage
|
||||
from turnstone.core.workstream import WorkstreamKind
|
||||
from turnstone.prompts import ClientType
|
||||
from turnstone.server import WebUI, create_app
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import MutableMapping
|
||||
|
||||
from turnstone.core.workstream import Workstream
|
||||
|
||||
_JWT_SECRET = "sse-recovery-e2e-jwt-secret-minimum-32-chars!"
|
||||
# Small server send buffer so a stalled consumer's in-flight backlog before
|
||||
# the listener-queue poison stays bounded (paired with the client's small
|
||||
# SO_RCVBUF in _sse_recovery_helpers). Harmless for prompt readers.
|
||||
_DEFAULT_SNDBUF = 8192
|
||||
|
||||
|
||||
def _fake_client(scripts: tuple[Any, ...]) -> Any:
|
||||
"""An SDK-shaped fake whose ``chat.completions.create`` follows a
|
||||
positional script (each a :func:`fake_chat_stream` kwargs dict)."""
|
||||
create_fn = scripted_chat_client(*scripts)
|
||||
client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create_fn)))
|
||||
client.calls = create_fn.calls
|
||||
return client
|
||||
|
||||
|
||||
class RecoveryServer:
|
||||
"""A booted interactive node the recovery scenarios drive."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
sndbuf: int = _DEFAULT_SNDBUF,
|
||||
listener_cap: int | None = None,
|
||||
extra_routes: list[Any] | None = None,
|
||||
port: int = 0,
|
||||
sock: socket.socket | None = None,
|
||||
) -> None:
|
||||
self._global_queue: _q.Queue[dict[str, Any]] = _q.Queue(maxsize=100000)
|
||||
self._global_listeners: list[_q.Queue[dict[str, Any]]] = []
|
||||
self._global_listeners_lock = threading.Lock()
|
||||
# Per-ws scripted client, resolved at factory-call time.
|
||||
self._pending_client: Any = _fake_client((dict(content="ok", finish_reason="stop"),))
|
||||
self._clients: dict[str, Any] = {}
|
||||
|
||||
WebUI._global_queue = self._global_queue
|
||||
|
||||
def session_factory(
|
||||
ui: Any,
|
||||
model_alias: str | None = None,
|
||||
ws_id: str | None = None,
|
||||
*,
|
||||
skill: Any = None,
|
||||
client_type: str = "",
|
||||
kind: WorkstreamKind = WorkstreamKind.INTERACTIVE,
|
||||
parent_ws_id: str | None = None,
|
||||
project_id: str = "",
|
||||
**_extra: Any,
|
||||
) -> ChatSession:
|
||||
client = self._pending_client
|
||||
if ws_id is not None:
|
||||
self._clients[ws_id] = client
|
||||
return ChatSession(
|
||||
client=client,
|
||||
model="test-model",
|
||||
ui=ui,
|
||||
instructions=None,
|
||||
temperature=None,
|
||||
max_tokens=1024,
|
||||
tool_timeout=30,
|
||||
ws_id=ws_id,
|
||||
user_id="recovery-user",
|
||||
client_type=ClientType.WEB,
|
||||
kind=kind,
|
||||
# Don't truncate large tool outputs: the harness tests
|
||||
# recovery, not the tool-result truncation budget, and a
|
||||
# truncated /history would diverge from the full live event
|
||||
# and defeat the convergence assertions.
|
||||
tool_truncation=10_000_000,
|
||||
)
|
||||
|
||||
self._adapter = InteractiveAdapter(
|
||||
global_queue=self._global_queue,
|
||||
ui_factory=lambda ws: WebUI(
|
||||
ws_id=ws.id, user_id=ws.user_id, kind=ws.kind, parent_ws_id=ws.parent_ws_id
|
||||
),
|
||||
session_factory=session_factory,
|
||||
)
|
||||
self._manager = SessionManager(
|
||||
self._adapter, storage=get_storage(), max_active=32, node_id="recovery-node"
|
||||
)
|
||||
self._adapter.attach(self._manager)
|
||||
WebUI._workstream_mgr = self._manager
|
||||
|
||||
# delay_load knob state (see the method): wraps the storage
|
||||
# singleton's load_messages; restored in stop().
|
||||
self._load_delay_ms = 0
|
||||
self._load_calls = 0
|
||||
# load_messages runs on asyncio.to_thread WORKERS, and delay_load
|
||||
# exists precisely to overlap two of them — unlike the
|
||||
# single-writer HTTP counters, this one has genuine concurrent
|
||||
# writers, so the increment takes a lock (a lost update would
|
||||
# false-FAIL G7's load_delta === 2, or mask a third load).
|
||||
self._load_calls_lock = threading.Lock()
|
||||
_storage_obj = get_storage()
|
||||
self._orig_load_messages = _storage_obj.load_messages
|
||||
|
||||
def _delayed_load(*a: Any, **k: Any) -> Any:
|
||||
with self._load_calls_lock:
|
||||
self._load_calls += 1
|
||||
result = self._orig_load_messages(*a, **k)
|
||||
# Sleep AFTER the load: the held flight must hold the data it
|
||||
# actually read (its transaction point), so a flight parked
|
||||
# across a rewind genuinely carries PRE-rewind rows — a
|
||||
# pre-load sleep would read post-rewind storage and mask a
|
||||
# wrongly-joined flight as fresh truth.
|
||||
d = self._load_delay_ms
|
||||
if d > 0:
|
||||
time.sleep(d / 1000.0)
|
||||
return result
|
||||
|
||||
_storage_obj.load_messages = _delayed_load # type: ignore[method-assign]
|
||||
self._patched_storage = _storage_obj
|
||||
|
||||
# Optional small listener-queue cap. The cap is a default arg on the
|
||||
# registration methods with no config/env override, so lower it by
|
||||
# patching their ``__defaults__`` (restored on stop). fix-3's
|
||||
# de-amplification makes a real 500-cap overflow need a pathological
|
||||
# storm; a small cap exercises the identical _ListenerOverflow ->
|
||||
# stream_overflow -> reconnect-replay path within a bounded storm.
|
||||
self._orig_defaults: list[tuple[Any, tuple[Any, ...] | None]] = []
|
||||
if listener_cap is not None:
|
||||
for meth in (
|
||||
SessionUIBase._register_listener,
|
||||
SessionUIBase.register_listener_with_in_progress_snapshot,
|
||||
SessionUIBase.register_listener_with_replay,
|
||||
):
|
||||
self._orig_defaults.append((meth, meth.__defaults__))
|
||||
meth.__defaults__ = (listener_cap,)
|
||||
|
||||
self._app = create_app(
|
||||
workstreams=self._manager,
|
||||
global_queue=self._global_queue,
|
||||
global_listeners=self._global_listeners,
|
||||
global_listeners_lock=self._global_listeners_lock,
|
||||
skip_permissions=True,
|
||||
jwt_secret=_JWT_SECRET,
|
||||
node_id="recovery-node",
|
||||
# /history + tenant checks read app.state.auth_storage.
|
||||
auth_storage=get_storage(),
|
||||
)
|
||||
# Same-origin extras (Tier 2 serves its recovery page here so the real
|
||||
# Pane's cookie auth + EventSource work without cross-origin plumbing).
|
||||
if extra_routes:
|
||||
self._app.router.routes.extend(extra_routes)
|
||||
|
||||
# Pre-bind a listening socket with a small SO_SNDBUF (accepted conns
|
||||
# inherit it), then hand it to uvicorn. ``sock`` injection: the
|
||||
# gap-free restart scenarios (roster-restart-native) bind a
|
||||
# placeholder BEFORE stopping the prior node and hand it in here —
|
||||
# a failed EventSource reconnect attempt is TERMINAL per WHATWG
|
||||
# (fail-the-connection → CLOSED, no further retries), so the
|
||||
# native-retry leg must never observe a refused-window; the
|
||||
# placeholder's listen backlog completes the TCP handshake during
|
||||
# the boot and uvicorn drains it once serving.
|
||||
self._sock = sock if sock is not None else make_listen_socket(port, sndbuf=sndbuf)
|
||||
self._port = int(self._sock.getsockname()[1])
|
||||
|
||||
# -- fault injection (public knobs below) ----------------------------
|
||||
# In-process arming: the Tier-2 runner holds this RecoveryServer and
|
||||
# arms a knob, THEN drives the browser request that consumes it.
|
||||
# Single-writer by construction — the runner never arms a knob while
|
||||
# the loop thread is mid-consume — and CPython makes each int read /
|
||||
# write atomic, so these need no lock even though the uvicorn loop
|
||||
# thread increments/decrements them while the runner thread reads.
|
||||
self.history_requests = 0
|
||||
# /history responses the PRODUCTION route answered 200 (not the
|
||||
# fault layer's injected 500s). See _fault_app for why arrival
|
||||
# counting cannot substitute.
|
||||
self.history_ok = 0
|
||||
self.rewind_requests = 0
|
||||
# Per-ws SSE connection opens (``GET …/events`` — the EventSource the
|
||||
# pane's connectSSE builds). A TRANSPORT-FREE heal (the #890 idle-edge
|
||||
# staleness backstop, a quiesced REST refetch) must leave this FLAT; a
|
||||
# reload-based backstop would bump it once per reconnect (the round-5
|
||||
# storm). Same lock-free single-writer int discipline as above.
|
||||
self.events_requests = 0
|
||||
# Global-lane SSE connection opens (``GET …/events/global`` — the
|
||||
# roster stream app.js's connectGlobalSSE builds). The
|
||||
# roster-restart scenario (#881) asserts the post-restart manual
|
||||
# reconnect actually reached the reborn node's real endpoint.
|
||||
self.global_events_requests = 0
|
||||
self._history_fail_remaining = 0
|
||||
self._history_delay_ms = 0
|
||||
# A thin pure-ASGI fault layer wrapping the REAL app (the production
|
||||
# app itself is untouched): count + optionally delay/fail
|
||||
# ``GET …/history``, count ``POST …/rewind``, count each per-ws SSE
|
||||
# connection open (``GET …/events``), forward everything else (SSE
|
||||
# bodies, /send, lifespan, static) verbatim.
|
||||
production_app = self._app
|
||||
|
||||
async def _fault_app(scope: dict[str, Any], receive: Any, send: Any) -> None:
|
||||
if scope.get("type") == "http":
|
||||
path = scope.get("path", "")
|
||||
method = scope.get("method", "")
|
||||
if path.endswith("/history") and method == "GET":
|
||||
# ARRIVAL, never move. Scenarios that hold a request open
|
||||
# use this bump as the IN-FLIGHT edge (E6/E7/G1/G7 say so
|
||||
# at their poll sites); counting on forward instead would
|
||||
# delay it past the hold and silently stop those scenarios
|
||||
# testing anything.
|
||||
self.history_requests += 1
|
||||
if self._history_delay_ms > 0:
|
||||
await asyncio.sleep(self._history_delay_ms / 1000.0)
|
||||
if self._history_fail_remaining > 0:
|
||||
self._history_fail_remaining -= 1
|
||||
await send(
|
||||
{
|
||||
"type": "http.response.start",
|
||||
"status": 500,
|
||||
"headers": [(b"content-type", b"application/json")],
|
||||
}
|
||||
)
|
||||
await send({"type": "http.response.body", "body": b'{"error": "injected"}'})
|
||||
return
|
||||
|
||||
# Successful-RESPONSE counter, distinct from the arrival
|
||||
# bump above. A scenario asserting that a render was
|
||||
# DECLINED needs to know a good payload actually existed —
|
||||
# otherwise "the client refused to render" and "there was
|
||||
# nothing to render" produce identical observables (no
|
||||
# wipe, latch held). Arrival cannot prove that, and
|
||||
# neither can an injected-fail budget: a PRODUCTION-side
|
||||
# 500/404 would slip through both. Reading the real
|
||||
# status off the response start is the only honest signal.
|
||||
async def _counting_send(message: MutableMapping[str, Any]) -> None:
|
||||
if (
|
||||
message.get("type") == "http.response.start"
|
||||
and message.get("status") == 200
|
||||
):
|
||||
self.history_ok += 1
|
||||
await send(message)
|
||||
|
||||
await production_app(scope, receive, _counting_send)
|
||||
return
|
||||
elif path.endswith("/rewind") and method == "POST":
|
||||
self.rewind_requests += 1
|
||||
elif path.endswith("/events") and method == "GET":
|
||||
# Per-ws SSE connection open — count it (readable on
|
||||
# RecoveryServer) and forward the long-lived stream
|
||||
# verbatim below. Uniquely the per-ws stream: the global
|
||||
# lane is ``…/events/global`` (ends ``/global``), and the
|
||||
# route the pane's EventSource hits is
|
||||
# ``…/workstreams/{ws_id}/events`` (session_routes).
|
||||
self.events_requests += 1
|
||||
elif path.endswith("/events/global") and method == "GET":
|
||||
self.global_events_requests += 1
|
||||
await production_app(scope, receive, send)
|
||||
|
||||
# ``timeout_graceful_shutdown``: an SSE stream that is still open
|
||||
# at ``stop()`` would otherwise park uvicorn's graceful drain
|
||||
# indefinitely (the 20s thread-join just expires and the browser
|
||||
# stays attached to the zombie server — the roster-restart-native
|
||||
# scenario is the one caller that stops a node mid-stream). A
|
||||
# bounded drain force-closes the stream after 2s and the lifespan
|
||||
# shutdown (#885's daemon-thread teardown) still runs after it.
|
||||
self._server = uvicorn.Server(
|
||||
uvicorn.Config(
|
||||
_fault_app,
|
||||
log_level="warning",
|
||||
lifespan="on",
|
||||
timeout_graceful_shutdown=2,
|
||||
)
|
||||
)
|
||||
self._thread = threading.Thread(
|
||||
target=self._serve, name=f"uvicorn-recovery-{self._port}", daemon=True
|
||||
)
|
||||
self._thread.start()
|
||||
if not _tcp_ready(self._port, 10.0):
|
||||
self.stop()
|
||||
raise AssertionError("recovery server did not accept TCP")
|
||||
|
||||
self._token = create_jwt(
|
||||
user_id="recovery-user",
|
||||
scopes=frozenset({"read", "write", "approve", "service"}),
|
||||
source="recovery",
|
||||
secret=_JWT_SECRET,
|
||||
audience=JWT_AUD_SERVER,
|
||||
)
|
||||
self._http = httpx.Client(base_url=self.base_url, timeout=httpx.Timeout(30.0))
|
||||
|
||||
def _serve(self) -> None:
|
||||
loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(loop)
|
||||
try:
|
||||
loop.run_until_complete(self._server.serve(sockets=[self._sock]))
|
||||
finally:
|
||||
pending = asyncio.all_tasks(loop)
|
||||
for task in pending:
|
||||
task.cancel()
|
||||
if pending:
|
||||
with contextlib.suppress(Exception):
|
||||
loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
|
||||
loop.close()
|
||||
|
||||
# -- properties ----------------------------------------------------------
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
return f"http://127.0.0.1:{self._port}"
|
||||
|
||||
@property
|
||||
def token(self) -> str:
|
||||
return self._token
|
||||
|
||||
@property
|
||||
def manager(self) -> SessionManager:
|
||||
return self._manager
|
||||
|
||||
# -- workstream lifecycle ------------------------------------------------
|
||||
|
||||
def create_workstream(self, *scripts: Any, name: str = "recovery-ws") -> str:
|
||||
"""Create a ws whose scripted LLM follows ``scripts`` (positional
|
||||
:func:`fake_chat_stream` kwargs). Auto-approves tools and suppresses
|
||||
the auto-title call so the positional script stays in sync."""
|
||||
self._pending_client = _fake_client(scripts)
|
||||
ws = self._manager.create(user_id="recovery-user", name=name)
|
||||
self._prime_ws(ws)
|
||||
return ws.id
|
||||
|
||||
def open_workstream(self, ws_id: str, *scripts: Any) -> None:
|
||||
"""Rehydrate a persisted ws on THIS node (the restart path). Fresh
|
||||
UI → empty ring + storage-seeded ``_event_id``."""
|
||||
if scripts:
|
||||
self._pending_client = _fake_client(scripts)
|
||||
ws = self._manager.open(ws_id)
|
||||
if ws is None:
|
||||
raise AssertionError(f"open_workstream: ws {ws_id} not resurrectable")
|
||||
self._prime_ws(ws)
|
||||
|
||||
def _prime_ws(self, ws: Workstream) -> None:
|
||||
if isinstance(ws.ui, SessionUIBase):
|
||||
ws.ui.auto_approve = True # blanket tool auto-approval
|
||||
if ws.session is not None:
|
||||
ws.session._title_generated = True # suppress the auto-title LLM call
|
||||
|
||||
def send(self, ws_id: str, message: str = "go") -> None:
|
||||
"""POST /send — spawns the worker thread and returns immediately."""
|
||||
r = self._http.post(
|
||||
f"/v1/api/workstreams/{ws_id}/send",
|
||||
headers={"Authorization": f"Bearer {self._token}"},
|
||||
json={"message": message},
|
||||
)
|
||||
r.raise_for_status()
|
||||
|
||||
def wait_turn(self, ws_id: str, *, timeout: float = 45.0) -> None:
|
||||
"""Block until the turn's worker thread finishes (the true
|
||||
turn-complete barrier) and the ws is idle."""
|
||||
deadline = time.monotonic() + timeout
|
||||
worker: threading.Thread | None = None
|
||||
while time.monotonic() < deadline:
|
||||
ws = self._manager.get(ws_id)
|
||||
worker = ws.worker_thread if ws is not None else None
|
||||
if worker is not None:
|
||||
break
|
||||
time.sleep(0.02)
|
||||
if worker is not None:
|
||||
worker.join(timeout=max(0.5, deadline - time.monotonic()))
|
||||
if worker.is_alive():
|
||||
raise AssertionError(f"turn worker for {ws_id} did not finish in {timeout}s")
|
||||
|
||||
def get_ws(self, ws_id: str) -> Workstream | None:
|
||||
return self._manager.get(ws_id)
|
||||
|
||||
def ws_state(self, ws_id: str) -> str:
|
||||
ws = self._manager.get(ws_id)
|
||||
return ws.state.value if ws is not None else ""
|
||||
|
||||
def ring_span(self, ws_id: str) -> tuple[int | None, int]:
|
||||
"""(earliest retained ring event_id or None, latest counter) — lets a
|
||||
scenario wait for the ring to evict a specific cursor."""
|
||||
ws = self._manager.get(ws_id)
|
||||
ui = ws.ui if ws is not None else None
|
||||
if not isinstance(ui, SessionUIBase):
|
||||
return None, 0
|
||||
buf = ui._event_buffer
|
||||
earliest = buf[0][0] if buf else None
|
||||
return earliest, ui._event_id
|
||||
|
||||
def listener_poisoned(self, ws_id: str) -> bool:
|
||||
"""True once any live SSE listener on the ws has poisoned (overflow)."""
|
||||
ws = self._manager.get(ws_id)
|
||||
ui = ws.ui if ws is not None else None
|
||||
if not isinstance(ui, SessionUIBase):
|
||||
return False
|
||||
return any(getattr(q, "poisoned", False) for q in list(ui._listeners))
|
||||
|
||||
def max_event_id(self, ws_id: str) -> int | None:
|
||||
"""The storage high-water ``MAX(conversations.event_id)`` — what a
|
||||
restarted node's fresh UI seeds ``_event_id`` from."""
|
||||
result: int | None = get_storage().get_max_event_id(ws_id)
|
||||
return result
|
||||
|
||||
def fetch_history(self, ws_id: str) -> dict[str, Any]:
|
||||
r = self._http.get(
|
||||
f"/v1/api/workstreams/{ws_id}/history",
|
||||
headers={"Authorization": f"Bearer {self._token}"},
|
||||
)
|
||||
r.raise_for_status()
|
||||
result: dict[str, Any] = r.json()
|
||||
return result
|
||||
|
||||
# -- fault-injection knobs -----------------------------------------------
|
||||
# Armed in-process by the Tier-2 runner (single writer at a time — see
|
||||
# __init__). A plain int is deliberate: CPython makes the loop thread's
|
||||
# increment/decrement and the runner thread's read each atomic, and the
|
||||
# arm-then-consume ordering means they never race.
|
||||
|
||||
def delay_load(self, ms: int) -> None:
|
||||
"""Hold ``storage.load_messages`` itself open for ``ms`` (0 = off).
|
||||
|
||||
``delay_history`` sleeps in the FAULT LAYER — before the route —
|
||||
so two delayed requests never overlap inside the #884 flight
|
||||
machinery (the first flight completes and pops before the second
|
||||
arrives at the route). This knob sleeps INSIDE the shared
|
||||
reconstruction's ``load_messages`` (sync, called via
|
||||
``asyncio.to_thread`` — the sleep parks only that worker), which
|
||||
is the same layer the unit tests gate, so held flights genuinely
|
||||
overlap and join/miss behavior is observable end to end via
|
||||
``load_calls``.
|
||||
"""
|
||||
self._load_delay_ms = ms
|
||||
|
||||
@property
|
||||
def load_calls(self) -> int:
|
||||
"""``load_messages`` entries (pre-sleep) — the flight-layer twin
|
||||
of ``history_requests`` (which counts HTTP arrivals): a JOINED
|
||||
request never enters ``load_messages``, so join=1 / miss=2."""
|
||||
return self._load_calls
|
||||
|
||||
def fail_history(self, count: int) -> None:
|
||||
"""Make the next ``count`` ``GET …/history`` responses a 500 — the
|
||||
failed refetch the #890 guard-before-wipe must survive."""
|
||||
self._history_fail_remaining = count
|
||||
|
||||
def delay_history(self, ms: int) -> None:
|
||||
"""Hold each ``GET …/history`` ``ms`` ms before forwarding (0
|
||||
clears). Opens the clear_ui-refetch quiesce window that the row
|
||||
affordance gate (``busy || _historyStale``) must close."""
|
||||
self._history_delay_ms = ms
|
||||
|
||||
@property
|
||||
def history_fail_remaining(self) -> int:
|
||||
"""Unconsumed forced-failure budget — 0 proves the armed failure
|
||||
actually fired (assert backend state, never scripted absence)."""
|
||||
return self._history_fail_remaining
|
||||
|
||||
# -- teardown ------------------------------------------------------------
|
||||
|
||||
def stop(self, *, hard: bool = False) -> None:
|
||||
"""Stop the node.
|
||||
|
||||
``hard=True`` skips the per-workstream ``manager.close`` sweep — a
|
||||
graceful close routes through ``cleanup_session_ui`` →
|
||||
``session.cancel()``, whose bash cancel path PERSISTS a
|
||||
synthesized "Cancelled by user" result while the old node is
|
||||
still alive, which masks crash states. A hard stop leaves any
|
||||
in-flight tool call genuinely unresulted in storage, modelling a
|
||||
SIGKILL/OOM death (the coord-orphan-rewind scenario's premise).
|
||||
The 2s graceful-shutdown timeout (uvicorn config) force-closes
|
||||
open SSE streams, and the lifespan teardown still runs, so the
|
||||
#885 daemon threads are joined on both paths.
|
||||
"""
|
||||
if not hard:
|
||||
with contextlib.suppress(Exception):
|
||||
for ws in list(self._manager.list_all()):
|
||||
with contextlib.suppress(Exception):
|
||||
self._manager.close(ws.id)
|
||||
# hard=True relies on ``timeout_graceful_shutdown=2`` (set in the
|
||||
# uvicorn config above) to force-close the pane's EventSource:
|
||||
# should_exit alone still runs the ASGI lifespan teardown, so the
|
||||
# #885 daemon threads and the sse_executor are joined either way
|
||||
# (``force_exit`` would SKIP the lifespan and leak them — the
|
||||
# fanout thread blocks on queue.get() forever). NOTE: the killed
|
||||
# workstream's in-flight tool keeps executing on this process's
|
||||
# session thread and persists its result at natural completion —
|
||||
# hard-kill scenarios must use a paced tool that outlives their
|
||||
# observation window.
|
||||
self._server.should_exit = True
|
||||
self._thread.join(timeout=20)
|
||||
with contextlib.suppress(Exception):
|
||||
self._http.close()
|
||||
with contextlib.suppress(OSError):
|
||||
self._sock.close()
|
||||
# Restore any patched cap defaults.
|
||||
for meth, defaults in self._orig_defaults:
|
||||
meth.__defaults__ = defaults
|
||||
# Restore the storage singleton's load_messages (delay_load knob).
|
||||
with contextlib.suppress(Exception):
|
||||
self._patched_storage.load_messages = self._orig_load_messages # type: ignore[method-assign]
|
||||
|
||||
|
||||
def make_listen_socket(port: int, *, sndbuf: int = _DEFAULT_SNDBUF) -> socket.socket:
|
||||
"""Bound + listening socket the way :class:`RecoveryServer` binds its own.
|
||||
|
||||
``SO_REUSEPORT`` on every listener (same process, same uid) is what
|
||||
lets a restart scenario bind the successor's socket while the prior
|
||||
node still holds the port — the seam behind the gap-free handoff
|
||||
documented at the ``sock`` parameter. ``SO_SNDBUF`` matches the
|
||||
server's small send buffer so accepted connections inherit identical
|
||||
backpressure behavior regardless of which side bound the socket.
|
||||
"""
|
||||
s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEPORT, 1)
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, sndbuf)
|
||||
s.bind(("127.0.0.1", port)) # port=0 -> ephemeral; fixed -> restart reuse
|
||||
s.listen(128)
|
||||
return s
|
||||
|
||||
|
||||
def _tcp_ready(port: int, timeout: float) -> bool:
|
||||
end = time.monotonic() + timeout
|
||||
while time.monotonic() < end:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
def bash_toolcall_script(
|
||||
call_id: str, command: str, *, finish_reason: str = "tool_calls"
|
||||
) -> dict[str, Any]:
|
||||
"""A scripted assistant turn issuing ONE bash tool call."""
|
||||
return dict(
|
||||
tool_calls=[{"id": call_id, "name": "bash", "arguments": json.dumps({"command": command})}],
|
||||
finish_reason=finish_reason,
|
||||
)
|
||||
|
||||
|
||||
def parallel_bash_script(commands: dict[str, str]) -> dict[str, Any]:
|
||||
"""A scripted assistant turn issuing SEVERAL bash tool calls at once
|
||||
(the parallel-pool storm), ``{call_id: command}``.
|
||||
|
||||
Each command is prefixed with a no-op ``: <call_id>;`` so the tool
|
||||
ARGUMENTS are distinct per call while the OUTPUT is unchanged (``:``
|
||||
ignores its args and prints nothing). Identical-argument parallel
|
||||
calls otherwise trip the session's repeat-tool-call guard, which
|
||||
appends a warning to the PERSISTED result only (not the live event) —
|
||||
an orthogonal divergence that would mask the recovery behavior the
|
||||
convergence assertions test.
|
||||
"""
|
||||
return dict(
|
||||
tool_calls=[
|
||||
{
|
||||
"id": cid,
|
||||
"name": "bash",
|
||||
"arguments": json.dumps({"command": f": {cid}; {cmd}"}),
|
||||
}
|
||||
for cid, cmd in commands.items()
|
||||
],
|
||||
finish_reason="tool_calls",
|
||||
)
|
||||
|
||||
|
||||
def final_text_script(content: str = "done") -> dict[str, Any]:
|
||||
"""The scripted assistant turn that ends the agent loop (no tools)."""
|
||||
return dict(content=content, finish_reason="stop")
|
||||
@@ -4,9 +4,6 @@ import asyncio
|
||||
import contextlib
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
@@ -219,98 +216,6 @@ def _seed_static_state(mgr: MCPClientManager, name: str, **overrides: Any) -> St
|
||||
return state
|
||||
|
||||
|
||||
def _run_on_loop(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 10) -> Any:
|
||||
"""Submit *coro* to *loop*, wait for the result.
|
||||
|
||||
The ONE copy shared by the MCP test files — four hand-synced copies
|
||||
had already drifted on the timeout (5s hardcoded vs a 10s default).
|
||||
The timeout is an upper bound on waiting, not a behavior assertion,
|
||||
so the most generous variant won the merge.
|
||||
"""
|
||||
fut = asyncio.run_coroutine_threadsafe(coro, loop)
|
||||
return fut.result(timeout=timeout)
|
||||
|
||||
|
||||
def _drain_background(mgr: MCPClientManager, loop: asyncio.AbstractEventLoop) -> None:
|
||||
"""Deterministically await ``mgr``'s tracked background tasks.
|
||||
|
||||
Replaces fixed sleeps for synchronizing with scheduled dead-grant
|
||||
drops / spawned refreshes: exact, and immune to slow-runner flake.
|
||||
"""
|
||||
|
||||
async def _drain() -> None:
|
||||
tasks = [t for t in list(mgr._background_tasks) if not t.done()]
|
||||
if tasks:
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
|
||||
_run_on_loop(loop, _drain())
|
||||
|
||||
|
||||
def _poll_until(predicate: Callable[[], bool], timeout: float, interval: float = 0.05) -> bool:
|
||||
"""Poll *predicate* until true or *timeout* elapses — the ONE wait loop.
|
||||
|
||||
Shared by the live MCP smoke tests' condition helpers so the
|
||||
deadline/poll pattern doesn't accrete per-file hand-synced copies.
|
||||
"""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if predicate():
|
||||
return True
|
||||
time.sleep(interval)
|
||||
return False
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
"""Grab an ephemeral localhost port for a live-server subprocess.
|
||||
|
||||
Shared by the live MCP smoke tests (flaky-server, push-refresh) so
|
||||
the socket-probe helpers stay in one place instead of drifting per
|
||||
file.
|
||||
"""
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return int(s.getsockname()[1])
|
||||
|
||||
|
||||
def _tcp_accepts(port: int) -> bool:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def _wait_tcp_ready(port: int, timeout: float) -> bool:
|
||||
"""Poll until something accepts TCP on 127.0.0.1:*port* (live tests)."""
|
||||
return _poll_until(lambda: _tcp_accepts(port), timeout)
|
||||
|
||||
|
||||
def _wait_session_live(mgr: MCPClientManager, name: str, timeout: float) -> bool:
|
||||
"""Poll until static server *name* has a live session (live tests)."""
|
||||
|
||||
def _live() -> bool:
|
||||
state = mgr._static_servers.get(name)
|
||||
return state is not None and state.session is not None
|
||||
|
||||
return _poll_until(_live, timeout)
|
||||
|
||||
|
||||
def _popen_mcp_server(script_path: Any, port: int) -> subprocess.Popen[bytes]:
|
||||
"""Start a FastMCP live-server subprocess, streams to DEVNULL.
|
||||
|
||||
The shared spawn primitive for the live MCP smoke tests
|
||||
(flaky-server flap loop, push-refresh) — the readiness wait and the
|
||||
skip-vs-raise-on-failure policy legitimately differ per test and
|
||||
stay at the call sites. ``sys.executable`` runs the same interpreter,
|
||||
so a server-side import gap surfaces as a failed TCP wait, not here.
|
||||
"""
|
||||
return subprocess.Popen(
|
||||
[sys.executable, str(script_path), str(port)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
|
||||
|
||||
def make_oidc_test_config(**overrides: Any) -> OIDCConfig:
|
||||
"""Build a test ``OIDCConfig`` with sensible defaults.
|
||||
|
||||
@@ -430,40 +335,6 @@ def mock_openai_client():
|
||||
return client
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def make_config_store():
|
||||
"""Factory for a lightweight ConfigStore double.
|
||||
|
||||
``make_config_store(**overrides)`` returns an object whose ``.get(key)``
|
||||
yields the override when present, else the registered SettingDef default —
|
||||
mirroring the real :meth:`ConfigStore.get` fail-open (a bool setting reads
|
||||
as its ``False`` default on a miss, never ``None``). Shared by the
|
||||
``server.require_project`` gate / advisory tests.
|
||||
"""
|
||||
|
||||
_unset = object()
|
||||
|
||||
def _make(**overrides: Any) -> Any:
|
||||
from turnstone.core.settings_registry import SETTINGS
|
||||
|
||||
class _ConfigStoreDouble:
|
||||
def get(self, key: str, default: Any = _unset) -> Any:
|
||||
# Mirror ConfigStore.get precedence exactly: cache (overrides)
|
||||
# first, then a caller-supplied default, then the registry
|
||||
# default, then None — so a reused caller passing an explicit
|
||||
# default for an unset key gets the same value production would.
|
||||
if key in overrides:
|
||||
return overrides[key]
|
||||
if default is not _unset:
|
||||
return default
|
||||
defn = SETTINGS.get(key)
|
||||
return defn.default if defn else None
|
||||
|
||||
return _ConfigStoreDouble()
|
||||
|
||||
return _make
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_policy_cache():
|
||||
"""Drop the in-process tool-policy cache between tests.
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "",
|
||||
"provider_content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": "{\"city\": \"Oslo\"}",
|
||||
"name": "get_weather"
|
||||
},
|
||||
"id": "synth-id-0",
|
||||
"type": "function"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scenario": "blank_id_tools",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Before tools",
|
||||
"provider_content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": "{\"city\": \"Nice\"}",
|
||||
"name": "get_weather"
|
||||
},
|
||||
"id": "call_1",
|
||||
"type": "function"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scenario": "combined_content_tools_finish",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Before tools"
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,35 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Redac",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "content_filter",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Redac"
|
||||
],
|
||||
[
|
||||
"error",
|
||||
"Warning: response blocked by content filter."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Hello world.",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "content_only",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Hello world."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "finish_only_no_content",
|
||||
"ui_events": [
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Answer with sources.",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "info_postfinish_footer",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Answer w"
|
||||
],
|
||||
[
|
||||
"info",
|
||||
"Sources:\n- example.com/page"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"ith sources."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Seals are pinnipeds.",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "info_prefinish",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"info",
|
||||
"[Searching: pinniped taxonomy]"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Seals ar"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"e pinnipeds."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,43 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Partial answer",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "length_with_tools",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Pa"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"rtial answer"
|
||||
],
|
||||
[
|
||||
"error",
|
||||
"Warning: response truncated (hit 4096 token limit). Use --max-tokens to increase, or /compact to free context."
|
||||
],
|
||||
[
|
||||
"error",
|
||||
"Discarding partial tool calls from truncated response."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 0,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 11
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Half an ans",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "no_finish_clean_exhaust",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Half an ans"
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,36 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Blocked.",
|
||||
"provider_content": [
|
||||
{
|
||||
"text": "captured",
|
||||
"type": "reasoning_text"
|
||||
}
|
||||
],
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "provider_blocks_on_terminal",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Blocked."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,44 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Answer.",
|
||||
"provider_content": [
|
||||
{
|
||||
"text": "think a think b",
|
||||
"type": "reasoning_text"
|
||||
}
|
||||
],
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "reasoning_then_content",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"reasoning",
|
||||
"think a"
|
||||
],
|
||||
[
|
||||
"reasoning",
|
||||
" think b"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Answer."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "\n\nAnswer",
|
||||
"provider_content": [
|
||||
{
|
||||
"text": "plan",
|
||||
"type": "reasoning_text"
|
||||
}
|
||||
],
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "think_tags_split_across_chunks",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"reasoning",
|
||||
"plan"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"\n\nAnswer"
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Calling.",
|
||||
"provider_content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": "{\"city\": \"Paris\"}",
|
||||
"name": "get_weather"
|
||||
},
|
||||
"id": "call_1",
|
||||
"type": "function"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scenario": "tools_simple",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Calling."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -57,6 +57,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -28,5 +28,6 @@
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b"
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -49,6 +49,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -42,6 +42,7 @@
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -28,5 +28,6 @@
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b"
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -40,6 +40,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -51,6 +51,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,6 +43,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -35,6 +35,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -51,6 +51,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,6 +43,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -39,6 +39,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,10 +43,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -18,8 +18,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -34,10 +34,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -15,8 +15,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -30,10 +30,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -43,10 +43,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user