mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-14 07:52:25 -06:00
Compare commits
62 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e99d3ee139 | |||
| 4f0fc3f219 | |||
| dc701986f7 | |||
| bedd25fbe7 | |||
| 251a912275 | |||
| d48902fd01 | |||
| 702ac43d0e | |||
| 01f83dc90f | |||
| 2463c480c2 | |||
| 2a3dfbc6fb | |||
| 6c3b3cc098 | |||
| 0dc52f05ee | |||
| 02929c0d00 | |||
| b2add19c56 | |||
| 5ce1873e9e | |||
| 7698a928c5 | |||
| 4e2eea2f86 | |||
| a6752cb645 | |||
| 06ba4e8d4f | |||
| 94dcaf34fd | |||
| 1a2a689033 | |||
| c02f960d0a | |||
| bfde387206 | |||
| 27d112ff60 | |||
| aeab2535b1 | |||
| 4638d22bd0 | |||
| ee3bd1dcf2 | |||
| ae3a83ccce | |||
| 0f17433e1f | |||
| 043554bb2f | |||
| 8389808add | |||
| 6cbef4f633 | |||
| 2b6dde4f7e | |||
| fbe31b9885 | |||
| ef13f40cf5 | |||
| bfa1b104cf | |||
| 324a1d1a35 | |||
| 95ab88ff6f | |||
| d29840f985 | |||
| 44c0b9c340 | |||
| cdbdf3dc2b | |||
| 8aabb061c2 | |||
| 8bd638569f | |||
| 251dc44a46 | |||
| efd0a1d000 | |||
| 2f93c39fd3 | |||
| f27ce104c6 | |||
| 20a61b692b | |||
| 1966107efe | |||
| d5b2fe6e45 | |||
| 1569819750 | |||
| 4107a30148 | |||
| d06d88b83f | |||
| c411aac939 | |||
| 104715b650 | |||
| 59a9899149 | |||
| 4da7c3b91c | |||
| 012f4e3e16 | |||
| 3636724848 | |||
| eeda5ac312 | |||
| be872b840f | |||
| ee94ae8ba1 |
+25
-27
@@ -14,8 +14,8 @@ jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install pre-commit
|
||||
@@ -25,8 +25,8 @@ jobs:
|
||||
typecheck:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install mypy
|
||||
@@ -35,31 +35,29 @@ jobs:
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
# Cap a hung run at 30 min instead of riding GitHub's 6-hour default
|
||||
# (a flaky-hang run otherwise streams -v output for hours). Was 20;
|
||||
# the suite's growth (~9.7k tests, coverage-instrumented, 3-version
|
||||
# matrix) started brushing the old cap on healthy runs.
|
||||
timeout-minutes: 30
|
||||
# Cap a hung run at 20 min instead of riding GitHub's 6-hour default
|
||||
# (a flaky-hang run otherwise streams -v output for hours).
|
||||
timeout-minutes: 20
|
||||
strategy:
|
||||
matrix:
|
||||
python-version: ["3.11", "3.12", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
# Node is required by tests/test_renderer_js.py — without
|
||||
# explicit setup, that suite silently skips if the runner
|
||||
# image happens not to ship Node, masking regressions in
|
||||
# the browser-side renderer.
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: pip install -e ".[test]"
|
||||
# -v lists each test id as it starts (pytest prints the nodeid at
|
||||
# logstart), so a hang names the culprit on the last line instead of
|
||||
# riding the job timeout with only a trail of "..." dots.
|
||||
- run: pytest tests/ -m "not live and not e2e_recovery" --cov=turnstone --cov-report=term-missing --cov-report=xml -v
|
||||
- run: pytest tests/ -m "not live" --cov=turnstone --cov-report=term-missing --cov-report=xml -v
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
if: always()
|
||||
with:
|
||||
@@ -68,7 +66,7 @@ jobs:
|
||||
|
||||
test-postgres:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
timeout-minutes: 20
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:18
|
||||
@@ -84,23 +82,23 @@ jobs:
|
||||
--health-timeout=5s
|
||||
--health-retries=5
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: pip install -e ".[test]"
|
||||
- run: pytest tests/ -m "not live and not e2e_recovery" --storage-backend=postgresql -v
|
||||
- run: pytest tests/ -m "not live" --storage-backend=postgresql -v
|
||||
env:
|
||||
TURNSTONE_TEST_PG_URL: postgresql+psycopg://postgres:postgres@localhost:5432/turnstone_test
|
||||
|
||||
wheel-completeness:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: pip install build
|
||||
@@ -153,8 +151,8 @@ jobs:
|
||||
lock-check:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- run: uv lock --check
|
||||
@@ -162,11 +160,11 @@ jobs:
|
||||
security:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- run: uv sync --frozen --all-extras
|
||||
@@ -190,8 +188,8 @@ jobs:
|
||||
run:
|
||||
working-directory: sdk/typescript
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
|
||||
with:
|
||||
node-version: "24"
|
||||
- run: npm ci
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
name: Claude Code Review
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, synchronize, ready_for_review, reopened]
|
||||
# Optional: Only run on specific file changes
|
||||
# paths:
|
||||
# - "src/**/*.ts"
|
||||
# - "src/**/*.tsx"
|
||||
# - "src/**/*.js"
|
||||
# - "src/**/*.jsx"
|
||||
|
||||
jobs:
|
||||
claude-review:
|
||||
if: github.event.pull_request.head.repo.full_name == github.repository
|
||||
# Optional: Filter by PR author
|
||||
# if: |
|
||||
# github.event.pull_request.user.login == 'external-contributor' ||
|
||||
# github.event.pull_request.user.login == 'new-developer' ||
|
||||
# github.event.pull_request.author_association == 'FIRST_TIME_CONTRIBUTOR'
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post the review + inline comments
|
||||
issues: read
|
||||
id-token: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
allowed_bots: 'renovate[bot]' # let Renovate PRs get reviewed
|
||||
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
|
||||
plugins: 'code-review@claude-code-plugins'
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ github.event.pull_request.number }}'
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
name: Claude Code
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
pull_request_review_comment:
|
||||
types: [created]
|
||||
issues:
|
||||
types: [opened, assigned]
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
jobs:
|
||||
claude:
|
||||
if: |
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review' &&
|
||||
contains(github.event.review.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.review.author_association)
|
||||
) || (
|
||||
github.event_name == 'issues' &&
|
||||
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.issue.author_association)
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post comments/reviews when @-mentioned on a PR
|
||||
issues: write # post comments when @-mentioned on an issue
|
||||
id-token: write
|
||||
actions: read # Required for Claude to read CI results on PRs
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
|
||||
# This is an optional setting that allows Claude to read CI results on PRs
|
||||
additional_permissions: |
|
||||
actions: read
|
||||
|
||||
# Optional: Give a custom prompt to Claude. If this is not specified, Claude will perform the instructions specified in the comment that tagged it.
|
||||
# prompt: 'Update the pull request description to include a summary of changes.'
|
||||
|
||||
# Optional: Add claude_args to customize behavior and configuration
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
# claude_args: '--allowed-tools Bash(gh pr *)'
|
||||
|
||||
@@ -33,7 +33,7 @@ jobs:
|
||||
startsWith(github.event.workflow_run.head_branch, 'v')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
@@ -54,7 +54,7 @@ jobs:
|
||||
|
||||
- name: Log in to GHCR
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
|
||||
uses: docker/login-action@af1e73f918a031802d376d3c8bbc3fe56130a9b0 # v4
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
|
||||
@@ -30,7 +30,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
echo "skip=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
with:
|
||||
python-version: "3.14"
|
||||
@@ -58,12 +58,12 @@ jobs:
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
- run: python -m build
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
- uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
- uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # release/v1
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
|
||||
- name: Create GitHub Release
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3
|
||||
uses: softprops/action-gh-release@718ea10b132b3b2eba29c1007bb80653f286566b # v3
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.tag }}
|
||||
generate_release_notes: true
|
||||
|
||||
@@ -31,8 +31,8 @@ jobs:
|
||||
# Floor and ceiling of the example's requires-python (>=3.11).
|
||||
python-version: ["3.11", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- run: pip install -e ".[test,dev]"
|
||||
|
||||
@@ -61,7 +61,7 @@ jobs:
|
||||
fi
|
||||
echo "head_ref=${ref}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ steps.ref.outputs.head_ref }}
|
||||
|
||||
|
||||
@@ -29,4 +29,3 @@ tools/skill_audit_analysis/output/
|
||||
design_ideas/
|
||||
.claude/
|
||||
docs/design/
|
||||
/.idea
|
||||
|
||||
-564
@@ -14,570 +14,6 @@ experimental line:
|
||||
|
||||
Earlier stable lines (`stable/1.6`, `stable/1.5`) are frozen.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- **`server_parses_reasoning` model capability.** Declare it on a model
|
||||
definition whose backend segregates reasoning into its own channel (a
|
||||
vLLM launched with a reasoning parser, a commercial provider): the
|
||||
inline think-tag scan turns off on every lane — interactive and
|
||||
drained alike — so content is trusted verbatim and prose that merely
|
||||
quotes a tag can no longer be misrouted into the reasoning lane, and
|
||||
the utility lanes stop suppressing reasoning they'd otherwise pin off.
|
||||
Default off for local lanes, preserving the passthrough-server
|
||||
behavior; the built-in capability tables declare it for every real
|
||||
commercial endpoint (known models and table-miss defaults alike),
|
||||
which also removes the quoted-tag false positive from those lanes.
|
||||
- **Per-model Entra gateway authentication.** Model definitions can bind either
|
||||
a caller-delegated OBO token (`entra_obo`) or a shared app-identity token
|
||||
(`entra_app`) through the provider SDK credential surface. Mints reuse the
|
||||
encrypted cluster token cache, refresh-rotation CAS, and advisory locking;
|
||||
add a host-local memo, failure cooldown, long-lived mint HTTP client, audience
|
||||
allow-list/permission boundary, identity-unlink purge, and optional
|
||||
`model.auth_fail_closed` refusal policy. Delegated identity now propagates
|
||||
through judge, output-guard, and principal-scoped perception lanes, and
|
||||
unattended watch restoration reacquires the persisted workstream owner.
|
||||
Ownerless OBO calls and dynamic aliases without a real static fallback always
|
||||
fail closed; grant modes are never silently switched. Static authentication
|
||||
remains the default.
|
||||
|
||||
- **Compaction is visible now: lifecycle events, a progress bar, and a
|
||||
persistent transcript card.** Context compaction (manual `/compact` and
|
||||
auto) emits a first-class `compaction` SSE event
|
||||
(`start` / `progress` / `end` — see the API reference) instead of loose
|
||||
info lines. The web UI renders an in-transcript card with a real progress
|
||||
bar (determinate `part k of N` during chunked summarization, indeterminate
|
||||
for single-call compactions) that settles into a result card — token delta
|
||||
plus the summary behind a fold — in both the interactive pane and the
|
||||
coordinator viewer. The result survives reloads: the persisted compaction
|
||||
marker now projects through `/history` as a `role="system"`,
|
||||
`source="compaction"` entry (resume/export/search unchanged), stamped with
|
||||
the end event's id so repaint and SSE replay can't double-render. The
|
||||
marker's `meta` additionally records `before_tokens` / `after_tokens` /
|
||||
`trigger`. Python and TypeScript SDKs gain a typed `CompactionEvent`.
|
||||
|
||||
- **One provider transport: every model call now streams (#831).**
|
||||
The per-adapter non-streaming entry (`create_completion`) is retired;
|
||||
single-shot lanes — judges, titles, compaction, web-fetch extraction,
|
||||
perception, eval, optimizer — sample through the same streaming entry
|
||||
the chat loop uses and accumulate via one shared drain, so request
|
||||
shaping can no longer drift between the two consumption styles. Two
|
||||
operator-visible consequences: long single-shot generations (a thinking
|
||||
model composing a title, a slow local judge) no longer sit in a single
|
||||
blocking read that can hit client read-timeouts — the same reason the
|
||||
Anthropic adapter already streamed internally — and judge timeouts now
|
||||
*abort* the underlying HTTP read instead of abandoning a worker thread
|
||||
on a dead call. Because every call now streams, an alias pointed at a
|
||||
model or org that cannot stream (OpenAI's verified-org streaming
|
||||
entitlement, a gateway api-version predating `stream_options` — e.g.
|
||||
older Azure OpenAI deployments) fails at request time where 1.7's
|
||||
non-streaming single-shot call succeeded; remediation is on the
|
||||
serving side (verify the org, bump the api-version/gateway) — there is
|
||||
deliberately no per-model non-streaming fallback left to configure. These lanes are also complete-or-error now: a stream
|
||||
that ends without any finish signal is treated as a generation that
|
||||
died mid-response and retried, instead of storing the partial text as
|
||||
a clean result (previously a half-generated compaction summary could
|
||||
silently replace real history). Caveats: these lanes now carry the
|
||||
same `stream_options: {include_usage: true}` the chat loop always
|
||||
sent — OpenAI-compatible servers old enough to *ignore* it stop
|
||||
producing usage rows on these lanes, and servers strict enough to
|
||||
*reject* unknown fields (pre-2024 llama.cpp/proxy builds) will 400 —
|
||||
such a server already couldn't serve turnstone's chat loop, but a
|
||||
judge/utility alias pointed at one worked on 1.7 and needs to move to
|
||||
a current server. Transient mid-stream deaths (connection drop, proxy
|
||||
hiccup) are re-issued in place up to twice with exponential backoff —
|
||||
the retry the SDK's request loop used to provide these lanes
|
||||
invisibly. Each lane accepts its own terminal marker (Anthropic
|
||||
`message_stop`, Responses terminal events); a lax server/gateway that
|
||||
never sends any terminal signal needs
|
||||
`{"finish_reason_optional": true}` in the model definition's
|
||||
capabilities JSON, which restores 1.7's tolerance (clean end-of-stream
|
||||
after output = completion) for that model on every lane — without it
|
||||
such streams fail as died-mid-generation, because SSE gives no way to
|
||||
tell the two apart and the default favors catching truncation. The
|
||||
unread `supports_streaming` capability flag (and its admin tile) is
|
||||
gone; the o-series models it described are dropped from the capability
|
||||
table entirely (see Removed).
|
||||
|
||||
- **One turn interface for every model call: `core/model_turn.py` (#827).**
|
||||
Judges (intent + output guard), perception, title generation, compaction,
|
||||
web-fetch extraction, the eval harness, the optimizer's meta lanes, and
|
||||
task agents all advance a trajectory through the same plant-call
|
||||
primitive the agent seam pioneered — Turn IR in, one shared lowering
|
||||
(argument sanitize → minted-id restore → vLLM reasoning attach), one
|
||||
shared re-ingest (blank-id repair → native-lane finalize). The judges'
|
||||
hand-built OpenAI-dict path is gone, and with it the Gemini judge's
|
||||
tool-blindness: evidence tools now work on Google models because the
|
||||
native lane round-trips `thought_signature` (with pairwise repair for
|
||||
blank-id compat responses). Provider adapters still take lowered wire
|
||||
dicts — the transport collapse and main-loop migration are tracked as
|
||||
#831 / #832.
|
||||
|
||||
- **task_agent keeps its model's reasoning across its own tool loop — on
|
||||
every provider lane.** A task agent's replayed turns now carry the
|
||||
provider-native reasoning lane the model produced — Anthropic thinking
|
||||
blocks with their signatures (commercial or an anthropic-compatible
|
||||
server), OpenAI Responses reasoning items, Gemini `thought_signature`
|
||||
fidelity blocks, and the reasoning text a vLLM `--reasoning-parser` /
|
||||
llama.cpp `reasoning_format` surfaces on the Chat Completions lane —
|
||||
instead of each turn being rebuilt from text + tool calls with the
|
||||
reasoning dropped. On a thinking model this restores reasoning continuity
|
||||
across the agent's own multi-turn tool use. On the wire the agent's
|
||||
session-minted sub-tool ids are mapped back to the provider's own ids
|
||||
(`restore_provider_tool_ids`), so the native block — replayed verbatim,
|
||||
its signature never touched — the `tool_calls` mirror, and each tool
|
||||
result always agree; internally the minted ids still key the live card,
|
||||
recall, and the cancel ledger unchanged. Replay honors the same per-model
|
||||
`replay_reasoning_to_model` flag the main loop uses on every lane: the
|
||||
vLLM Chat-Completions field replay keeps its server-type gate, and
|
||||
llama.cpp stays capture-only, matching main-loop behavior. The native
|
||||
lane is finalized by the same shared builder as the main loop's, so the
|
||||
two harnesses cannot drift.
|
||||
|
||||
- **Background shells: `bash` gains `run_in_background`, plus `bash_output` /
|
||||
`kill_shell`.** Setting `run_in_background=true` starts the command as a
|
||||
detached shell and returns immediately with a `bash_N` handle — "start a dev
|
||||
server, use it in a later call" is back as an explicit opt-in (the shape
|
||||
follows the convention the major coding agents converged on). `bash_output`
|
||||
returns only output produced since the previous read (optionally filtered by
|
||||
a regex) plus status and exit code; `kill_shell` terminates the shell's
|
||||
whole process group. Output is buffered per shell with a drop-oldest cap, so
|
||||
a chatty server can't grow memory unbounded. When a background shell exits,
|
||||
a system notice lands at the next seam (waking an idle workstream if
|
||||
needed). Shells survive a generation cancel, die with the workstream, and
|
||||
never outlive a task_agent that started them; anything a background shell
|
||||
itself backgrounds is still reaped when that shell exits — the no-leak
|
||||
guarantee below is unchanged.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Log event rename: `drain_stream.post_finish_blip` is now
|
||||
`stream.post_finish_blip`; its `usage_captured` field is retained.** The
|
||||
single-shot drain normalizes mid-body transport deaths through the same
|
||||
`transport_guarded` wrapper the interactive loop uses, so its
|
||||
post-finish-blip tolerance logs under the wrapper's event name. Update
|
||||
any external log filters pinned to the old name; the drained result's
|
||||
possible `usage=None` on a post-finish blip is unchanged and documented
|
||||
on `drain_stream`.
|
||||
|
||||
- **Breaking (1.8): compaction feedback moved from `info` events to the
|
||||
typed `compaction` SSE event.** Pre-1.8 SSE/SDK clients that ignore
|
||||
unknown event types no longer see compaction lines (they are
|
||||
deliberately not dual-emitted — dual emission would double-render on
|
||||
every current client). Consume the `compaction` lifecycle event (see
|
||||
the API reference and the `CompactionEvent` SDK type); embedders
|
||||
driving `ChatSession` through a duck-typed `SessionUI` are unaffected
|
||||
(the classic `on_info` lines are restored for them — see Fixed).
|
||||
|
||||
- **Sampling knobs (temperature, reasoning effort) now ride one assignment
|
||||
scheme: per-model alias value → operator-stored global setting → the
|
||||
model definition's declared default (effort only) → field omitted.**
|
||||
Turnstone previously manufactured values onto every unconfigured
|
||||
request — a hidden `temperature: 0.5` and a `reasoning_effort: "medium"`
|
||||
baked in at three layers — overriding serving-side defaults like a vLLM
|
||||
model's `generation_config`. Unconfigured installs now send neither
|
||||
field and the inference engine's own defaults rule; `model.temperature`
|
||||
is blank by default ("inherit each model's own default") and
|
||||
`model.reasoning_effort` defaults to the empty "inherit" choice. The
|
||||
per-model → global resolution lives in one shared resolver used by the
|
||||
session factories, the `/model` switch, and every `model_turn` lane, so
|
||||
the same alias samples identically on every surface. CLI
|
||||
`--temperature` / `--reasoning-effort` likewise default to inherit.
|
||||
|
||||
**Upgrade notes:**
|
||||
- The empty (`""`) reasoning-effort choice changed meaning from
|
||||
"explicitly disable thinking" to "inherit the model/serving default".
|
||||
On local manual-thinking models (e.g. Qwen templates with
|
||||
`enable_thinking`), a stored `""` previously sent
|
||||
`enable_thinking: false`; it now sends nothing, so the template's own
|
||||
default (often thinking ON) applies. Use **`none`** to actually
|
||||
disable reasoning.
|
||||
- Workstreams saved by earlier versions carry the old defaults
|
||||
(`temperature=0.5`, `reasoning_effort=medium`) in their persisted
|
||||
config and keep that exact behavior on resume; they pick up the new
|
||||
inherit semantics the next time you change the model or a sampling
|
||||
knob in that workstream. New workstreams inherit from the start.
|
||||
|
||||
### Removed
|
||||
|
||||
- **O-series and pre-5.4 GPT-5 rows dropped from the OpenAI capability
|
||||
table.** `o1`, `o1-mini`, `o3`, `o3-mini`, `o3-pro`, `o4-mini`,
|
||||
`gpt-5`, `gpt-5-mini`, `gpt-5-nano`, `gpt-5-pro`, `gpt-5.1`,
|
||||
`gpt-5.1-codex-max`, `gpt-5.2`, `gpt-5.2-pro`, and `gpt-5.3` no longer
|
||||
have built-in capability rows — OpenAI has retired these model ids
|
||||
from the API, so the rows described contracts no request can reach
|
||||
anymore. The table floor is now `gpt-5.4`; the search-api and
|
||||
audio/STT/TTS rows are unchanged. An alias still pinning a retired id
|
||||
fails at OpenAI itself; any other unlisted commercial id resolves to
|
||||
the generic commercial defaults (temperature sent, no declared
|
||||
reasoning-effort vocabulary, 200K window) — declare the contract on
|
||||
the model definition's capabilities JSON if you run one, or move to a
|
||||
current model.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **A cancelled judge, guard, or compaction call can now stop before its
|
||||
request goes out (#972).** Previously it could not: `model_turn` refused
|
||||
to *re-issue* an abandoned call after a mid-stream death, but nothing
|
||||
checked before a first dispatch, so a call whose caller had already gone
|
||||
away still sent — and the reply was discarded unread after the endpoint
|
||||
had accepted the work. It now checks immediately before sending, so a
|
||||
Stop observed by that point costs no request, and again on entry, so a
|
||||
call already cancelled when it arrives also skips credential resolution.
|
||||
Cancellation is cooperative, which bounds what that buys: a Stop only
|
||||
saves the request if it lands before dispatch — sending is a moment, the
|
||||
response streaming back is the rest of the call, and an abort arriving
|
||||
then still meets a request in flight, closed in place exactly as before.
|
||||
The window that did widen usefully is a delegated-auth alias whose token
|
||||
mint blocks; a Stop during that mint now costs no request (though a mint
|
||||
already under way still completes). What a stopped call saves is the
|
||||
request, its prompt-side billing, and — on a capacity-bounded
|
||||
self-hosted endpoint — a slot a live request wanted. Unchanged: the
|
||||
interactive turn, which has its own pre-send cancellation check on a
|
||||
different path, and the lanes that thread no cancellation handle
|
||||
(attachment perception, title generation, web-fetch extraction,
|
||||
sub-agents, optimizer, eval) — and web-fetch extraction deliberately
|
||||
never will, since it runs on parallel tool threads where registering one
|
||||
would clobber the main stream's.
|
||||
- **Unmarked chain-of-thought no longer leaks into titles, summaries, or
|
||||
web-fetch tool results (#940).** Some serving setups emit reasoning
|
||||
inline with no tags and no `reasoning_content` at all — nothing any
|
||||
parser can segregate. The bounded-artifact lanes (title, compaction,
|
||||
web-fetch extraction) now ask the model for no reasoning instead:
|
||||
the model definition's declared thinking toggle is pinned off for that
|
||||
call — the same suppression transcription already used — and the
|
||||
reasoning-effort channels (the relayed session knob, the definition's
|
||||
default, the graded template key) are withheld with it, since an
|
||||
effort value beside a pinned-off toggle re-requests the reasoning the
|
||||
pin declined. A no-op on backends that segregate reasoning
|
||||
server-side. Title generation additionally stopped trusting line
|
||||
position: it takes the last line that reads as a title (within the
|
||||
word cap and ending in a word character, so explanation sentences,
|
||||
sign-offs, and reasoning headings lose in any script) rather than the
|
||||
first non-empty line, which unmarked reasoning turned into titles
|
||||
like "Thinking Process:".
|
||||
- **A think tag split across a reasoning delta now reassembles.** The
|
||||
non-streaming drain closes content runs at interleaving signals; a
|
||||
partial-tag tail is carried across reasoning-delta boundaries (a
|
||||
reasoning delta cannot terminate a tag) so the tag is consumed instead
|
||||
of its halves passing through as visible content. Tool-call boundaries
|
||||
still flush — no tag spans a tool call.
|
||||
- **Streaming consumers follow the ACTIVE model's capabilities.** The
|
||||
interactive tag-scan posture and the drain's scan gate now read the
|
||||
capabilities of the lane that owns the stream being consumed (fallback
|
||||
walks included) instead of the session's primary alias.
|
||||
- **Notification bodies no longer fuse multi-block answers.** `Turn.text`
|
||||
joins text blocks with a newline; a final assistant turn stored as
|
||||
multiple text blocks previously concatenated the last word of one
|
||||
block to the first word of the next in completion notifications and
|
||||
every other flattened read.
|
||||
- **String-typed boolean capability overrides coerce instead of
|
||||
truthiness-flipping.** A hand-edited `"false"`/`"0"` in a model
|
||||
definition's capabilities JSON now means false; unrecognized values
|
||||
drop the key and keep the field's default.
|
||||
- **Inline `<think>`/`<reasoning>` blocks no longer leak into drained
|
||||
results (#965, #940).** On servers without a reasoning parser
|
||||
(parserless vLLM/llama.cpp, LM Studio, bare gateways), reasoning
|
||||
arrives as literal tags inside content; segregation now happens once
|
||||
at the drain seam, so web-fetch tool results, sub-agent syntheses,
|
||||
judge verdicts, titles, summaries, and optimizer prompts receive
|
||||
tag-free content and the extracted reasoning rides the native lane.
|
||||
Two behavior notes: a web-fetch extraction whose whole response was
|
||||
reasoning now returns an explicit `Error: extraction returned no
|
||||
answer` tool result (previously the raw reasoning text persisted as a
|
||||
successful result and was replayed every following turn), and a
|
||||
mismatched-vocabulary close tag (`<think>…</reasoning>`) now closes
|
||||
the block — matching the interactive lane's long-standing rule —
|
||||
where the old per-lane strips treated it as unterminated.
|
||||
|
||||
- **A transport failure mid-generation no longer kills the interactive
|
||||
turn (#937).** A wire death during body streaming (TLS record failure,
|
||||
connection reset — `httpx.ReadError` and kin) surfaces after the
|
||||
request has already returned its stream handle, so neither the SDK's
|
||||
request retries nor the creation-time retry ladder ever saw it: the
|
||||
turn died with a bare `ReadError: …`, the partial output was
|
||||
discarded, and nothing was logged. The interactive loop now normalizes
|
||||
mid-body transport deaths exactly like the single-shot lanes and
|
||||
re-issues the turn (bounded, cancel-aware, exponential backoff),
|
||||
finalizing the dead attempt across every UI surface first so retried
|
||||
text never double-renders (web transcript, CLI markdown fences,
|
||||
Slack/Discord streamed messages). Before re-creating the stream the
|
||||
session re-resolves its registry binding, so a concurrent model-registry
|
||||
reload that closed the old client cannot turn the retry into a
|
||||
misleading closed-client error. On exhaustion the surfaced error names
|
||||
the provider, endpoint, and model with a stream-death message instead
|
||||
of a bare exception string, and every fatal turn now leaves a
|
||||
`session.fatal.recorded` log line (INFO for a user Ctrl-C, ERROR
|
||||
otherwise).
|
||||
|
||||
- **A failed worker-thread spawn no longer wedges the workstream — at
|
||||
either spawn site — and never masquerades as success.** If
|
||||
`Thread.start()` itself raised (thread exhaustion, out-of-memory), the
|
||||
dispatcher had already claimed the worker slot but the flag's only
|
||||
clearer lived in the never-started thread — the workstream looked idle
|
||||
forever while every subsequent message queued behind a worker that
|
||||
didn't exist, until an operator force-cancel. The claim is now rolled
|
||||
back under the lock and the error propagates, so the workstream is
|
||||
dispatchable again as soon as resources recover. Affected every
|
||||
dispatch path (sends, wakes, retries, deferred-send drain, init). The
|
||||
same failure at the deferred-send drain's own spawn rolls back the
|
||||
just-accepted entry and answers the retryable `queue_full` (previously
|
||||
a 500 landed *after* the entry was registered — an invisible,
|
||||
unretractable phantom that later dispatched as duplicate turns), and a
|
||||
`/command` whose worker never spawned now answers **503**
|
||||
`{"status": "error"}` instead of the generic 200 ok that told SDK
|
||||
callers their `/clear` or `/resume` had applied.
|
||||
|
||||
- **Manual `/compact` from the web UI: no phantom user turn, no frozen
|
||||
server, cancellable.** A slash command typed into the web composer no
|
||||
longer renders as a user chat bubble (it echoes as a distinct command
|
||||
chip — commands aren't conversation turns and were never persisted as
|
||||
such). `/compact` itself now dispatches onto the workstream's worker
|
||||
slot instead of running inline on the server's event loop — previously a
|
||||
long compaction froze every SSE stream on the node for its whole
|
||||
duration, which is also why its own progress only ever arrived as one
|
||||
burst after the fact. The manual path carries `send()`'s full generation
|
||||
discipline (`compact_now()`): a force-abandoned compaction goes stale
|
||||
instead of swapping history under a successor turn — and retires at its
|
||||
next checkpoint instead of running out its remaining summary calls,
|
||||
with its late lifecycle events fenced off (`compaction_id` on every
|
||||
event, `superseded` on end events — both in the SDKs) so they can't
|
||||
animate, tear down, re-title, or falsely narrate a successor's card or
|
||||
activity pill; a cancel aimed at it is consumed on exit (previously it
|
||||
bricked every `/compact` retry until the next message); a Stop click on
|
||||
an idle session can't pre-abort the next compaction; a Stop that lands
|
||||
in the completion tail — after the last cancel check, or during a retry
|
||||
backoff (which now aborts immediately instead of sleeping it out) — is
|
||||
honored rather than silently eaten; and Stop now aborts the in-flight
|
||||
summary HTTP call itself (the compaction lane registers its stream in
|
||||
the same abort seam the main loop uses), so cancelling a compaction is
|
||||
immediate instead of waiting out a model call.
|
||||
|
||||
- **Sends during a command window are deferred, ordered, bounded, and
|
||||
honestly rendered — never silently truncated or lost.** Messages sent
|
||||
while any slash command holds the worker slot are **deferred**: answered
|
||||
`{"status": "queued", "msg_id"}` immediately and dispatched as ordinary
|
||||
full-fidelity sends (attachments and sender identity included) when the
|
||||
command finishes — never routed through the mid-turn interjection
|
||||
queue, whose semantics are turn-shaped: previously a send during a
|
||||
manual `/compact` was silently truncated to 2,000 characters, a second
|
||||
participant in a shared workstream was locked out with a misleading
|
||||
"another participant's turn" 409 for the whole compaction, and a
|
||||
message queued across a `/resume`/`/new` could be answered into the
|
||||
post-swap workstream. Because the response is immediate,
|
||||
timeout-bounded callers — the coordinator's `send_message`, the console
|
||||
proxy, SDKs, anything behind a stock reverse proxy — can no longer lose
|
||||
a message to a multi-minute command window; the deferred send is
|
||||
retractable until dispatch via the same `DELETE .../send` used for
|
||||
queued interjections (node-local, in-memory — the API reference
|
||||
documents the at-most-once durability contract). Deferred responses
|
||||
carry `"deferred": true`; the pending list is the **order authority**
|
||||
(a fresh send — or a coordinator dispatch, or a queued-nudge wake —
|
||||
lines up behind acknowledged entries instead of overtaking them, with
|
||||
the two-term barrier defined once on the workstream so the wake gate
|
||||
also honors a claimed entry whose dispatch is mid-flight, and the gate
|
||||
re-arms at the drain's exit even when everything pending was
|
||||
retracted); acceptance is **bounded** (10 pending per workstream — the
|
||||
interjection queue's own backpressure contract; the 11th answers the
|
||||
retryable `queue_full` instead of pinning attachment bytes without
|
||||
limit and then running one unattended turn per entry); a dispatch
|
||||
crash re-queues the entry instead of eating an acknowledged message,
|
||||
and a drain thread that fails to *start* rolls the acceptance back and
|
||||
answers `queue_full` rather than parking a phantom the client can
|
||||
neither see nor retract; each dispatch emits a pane-tier
|
||||
`message_dispatched` event (`folded: true` for interjection fold-ins)
|
||||
so queued-bubble UI keeps its retract affordance exactly until the
|
||||
message truly leaves — including when the send was accepted by a pane
|
||||
that believed the workstream idle, which now renders a real queued
|
||||
chip instead of a sent-looking bubble, releases the composer (a
|
||||
deferred send has no running worker to wait on), and cleans up fully
|
||||
when the send is refused or the chip retracted instead of stranding
|
||||
the pane in Stop mode. Dismissing a queued bubble — interjection or
|
||||
deferred — is a server-confirmed `DELETE`, and retracting a deferred
|
||||
send that carried attachments tells the user they were discarded
|
||||
instead of silently expiring them.
|
||||
|
||||
- **Slash commands hold the worker slot with a loud contract.**
|
||||
A `/compact` raced against an in-flight turn is refused with an
|
||||
explicit busy response. Every other slash command runs through the same
|
||||
worker slot too — mutual exclusion against sends, a running compaction,
|
||||
and each other, with a busy answer replacing the old silent interleave —
|
||||
while the endpoint still awaits quick commands' completion off-loop
|
||||
(without parking an executor thread per request); the post-command pane
|
||||
refreshes (`clear_ui` after `/clear`/`/new`/`/resume`, the
|
||||
workstream-name sync) ride the worker itself, so a command that
|
||||
outlives the endpoint's 25s response backstop still refreshes every
|
||||
pane on completion (the backstop sits under the console proxy's 30s
|
||||
client timeout so the degraded `running` answer can actually traverse
|
||||
a proxied pane, which now surfaces it instead of silence; the
|
||||
`/command` response contract — `ok` / `running`, with busy refusals
|
||||
answering a loud HTTP 409 rather than a silent 200 — is now documented
|
||||
in the API reference and the OpenAPI spec).
|
||||
|
||||
- **Compaction status stays truthful across every UI surface.** Manual
|
||||
compaction
|
||||
success also refreshes the status line/context pill immediately (parity
|
||||
with auto-compaction), compaction failures keep feeding the typed
|
||||
`error` event and the node error counter (while a CLI Ctrl-C reports as
|
||||
cancelled, not a failure), one Stop prints one notice (a cancelled
|
||||
auto-compaction no longer stacks "Compaction cancelled." on top of
|
||||
send's own "[Generation cancelled]"), the workstream activity pill
|
||||
shows "Compacting context…" for the whole summarize phase, restores
|
||||
cleanly afterwards, and can no longer be stranded by a force-stopped
|
||||
compaction (a new turn's generation claim breaks a stale latch). Every
|
||||
retry backoff on the session (stream retries, task agents, notify
|
||||
delivery, compaction) now aborts immediately on Stop via one shared
|
||||
cancel-aware helper instead of sleeping out its exponential delay.
|
||||
|
||||
- **Compaction failures report exactly once, to the right owner.** A
|
||||
compaction failure reports
|
||||
exactly once (auto-compaction errors defer to the turn's fatal handler
|
||||
instead of doubling the red row and the error metric), failed-end
|
||||
notice suppression is computed once by the emitter (a `notice` bool on
|
||||
the end event — in the SDKs — replaces hand-synced client policy), and
|
||||
a manual `/compact` failure no longer crashes the CLI REPL. `/compact`
|
||||
on a workstream showing the `error` badge restores the badge on exit
|
||||
instead of stamping `idle` over it (the compaction neither retried nor
|
||||
resolved the failed turn). A force-cancelled initial send that
|
||||
completes late still delivers its scheduled-run completion
|
||||
notification (the only completion signal unattended workstreams have);
|
||||
the other post-command pane refreshes and error notices remain
|
||||
owner-guarded, so a force-cancelled wedged command that unwedges late
|
||||
can't wipe panes or inject stray notices into a successor turn.
|
||||
|
||||
- **Pre-1.8 embedder UIs keep their compaction lines.** Embedders
|
||||
driving `ChatSession` with a pre-1.8 duck-typed `SessionUI`
|
||||
(no `on_compaction` hook) get the classic `on_info` compaction lines
|
||||
back — threshold notice, `part k/N`, retry waits, token delta +
|
||||
summary box — instead of silent history swaps. (See the breaking
|
||||
event-contract note under **Changed** for SSE/SDK clients.)
|
||||
|
||||
- **Static MCP servers: a pushed catalog change no longer wedges the shared
|
||||
session (#839).** The static-path `*/list_changed` handler awaited its
|
||||
catalog refresh inline in the SDK's receive loop, but the refresh's own
|
||||
request can only be answered by that (now parked) loop — the refresh never
|
||||
completed, and every user's in-flight calls on the shared per-node session
|
||||
stalled behind it, unbounded, until the health loop's ping timeout tore the
|
||||
transport down (which was also the only way the changed catalog ever
|
||||
landed). Push refreshes now run as spawned tasks — debounced, coalesced per
|
||||
(server, kind), bounded by the connect timeout, and serialized on the
|
||||
per-server connect lock — and the manual and post-reconnect refreshes
|
||||
publish under that same lock, so a slower publisher can no longer land a
|
||||
staler catalog over a fresher one. Every teardown path now also clears the
|
||||
notification debounce stamp, so a reconnected server's first push refreshes
|
||||
immediately. Push-refresh debouncing is now per (server, kind) on BOTH the
|
||||
static and per-user pool paths — a tools push no longer swallows a prompts
|
||||
push arriving in the same 5-second window. A change genuinely lost to the
|
||||
debounce window (a same-kind push landing after the prior refresh finished,
|
||||
which the server will never re-announce) is recovered by an automatic
|
||||
health-tick retry rather than staying invisible until an unrelated push or
|
||||
a reconnect. The resource-refresh fan-out on both paths no longer orphans
|
||||
its sibling list call when one of the pair fails fast — the real error
|
||||
surfaces immediately (not masked as a 30-second timeout) and the surviving
|
||||
sibling is cancelled and reaped, under a bounded grace, inside the scope. A
|
||||
push refresh that fails while the connection stays up is likewise retried on
|
||||
the next health-loop tick until one completes — previously a single
|
||||
transient blip left the shared catalog stale for every user on the node
|
||||
until an operator intervened. An operator `/mcp refresh` no longer parks
|
||||
behind a busy per-server connect lock (a slow reconnect attempt could eat
|
||||
the whole 30-second refresh budget and fail the pass for every healthy
|
||||
server behind it) — the busy server is skipped on both the connected and
|
||||
disconnected branches, reported distinctly as "skipped" rather than as a
|
||||
false "no changes", the skip arms the automatic retry, and a
|
||||
force-reconnect drops the session up front so queued push refreshes can't
|
||||
starve it. Static-path resource and prompt catalogs are now size-capped
|
||||
like the pool path's (and like static tools) at discovery and on every
|
||||
refresh, so a misbehaving server's push can't balloon the node's merged
|
||||
catalogs. Deleting or reconfiguring a server can no longer leave it
|
||||
half-removed: the config removal and all cleanup are serialized under the
|
||||
connect lock (a cancelled removal completes its cleanup rather than
|
||||
stranding a live session and published catalog with the config already
|
||||
gone), and `reconcile_sync` retries a removal that timed out instead of
|
||||
marking it done — previously a DB-driven delete of a busy server could be a
|
||||
silent, permanent no-op until process restart. A refresh outcome now
|
||||
threads consistently to every operator surface off one source of truth
|
||||
(the per-server `last_refresh_outcome`): a busy-skip and a genuine failure
|
||||
are each reported distinctly from a real "no changes" — `/mcp refresh`
|
||||
prints "skipped" or "failed" rather than a false "no changes", and the
|
||||
node-internal refresh endpoint returns `202 skipped` instead of a
|
||||
misleading `200 ok` for a refresh that never ran. A single-kind push
|
||||
refresh no longer paints the whole server healthy: because the
|
||||
error/outcome state is server-scoped, a successful tools push while the
|
||||
prompts catalog is still broken (or vice versa) no longer clears the
|
||||
failure — only a full refresh pass declares "ok".
|
||||
|
||||
- **OpenAI Responses streaming: truncated and refused responses no longer
|
||||
vanish.** A response that hit `max_output_tokens` terminates the stream
|
||||
with `response.incomplete`, which the stream consumer did not handle —
|
||||
the turn was mislabeled `finish_reason: stop` and its final usage and
|
||||
collected output items were dropped. Refusal parts had no streaming
|
||||
handler at all, so a refusal rendered as empty content instead of the
|
||||
`[Refused: …]` text the non-streaming path produced. Both now match:
|
||||
truncation maps to `length` with usage/items intact, refusals render
|
||||
in content. Applies to the chat loop and every drained single-shot
|
||||
lane (#831).
|
||||
|
||||
- **task_agent: sub-tool ids no longer alias across a local model's reused
|
||||
ids.** A local model that reissues per-response sequential tool-call ids
|
||||
(`call_0` every turn) made two of a task agent's steps share one id — the
|
||||
live card collapsed both onto one DOM row while `/history` recall kept them
|
||||
apart, so the two views disagreed. Sub-tool ids are now minted
|
||||
`{parent}::r{run}s{step}::{id}`, unique within the session (across an
|
||||
agent's turns and across concurrent or sequential runs), and that one id
|
||||
keys the nesting registry, the live rows, recall, and the cancel ledger.
|
||||
On the wire the agent's self-built history carries the provider's own ids,
|
||||
restored from the mint map (see the reasoning-lane entry under Added), and
|
||||
malformed tool-call arguments are legalized the same way the main loop's
|
||||
wire prep does.
|
||||
|
||||
- **bash tool: never hang on a backgrounded child.** A command that left a
|
||||
long-lived process running (`server &`, a daemon) could wedge the whole
|
||||
workstream forever — the tool read stdout/stderr to EOF, which never arrived
|
||||
because the child inherited the pipe, and the timeout watchdog bailed once the
|
||||
foreground `bash` had exited. The tool now waits on the tracked process
|
||||
(bounded by the tool timeout) and terminates its whole process group on
|
||||
return, so the call always completes. Undecodable output is preserved
|
||||
(`errors="replace"`) instead of being dropped as a spurious error.
|
||||
- **Behavior change:** a process the command backgrounds no longer survives
|
||||
the call — nothing persists across bash invocations. (First-class
|
||||
"run this in the background" support landed separately — see
|
||||
`run_in_background` under Added.)
|
||||
|
||||
## [1.7.3]
|
||||
|
||||
A small feature and maintenance patch for the 1.7 line. No schema migrations
|
||||
and no new configuration knobs.
|
||||
|
||||
### Added
|
||||
|
||||
- **OpenAI GPT-5.6 (Sol/Terra/Luna) support** — the Responses provider
|
||||
understands the GPT-5.6 family: the `reasoning.mode` control, the new
|
||||
`max` effort tier, and `text.verbosity`, with golden wire payloads pinning
|
||||
the request shapes. The `openai` dependency floor moves to `>=2.44`.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Engineer base prompt hardened with process discipline** — the default
|
||||
base prompt for non-coordinator sessions now works in phases scaled to the
|
||||
size of the change, defaults to red-green for testable work, scopes to the
|
||||
smallest sufficient diff, stops to report after repeated failed attempts
|
||||
instead of thrashing, reports only observed results, and delegates
|
||||
exploration to `task_agent`. Persona prompts freeze into the workstream
|
||||
stamp at creation, so this reaches new workstreams only.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Unknown reasoning-mode warnings name the allowed modes** — a model
|
||||
definition with an unrecognized reasoning mode now logs the valid options
|
||||
instead of leaving the operator to guess.
|
||||
|
||||
### Documentation
|
||||
|
||||
- **HYPOTHESIS.md / PRIMER.md** — the control normal form is tightened and
|
||||
the factored Q_E reading is carried into the glossary; the plain-language
|
||||
PRIMER stays in sync.
|
||||
|
||||
## [1.7.2]
|
||||
|
||||
A feature-bearing patch for the 1.7 line. Rather than hold this work for the
|
||||
|
||||
@@ -8,11 +8,6 @@ The following people have contributed code to the project — thank you:
|
||||
- Burhan ([@Burhan-Q](https://github.com/Burhan-Q))
|
||||
- chrismuzyn ([@chrismuzyn](https://github.com/chrismuzyn))
|
||||
- daoxley ([@daoxley](https://github.com/daoxley))
|
||||
- metaclassing ([@metaclassing](https://github.com/metaclassing))
|
||||
- posixpositive ([@bensonjohnson](https://github.com/bensonjohnson))
|
||||
- Robert DeAngelis ([@OriginalOrangeXD](https://github.com/OriginalOrangeXD))
|
||||
- Sanjay Santhanam ([@Sanjays2402](https://github.com/Sanjays2402))
|
||||
- Stefano Maffeis ([@lesbass](https://github.com/lesbass))
|
||||
- William ([@sillyWillieBilly](https://github.com/sillyWillieBilly))
|
||||
- [@BlackMyrmidon](https://github.com/BlackMyrmidon)
|
||||
- [@pizzaandcheese](https://github.com/pizzaandcheese)
|
||||
|
||||
+3
-7
@@ -8,7 +8,7 @@ FROM python:3.14-slim
|
||||
LABEL org.opencontainers.image.title="turnstone" \
|
||||
org.opencontainers.image.description="Multi-node AI orchestration platform"
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.12.1 /uv /usr/local/bin/uv
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.27 /uv /usr/local/bin/uv
|
||||
|
||||
# Remove the slim image's man page exclusion so man-db has actual content
|
||||
RUN rm -f /etc/dpkg/dpkg.cfg.d/docker
|
||||
@@ -55,17 +55,13 @@ COPY docker/healthcheck.py /usr/local/bin/healthcheck.py
|
||||
|
||||
# Entrypoint script — runs migrations before starting
|
||||
COPY docker/entrypoint.sh /usr/local/bin/entrypoint.sh
|
||||
RUN chmod +x /usr/local/bin/entrypoint.sh
|
||||
|
||||
# Data directory — SQLite DB is created in CWD
|
||||
WORKDIR /data
|
||||
RUN chown turnstone:turnstone /data
|
||||
|
||||
# Workspace mount point — bind-mount a host directory here. The env var
|
||||
# surfaces the path in the model's shell/file tool descriptions
|
||||
# (config.get_workspace_dir); without it the mount is invisible to the
|
||||
# model, whose cwd is /data below.
|
||||
# Workspace mount point — bind-mount a host directory here
|
||||
RUN mkdir -p /workspace && chown turnstone:turnstone /workspace
|
||||
ENV TURNSTONE_WORKSPACE=/workspace
|
||||
|
||||
USER turnstone
|
||||
|
||||
|
||||
+21
-21
File diff suppressed because one or more lines are too long
@@ -16,7 +16,7 @@ One sentence to keep: **the model proposes; the gate disposes.** The model's out
|
||||
|
||||
| Plain name | What it does | In the formal doc |
|
||||
|---|---|---|
|
||||
| The owner | The human — or sign-off group — the run acts for; the only place new permissions can come from | the trusted principal |
|
||||
| The owner | The human or account the run acts for; the only party who can grant new permissions | the trusted principal |
|
||||
| The memory | Everything the run knows: task, plan, transcript, and the ledger of what has been done | the state, *s* |
|
||||
| The prompt builder | Decides which slice of memory the model gets to see this step | the lowering, π |
|
||||
| The model | The black box that reads the prompt and writes a proposal | the plant, M_W |
|
||||
@@ -67,7 +67,7 @@ Three consequences people miss:
|
||||
|
||||
**Validation must not act.** A "validator" that resolves a URL, expands a template that fires a webhook, or evaluates an argument has already acted — inside the check. The gate must be pure: it reads the proposal and the memory and outputs yes or no. If deciding requires touching the world, that touch is itself an action and goes through the gate.
|
||||
|
||||
**Anything irreversible is decided at the gate.** The verifier can reject a bad *result*; it cannot unsend the email. So the question "can we take this back, and until when?" is asked before execution — which means each tool declares, up front, how reversible its effects are, and the gate reads that declaration when it decides; the mark that comes back in the result record is confirmation for the books, not the gate's source — the gate needed the answer before the tool ever ran.
|
||||
**Anything irreversible is decided at the gate.** The verifier can reject a bad *result*; it cannot unsend the email. So the question "can we take this back, and until when?" is asked before execution — which means each tool's effect record has to carry a reversibility mark, or the gate can't ask it.
|
||||
|
||||
Two honest asterisks. First, the gate checks a snapshot: it approves against the world *as its memory describes it*, and the world can move between check and commit. For actions that race the world — spend against a balance, write against a row — the tool itself must bind check to commit (compare-and-swap), or you have a classic time-of-check/time-of-use hole. The gate decides; for those effects, the tool enforces. Second, a gate is only as binding as the authority behind the tools. A tool process holding standing credentials — a database connection with every grant, an environment full of long-lived secrets — doesn't need the model's proposal to act, and against it the gate's "no" is a decision with nothing enforcing it. **A gate in front of an omnipotent tool is a suggestion.** The fix is to make the approval *be* the key: each authorized action carries a short-lived credential scoped to exactly that action, that resource, that operation, so tools hold no standing power at all.
|
||||
|
||||
@@ -85,13 +85,13 @@ A measurement is a risk metric. A proof is a certificate. Keeping those two word
|
||||
|
||||
Formally, security here is a *reach-avoid* problem: reach a good stop, never touch the danger zone, **while an adversary picks the worst tool outputs your setup permits**. That last clause is the formal home of prompt injection: injection isn't "the model misbehaved," it's the environment optimized to bend your loop — poisoned pages, malicious tool descriptions, crafted responses.
|
||||
|
||||
Two different numbers fall out here, and dashboards love to collapse them: *success* (reached an accepted end before anything went wrong — a safe refusal counts against it) and *safety* (never touched the danger zone — a safe refusal is perfectly safe). Track both. They move independently. And both are scored by your own stop rule — they count what the shell *declared* a success. Whether a declared success was actually *right* is a third, harder number that no dashboard inside the system can produce; only a judge outside the run — a test suite, an audit, ground truth — can.
|
||||
Two different numbers fall out here, and dashboards love to collapse them: *success* (reached the right end before anything went wrong — a safe refusal counts against it) and *safety* (never touched the danger zone — a safe refusal is perfectly safe). Track both. They move independently.
|
||||
|
||||
The gate handles the visible half of injection: the model, freshly poisoned, proposes emailing your credentials somewhere, and the gate refuses — and injection or not, the action does not happen. But the deeper attack doesn't propose a bad action today. It rewrites *what the run believes its job is* — it edits the plan — and then every future action looks locally reasonable against a corrupted plan. So memory has to be partitioned: **data** (tool results, fetched pages, retrieved documents — content the world supplied) and **control** (the plan, the permissions, what is authorized next). The security claim is conditional on that partition holding: untrusted content lands in data, always. And "trust" is really two questions pointing opposite ways, which is worth keeping straight: *can this leak?* (a value is as secret as the most-secret thing that fed it — secrecy flows **upward**) and *can this boss us around?* (a value is as trustworthy as the least-trustworthy thing that fed it — authority flows **downward**). Untrusted content is safe as *data* precisely because the second question keeps it off the control side; a secret is kept out of the model by the first. Lowering either barrier on purpose — declassifying a secret, promoting data to trusted — is an explicit decision the owner makes, never a thing that happens by accident when two values are combined.
|
||||
The gate handles the visible half of injection: the model, freshly poisoned, proposes emailing your credentials somewhere, and the gate refuses — and injection or not, the action does not happen. But the deeper attack doesn't propose a bad action today. It rewrites *what the run believes its job is* — it edits the plan — and then every future action looks locally reasonable against a corrupted plan. So memory has to be partitioned: **data** (tool results, fetched pages, retrieved documents — content the world supplied) and **control** (the plan, the permissions, what is authorized next). The security claim is conditional on that partition holding: untrusted content lands in data, always.
|
||||
|
||||
Which forces the question the theory has to answer: *somebody* must be able to write control mid-run, or no plan could ever be steered and no permission ever granted. The answer is a small hierarchy with a top the model can't reach. The simplest top is one owner — but it needn't be a single person: a two-person sign-off, a quorum, several authenticated people each holding different scopes all work equally well, because the one property that matters is the same for all of them — the thing that can grant new power is a *human decision*, never a model:
|
||||
Which forces the question the theory has to answer: *somebody* must be able to write control mid-run, or no plan could ever be steered and no permission ever granted. The answer is a small hierarchy with exactly one party at the top:
|
||||
|
||||
- **The top alone widens.** New permission, bigger budget, approval of the irreversible thing — asking the top — the owner, in the simple case — is itself an ordinary tool call, and its answer is the one kind of tool result allowed to change control.
|
||||
- **The owner alone widens.** New permission, bigger budget, approval of the irreversible thing — asking the owner is itself an ordinary tool call, and the owner's answer is the one kind of tool result allowed to change control.
|
||||
- **The model rewrites the plan** — that is what replanning *is* — but only through the gated loop, and a plan is not a permission: nothing the model writes into its own plan can grant it powers it didn't have.
|
||||
- **Everything else is data.** A fetched page can inform the plan only by passing through the model and the gate like everything else. It can suggest. It cannot promote itself to boss.
|
||||
- **AI judges only tighten.** Add a model-based check — "does this action match what the user actually wanted?" — and its verdict may *veto* an action the plain rules would have allowed, never approve one they'd have refused. A judge that can approve is a tricked judge that can open the vault. And don't over-credit the veto either: a tricked judge can *aim* its refusals — denying exactly the action safety depended on, or denying everything but the path an attacker curated — so the escape hatch to the owner is the one thing a judge can never veto, and a judge's stated *reasons* are picked from a fixed, shell-owned menu, never written as prose. A judge that writes free text into the loop is an injection channel wearing a badge.
|
||||
@@ -102,9 +102,9 @@ One more rule closes the loop: transformations don't launder trust. A *summary*
|
||||
|
||||
The formal document's appendix works the operational cases in full; here they are at speed.
|
||||
|
||||
**The ledger, and the three-way distinction that keeps it honest.** Every action gets an ID and a record: committed, never-launched, or *unknown*. "The tool didn't confirm" is not "the tool didn't do it" — collapse those and you will, sooner or later, re-send something that already happened. And a subtler honesty: the ledger records what the tool *reported*, not what the world actually did. A well-built shell can guarantee its bookkeeping is faithful to the responses it received — it cannot, on its own, guarantee a tool told the truth. A tool that returns a clean "done!" for something it never did puts a clean "done!" in your ledger. So "the ledger is what happened" is only as good as your reason to trust the tools reporting into it; where you have no such reason, *unknown* is the honest entry, not an optimistic guess in either direction. The double-send bug has one reliable cure: **journal before dispatch.** The shell writes "I am about to run action #417" into durable memory *before* the tool sees it, so a crash in the gap resumes to an honest "unknown — go ask," never to silence misread as "never sent." Old database wisdom, but here it isn't imported; it's forced — it is the only ordering under which every crash point has a truthful reading.
|
||||
**The ledger, and the three-way distinction that keeps it honest.** Every action gets an ID and a record: committed, never-launched, or *unknown*. "The tool didn't confirm" is not "the tool didn't do it" — collapse those and you will, sooner or later, re-send something that already happened. The double-send bug has one reliable cure: **journal before dispatch.** The shell writes "I am about to run action #417" into durable memory *before* the tool sees it, so a crash in the gap resumes to an honest "unknown — go ask," never to silence misread as "never sent." Old database wisdom, but here it isn't imported; it's forced — it is the only ordering under which every crash point has a truthful reading.
|
||||
|
||||
**Crashes aren't finishes.** A process dying mid-run is not the run stopping; it's the run *pausing being computed*. Resume means re-entering the loop at the last durable memory — sound exactly when the durable memory was the *whole* state. Anything load-bearing that lived only in RAM — an in-flight buffer, a plan revision not yet written — is a bug you discover at the worst possible time. Recovery is where you find out whether your state was really your state. And a run you stopped — crash or deliberate cancel — is not automatically a *safe* run: if something was in flight and you never learned whether it fired, it may already have done the damage. "We stopped in time" is only true when everything in flight resolved to something safe; an outstanding *unknown* has to be treated as possibly-bad, the same optimism the ledger warns against, one level up.
|
||||
**Crashes aren't finishes.** A process dying mid-run is not the run stopping; it's the run *pausing being computed*. Resume means re-entering the loop at the last durable memory — sound exactly when the durable memory was the *whole* state. Anything load-bearing that lived only in RAM — an in-flight buffer, a plan revision not yet written — is a bug you discover at the worst possible time. Recovery is where you find out whether your state was really your state.
|
||||
|
||||
**Two innocent actions can be guilty together.** Models emit several tool calls per turn. "Read the secret" passes review. "Post to the web" passes review. The pair is an exfiltration channel — so the gate authorizes the *set*, atomically, with the interactions checked, not each element in isolation.
|
||||
|
||||
@@ -140,7 +140,7 @@ This is a hypothesis, and it says out loud what would kill it. The tests, in pla
|
||||
- **The red-team test.** Swap sampled tool outputs for worst-case ones: injected pages, poisoned metadata, malformed replies. The design must survive the worst permitted world, not the average one.
|
||||
- **Gates versus begging.** The theory predicts deterministic gating beats prompt-level pleading. If "please be careful" alone matches real gates on security outcomes, the controller-versus-model story is wrong.
|
||||
- **The compression hunt.** Exhibit a compact, provably sound progress certificate for a frontier-scale model on a nontrivial task family, and the central conjecture falls — constructively.
|
||||
- **The desk probe.** Take a task family with a *proven* memory floor — so "it needed the whole picture at once" is someone else's theorem, not our excuse — scale it past the window, and watch: the wall predicts a *ceiling*, not a cliff — past the boundary, a success rate that stays capped no matter how many retries you buy. A family solved reliably out there, without new shell tricks for splitting the work, kills the wall.
|
||||
- **The desk probe.** Take a task family with a *proven* memory floor — so "it needed the whole picture at once" is someone else's theorem, not our excuse — scale it past the window, and watch: the wall predicts collapse at the boundary, not graceful degradation.
|
||||
|
||||
## Who else landed here
|
||||
|
||||
@@ -148,7 +148,7 @@ The formal document keeps three honesty tiers. **Borrowed**: real theorems, cite
|
||||
|
||||
## What to remember
|
||||
|
||||
The model proposes; the gate disposes. No is the default, and a refusal must be safe. Only the top of the trust hierarchy widens permissions — a human decision, never the model, a tool result, a summary, or a judge. "Didn't confirm" is not "didn't happen." The desk is finite and the proof doesn't compress, so you measure — and you say *measurement* when you mean measurement. A robot that never stops leaks safety slowly, so it needs scheduled resets — and when it can't reach you, it must be able to stop. A loop that runs robots for you is just a bigger robot with the same rules and a further-away owner. And all of it is a hypothesis wearing its own kill-conditions on its sleeve.
|
||||
The model proposes; the gate disposes. No is the default, and a refusal must be safe. Exactly one party widens permissions — and it is not the model, a tool result, a summary, or a judge. "Didn't confirm" is not "didn't happen." The desk is finite and the proof doesn't compress, so you measure — and you say *measurement* when you mean measurement. A robot that never stops leaks safety slowly, so it needs scheduled resets — and when it can't reach you, it must be able to stop. A loop that runs robots for you is just a bigger robot with the same rules and a further-away owner. And all of it is a hypothesis wearing its own kill-conditions on its sleeve.
|
||||
|
||||
The formal version — the objects, the certificates, the falsifiers, the citations — is [HYPOTHESIS.md](HYPOTHESIS.md). It wins every disagreement with this file, including this sentence.
|
||||
|
||||
|
||||
@@ -17,17 +17,11 @@ Named after the [Ruddy Turnstone](https://en.wikipedia.org/wiki/Ruddy_turnstone)
|
||||
|
||||
**What is a harness?**
|
||||
|
||||
<p align="center">
|
||||
<a href="https://media.githubusercontent.com/media/turnstonelabs/turnstone/main/docs/diagrams/harness.png">
|
||||
<img src="https://media.githubusercontent.com/media/turnstonelabs/turnstone/main/docs/diagrams/harness.png" alt="ℋ : s_{n+1} ~ T(s_n) for n < τ_H — the whole controlled loop: π lowers state to context, M_W proposes a readout, γ authorizes it, Q_E acts on the world, ρ verifies and folds back" width="960"/>
|
||||
</a>
|
||||
</p>
|
||||
|
||||
```
|
||||
ℋ : s_{n+1} ~ T(s_n) for n < τ_H
|
||||
ℋ : s_{n+1} ~ T(s_n) for n < τ*, T = ρ ∘ (M_W ∘ π, E)
|
||||
```
|
||||
|
||||
[**the primer →**](PRIMER.md) · [**the formalism →**](HYPOTHESIS.md)
|
||||
[**the primer →**](PRIMER.md)
|
||||
|
||||
### Release Tracks
|
||||
|
||||
|
||||
@@ -2,11 +2,11 @@ apiVersion: v2
|
||||
name: turnstone
|
||||
description: Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation
|
||||
type: application
|
||||
version: 0.2.0
|
||||
version: 0.1.0
|
||||
appVersion: "0.3.0"
|
||||
|
||||
dependencies:
|
||||
- name: postgresql
|
||||
version: ~18.8.0
|
||||
version: ~18.7.0
|
||||
repository: https://charts.bitnami.com/bitnami
|
||||
condition: postgresql.enabled
|
||||
|
||||
@@ -110,153 +110,6 @@ Determine the PostgreSQL username.
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
The PostgreSQL password when the chart stores it itself, empty when it
|
||||
does not. Doubles as the predicate for "does <fullname>-secrets need to
|
||||
carry POSTGRES_PASSWORD", so an inline password is never written
|
||||
anywhere but <fullname>-secrets, and an operator-supplied Secret is
|
||||
never duplicated into it.
|
||||
|
||||
An operator-supplied existingSecret wins outright: writing the value
|
||||
into a second Secret nothing reads would only duplicate a credential.
|
||||
|
||||
Both branches need "default" because this is reached through include,
|
||||
which captures rendered text rather than a value: a key that is unset
|
||||
rather than empty — "password:" with nothing after it — renders as the
|
||||
literal "<no value>", and a ten-character string is truthy. Without the
|
||||
default that lands base64-encoded in POSTGRES_PASSWORD and the workloads
|
||||
authenticate with it.
|
||||
*/}}
|
||||
{{- define "turnstone.db.inlinePassword" -}}
|
||||
{{- if .Values.postgresql.enabled }}
|
||||
{{- .Values.postgresql.auth.password | default "" }}
|
||||
{{- else if not .Values.database.external.existingSecret }}
|
||||
{{- .Values.database.external.password | default "" }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
The name of the bundled subchart's own Secret.
|
||||
|
||||
Mirrors the subchart's naming rather than calling its helpers, which
|
||||
expect a context scoped to the subchart that this chart cannot hand
|
||||
them. Release-derived, so deliberately not turnstone.fullname: a
|
||||
fullnameOverride here renames this chart's resources and leaves the
|
||||
subchart's alone, and pointing at "<fullname>-postgresql" would then
|
||||
name a Secret that does not exist.
|
||||
|
||||
The subchart also normalises the release name through a regex before
|
||||
using it, which is a no-op for the DNS-1123 names Helm accepts, so it is
|
||||
not reproduced.
|
||||
*/}}
|
||||
{{- define "turnstone.postgresql.fullname" -}}
|
||||
{{- $global := ((.Values.global).postgresql).fullnameOverride }}
|
||||
{{- if $global }}
|
||||
{{- $global | trunc 63 | trimSuffix "-" }}
|
||||
{{- else if .Values.postgresql.fullnameOverride }}
|
||||
{{- .Values.postgresql.fullnameOverride | trunc 63 | trimSuffix "-" }}
|
||||
{{- else }}
|
||||
{{- $name := .Values.postgresql.nameOverride | default "postgresql" }}
|
||||
{{- if contains $name .Release.Name }}
|
||||
{{- .Release.Name | trunc 63 | trimSuffix "-" }}
|
||||
{{- else }}
|
||||
{{- printf "%s-%s" .Release.Name $name | trunc 63 | trimSuffix "-" }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{- define "turnstone.postgresql.secretName" -}}
|
||||
{{- $existing := coalesce (((.Values.global).postgresql).auth).existingSecret .Values.postgresql.auth.existingSecret }}
|
||||
{{- if $existing }}
|
||||
{{- tpl $existing . }}
|
||||
{{- else }}
|
||||
{{- include "turnstone.postgresql.fullname" . }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
The subchart stores the named user's password under "password" and the
|
||||
superuser's under "postgres-password", and lets an operator rename
|
||||
either through auth.secretKeys.
|
||||
*/}}
|
||||
{{- define "turnstone.postgresql.passwordKey" -}}
|
||||
{{- $user := .Values.postgresql.auth.username | default "" }}
|
||||
{{- $keys := .Values.postgresql.auth.secretKeys | default dict }}
|
||||
{{- if or (empty $user) (eq $user "postgres") }}
|
||||
{{- $keys.adminPasswordKey | default "postgres-password" }}
|
||||
{{- else }}
|
||||
{{- $keys.userPasswordKey | default "password" }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
Determine the secret holding the PostgreSQL password, and the key within
|
||||
it. Three sources, and the two helpers agree by construction because
|
||||
they branch identically:
|
||||
|
||||
- an external database pointed at a Secret the chart does not own (a
|
||||
CloudNativePG-generated secret, an External Secrets target, ...), in
|
||||
which case the key is rarely "POSTGRES_PASSWORD" — hence the
|
||||
companion existingSecretPasswordKey
|
||||
- the bundled subchart's own Secret, when it generates the password
|
||||
- <fullname>-secrets, when the password is supplied inline in values
|
||||
|
||||
Note the last is deliberately not turnstone.llm.secretName: that
|
||||
resolves to llm.existingSecret when the operator supplies one, which
|
||||
holds LLM API keys and has no reason to carry a database password.
|
||||
*/}}
|
||||
{{- define "turnstone.db.secretName" -}}
|
||||
{{- if not .Values.postgresql.enabled }}
|
||||
{{- if .Values.database.external.existingSecret }}
|
||||
{{- .Values.database.external.existingSecret }}
|
||||
{{- else }}
|
||||
{{- printf "%s-secrets" (include "turnstone.fullname" .) }}
|
||||
{{- end }}
|
||||
{{- else if include "turnstone.db.inlinePassword" . }}
|
||||
{{- printf "%s-secrets" (include "turnstone.fullname" .) }}
|
||||
{{- else }}
|
||||
{{- include "turnstone.postgresql.secretName" . }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{- define "turnstone.db.passwordKey" -}}
|
||||
{{- if not .Values.postgresql.enabled }}
|
||||
{{- if .Values.database.external.existingSecret }}
|
||||
{{- .Values.database.external.existingSecretPasswordKey | default "password" }}
|
||||
{{- else }}
|
||||
{{- printf "POSTGRES_PASSWORD" }}
|
||||
{{- end }}
|
||||
{{- else if include "turnstone.db.inlinePassword" . }}
|
||||
{{- printf "POSTGRES_PASSWORD" }}
|
||||
{{- else }}
|
||||
{{- include "turnstone.postgresql.passwordKey" . }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
Database environment shared by the server, console and migrate Job.
|
||||
|
||||
Every value except the password is rendered inline rather than pulled
|
||||
from the ConfigMap via envFrom, so that one definition serves all three
|
||||
workloads and the URL is assembled in exactly one place.
|
||||
|
||||
POSTGRES_PASSWORD must still precede TURNSTONE_DB_URL: the kubelet
|
||||
expands $(VAR) only against env entries declared earlier in the list, so
|
||||
a later definition would leave a literal "$(POSTGRES_PASSWORD)" in the
|
||||
URL.
|
||||
*/}}
|
||||
{{- define "turnstone.db.env" -}}
|
||||
- name: TURNSTONE_DB_BACKEND
|
||||
value: {{ .Values.database.backend | quote }}
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: {{ include "turnstone.db.secretName" . }}
|
||||
key: {{ include "turnstone.db.passwordKey" . }}
|
||||
- name: TURNSTONE_DB_URL
|
||||
value: "postgresql+psycopg://{{ include "turnstone.postgresql.username" . }}:$(POSTGRES_PASSWORD)@{{ include "turnstone.postgresql.host" . }}:{{ include "turnstone.postgresql.port" . }}/{{ include "turnstone.postgresql.database" . }}{{ if and (not .Values.postgresql.enabled) .Values.database.external.sslmode }}?sslmode={{ .Values.database.external.sslmode }}{{ end }}"
|
||||
{{- end }}
|
||||
|
||||
{{/*
|
||||
Determine the secret name for LLM API keys.
|
||||
*/}}
|
||||
|
||||
@@ -7,17 +7,6 @@ metadata:
|
||||
app.kubernetes.io/component: console
|
||||
spec:
|
||||
replicas: {{ .Values.console.replicas }}
|
||||
{{- if eq (int .Values.console.replicas) 1 }}
|
||||
# The console registers itself under the fixed service_id "console" and
|
||||
# deregisters on shutdown. Under RollingUpdate the outgoing pod's
|
||||
# deregister runs *after* the incoming pod registers and deletes its
|
||||
# row -- and the heartbeat only touches last_heartbeat, so the row is
|
||||
# never recreated and the console stays invisible in the registry until
|
||||
# the next clean start. Recreate orders shutdown strictly before
|
||||
# startup. Only valid at one replica; see console.replicas.
|
||||
strategy:
|
||||
type: Recreate
|
||||
{{- end }}
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- include "turnstone.selectorLabels" . | nindent 6 }}
|
||||
@@ -29,18 +18,6 @@ spec:
|
||||
app.kubernetes.io/component: console
|
||||
spec:
|
||||
serviceAccountName: {{ include "turnstone.serviceAccountName" . }}
|
||||
{{- with .Values.console.nodeSelector }}
|
||||
nodeSelector:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.console.affinity }}
|
||||
affinity:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.console.tolerations }}
|
||||
tolerations:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: console
|
||||
image: {{ include "turnstone.image" . }}
|
||||
@@ -59,18 +36,8 @@ spec:
|
||||
- secretRef:
|
||||
name: {{ include "turnstone.llm.secretName" . }}
|
||||
optional: true
|
||||
env:
|
||||
{{- include "turnstone.db.env" . | nindent 12 }}
|
||||
# Self-registration URL for the service registry. Unlike a
|
||||
# server node the console is one logical endpoint behind its
|
||||
# Service, so the Service DNS name is correct here. Without
|
||||
# it the console registers gethostname() (its pod name),
|
||||
# which no server node can resolve. Stops at ".svc" rather
|
||||
# than assuming a "cluster.local" DNS domain, which is
|
||||
# configurable per cluster.
|
||||
- name: TURNSTONE_CONSOLE_URL
|
||||
value: "http://{{ include "turnstone.fullname" . }}-console.{{ .Release.Namespace }}.svc:{{ .Values.console.service.port }}"
|
||||
{{- if or .Values.auth.existingSecret .Values.auth.jwtSecret }}
|
||||
env:
|
||||
- name: TURNSTONE_JWT_SECRET
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
|
||||
@@ -18,18 +18,6 @@ spec:
|
||||
app.kubernetes.io/component: server
|
||||
spec:
|
||||
serviceAccountName: {{ include "turnstone.serviceAccountName" . }}
|
||||
{{- with .Values.server.nodeSelector }}
|
||||
nodeSelector:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.server.affinity }}
|
||||
affinity:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.server.tolerations }}
|
||||
tolerations:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: server
|
||||
image: {{ include "turnstone.image" . }}
|
||||
@@ -51,20 +39,8 @@ spec:
|
||||
name: {{ include "turnstone.llm.secretName" . }}
|
||||
optional: true
|
||||
env:
|
||||
{{- include "turnstone.db.env" . | nindent 12 }}
|
||||
# Each replica is a distinct node in the rendezvous ring, so it
|
||||
# must advertise an address that reaches *itself*. The Service
|
||||
# DNS name would load-balance across every replica, sending
|
||||
# console traffic routed for node A to an arbitrary pod; the
|
||||
# default (gethostname(), i.e. the pod name) is not resolvable
|
||||
# at all. The pod IP is unique, routable in-cluster, and
|
||||
# re-registered on every start, so churn is self-healing.
|
||||
- name: POD_IP
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: status.podIP
|
||||
- name: TURNSTONE_ADVERTISE_URL
|
||||
value: "http://$(POD_IP):{{ .Values.server.service.port }}"
|
||||
- name: TURNSTONE_DB_URL
|
||||
value: "postgresql+psycopg://$(TURNSTONE_DB_USER):$(POSTGRES_PASSWORD)@$(TURNSTONE_DB_HOST):$(TURNSTONE_DB_PORT)/$(TURNSTONE_DB_NAME)"
|
||||
{{- if or .Values.auth.existingSecret .Values.auth.jwtSecret }}
|
||||
- name: TURNSTONE_JWT_SECRET
|
||||
valueFrom:
|
||||
|
||||
@@ -6,23 +6,11 @@ metadata:
|
||||
{{- include "turnstone.labels" . | nindent 4 }}
|
||||
app.kubernetes.io/component: migrate
|
||||
annotations:
|
||||
# post-install, not pre-install: on a first install nothing the
|
||||
# migration needs exists yet — not the ConfigMap, not the Secret, and
|
||||
# with the bundled subchart not the database either, since Helm
|
||||
# creates ordinary resources only once hooks have finished. On an
|
||||
# upgrade all of it is already running, so pre-upgrade is both safe
|
||||
# and preferable: migrations land before the new code rolls out
|
||||
# rather than after.
|
||||
"helm.sh/hook": post-install,pre-upgrade
|
||||
"helm.sh/hook": pre-install,pre-upgrade
|
||||
"helm.sh/hook-weight": "-1"
|
||||
"helm.sh/hook-delete-policy": before-hook-creation,hook-succeeded
|
||||
spec:
|
||||
# Helm does not wait for the database to be ready before running
|
||||
# post-install hooks, so on a first install this Job is what waits: it
|
||||
# exits non-zero until PostgreSQL accepts connections, and the retry
|
||||
# budget has to cover a cold StatefulSet pulling its image and
|
||||
# initialising.
|
||||
backoffLimit: 10
|
||||
backoffLimit: 3
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -31,18 +19,6 @@ spec:
|
||||
spec:
|
||||
serviceAccountName: {{ include "turnstone.serviceAccountName" . }}
|
||||
restartPolicy: OnFailure
|
||||
{{- with .Values.migrate.nodeSelector }}
|
||||
nodeSelector:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.migrate.affinity }}
|
||||
affinity:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.migrate.tolerations }}
|
||||
tolerations:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: migrate
|
||||
image: {{ include "turnstone.image" . }}
|
||||
@@ -51,5 +27,12 @@ spec:
|
||||
- python
|
||||
- -m
|
||||
- turnstone.core.storage._migrate
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: {{ include "turnstone.fullname" . }}-config
|
||||
- secretRef:
|
||||
name: {{ include "turnstone.llm.secretName" . }}
|
||||
optional: true
|
||||
env:
|
||||
{{- include "turnstone.db.env" . | nindent 12 }}
|
||||
- name: TURNSTONE_DB_URL
|
||||
value: "postgresql+psycopg://$(TURNSTONE_DB_USER):$(POSTGRES_PASSWORD)@$(TURNSTONE_DB_HOST):$(TURNSTONE_DB_PORT)/$(TURNSTONE_DB_NAME)"
|
||||
|
||||
@@ -1,19 +1,4 @@
|
||||
{{/*
|
||||
This Secret backs every credential supplied inline in values, so it is
|
||||
rendered whenever any one of them is set — not, as it once was, only
|
||||
when llm.existingSecret is empty. Under that older gate an operator who
|
||||
supplied an LLM Secret lost the unrelated inline values with it: both
|
||||
POSTGRES_PASSWORD and TURNSTONE_JWT_SECRET silently went unrendered
|
||||
while the workloads went on referencing them, so every pod stalled in
|
||||
CreateContainerConfigError.
|
||||
|
||||
Each key keeps its own condition, so an operator-supplied Secret still
|
||||
suppresses the value it replaces and nothing else.
|
||||
*/}}
|
||||
{{- $apiKey := and .Values.llm.apiKey (not .Values.llm.existingSecret) }}
|
||||
{{- $dbPassword := include "turnstone.db.inlinePassword" . }}
|
||||
{{- $jwtSecret := and .Values.auth.jwtSecret (not .Values.auth.existingSecret) }}
|
||||
{{- if or $apiKey $dbPassword $jwtSecret }}
|
||||
{{- if not .Values.llm.existingSecret }}
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
@@ -22,13 +7,15 @@ metadata:
|
||||
{{- include "turnstone.labels" . | nindent 4 }}
|
||||
type: Opaque
|
||||
data:
|
||||
{{- if $apiKey }}
|
||||
{{- if .Values.llm.apiKey }}
|
||||
OPENAI_API_KEY: {{ .Values.llm.apiKey | b64enc | quote }}
|
||||
{{- end }}
|
||||
{{- if $dbPassword }}
|
||||
POSTGRES_PASSWORD: {{ $dbPassword | b64enc | quote }}
|
||||
{{- if and .Values.postgresql.enabled .Values.postgresql.auth.password }}
|
||||
POSTGRES_PASSWORD: {{ .Values.postgresql.auth.password | b64enc | quote }}
|
||||
{{- else if and (not .Values.postgresql.enabled) .Values.database.external.password }}
|
||||
POSTGRES_PASSWORD: {{ .Values.database.external.password | b64enc | quote }}
|
||||
{{- end }}
|
||||
{{- if $jwtSecret }}
|
||||
{{- if and .Values.auth.jwtSecret (not .Values.auth.existingSecret) }}
|
||||
TURNSTONE_JWT_SECRET: {{ .Values.auth.jwtSecret | b64enc | quote }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
@@ -14,13 +14,7 @@ database:
|
||||
port: 5432
|
||||
database: turnstone
|
||||
username: turnstone
|
||||
# Secret holding the password for `username`. Leave empty to supply
|
||||
# `password` inline below instead.
|
||||
existingSecret: ""
|
||||
# Key within existingSecret holding the password. CloudNativePG
|
||||
# generates "password"; other operators differ.
|
||||
existingSecretPasswordKey: password
|
||||
password: ""
|
||||
sslmode: prefer
|
||||
|
||||
# -- Bitnami PostgreSQL subchart
|
||||
@@ -43,10 +37,6 @@ server:
|
||||
service:
|
||||
type: ClusterIP
|
||||
port: 8080
|
||||
# -- Node scheduling constraints
|
||||
nodeSelector: {}
|
||||
affinity: {}
|
||||
tolerations: []
|
||||
|
||||
# -- Turnstone console (cluster dashboard)
|
||||
console:
|
||||
@@ -61,17 +51,6 @@ console:
|
||||
service:
|
||||
type: ClusterIP
|
||||
port: 8090
|
||||
# -- Node scheduling constraints
|
||||
nodeSelector: {}
|
||||
affinity: {}
|
||||
tolerations: []
|
||||
|
||||
# -- Database migration Job (post-install/pre-upgrade hook)
|
||||
migrate:
|
||||
# -- Node scheduling constraints
|
||||
nodeSelector: {}
|
||||
affinity: {}
|
||||
tolerations: []
|
||||
|
||||
# -- LLM provider configuration
|
||||
llm:
|
||||
|
||||
+29
-133
@@ -458,7 +458,7 @@ Each item in `items` (shared by `tool_info` and `approve_request`):
|
||||
| `context_window` | int | Total context window size in tokens |
|
||||
| `pct` | float | Percentage of context window used |
|
||||
| `effort` | string | Reasoning effort level (`low`/`medium`/`high`) |
|
||||
| `cache_creation_tokens` | int | Tokens written to prompt cache (Anthropic + OpenAI) |
|
||||
| `cache_creation_tokens` | int | Tokens written to prompt cache (Anthropic) |
|
||||
| `cache_read_tokens` | int | Tokens served from prompt cache (Anthropic + OpenAI) |
|
||||
|
||||
**`info`** -- an informational message (e.g. command output).
|
||||
@@ -467,46 +467,6 @@ Each item in `items` (shared by `tool_info` and `approve_request`):
|
||||
{"type": "info", "message": "Session cleared."}
|
||||
```
|
||||
|
||||
**`compaction`** -- context-compaction lifecycle (manual `/compact` and
|
||||
auto-compaction). `phase: "start"` opens the operation (`trigger` is
|
||||
`"manual"` or `"auto"`; auto adds `where` — e.g. `"mid-turn"` — and, when
|
||||
the percentage threshold actually fired, `pct`; the context-overflow retry
|
||||
path compacts without a `pct` since no threshold was evaluated).
|
||||
`phase: "progress"` reports chunked summarization (`part`/`total`/`depth`,
|
||||
where depth 0 summarizes transcript batches and deeper levels merge partial
|
||||
summaries), a transient-error retry wait (`retry_in` seconds + `error`), or
|
||||
`warning: "summary_truncated"`. `phase: "end"` settles it: `ok: true`
|
||||
carries `before_tokens`/`after_tokens` and the produced `summary`;
|
||||
`ok: false` carries a `reason`
|
||||
(`"not_enough_messages"` / `"irreducible"` / `"empty_summary"` /
|
||||
`"cancelled"` / `"error"`) and a human-readable `message` — for
|
||||
`reason: "error"` the same message is also emitted as a paired typed
|
||||
`error` event (that is the renderable error surface; the end event is
|
||||
card-teardown). Failed ends also carry `notice`: the emitter-computed
|
||||
display verdict — show `message` only when it is `true` (the server
|
||||
suppresses error-reason, superseded, and cancelled-auto notices once,
|
||||
centrally, so clients don't re-derive that policy). Every end (ok or
|
||||
failed) carries `trigger`, and every event carries `compaction_id` — an
|
||||
opaque integer correlating the start/progress/end of one compaction run (a
|
||||
client that force-stopped one compaction can use it to ignore stragglers
|
||||
from the abandoned run). End events also carry `superseded`: `true` marks
|
||||
a force-abandoned compaction retiring after a successor generation took
|
||||
over (an OK end's result card still stands: the history swap happened).
|
||||
Superseded start/progress events are never emitted.
|
||||
Exactly one `start` and one `end` are emitted per attempt,
|
||||
so clients can key an in-progress affordance (progress bar) on the pair. A
|
||||
successful end is also persisted: the summary replays from `/history` as a
|
||||
`role: "system"`, `source: "compaction"` entry whose `meta` carries
|
||||
`{watermark, before_tokens, after_tokens, trigger}` and whose `event_id`
|
||||
matches the end event's id (dedup across repaint + replay).
|
||||
|
||||
```json
|
||||
{"type": "compaction", "phase": "start", "compaction_id": 7, "trigger": "auto", "where": "mid-turn", "pct": 80}
|
||||
{"type": "compaction", "phase": "progress", "compaction_id": 7, "part": 2, "total": 5, "depth": 0}
|
||||
{"type": "compaction", "phase": "end", "ok": true, "compaction_id": 7, "trigger": "auto",
|
||||
"before_tokens": 128400, "after_tokens": 9200, "summary": "## Decisions\n..."}
|
||||
```
|
||||
|
||||
**`error`** -- an error message.
|
||||
|
||||
```json
|
||||
@@ -788,41 +748,32 @@ Sends a user message to a workstream. Spawns a daemon worker thread that calls
|
||||
**Request body:**
|
||||
|
||||
```json
|
||||
{"message": "Explain how the server works", "attachment_ids": ["a1"]}
|
||||
{"message": "Explain how the server works"}
|
||||
```
|
||||
|
||||
| Field | Type | Required | Description |
|
||||
|------------------|------------|----------|------------------------------------------------------|
|
||||
| `message` | string | yes | The user's message text |
|
||||
| `attachment_ids` | string[] | no | Staged uploads to attach (omit = auto-consume; `[]` = none) |
|
||||
| Field | Type | Required | Description |
|
||||
|-----------|--------|----------|-------------------------|
|
||||
| `message` | string | yes | The user's message text |
|
||||
|
||||
**Response.** Every 200 body carries `attached_ids` and
|
||||
`dropped_attachment_ids` (empty lists when no attachments are involved):
|
||||
**Response (success):**
|
||||
|
||||
- `{"status": "ok", ...}` — a fresh turn was dispatched.
|
||||
- `{"status": "queued", "priority", "msg_id", ...}` — folded into the live
|
||||
turn's interjection queue; delivered at the next tool-result seam.
|
||||
`DELETE .../send` with the `msg_id` retracts it before delivery.
|
||||
- `{"status": "queued", "deferred": true, ...}` — parked on the deferred-send
|
||||
list (a command window holds the slot, or earlier deferred sends are
|
||||
pending) and dispatched as its own full-fidelity send afterwards; see the
|
||||
defer contract under `POST /v1/api/command`.
|
||||
- `{"status": "queue_full", ...}` — the send was refused with retry-shortly
|
||||
semantics: the live worker's interjection queue is at capacity, the
|
||||
deferred-send list hit its saturation bound (10 pending — the same
|
||||
backpressure contract), or the deferred-send drain could not be started
|
||||
under resource exhaustion (the message was **not** accepted; nothing is
|
||||
parked).
|
||||
- `{"status": "attachments_busy", ...}` — attachments can't ride a queued
|
||||
turn; the staged uploads survive for a retry once the worker idles.
|
||||
```json
|
||||
{"status": "ok"}
|
||||
```
|
||||
|
||||
**Response (busy):** Returned if the workstream's worker thread is still alive
|
||||
from a previous request. Also pushes a `busy_error` event to the SSE stream.
|
||||
|
||||
```json
|
||||
{"status": "busy"}
|
||||
```
|
||||
|
||||
**Error responses:**
|
||||
|
||||
| Status | Body | Condition |
|
||||
|--------|-------------------------------------------------|----------------------------------------|
|
||||
| 400 | `{"error": "message is required"}` | Message is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found (or closed mid-send) |
|
||||
| 409 | `{"status": "cross_user_interjection", ...}` | Another participant's turn is in flight |
|
||||
| Status | Body | Condition |
|
||||
|--------|------------------------------------|------------------------|
|
||||
| 400 | `{"error": "Empty message"}` | Message is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
|
||||
---
|
||||
|
||||
@@ -865,56 +816,7 @@ automatically approved without prompting.
|
||||
|
||||
### `POST /v1/api/command`
|
||||
|
||||
Executes a slash command in the given workstream. Commands run on the
|
||||
workstream's worker slot (mutual exclusion against sends, a running
|
||||
compaction, and each other) — the endpoint is **not** unconditionally
|
||||
synchronous:
|
||||
|
||||
- **Quick commands** (everything except `/compact`): the endpoint waits for
|
||||
completion, so `{"status": "ok"}` means the command ran. A command still
|
||||
running after 25 s answers `{"status": "running"}` — the worker keeps
|
||||
going, its output reaches the pane via SSE, and the post-command pane
|
||||
refreshes below still fire when it completes. (The bound sits under
|
||||
common 30 s client/proxy timeouts — the console proxy's included — so
|
||||
the degraded answer actually reaches bounded callers.)
|
||||
- **`/compact`**: dispatched fire-and-forget — `{"status": "ok"}` means the
|
||||
compaction *started*. A large context can legitimately compact for many
|
||||
minutes; progress streams as `compaction` SSE events (see the event
|
||||
reference) and the persisted marker row lands on completion. Do not read
|
||||
`/history` expecting the compacted transcript immediately after the
|
||||
response.
|
||||
- **Busy refusal**: if a turn or another command holds the worker slot, the
|
||||
command is refused with HTTP **409** `{"status": "busy", "error": ...}` and
|
||||
did **not** run. Retry after the current turn finishes. (The old inline
|
||||
endpoint executed commands unconditionally mid-turn; the 409 makes the
|
||||
refusal loud for callers that only check the HTTP status.)
|
||||
|
||||
While a command holds the slot — and afterwards, while earlier deferred
|
||||
sends are still waiting (the pending list is the order authority: a fresh
|
||||
send never overtakes a message already acknowledged) — `POST .../send`
|
||||
requests are **deferred**: the server answers `{"status": "queued",
|
||||
"deferred": true, "msg_id": ...}` immediately and dispatches the message
|
||||
as an ordinary full-fidelity send (attachments and sender identity
|
||||
included) in arrival order once the slot frees — it is never routed
|
||||
through the mid-turn interjection queue (no length cap, no cross-user
|
||||
rejection). The response arrives within normal round-trip time, so
|
||||
timeout-bounded clients (SDKs, proxies, the coordinator) need no special
|
||||
handling. To retract a deferred send before it dispatches, issue the same
|
||||
`DELETE .../send` with its `msg_id` used for queued interjections —
|
||||
`{"status": "removed"}` confirms it will not dispatch; `"not_found"` means
|
||||
it already dispatched (or is dispatching). Retracting a deferred send
|
||||
discards any attachments it carried; re-attach to send them again. When a
|
||||
deferred send dispatches, panes receive a `message_dispatched` event
|
||||
(`msg_id`, plus `folded: true` when it folded into a live turn's
|
||||
interjection queue rather than spawning its own turn) so queued-message
|
||||
UI can settle the right way.
|
||||
|
||||
Durability: deferred sends are **node-local and in-memory** (the same
|
||||
lifetime as the interjection queue). `"queued"` is at-most-once intake, not
|
||||
durable acceptance — if the workstream is closed or the node restarts before
|
||||
the window ends, the message is dropped. Anything that must survive a
|
||||
restart should be re-sent after confirming dispatch (the turn appears on the
|
||||
SSE stream / in `/history`).
|
||||
Executes a slash command in the given workstream.
|
||||
|
||||
**Request body:**
|
||||
|
||||
@@ -927,12 +829,10 @@ SSE stream / in `/history`).
|
||||
| `command` | string | yes | The slash command (e.g. `/clear`) |
|
||||
| `ws_id` | string | yes | Target workstream ID |
|
||||
|
||||
If the command is `/clear`, `/new`, or `/resume`, the server pushes a
|
||||
`clear_ui` SSE event to instruct the client to reset its message display and
|
||||
re-fetch the transcript via `GET .../history` (there is no SSE event that
|
||||
carries the messages themselves). These follow-ups are emitted by the
|
||||
command worker itself, so they fire even when the endpoint already answered
|
||||
`{"status": "running"}`.
|
||||
If the command is `/clear` or `/new`, the server pushes a `clear_ui` SSE event
|
||||
to instruct the client to reset its message display. If the command is
|
||||
`/resume`, the server pushes `clear_ui` followed by a `history` event
|
||||
containing the resumed session's messages.
|
||||
|
||||
**Response:**
|
||||
|
||||
@@ -940,16 +840,12 @@ command worker itself, so they fire even when the endpoint already answered
|
||||
{"status": "ok"}
|
||||
```
|
||||
|
||||
or `{"status": "running"}` as above.
|
||||
|
||||
**Error responses:**
|
||||
|
||||
| Status | Body | Condition |
|
||||
|--------|-------------------------------------|--------------------------------------------------|
|
||||
| 400 | `{"error": "Empty command"}` | Command is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
| 409 | `{"status": "busy", "error": ...}` | A turn/command holds the worker |
|
||||
| 503 | `{"status": "error", "error": ...}` | The command worker could not be started (resource exhaustion) — the command did **not** run; retry shortly |
|
||||
| Status | Body | Condition |
|
||||
|--------|------------------------------------|----------------------|
|
||||
| 400 | `{"error": "Empty command"}` | Command is empty |
|
||||
| 404 | `{"error": "Unknown workstream"}` | `ws_id` not found |
|
||||
|
||||
---
|
||||
|
||||
|
||||
+59
-109
@@ -91,7 +91,7 @@ turnstone/
|
||||
discord/ Discord adapter (bot, cog, views, streaming, config)
|
||||
slack/ Slack adapter (Socket Mode bot, DM routing, approval buttons)
|
||||
shared_static/ Shared design system (base.css, auth.js, theme.js, toast.js, utils.js, kb.js)
|
||||
katex-0.18.1/ Vendored KaTeX math rendering library (MIT, woff2 fonts)
|
||||
katex-0.17.0/ Vendored KaTeX math rendering library (MIT, woff2 fonts)
|
||||
ui/
|
||||
colors.py ANSI color constants with NO_COLOR support
|
||||
markdown.py Streaming terminal markdown renderer (line-buffered)
|
||||
@@ -128,16 +128,13 @@ A user message flows through the system as follows:
|
||||
_emit_state("thinking")
|
||||
|
|
||||
v
|
||||
_stream_response() -------------> model_turn(lane, turns, on_chunk=...) per attempt
|
||||
| lane-swap fallback walk; per-lane ladder:
|
||||
_create_stream_with_retry() ----> provider.create_streaming(client, model, messages, ...)
|
||||
| up to 3 retries (4 total attempts), exponential backoff
|
||||
v
|
||||
the on_chunk consumer -----------> display grid ONLY:
|
||||
_stream_response(stream) --------> dispatch tokens to UI:
|
||||
| on_reasoning_token() / on_content_token()
|
||||
| tool-call deltas just flush the splitter
|
||||
| (assembly lives in drain_stream, inside
|
||||
| model_turn — the consumer never accumulates)
|
||||
| track finish_reason (citations-footer gate)
|
||||
| accumulate tool_calls from deltas
|
||||
| track finish_reason
|
||||
| _check_cancelled() per chunk (cooperative cancel)
|
||||
v
|
||||
finish_reason check:
|
||||
@@ -246,9 +243,7 @@ class SessionUI(Protocol):
|
||||
def on_content_token(self, text: str) -> None: ...
|
||||
def on_stream_end(self) -> None: ...
|
||||
def approve_tools(self, items: list[dict]) -> tuple[bool, str | None]: ...
|
||||
def on_tool_result(
|
||||
self, call_id: str, name: str, output: str, *, is_error: bool = False
|
||||
) -> None: ...
|
||||
def on_tool_result(self, call_id: str, name: str, output: str, *, is_error: bool = False) -> None: ...
|
||||
def on_tool_output_chunk(self, call_id: str, chunk: str) -> None: ...
|
||||
def on_status(self, usage: dict, context_window: int, effort: str) -> None: ...
|
||||
def on_info(self, message: str) -> None: ...
|
||||
@@ -318,15 +313,15 @@ ERROR last operation failed
|
||||
```python
|
||||
@dataclass
|
||||
class Workstream:
|
||||
id: str # uuid hex, 8 chars
|
||||
name: str # user-visible label
|
||||
state: WorkstreamState # current state
|
||||
session: ChatSession | None # the conversation engine
|
||||
ui: SessionUI | None # frontend adapter
|
||||
id: str # uuid hex, 8 chars
|
||||
name: str # user-visible label
|
||||
state: WorkstreamState # current state
|
||||
session: ChatSession | None # the conversation engine
|
||||
ui: SessionUI | None # frontend adapter
|
||||
worker_thread: threading.Thread | None
|
||||
error_message: str
|
||||
last_active: float # time.monotonic() timestamp, updated on every state change
|
||||
_lock: threading.Lock # per-workstream state lock
|
||||
last_active: float # time.monotonic() timestamp, updated on every state change
|
||||
_lock: threading.Lock # per-workstream state lock
|
||||
```
|
||||
|
||||
### WorkstreamManager
|
||||
@@ -338,9 +333,7 @@ class WorkstreamManager:
|
||||
def __init__(self, session_factory: Callable[[SessionUI], ChatSession]): ...
|
||||
def create(self, name="", ui_factory=None) -> Workstream: ...
|
||||
def close(self, ws_id: str) -> bool: ...
|
||||
def close_idle(
|
||||
self, max_age_seconds: float
|
||||
) -> list[str]: ... # auto-close stale IDLE workstreams
|
||||
def close_idle(self, max_age_seconds: float) -> list[str]: ... # auto-close stale IDLE workstreams
|
||||
def get(self, ws_id: str) -> Workstream | None: ...
|
||||
def get_active(self) -> Workstream | None: ...
|
||||
def list_all(self) -> list[Workstream]: ...
|
||||
@@ -515,7 +508,7 @@ then returns the final content as the tool result.
|
||||
response without tools. When unlimited, the loop only exits when the model
|
||||
stops calling tools or hits `finish_reason: "length"`.
|
||||
- **Retry**: each API call in the agent loop uses the same retry+backoff logic
|
||||
as the main loop's per-lane ladder (`_model_turn_with_retry`).
|
||||
as the main `_create_stream_with_retry()`.
|
||||
- **Finish reason handling**: `finish_reason: "length"` stops the agent early
|
||||
and returns whatever content was generated. `finish_reason: "content_filter"`
|
||||
returns a placeholder.
|
||||
@@ -616,7 +609,8 @@ LLMProvider (protocol)
|
||||
|
||||
| Method | Purpose |
|
||||
|--------|---------|
|
||||
| `create_streaming()` | The one transport: streaming request, yields normalized `StreamChunk` objects (single-shot callers accumulate via `drain_stream()` into a `CompletionResult`) |
|
||||
| `create_streaming()` | Streaming request, yields normalized `StreamChunk` objects |
|
||||
| `create_completion()` | Non-streaming request, returns `CompletionResult` |
|
||||
| `get_capabilities()` | Per-model flags (`ModelCapabilities`) |
|
||||
| `convert_tools()` | Translate OpenAI tool schemas to provider format |
|
||||
| `retryable_error_names` | Exception class names that trigger retry |
|
||||
@@ -628,19 +622,19 @@ LLMProvider (protocol)
|
||||
|------|--------|
|
||||
| `StreamChunk` | `content_delta`, `reasoning_delta`, `tool_call_deltas`, `info_delta`, `usage`, `finish_reason`, `provider_blocks` |
|
||||
| `CompletionResult` | `content`, `tool_calls`, `finish_reason`, `usage`, `provider_blocks` |
|
||||
| `ModelCapabilities` | `context_window`, `max_output_tokens`, `supports_temperature`, `token_param`, `thinking_mode`, `supports_effort`, `supports_web_search`, `supports_tool_search`, `supports_vision`, `supports_reasoning_replay`, `supports_verbosity`, `verbosity`, `supports_pro_mode`, `reasoning_mode` |
|
||||
| `ModelCapabilities` | `context_window`, `max_output_tokens`, `supports_temperature`, `token_param`, `thinking_mode`, `supports_effort`, `supports_web_search`, `supports_tool_search`, `supports_vision`, `supports_reasoning_replay` |
|
||||
| `UsageInfo` | `prompt_tokens`, `completion_tokens`, `total_tokens`, `cache_creation_tokens`, `cache_read_tokens` |
|
||||
|
||||
**OpenAIProvider** (`_openai.py`): passes messages through unchanged (they are
|
||||
already in OpenAI format), including multi-part content blocks (text + images)
|
||||
in tool results. Model capability lookup covers GPT-5 through GPT-5.6,
|
||||
in tool results. Model capability lookup table covers GPT-5/5.1/5.2/5.3/5.4,
|
||||
O-series, and search models (`gpt-5-search-api`) — all with `supports_vision`.
|
||||
For search models, injects `web_search_options` and removes the `web_search`
|
||||
function tool (the model always searches). Citations from `url_citation`
|
||||
annotations are formatted as footnotes. Pre-5.6 GPT-5 models request extended
|
||||
prompt-cache retention (`prompt_cache_retention: "24h"`); GPT-5.6 uses
|
||||
`prompt_cache_options.ttl: "30m"`. Cache reads and writes are extracted from
|
||||
`cached_tokens` and `cache_write_tokens`. Unknown models get permissive
|
||||
annotations are formatted as footnotes. Extended prompt cache retention
|
||||
(`prompt_cache_retention: "24h"`) is enabled for GPT-5.x models at no
|
||||
additional cost. Cached token counts are extracted from
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models get permissive
|
||||
defaults with `supports_vision=False` and use SearxNG for web search. The
|
||||
`openai-compatible` lane never consults this table at all — on either API
|
||||
surface (the responses pin is served by a compat-mode
|
||||
@@ -648,9 +642,8 @@ surface (the responses pin is served by a compat-mode
|
||||
local server serves whatever the operator named it (vLLM
|
||||
`--served-model-name` is a free string), so a prefix collision with a cloud
|
||||
model id must not inherit that model's sampling/effort contract — every
|
||||
local model gets the plain defaults, commercial prompt-cache controls are not
|
||||
injected by model-name prefix, and anything beyond those defaults is declared
|
||||
on the model definition (capabilities JSON + `server_compat`), matching the
|
||||
local model gets the plain defaults, and anything beyond them is declared on
|
||||
the model definition (capabilities JSON + `server_compat`), matching the
|
||||
`anthropic-compatible` lane.
|
||||
|
||||
**AnthropicProvider** (`_anthropic.py`): converts OpenAI-format messages to
|
||||
@@ -669,7 +662,7 @@ display). Automatic prompt caching is enabled via top-level `cache_control:
|
||||
cacheable block and advances it as conversations grow (90% input cost
|
||||
reduction on cache hits, 1.25x write on first turn). Cache metrics
|
||||
(`cache_creation_input_tokens`, `cache_read_input_tokens`) are extracted from
|
||||
the stream's usage events. The `anthropic` SDK is a core
|
||||
both streaming and non-streaming responses. The `anthropic` SDK is a core
|
||||
dependency — the Anthropic provider is first-class alongside OpenAI.
|
||||
|
||||
**GoogleProvider** (`_google.py`): extends `OpenAIChatCompletionsProvider` for
|
||||
@@ -948,8 +941,8 @@ with the same alias in-memory (the DB rows are never modified).
|
||||
5. `/model` command shows available models; `/model <alias>` switches the
|
||||
active workstream's client, model, context window, and per-model sampling
|
||||
parameters
|
||||
6. `_model_turn_with_fallback()` tries the primary lane, then each fallback
|
||||
alias's lane in order if the primary is unreachable
|
||||
6. `_create_stream_with_retry()` tries the primary model, then each fallback
|
||||
alias in order if the primary is unreachable
|
||||
7. `_run_agent()` resolves `registry.agent_model` (if set) for task
|
||||
sub-agents, allowing a cheaper model for autonomous loops
|
||||
|
||||
@@ -974,33 +967,6 @@ The default limit is 50% of the context window in characters (computed as
|
||||
|
||||
This truncation message is visible to the model, so it knows output was cut.
|
||||
|
||||
During the send loop the limit is additionally capped by the remaining
|
||||
context budget, and three guarantees apply when that budget reaches zero
|
||||
(#883):
|
||||
|
||||
- **Structural floor** — orchestration handles (`spawn_workstream`,
|
||||
`spawn_batch`, `wait_for_workstream`, `tasks`) and error results are
|
||||
always admitted up to a guaranteed floor (2048 chars, head+tail beyond
|
||||
it), because a lost `ws_id` or a masked failure wedges the session.
|
||||
- **Small-result pass** — results at or under the floor pass verbatim,
|
||||
funded from a bounded per-batch grace pool (2× the floor) so a wide
|
||||
batch of small results cannot collectively bypass budget accounting;
|
||||
past the pool they get the drop notice instead.
|
||||
- **Honest drop notice** — a bulky non-structural result is replaced by an
|
||||
explicit `Error: tool result dropped — context budget exhausted…` notice
|
||||
stating the call ran but its output could not be admitted (never a
|
||||
successful-looking trim).
|
||||
|
||||
A zero budget also triggers one mid-turn auto-compaction before results are
|
||||
sized. With `max_tokens ≥ context_window/4` the response reserve zeroes the
|
||||
budget near 70% fullness — below the default 80% auto-compact threshold —
|
||||
and without this trigger a session could idle in that band indefinitely
|
||||
with every tool result floored or dropped. The trigger keys on the
|
||||
exhausted budget itself, not on any threshold, so it composes with any
|
||||
operator-set `auto_compact_pct`: with thresholds below the zero point the
|
||||
ordinary owed-compaction paths fire first and this trigger degrades to a
|
||||
backstop for the cases where they bailed or freed too little.
|
||||
|
||||
---
|
||||
|
||||
## Persistence
|
||||
@@ -1183,30 +1149,20 @@ Named (aliased) workstreams are never age-pruned. Configure with
|
||||
|
||||
### API Retry
|
||||
|
||||
Every model call streams (#831); retry lives at two stacked layers:
|
||||
`ChatSession._create_stream_with_retry()` (streaming path) and the agent
|
||||
`_api_call()` (non-streaming) both use the same retry pattern:
|
||||
|
||||
- **Caller ladders** — `ChatSession._model_turn_with_retry()` (chat
|
||||
loop, one ladder per lane) and the agent `_api_call()` (drained via
|
||||
`model_turn`) use the same pattern: 4 total attempts (1 initial + 3 retries,
|
||||
`_MAX_RETRIES = 3`), exponential backoff base 1 second
|
||||
(`delay = 1s * 2^attempt`), `ui.on_info()` on retry, exception
|
||||
propagates on final failure. `_compact_messages()` wraps its drained
|
||||
call in the same loop.
|
||||
- **`model_turn`'s drain ladder** — inside every single-shot call,
|
||||
mid-stream deaths (errors raised while draining, e.g.
|
||||
`IncompleteStreamError`) are re-issued up to 2 more times with a
|
||||
0.5s-base exponential backoff (±50% jitter); request-time failures
|
||||
keep the SDK's own retry policy. The two ladders stack
|
||||
multiplicatively on transient-shaped failures.
|
||||
- **Retryable errors** are matched by class name against each
|
||||
provider's `retryable_error_names` (avoids importing
|
||||
backend-specific exception hierarchies): `RateLimitError`,
|
||||
`APITimeoutError`, `APIConnectionError`, `InternalServerError`,
|
||||
`ServiceUnavailableError`, `APIError`, plus the drained-transport
|
||||
errors `IncompleteStreamError` (stream ended with no terminal
|
||||
signal — for servers that never send one, declare
|
||||
`finish_reason_optional` in the model's capabilities JSON) and
|
||||
`ResponsesStreamFailedError` (transient in-band Responses failure).
|
||||
- **Retries**: 4 total attempts (1 initial + 3 retries, `_MAX_RETRIES = 3`)
|
||||
- **Backoff**: exponential, base 1 second (`delay = 1s * 2^attempt`)
|
||||
- **Retryable errors**: `RateLimitError`, `APITimeoutError`,
|
||||
`APIConnectionError`, `InternalServerError`, `ServiceUnavailableError`,
|
||||
`APIError` (matched by class name to avoid importing backend-specific
|
||||
exception hierarchies)
|
||||
- On retry: `ui.on_info()` notification
|
||||
- On final failure: exception propagates
|
||||
|
||||
`_compact_messages()` also wraps its non-streaming API call in the same
|
||||
retry loop.
|
||||
|
||||
### Finish Reason Handling
|
||||
|
||||
@@ -1219,7 +1175,7 @@ Every model call streams (#831); retry lives at two stacked layers:
|
||||
blocked.
|
||||
|
||||
Agent sub-sessions (`_run_agent()`) check `finish_reason` on each
|
||||
drained turn and stop the agent early on `"length"` or
|
||||
non-streaming response and stop the agent early on `"length"` or
|
||||
`"content_filter"`.
|
||||
|
||||
`_compact_messages()` checks `finish_reason` on the compaction response and
|
||||
@@ -1263,34 +1219,28 @@ warns if the summary was truncated.
|
||||
`_run_single_test()`: wraps `session.send_headless()` in a retry loop (3
|
||||
attempts) to avoid transient API errors from poisoning evaluation scores.
|
||||
|
||||
### Backend Health Tracking
|
||||
### Health Monitor & Circuit Breaker
|
||||
|
||||
`BackendHealthTracker` (`turnstone/core/healthcheck.py`) records LLM backend
|
||||
health passively from real request outcomes — there is no probe thread and no
|
||||
circuit breaker, and requests are never blocked. Two states:
|
||||
`BackendHealthMonitor` (`turnstone/core/healthcheck.py`) runs a daemon thread
|
||||
that probes the LLM backend by calling `client.models.list()` every
|
||||
`backend_probe_interval` seconds (default 30). Probe results drive a three-state
|
||||
circuit breaker:
|
||||
|
||||
```
|
||||
healthy ──(failure_threshold consecutive failures)──> degraded
|
||||
degraded ──(any success)───────────────────────────> healthy
|
||||
CLOSED ──(N consecutive failures)──> OPEN
|
||||
OPEN ──(cooldown expires)────────> HALF_OPEN
|
||||
HALF_OPEN ──(probe succeeds)────────> CLOSED
|
||||
HALF_OPEN ──(probe fails)──────────> OPEN
|
||||
```
|
||||
|
||||
- `record_success()` fires at the request-accepted instant: the streaming
|
||||
consumer's `on_stream_armed` hook, driven by the eager `cancel_ref` append
|
||||
every adapter performs at HTTP-response time.
|
||||
- `record_failure()` fires once per lane's whole creation ladder, in
|
||||
`ChatSession._model_turn_with_fallback` / `_try_fallback_lane`. A mid-stream
|
||||
death (the stream armed, then died) records neither — it belongs to the
|
||||
re-issue ladder, not the fallback walk. `BackendAuthUnavailableError` and
|
||||
`WirePreparationError` also record nothing: an auth refusal is fail-closed
|
||||
configuration policy and a wire-preparation fault is session data — neither
|
||||
says anything about the backend.
|
||||
- `is_degraded` is advisory ordering, not admission: the fallback walk tries
|
||||
non-degraded aliases first and degraded ones as a last resort, and the
|
||||
primary lane is always dialed.
|
||||
- `HealthTrackerRegistry` keys trackers by `(provider, base_url)` so aliases
|
||||
sharing a backend share one tracker. The `/health` endpoint projects the
|
||||
same trackers: `"status": "ok"` when the backend is healthy, `"degraded"`
|
||||
otherwise.
|
||||
- `record_success()` / `record_failure()` update `_consecutive_failures` and
|
||||
transition the `_state` (`CircuitState` enum: `CLOSED`, `OPEN`, `HALF_OPEN`).
|
||||
- `acquire_request_permit()` returns `False` when the circuit is `OPEN` or when
|
||||
in `HALF_OPEN` and the single probe permit has already been consumed. Causes
|
||||
`ChatSession._create_stream_with_retry` to skip the backend and surface an
|
||||
error immediately.
|
||||
- The `/health` endpoint reads the monitor's state: `"status": "ok"` when the
|
||||
circuit is closed, `"status": "degraded"` when open or half-open.
|
||||
|
||||
### Rate Limiting
|
||||
|
||||
|
||||
+11
-21
@@ -126,11 +126,10 @@ the skill should end on.
|
||||
|
||||
`tasks` is the coordinator's scratchpad — a persisted, ordered
|
||||
list of rows with fields `{id, title, status, child_ws_id, created,
|
||||
updated}`, plus `note` on rows where one has been set (the key is
|
||||
absent otherwise), that only this coordinator sees. Children don't
|
||||
see it; the user does via the sidebar. Five actions: `add`,
|
||||
`update`, `remove`, `reorder`, `list` (only `list` is auto-approved;
|
||||
the mutators go through the approval flow).
|
||||
updated}` that only this coordinator sees. Children don't see it;
|
||||
the user does via the sidebar. Five actions: `add`, `update`,
|
||||
`remove`, `reorder`, `list` (only `list` is auto-approved; the
|
||||
mutators go through the approval flow).
|
||||
|
||||
The input schema refers to rows by `task_id`; the persisted row
|
||||
object exposes the same id as `id`. The `child_ws_id` field is a
|
||||
@@ -143,20 +142,11 @@ A skill's initial prompt can seed the task list by calling
|
||||
`tasks(action="add", title=...)` as its very first tool calls —
|
||||
the user gets a visible plan before any child is spawned, and the
|
||||
coordinator's future self has something concrete to iterate on.
|
||||
Status transitions (`pending` → `in_progress` → `done` / `blocked` /
|
||||
`needs_user`) are the skill's main feedback loop: mutate the task
|
||||
when the child covering it finishes, not when the child starts.
|
||||
`blocked` and `needs_user` are not interchangeable — `blocked` is a
|
||||
dependency the coordinator may be able to clear itself, while
|
||||
`needs_user` marks a task that cannot move without a decision,
|
||||
approval, or grant only the user can give. The distinction is
|
||||
load-bearing: a coordinator that goes idle holding open tasks gets
|
||||
nudged to pick them back up — even when children are still running, so
|
||||
keep the matrix honest rather than expecting the reminder to wait for
|
||||
an all-clear — and `needs_user` is what tells that nudge the stop was
|
||||
deliberate. Pair it with `note` to record what is being asked for.
|
||||
Use `tasks(action="update", task_id=..., child_ws_id=<ws_id>)` to link
|
||||
a task to the child that owns it once spawn returns.
|
||||
Status transitions (`pending` → `in_progress` → `done` / `blocked`)
|
||||
are the skill's main feedback loop: mutate the task when the child
|
||||
covering it finishes, not when the child starts. Use
|
||||
`tasks(action="update", task_id=..., child_ws_id=<ws_id>)` to
|
||||
link a task to the child that owns it once spawn returns.
|
||||
|
||||
A final gotcha: parallel tool dispatch does NOT serialise reads
|
||||
after writes in the same batch. If a skill issues an `update` and
|
||||
@@ -307,11 +297,11 @@ and the coordinator's planning step is itself valuable.
|
||||
tasks(action='add', title='...') × N # the plan, visible in the sidebar
|
||||
for task in tasks:
|
||||
spawn_workstream(skill=..., initial_message=task.brief)
|
||||
tasks(action='update', task_id=task.id, note='ws=<child_ws_id>')
|
||||
tasks(action='update', task_id=task.id, notes='ws=<child_ws_id>')
|
||||
wait_for_workstream(ws_ids=[...], mode='all', timeout=...)
|
||||
for child in children:
|
||||
inspect_workstream(ws_id=child)
|
||||
tasks(action='update', task_id=..., status='done', note='result summary')
|
||||
tasks(action='update', task_id=..., status='done', notes='result summary')
|
||||
→ synthesise
|
||||
```
|
||||
|
||||
|
||||
@@ -66,7 +66,8 @@ class "NullUI" as NullUI {
|
||||
interface "LLMProvider" as LLMProvider <<Protocol>> {
|
||||
+ provider_name: str {property}
|
||||
+ get_capabilities(model) → ModelCapabilities
|
||||
+ create_streaming(client, model, messages, ..., cancel_ref, replay_reasoning_to_model) → Iterator[StreamChunk]
|
||||
+ create_streaming(client, model, messages, ..., replay_reasoning_to_model) → Iterator[StreamChunk]
|
||||
+ create_completion(client, model, messages, ..., replay_reasoning_to_model) → CompletionResult
|
||||
+ convert_tools(tools) → list[dict]
|
||||
+ extract_reasoning_text(provider_blocks) → str
|
||||
+ retryable_error_names: frozenset[str] {property}
|
||||
@@ -148,9 +149,9 @@ class "ChatSession" as ChatSession {
|
||||
+ handle_command(command: str)
|
||||
+ resume(ws_id: str)
|
||||
- _save_config()
|
||||
- _stream_response(my_generation) → ModelTurnResult
|
||||
- _model_turn_with_fallback(consumer, prepare_wire) → ModelTurnResult
|
||||
- _model_turn_with_retry(lane, tracker, ...) → ModelTurnResult
|
||||
- _stream_response(stream) → dict
|
||||
- _create_stream_with_retry(msgs) → Stream (+ fallback)
|
||||
- _try_stream(client, model, msgs) → Stream
|
||||
- _execute_tools(tool_calls) → (results, feedback)
|
||||
- _prepare_tool(tc) → item dict
|
||||
- _prepare_mcp_tool(call_id, name, args) → item dict
|
||||
@@ -176,7 +177,7 @@ class "HeadlessSession" as HeadlessSession {
|
||||
+ send_headless(input, max_turns, ...)
|
||||
- _override_system_prompt(content)
|
||||
--
|
||||
eval.py: drained single-shot turns,
|
||||
eval.py: non-streaming,
|
||||
records all tool calls
|
||||
}
|
||||
|
||||
|
||||
@@ -84,8 +84,8 @@ end note
|
||||
|
||||
loop up to 3 turns (timeout budget)
|
||||
|
||||
Judge -> LLM : model_turn(lane, judge_turns,\ntools=[read_file, list_directory])\nvia drained create_streaming
|
||||
LLM --> Judge : ModelTurnResult
|
||||
Judge -> LLM : create_completion(\nmodel, judge_messages,\ntools=[read_file, list_directory])
|
||||
LLM --> Judge : CompletionResult
|
||||
|
||||
alt tool_calls present (turn < 3)
|
||||
Judge -> Judge : _exec_read_only_tool()
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:6e2bfdf968e96f3720ed58674103288e2f57e9c056f5c479a57f37a849f3e69c
|
||||
size 821878
|
||||
@@ -252,7 +252,6 @@ interface, or anyone who can reach it can search through your instance.
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `WORKSPACE_MOUNT` | empty volume | Host directory bind-mounted at `/workspace` for the model to read/write |
|
||||
| `TURNSTONE_WORKSPACE` | `/workspace` (image env) | Directory named as the user's workspace in the model's tool descriptions; informational only — see [Working directory](#working-directory) |
|
||||
| `SKIP_PERMISSIONS` | — | Set to any value to auto-approve all tool calls (dev only) |
|
||||
| `MCP_CONFIG` | — | Path to an MCP server config file |
|
||||
| `TURNSTONE_IMAGE_TAG` | `latest` | ghcr.io image tag — production stack |
|
||||
@@ -277,35 +276,6 @@ docker compose build --no-cache # rebuild from scratch
|
||||
| `workspace` | `/workspace` (unless `WORKSPACE_MOUNT` is set) |
|
||||
| `caddy-data` / `caddy-config` | Caddy's local CA and config (dev stack) |
|
||||
|
||||
## Working directory
|
||||
|
||||
Node processes run with `/data` as their working directory (the image's
|
||||
`WORKDIR`), and that is where the model's shell commands execute and
|
||||
relative file paths resolve — **not** `/workspace`. The shell and file
|
||||
tool descriptions state both paths (the working directory, and the
|
||||
workspace named by `TURNSTONE_WORKSPACE`), so the model knows to look in
|
||||
`/workspace` for your files without being told each session.
|
||||
|
||||
To make tools start inside the mount instead, override the working
|
||||
directory on the node services:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
turnstone-node:
|
||||
working_dir: /workspace
|
||||
```
|
||||
|
||||
Two caveats before overriding:
|
||||
|
||||
- **SQLite fallback**: when a node runs without PostgreSQL, its fallback
|
||||
database `.turnstone.db` is created in the process working directory.
|
||||
Changing `working_dir` on an existing SQLite-fallback deployment makes
|
||||
the node create a fresh database inside the mount and your prior state
|
||||
appears lost (it is still in the `turnstone-data` volume under `/data`).
|
||||
The stock compose stacks use PostgreSQL and are unaffected.
|
||||
- Migrations (`entrypoint.sh`) run in the same working directory, so the
|
||||
same SQLite caveat applies to them.
|
||||
|
||||
## Cleanup
|
||||
|
||||
```bash
|
||||
|
||||
+3
-6
@@ -131,12 +131,9 @@ Per-LLM-request token and tool call metrics:
|
||||
LLM response with prompt/completion tokens, cache tokens, tool call count,
|
||||
model, ws_id
|
||||
- **Prompt caching**: Anthropic automatic caching (`cache_control: ephemeral`)
|
||||
and OpenAI caching are enabled by default. Pre-5.6 GPT-5 models request
|
||||
`prompt_cache_retention: 24h`; GPT-5.6 uses
|
||||
`prompt_cache_options: {"ttl": "30m"}`. GPT-5.6 cache writes use the
|
||||
provider's 1.25× input-token rate. `cache_creation_tokens` and
|
||||
`cache_read_tokens` are tracked per request in `usage_events` and surfaced
|
||||
in the Usage admin tab
|
||||
and OpenAI extended retention (`prompt_cache_retention: 24h` for GPT-5.x)
|
||||
are enabled by default. `cache_creation_tokens` and `cache_read_tokens` are
|
||||
tracked per request in `usage_events` and surfaced in the Usage admin tab
|
||||
- **Querying**: `GET /v1/api/admin/usage` with `group_by` (day/hour/model/user)
|
||||
and time range filtering — includes cache token aggregates
|
||||
- **Prometheus**: `turnstone_tokens_total{type="cache_creation|cache_read"}`
|
||||
|
||||
+2
-8
@@ -249,14 +249,8 @@ are withheld from the live surfaces (a reused call_id must never ride a stale
|
||||
`approve` into Smart Approvals) but still persist with
|
||||
`user_decision = "superseded"` so the audit trail records the judge's answer.
|
||||
|
||||
Sub-agent (task agent) tool calls are judge-gated too. Each runs the same
|
||||
intent pipeline as its own `agent_gate` generation, grounded in that sub-agent's
|
||||
own trajectory -- its task prompt is the delegation contract the operator
|
||||
approved, so "does this call serve the task" is the right local question.
|
||||
Agent-gate generations never occupy the main loop's supersede slot (parallel
|
||||
siblings would otherwise make each other's verdicts look stale); per-cycle
|
||||
generation checks enforce staleness instead, and `judge.cancel_on_approval`
|
||||
fires per gate exactly like the main loop.
|
||||
Sub-agents (plan agent, task agent) are exempt from intent validation -- they
|
||||
always get full tool visibility without judge evaluation.
|
||||
|
||||
---
|
||||
|
||||
|
||||
+4
-65
@@ -17,9 +17,8 @@ The MCP server admin form exposes three authorization modes ("Multitenant Author
|
||||
| `none` | No headers attached. Open MCP server (or one gated by network policy only). | Internal MCP servers on a trusted network. |
|
||||
| `static` | One static bearer token, configured per server, sent on every request from every user. | Service-to-service MCP servers where per-user attribution doesn't matter, or single-tenant deployments. |
|
||||
| `oauth_user` *(recommended for user-data servers)* | Each user authorizes separately via OAuth 2.1 + PKCE; Turnstone stores per-user tokens encrypted at rest. | MCP servers that expose user-specific data or that want per-user audit attribution. |
|
||||
| `oauth_obo` *(sign-in passthrough)* | Each user's Turnstone **org sign-in** (OIDC) mints a per-server access token on demand — no separate per-server consent. One captured credential per user covers every `oauth_obo` server. | Enterprise deployments where the identity provider governs access (Entra, Keycloak) and you want zero per-user connect clicks. See the dedicated section below. |
|
||||
|
||||
Switching `auth_type` away from `oauth_user` / `oauth_obo` **deletes** that server's per-user rows (consents / minted cache) — see the transition table below. Switching back later starts clean: users re-consent (or re-mint) on next use. The admin **bulk-revoke** / **flush cache** affordance clears rows without an auth-type change.
|
||||
Switching `auth_type` away from `oauth_user` orphans existing per-user tokens. Use the admin **bulk-revoke** affordance on the server row (Phase 9) to clear them, or let them expire naturally — they're inert without the matching `auth_type` value.
|
||||
|
||||
---
|
||||
|
||||
@@ -66,59 +65,6 @@ Keep this in `config.toml` rather than environment variables. An in-process LLM
|
||||
|
||||
---
|
||||
|
||||
## `auth_type=oauth_obo` — single-credential sign-in passthrough
|
||||
|
||||
Where `oauth_user` makes each user complete a **separate** browser consent per MCP server, `oauth_obo` reuses the user's Turnstone **org sign-in** (OIDC). Turnstone captures one refresh credential per user at login and, on each tool call, mints a short-lived access token scoped to that server's audience. There is no per-server connect step, and one credential covers every `oauth_obo` server. This is the right shape when your identity provider already governs who may reach each backend (an Entra tenant with Entra-protected MCP servers; a Keycloak realm with token exchange).
|
||||
|
||||
Access is governed **downstream** by the IdP: a user can only mint a token for a server their delegated permissions allow. Removing that grant at the IdP cuts the user off regardless of their Turnstone state.
|
||||
|
||||
### Deployment configuration (`[oidc]` in `config.toml`)
|
||||
|
||||
`oauth_obo` requires OIDC SSO to be configured (it is the credential source), plus:
|
||||
|
||||
```toml
|
||||
[oidc]
|
||||
# ... your existing issuer / client_id / client_secret ...
|
||||
capture_user_credential = true # persist the IdP refresh token at login
|
||||
obo_grant_profile = "entra" # "entra" | "rfc8693" — how tokens are minted
|
||||
```
|
||||
|
||||
- **`capture_user_credential`** (default `false`): when enabled, Turnstone appends `offline_access` to the login scopes and stores the returned refresh token, encrypted with the same `[security] mcp_token_encryption_key` as `oauth_user` tokens. **The encryption key is required** — Turnstone refuses to start with an `oauth_obo` row (or capture enabled) and no key.
|
||||
- **`obo_grant_profile`** picks the mint mechanism (the IdP determines which one is valid; this is deployment-wide, not per-server):
|
||||
- **`entra`** — redeems the user's refresh token directly for a token scoped to `<audience>/.default`. `oauth_scopes` on the server row is **not used** (the admin form rejects it under this profile).
|
||||
- **`rfc8693`** — a refresh grant for a subject token, then an RFC 8693 token exchange for the server audience. Per-server `oauth_scopes` **are** sent on the exchange (some IdPs require the audience scope explicitly).
|
||||
|
||||
### Adding an `oauth_obo` server
|
||||
|
||||
In the admin MCP form, choose **Sign-in passthrough** and set **Audience** (required — the downstream resource the token is minted for, e.g. `api://<app-id>` on Entra or the client id on Keycloak). The client-id / secret / registration fields do not apply and are hidden.
|
||||
|
||||
`oauth_obo` servers are accepted only when **OIDC sign-in is configured and enabled** and `[oidc] obo_grant_profile` is a valid profile — the write is rejected otherwise, since a row that can never mint would surface to users as a permanent "please retry" that never heals.
|
||||
|
||||
### Identity-provider setup
|
||||
|
||||
**Entra (`obo_grant_profile = "entra"`):**
|
||||
1. Turnstone's app registration must hold **delegated permissions** to each MCP server's exposed API, with **admin consent granted** (or the MCP app listed in Turnstone's `preAuthorizedApplications`).
|
||||
2. Set the server row's Audience to the MCP app's Application ID URI (`api://<guid>`).
|
||||
3. **Gotcha (verified):** admin-consent issued *immediately* after creating the app/service principal can silently skip a not-yet-propagated resource — the only symptom is `AADSTS65001` at mint time. Verify the delegated grant landed (`az ad app permission list-grants` / the portal's *API permissions* blade shows *Granted*), or grant it explicitly per resource. A missing grant surfaces in Turnstone as a re-login prompt on the affected server (same rail as a revoked credential), and the `mcp_server.oauth.obo_mint_rejected` log line carries the raw `AADSTS…` text.
|
||||
|
||||
**Keycloak / RFC 8693 (`obo_grant_profile = "rfc8693"`):**
|
||||
1. Enable **standard token exchange** on Turnstone's client.
|
||||
2. Grant the audience: add an audience client scope for each MCP client and attach it to Turnstone's client (optional scopes must be requested — set the server row's Scopes to that scope, or the exchange returns *"Requested audience not available"*).
|
||||
3. Set the server row's Audience to the downstream client id.
|
||||
|
||||
### Revocation & custody
|
||||
|
||||
The captured credential is a single per-user secret that can mint for every `oauth_obo` server, so treat it like any long-lived credential:
|
||||
|
||||
- **Cut off one user:** unlink their OIDC identity in the admin console (**Users → OIDC identities → delete**). This revokes the captured credential **and** purges their minted cache rows, so future mints fail and cached tokens are dropped. (Warmed in-memory sessions on server nodes self-expire at the access-token TTL; there is no cross-node per-user session-kill.) Removing the user's access at the IdP is the authoritative cut-off.
|
||||
- The same unlink also purges that user's synthetic `__model_obo__:` gateway-token rows and requests eviction from every registered host's in-process mint memo. Shared `entra_app` model tokens live under the `__app__` pseudo-user and are intentionally not user-deprovisioned; revoking the app credential prevents new mints, while a cached app bearer lasts until `expires_at`.
|
||||
- **Flush a server's minted tokens** (e.g. after narrowing its audience): the server row's **flush cache** action drops all users' cached tokens for that server. This is **not** a revocation — users re-mint on next use from their still-valid sign-in. It is surfaced honestly (audit `mcp_server.oauth.obo_cache_flushed`, response `effect: cache_flush_remints`) so it is never mistaken for cutting access.
|
||||
- Per-server revocation in the `oauth_user` sense does not exist for `oauth_obo` — the credential is issuer-scoped and IdP-governed. Revoke at the IdP.
|
||||
|
||||
> **Interim for Entra without OBO:** if you don't want host-side minting, admin consent + `preAuthorizedApplications` on each MCP app registration removes the second consent prompt for the plain `oauth_user` flow too (a tenant-config change, no Turnstone code). Tracked in issue #682. It does not remove the per-server connect clicks or per-(user, server) token custody — that is what `oauth_obo` is for.
|
||||
|
||||
---
|
||||
|
||||
## Lifecycle
|
||||
|
||||
1. **First tool call** for a user against an `oauth_user` MCP server: pool dispatch finds no stored token, returns `mcp_consent_required` to the agent. Dashboard renders an inline "Connect" action card.
|
||||
@@ -129,7 +75,7 @@ The captured credential is a single per-user secret that can mint for every `oau
|
||||
|
||||
4. **Step-up scope**: when a tool call hits `403` with `WWW-Authenticate: error="insufficient_scope"`, Turnstone emits `mcp_insufficient_scope` with the parsed scope set; the dashboard offers a "Connect with additional scopes" affordance that opens `/v1/api/mcp/oauth/start?server=<name>&scopes=<extra>` so the union of original + new scopes flows into the AS authorize request.
|
||||
|
||||
5. **User revoke** (settings modal): `DELETE /v1/api/mcp/oauth/connections/{server_name}` runs the authoritative local delete + best-effort RFC 7009 upstream revoke (fire-and-forget, capped at 256 concurrent in-flight tasks). `oauth_obo` servers and synthetic model-auth rows are excluded: their rows are mint caches, not consents — deleting one only forces a re-mint — so the connections list hides them and the endpoint refuses them with `409` (revocation for sign-in passthrough happens at the identity layer: unlink the identity or revoke at the IdP).
|
||||
5. **User revoke** (settings modal): `DELETE /v1/api/mcp/oauth/connections/{server_name}` runs the authoritative local delete + best-effort RFC 7009 upstream revoke (fire-and-forget, capped at 256 concurrent in-flight tasks).
|
||||
|
||||
6. **Admin bulk-revoke** (Phase 9): `POST /v1/api/admin/mcp-servers/{name}/bulk-revoke` drops every user's token for the server. Upstream RFC 7009 revoke is intentionally **not** attempted in bulk (avoids N upstream HTTP calls per admin click); tokens at the AS expire naturally. Use the per-user revoke endpoint if you need guaranteed upstream invalidation.
|
||||
|
||||
@@ -151,13 +97,10 @@ Additional indicators (circuit-breaker state, encryption-key mismatch) are expos
|
||||
| From | To | What happens |
|
||||
|---|---|---|
|
||||
| `none` / `static` → `oauth_user` | — | New code path activates for this server. Existing static headers (if any) are no longer sent. Users must authorize on first use. |
|
||||
| `oauth_user` → `none` / `static` | — | Existing `mcp_user_tokens` rows are **deleted**: the tokens are bound to the auth model + URL active at consent time, and rows left behind could silently rebind if a row with the old name/URL reappears. Switching back to `oauth_user` later starts clean — users re-consent on next use. This is **not reversible**; the AS-side grants are untouched (revoke upstream via the AS if needed). |
|
||||
| `oauth_user` → `none` / `static` | — | Existing `mcp_user_tokens` rows are **orphaned** — inert without a matching `auth_type`. Use admin bulk-revoke to drop them, or let them expire. Switching back to `oauth_user` later re-activates the orphaned rows if they haven't been deleted. |
|
||||
| OAuth `client_id` or `client_secret` rotated | — | Existing tokens may stop refreshing if the AS treats them as bound to the previous client. Bulk-revoke after rotation. |
|
||||
| `oauth_user` ↔ `oauth_obo` | — | The per-user rows are **deleted** on the flip (they mean different things: per-server AS refresh tokens vs. minted cache). `oauth_audience` and `oauth_scopes` mean different things in each model (a resource indicator vs. an IdP app identifier; AS-consent scopes vs. an rfc8693 exchange scope), so on a flip they **never carry** — each is taken from the request for the target model or set NULL. The admin console clears these fields when you change the auth type, so re-enter the correct values for the new mode; via the API, supply them explicitly (a flip into `oauth_obo` with no `oauth_audience` is rejected, and a non-empty `oauth_scopes` under the `entra` profile is rejected since that leg pins `<audience>/.default`). |
|
||||
| `oauth_obo` → `none` / `static` | — | Minted cache rows are deleted. |
|
||||
| `oauth_obo` **audience**, **URL**, or **`oauth_scopes`** changed | — | Minted cache rows are **deleted** (tokens are bound to the audience/URL/scopes at mint time), forcing a fresh mint — so an audience or scope narrowing takes effect immediately, not at token expiry. |
|
||||
|
||||
Every transition that changes what a stored row *means* deletes the rows outright — a stale consent or minted token must never be served under new semantics. There is no orphan-and-reactivate path.
|
||||
The orphan-by-default behavior is chosen so switching back to `oauth_user` is non-destructive. Bulk-revoke is the explicit cleanup path.
|
||||
|
||||
---
|
||||
|
||||
@@ -170,9 +113,5 @@ Every transition that changes what a stored row *means* deletes the rows outrigh
|
||||
| `mcp_oauth_url_insecure` | MCP server URL is `http://` (not `https://`) on a non-loopback host | Use `https://`. Per-user bearers must not transit cleartext. |
|
||||
| Tools fail in scheduled / Discord / Slack runs | OAuth-MCP requires browser-based consent | Users must pre-consent via the web UI. Phase 9 dashboard badge surfaces deferred consents from these runs on next login. |
|
||||
| Circuit breaker open repeatedly | Transport-level errors on the MCP server (DNS, TLS, 5xx) | Check the per-server error pill; auth errors do not trip the breaker. |
|
||||
| **`oauth_obo`**: every tool call fails, log shows `obo_misconfigured` | Server row has no Audience, or `obo_grant_profile` is unset/unknown | Set the Audience on the server row; set `[oidc] obo_grant_profile` to `entra` or `rfc8693`. |
|
||||
| **`oauth_obo`**: `obo_mint_rejected` with `AADSTS65001` | Turnstone's app lacks the (admin-consented) delegated grant to this MCP app — often admin consent that didn't propagate | Grant + admin-consent the delegated permission for this resource; verify it shows *Granted*. See the Entra gotcha above. |
|
||||
| **`oauth_obo`**: "Sign in to Turnstone again" on one server | Captured credential missing/rejected, or a Conditional Access challenge | User re-logs into Turnstone (re-captures the credential). If it persists, check the IdP grant / CA policy. |
|
||||
| **`oauth_obo`**: tools don't appear at all for a user | User has not signed in since `capture_user_credential` was enabled (no credential captured) | User logs out and back in via OIDC so the refresh credential is captured. |
|
||||
|
||||
See also: `docs/operations/mcp-oauth-headless.md` for the cron / channel-driven run caveat.
|
||||
|
||||
+9
-68
@@ -77,17 +77,17 @@ IdP from redirecting the token-exchange POST (which carries
|
||||
being aimed at internal services.
|
||||
|
||||
A few public IdPs legitimately split endpoints across hostnames. Google
|
||||
and Microsoft Entra ID are the canonical examples:
|
||||
is the canonical example:
|
||||
|
||||
| IdP | Issuer host | Cross-host endpoint(s) |
|
||||
|-----|-------------|------------------------|
|
||||
| Google | `accounts.google.com` | `oauth2.googleapis.com`, `www.googleapis.com`, `openidconnect.googleapis.com` |
|
||||
| Microsoft Entra | `login.microsoftonline.com` | `graph.microsoft.com` (userinfo) |
|
||||
| Field | Hostname |
|
||||
|-------|----------|
|
||||
| issuer | `accounts.google.com` |
|
||||
| token_endpoint | `oauth2.googleapis.com` |
|
||||
| jwks_uri | `www.googleapis.com` |
|
||||
| userinfo_endpoint | `openidconnect.googleapis.com` |
|
||||
|
||||
Both sets are built in — operators using `https://accounts.google.com` or
|
||||
`https://login.microsoftonline.com/<tenant>/v2.0` need no extra
|
||||
configuration. (Entra's discovery document advertises `userinfo_endpoint`
|
||||
on `graph.microsoft.com`, distinct from the issuer host.)
|
||||
Google's set is built in — operators using `https://accounts.google.com`
|
||||
need no extra configuration.
|
||||
|
||||
For other IdPs whose discovery document references a non-issuer host,
|
||||
extend the allow-list explicitly:
|
||||
@@ -134,65 +134,6 @@ This knob only affects the login-flow IdP configured here. OAuth
|
||||
endpoints advertised by remote MCP servers are untrusted input and are
|
||||
always held to the strict public-address rule.
|
||||
|
||||
### Model gateway credentials
|
||||
|
||||
The same OIDC registration can authenticate model gateways. A model definition
|
||||
with `auth_mode = "entra_obo"` (Entra grant profile) or `auth_mode =
|
||||
"rfc8693_obo"` (RFC 8693 token-exchange profile) redeems the driving user's
|
||||
captured credential for its exact `obo_audience`; `auth_mode = "entra_app"`
|
||||
uses the registration's client ID and secret with Entra client credentials.
|
||||
All three bind the result through the provider SDK's native credential option
|
||||
rather than injecting an override header. The grant mode is never inferred:
|
||||
missing user context or a failed OBO mint cannot switch a delegated definition
|
||||
to client credentials.
|
||||
|
||||
Each dynamic mode pairs with the grant profile whose dialect it names:
|
||||
`entra_obo` and `entra_app` require `obo_grant_profile = "entra"`;
|
||||
`rfc8693_obo` requires `obo_grant_profile = "rfc8693"`. The pairing is
|
||||
enforced when a write chooses a `(auth_mode, obo_audience)` pair — a same-pair
|
||||
edit of a row saved before the pairing rule keeps working — and at runtime a
|
||||
mismatched legacy row refuses to mint with `cause=grant_profile_mismatch` and
|
||||
no IdP traffic. RFC 8693 client-credentials is not implemented.
|
||||
|
||||
The delegated modes need the MCP encryption key, a credential captured for the
|
||||
driving user, and delegated/admin-consented permission to the audience.
|
||||
`rfc8693_obo` additionally carries `obo_scopes`, the space-separated scope
|
||||
list its exchange leg requests: exchange-capable IdPs that gate audiences
|
||||
behind optional scopes refuse the exchange without it ("Requested audience not
|
||||
available"), which is why the scope-less Entra-named mode could never mint on
|
||||
that profile (issue #955). Scopes are stored shape-checked only — whether a
|
||||
value satisfies the IdP stays the IdP's call at mint time. Turning
|
||||
`capture_user_credential` off later stops *new* captures but does not
|
||||
invalidate credentials already stored, so existing users keep minting.
|
||||
`entra_app` requires a confidential-client secret. Configure the permitted
|
||||
resource IDs in the runtime setting `model.auth_audience_allowlist` before
|
||||
saving dynamic model definitions. De-listing an audience later blocks every
|
||||
write that would arm or re-aim a definition at it, but does not stop aliases
|
||||
already configured from minting — disabling the row (the `admin.models` disarm
|
||||
lever) is what stops minting. See
|
||||
[Settings](settings.md#model-backend-authentication) for permissions, failure
|
||||
policy, and lane identity rules.
|
||||
|
||||
An unrecognised `obo_grant_profile` is warned about at startup and **rejected
|
||||
at the write choke points**: configuring an `oauth_obo` MCP server or a dynamic
|
||||
model alias returns a 400 that echoes the configured value, so the typo is the
|
||||
diagnosis. At runtime an unknown profile never mints — the mint legs resolve by
|
||||
exact name; the full cause detail is logged once per audience, and every
|
||||
affected call still logs its per-turn fallback or refusal naming the alias,
|
||||
the target audience, and the last recorded cause (`cause=` — for example
|
||||
`unsupported_grant_profile` or `oidc_not_enabled`) — so a pre-existing row
|
||||
degrades loudly, with the reason visible mid-incident even after the
|
||||
once-per-process line has rotated out of retained logs, rather than silently
|
||||
swapping per-user attribution for the shared static key.
|
||||
|
||||
The `[security]` token encryption key is deployment-wide, not per-host: rows are
|
||||
encrypted with `MultiFernet` and carry no key id, so every host that reads them
|
||||
needs the same keyring. That includes the console, which mints for
|
||||
coordinator-hosted sessions. A node that needs the key and lacks it refuses to
|
||||
start; the console starts but withholds its coordinator subsystem and shows
|
||||
the key requirement as the remediation error instead of failing silently at
|
||||
call time.
|
||||
|
||||
### config.toml alternative
|
||||
|
||||
```toml
|
||||
|
||||
+1
-103
@@ -54,108 +54,6 @@ When a per-model override is `NULL` (empty in the UI), the global default is
|
||||
used. Switching models via `/model <alias>` re-resolves sampling parameters
|
||||
from the new model's overrides or global defaults.
|
||||
|
||||
### Model backend authentication
|
||||
|
||||
Model definitions support four backend credential modes:
|
||||
|
||||
| `auth_mode` | Identity sent to the model gateway |
|
||||
|-------------|------------------------------------|
|
||||
| `static` | The definition's stored `api_key`. |
|
||||
| `entra_obo` | A caller-delegated Entra access token minted from that user's captured OIDC credential. |
|
||||
| `entra_app` | A shared app-identity token minted with Turnstone's OIDC client credentials. |
|
||||
| `rfc8693_obo` | A caller-delegated access token minted from the captured credential via RFC 8693 token exchange, requesting the definition's `obo_scopes`. |
|
||||
|
||||
Dynamic modes require an exact `obo_audience` resource identifier. Before an
|
||||
admin can save one, an operator must add that literal audience to
|
||||
`model.auth_audience_allowlist` (comma- or newline-separated). Wildcards and
|
||||
base-URL host matching are intentionally unsupported, and a row whose
|
||||
effective mode is `static` refuses to store a new non-empty `obo_audience` on
|
||||
either create or update — an audience cannot be staged for a later flip
|
||||
(clearing a stale value, or re-saving it unchanged, stays allowed).
|
||||
`obo_scopes` follows the same staging rule with the mode set inverted: only
|
||||
`rfc8693_obo` reads it, so every other effective mode refuses to store a new
|
||||
non-empty value, while clearing or re-saving one unchanged stays open. The
|
||||
value itself is optional and shape-checked only — whether it satisfies the
|
||||
IdP is decided at mint time. On a row that is (or becomes) dynamic, every
|
||||
change except the tuning fields — context window, temperature, max tokens,
|
||||
reasoning effort, and the two reasoning-persistence toggles — also requires
|
||||
`admin.mcp`; service tokens do not bypass this capability-escalation gate.
|
||||
The one exception is de-escalation: a save whose only gated change is
|
||||
switching `enabled` off is a pure disable, needs only `admin.models`, and
|
||||
skips validation — a de-listed audience must never block disarming its own
|
||||
row. The gate is deny-by-default: a field counts as auth-relevant unless it
|
||||
is provably neutral, so re-enabling a disabled dynamic row, re-pointing its
|
||||
`base_url`, or swapping its provider or alias all escalate.
|
||||
|
||||
Validation runs in two tiers, matching the MCP `oauth_obo` write rules. Row
|
||||
validity — the audience is allow-listed — applies to every gated write that
|
||||
touches a dynamic configuration, so a revoked audience can be neither silently
|
||||
re-pointed at a new `base_url` nor re-armed by an enable flip. Deployment
|
||||
posture — the token encryption key installed, single sign-on configured, and
|
||||
the grant profile valid and able to carry the mode — is checked when a write
|
||||
*chooses* the mode/audience pair and when it re-enables a disabled dynamic
|
||||
row (arming is the flip that resumes minting, so it must meet what minting
|
||||
needs); other edits to an existing row stay open if the deployment's posture
|
||||
changed after it was saved (its mints warn at runtime instead). Refusals name
|
||||
their cause and echo the configured value.
|
||||
|
||||
One asymmetry to be aware of: the write path counts a transient discovery
|
||||
outage (`enabled=false`, retryable) as configured, but the mints themselves
|
||||
require discovery to have completed — a config saved during an outage starts
|
||||
minting only once any authenticated request heals discovery. Until then calls
|
||||
warn and follow the fail-open/fail-closed policy above.
|
||||
|
||||
Every dynamic mode pairs with exactly one grant profile: `entra_obo` and
|
||||
`entra_app` require `[oidc] obo_grant_profile = "entra"`, and `rfc8693_obo`
|
||||
requires `"rfc8693"`. The pairing is enforced at the posture tier, so a row
|
||||
saved before the rule existed keeps accepting same-pair edits; its mints
|
||||
refuse at runtime with `cause=grant_profile_mismatch` and no IdP traffic.
|
||||
Judge, output-guard, perception, utility, and sub-agent lanes inherit the
|
||||
session's effective user for the delegated modes. The perception memo is
|
||||
partitioned by that principal as well as alias and content hash, so a result
|
||||
authorized as one user cannot be served to another. Scheduled and wake-driven
|
||||
work retains the workstream owner even when no user is connected. Eval and
|
||||
optimizer lanes are registry-less development tools and therefore do not use
|
||||
dynamic model authentication.
|
||||
|
||||
`entra_app` is an explicit model-definition choice; Turnstone never changes a
|
||||
failed or ownerless delegated call into a client-credentials grant. A
|
||||
delegated-mode call with no effective user always refuses. A dynamic alias
|
||||
without a real static key also always refuses instead of issuing its
|
||||
SDK-construction placeholder. When a real static key is explicitly configured,
|
||||
mint failures may use it by default; set `model.auth_fail_closed = true` to
|
||||
prohibit even that fallback. A refusal is not routed through the model
|
||||
fallback chain.
|
||||
|
||||
Dynamic token caches are encrypted in `mcp_user_tokens`, shared across nodes,
|
||||
and memoized on each host. Unlinking a user's OIDC identity purges their
|
||||
delegated-mode rows and memo entries. `entra_app` rows belong to the shared
|
||||
`__app__` identity and are not user-deprovisioned; after client-credential
|
||||
revocation, an already-minted app bearer remains usable until its recorded
|
||||
expiry.
|
||||
|
||||
`obo_audience` and `obo_scopes` are literal and capped at 2048 characters
|
||||
each. Environment-variable expansion is deliberately not applied, so the
|
||||
allow-list decision cannot vary by node or expand beyond the persisted
|
||||
boundary.
|
||||
|
||||
### Responses output controls (per-model)
|
||||
|
||||
Models whose capability table declares Responses output controls expose two
|
||||
additional fields in the Models create/edit shelf:
|
||||
|
||||
| Field | Stored capability | Values | Effect |
|
||||
|-------|-------------------|--------|--------|
|
||||
| Output verbosity | `verbosity` | `low`, `medium`, `high` | Controls answer length independently of reasoning effort. |
|
||||
| Reasoning mode | `reasoning_mode` | `standard`, `pro` | Selects standard or higher-compute Pro execution without changing the model ID. |
|
||||
|
||||
An empty selection means provider default and omits the capability key. Known
|
||||
GPT-5.6 models inherit support from the built-in table without persisting
|
||||
redundant support flags. An OpenAI-compatible model pinned to the Responses API
|
||||
can opt in with the `supports_verbosity` and `supports_pro_mode` capability
|
||||
tiles. Chat Completions and non-Responses providers do not surface or submit
|
||||
these controls.
|
||||
|
||||
**Removed settings:** `model.name` and `model.context_window` have been removed
|
||||
from ConfigStore. Model names and context windows are now configured per-model
|
||||
in the Models tab. A startup warning is logged if these keys appear in
|
||||
@@ -209,7 +107,7 @@ initialization:
|
||||
|
||||
| Section | Settings |
|
||||
|---------|----------|
|
||||
| `model` | default_alias, auth_audience_allowlist, auth_fail_closed, temperature, max_tokens, reasoning_effort, task_alias, task_effort |
|
||||
| `model` | default_alias, temperature, max_tokens, reasoning_effort, task_alias, task_effort |
|
||||
| `session` | instructions, retention_days, compact_max_tokens, auto_compact_pct |
|
||||
| `tools` | timeout, truncation, agent_max_turns, skip_permissions, search, search_threshold, search_max_results |
|
||||
| `server` | workstream_idle_timeout, max_workstreams |
|
||||
|
||||
+12
-28
@@ -28,19 +28,13 @@ schema plus turnstone-specific metadata keys:
|
||||
}
|
||||
```
|
||||
|
||||
**Metadata keys** (stripped before sending the schema to the model; the full
|
||||
set lives in `_META_KEYS` in `turnstone/core/tools.py`):
|
||||
**Metadata keys** (stripped before sending the schema to the model):
|
||||
|
||||
| Key | Type | Meaning |
|
||||
|------------------|------|---------|
|
||||
| `task_agent` | bool | Tool is available to task sub-agents. |
|
||||
| `coordinator` | bool | Tool is available to coordinator sessions. Without `interactive: true` alongside it, this reads as coord-only and the tool is stripped from interactive sessions. |
|
||||
| `interactive` | bool | Opt a `coordinator: true` tool back into interactive sessions (dual-kind tools like `memory`). |
|
||||
| `auto_approve` | bool | Tool runs without user confirmation (read-only, safe operations). |
|
||||
| `primary_key` | str | When the model sends a bare string instead of JSON args, map it to this parameter name. |
|
||||
| `kind_variants` | dict | Per-kind description / parameter-schema overlays so each session kind sees only the surface it can use (see `memory.json`). |
|
||||
| `cwd_note` | str | Sentence appended to the description at session build time with `{working_dir}` substituted — declare on tools whose semantics depend on the process working directory (see `bash.json`, `apply_cwd_context`). |
|
||||
| `workspace_note` | str | Companion sentence naming the operator-configured workspace directory, `{workspace_dir}` substituted; dropped when no workspace is configured. |
|
||||
| Key | Type | Meaning |
|
||||
|----------------|------|---------|
|
||||
| `task_agent` | bool | Tool is available to task sub-agents. |
|
||||
| `auto_approve` | bool | Tool runs without user confirmation (read-only, safe operations). |
|
||||
| `primary_key` | str | When the model sends a bare string instead of JSON args, map it to this parameter name. |
|
||||
|
||||
---
|
||||
|
||||
@@ -785,10 +779,7 @@ MCP tool lists stay up-to-date without restart through two mechanisms:
|
||||
1. **Push notifications** -- MCP servers that declare `tools.listChanged: true` in
|
||||
their capabilities send `notifications/tools/list_changed` when their tool list
|
||||
changes. `MCPClientManager` registers a `message_handler` on each `ClientSession`
|
||||
that triggers an immediate refresh for that server (debounced per server and
|
||||
notification kind, and run off the receive loop). A refresh that fails while
|
||||
the connection stays up is retried automatically on the next health-loop tick
|
||||
until one completes.
|
||||
that triggers an immediate refresh for that server.
|
||||
|
||||
2. **Manual** -- `/mcp refresh` re-fetches tools from all servers immediately.
|
||||
`/mcp refresh <server>` targets a single server. If a server has disconnected,
|
||||
@@ -796,10 +787,6 @@ MCP tool lists stay up-to-date without restart through two mechanisms:
|
||||
same controls (refresh / reconnect buttons per server) for cluster-wide
|
||||
fan-out.
|
||||
|
||||
Reconnects (health-loop, dispatch-driven, or operator-forced) always end in a
|
||||
full catalog rediscovery, so a server that changed its tools while disconnected
|
||||
comes back current.
|
||||
|
||||
When tools change, `MCPClientManager` rebuilds its merged tool list using copy-on-write
|
||||
(new list/dict objects assigned atomically) and notifies all active `ChatSession`
|
||||
instances via registered listener callbacks. Each session rebuilds its `_tools`,
|
||||
@@ -870,16 +857,13 @@ catalog.
|
||||
|
||||
### Refresh
|
||||
|
||||
Resource lists stay current through the same mechanisms as tool lists:
|
||||
Resource lists stay current through the same three-tier mechanism as tool lists:
|
||||
|
||||
1. **Push** -- Servers declaring `resources.listChanged: true` send
|
||||
`notifications/resources/list_changed`, triggering an immediate refresh
|
||||
(with the same failed-refresh retry on the health-loop tick).
|
||||
2. **Manual** -- `/mcp refresh` re-fetches resources alongside tools.
|
||||
|
||||
Servers without push support are refreshed whenever they reconnect (every
|
||||
reconnect ends in full rediscovery) or when an operator refreshes manually;
|
||||
there is no periodic polling.
|
||||
`notifications/resources/list_changed`, triggering an immediate refresh.
|
||||
2. **Periodic** -- Servers without push are polled on the configured refresh
|
||||
interval (default 4 hours, same timer as tools).
|
||||
3. **Manual** -- `/mcp refresh` re-fetches resources alongside tools.
|
||||
|
||||
---
|
||||
|
||||
|
||||
+6
-7
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "turnstone"
|
||||
version = "1.8.0a6"
|
||||
version = "1.7.2"
|
||||
description = "Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
@@ -23,8 +23,8 @@ classifiers = [
|
||||
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
||||
]
|
||||
dependencies = [
|
||||
"openai>=2.45", # GPT-5.6: typed reasoning.mode, prompt_cache_options, and cache_write_tokens
|
||||
"anthropic>=0.117", # tracks the release current at claude-opus-5 onboarding; hard runtime floor is still 0.105 (mid-conversation system blocks) — Opus 5 itself needs no new SDK surface (model ids are opaque strings; "refusal" has been in the StopReason literal since ~0.95). Raise this when adopting fast mode / server-side fallbacks / advisor / mid-conversation tool changes, which DO need newer typed params.
|
||||
"openai>=2.37",
|
||||
"anthropic>=0.108", # claude-fable-5 support; hard runtime floor is 0.105 (mid-conversation system blocks)
|
||||
"httpx>=0.28",
|
||||
"mcp>=1.27,<2", # v2 is a breaking rewrite (2.0.0a1 live 2026-06-11; stable ~2026-07-27) — streamablehttp_client removed, 2-tuple transport, snake_case types; migrate deliberately
|
||||
"starlette>=1.3.1", # CVE-2026-54282 (path->authority host spoof) + CVE-2026-54283 (url-encoded form DoS); supersedes the PYSEC-2026-161 host-header path-injection floor
|
||||
@@ -88,10 +88,10 @@ include = [
|
||||
"turnstone/console/static/coordinator/*.js",
|
||||
"turnstone/shared_static/*.css",
|
||||
"turnstone/shared_static/*.js",
|
||||
"turnstone/shared_static/katex-0.18.1/**/*",
|
||||
"turnstone/shared_static/katex-0.17.0/**/*",
|
||||
"turnstone/shared_static/hljs-11.11.1/**/*",
|
||||
"turnstone/shared_static/mermaid-11.16.1/**/*",
|
||||
"turnstone/shared_static/hls-1.6.17/**/*",
|
||||
"turnstone/shared_static/mermaid-11.16.0/**/*",
|
||||
"turnstone/shared_static/hls-1.6.16/**/*",
|
||||
"turnstone/sdk/py.typed",
|
||||
"turnstone/deploy/*.yaml",
|
||||
"turnstone/deploy/Caddyfile",
|
||||
@@ -103,7 +103,6 @@ testpaths = ["tests"]
|
||||
markers = [
|
||||
"live: requires a running LLM backend",
|
||||
"allow_thread_leak: test intentionally leaves a background thread running (opts out of the leaked-thread guard)",
|
||||
"e2e_recovery: opt-in end-to-end SSE recovery harness (real server + real SSE consumers, scripted provider — NOT live, no LLM backend needed); tens of seconds each. CI lanes run ``-m 'not live and not e2e_recovery'``; select with ``-m e2e_recovery``.",
|
||||
]
|
||||
filterwarnings = [
|
||||
# mcp v1 deprecates streamablehttp_client for an entry point whose call
|
||||
|
||||
@@ -4,9 +4,8 @@
|
||||
#
|
||||
# curl -fsSL https://raw.githubusercontent.com/turnstonelabs/turnstone/main/run.sh | bash
|
||||
#
|
||||
# Autodetects your distro — Ubuntu/Debian, Fedora/RHEL, Arch, their common
|
||||
# derivatives (Mint, Pop!_OS, Nobara, AlmaLinux, …), and WSL on any of them —
|
||||
# and:
|
||||
# Autodetects your distro (Ubuntu/Debian, Fedora/RHEL, Arch, and WSL on any of
|
||||
# them) and:
|
||||
# 1. ensures git is installed, then clones the repo
|
||||
# 2. ensures Docker + the compose plugin are installed and the daemon is usable
|
||||
# 3. asks how many server nodes to run (1-10)
|
||||
@@ -66,18 +65,12 @@ ask() {
|
||||
|
||||
# -- distro / package manager detection --------------------------------------
|
||||
OS_ID=""; OS_LIKE=""; PKG=""; IS_WSL=0; SUDO=""
|
||||
# Extra os-release fields, captured only to pick Docker's upstream repo when
|
||||
# get.docker.com refuses a derivative it doesn't recognize (see install_docker).
|
||||
OS_PLATFORM_ID=""; OS_CODENAME=""; OS_UBUNTU_CODENAME=""
|
||||
|
||||
detect_os() {
|
||||
if [ -r /etc/os-release ]; then
|
||||
# shellcheck disable=SC1091
|
||||
. /etc/os-release
|
||||
OS_ID="${ID:-}"; OS_LIKE="${ID_LIKE:-}"
|
||||
OS_PLATFORM_ID="${PLATFORM_ID:-}"
|
||||
OS_CODENAME="${VERSION_CODENAME:-}"
|
||||
OS_UBUNTU_CODENAME="${UBUNTU_CODENAME:-}"
|
||||
fi
|
||||
if grep -qiE 'microsoft|wsl' /proc/version 2>/dev/null || [ -n "${WSL_DISTRO_NAME:-}" ]; then
|
||||
IS_WSL=1
|
||||
@@ -137,83 +130,11 @@ clone_repo() {
|
||||
# -- docker -------------------------------------------------------------------
|
||||
DOCKER="docker"
|
||||
|
||||
# Fallback when get.docker.com won't install here. That script keys off $ID alone
|
||||
# (never ID_LIKE), so it aborts with "Unsupported distribution '<id>'" on every
|
||||
# derivative — Nobara, Linux Mint, Pop!_OS, AlmaLinux, Oracle Linux, … — even
|
||||
# though the family is clear. We already know the family from detect_os, so we add
|
||||
# Docker's official CE repo for the matching upstream and install the same
|
||||
# packages get.docker.com would (including the compose plugin the rest of run.sh
|
||||
# relies on).
|
||||
install_docker_ce_repo() {
|
||||
local up
|
||||
case "$PKG" in
|
||||
apt)
|
||||
local codename arch
|
||||
# UBUNTU_CODENAME is set by Ubuntu and every Ubuntu-derived distro
|
||||
# (Mint/Pop!_OS/Zorin/…) and never by pure Debian, so it both routes
|
||||
# the family and gives the exact codename Docker's repo expects.
|
||||
if [ -n "$OS_UBUNTU_CODENAME" ]; then
|
||||
up=ubuntu; codename="$OS_UBUNTU_CODENAME"
|
||||
else
|
||||
up=debian; codename="$OS_CODENAME"
|
||||
fi
|
||||
[ -n "$codename" ] || die "couldn't determine the $up release codename for Docker's repo — install Docker manually and re-run."
|
||||
arch="$(dpkg --print-architecture 2>/dev/null || echo amd64)"
|
||||
info "Adding Docker's $up repository ($codename)."
|
||||
$SUDO install -m 0755 -d /etc/apt/keyrings
|
||||
curl -fsSL "https://download.docker.com/linux/$up/gpg" | $SUDO tee /etc/apt/keyrings/docker.asc >/dev/null
|
||||
$SUDO chmod a+r /etc/apt/keyrings/docker.asc
|
||||
printf 'deb [arch=%s signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/%s %s stable\n' \
|
||||
"$arch" "$up" "$codename" | $SUDO tee /etc/apt/sources.list.d/docker.list >/dev/null
|
||||
$SUDO apt-get update -y
|
||||
$SUDO apt-get install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin
|
||||
;;
|
||||
dnf|yum)
|
||||
# A Fedora spin and a RHEL clone can both carry "fedora" in ID_LIKE
|
||||
# (Nobara's is "rhel centos fedora"), so ID_LIKE can't separate them.
|
||||
# PLATFORM_ID can: Fedora is platform:fNN, Enterprise Linux platform:elN.
|
||||
case "$OS_PLATFORM_ID" in
|
||||
platform:f*) up=fedora ;;
|
||||
platform:el*) up=centos ;;
|
||||
*) if [ -e /etc/fedora-release ]; then up=fedora; else up=centos; fi ;;
|
||||
esac
|
||||
info "Adding Docker's $up repository."
|
||||
$SUDO curl -fsSL "https://download.docker.com/linux/$up/docker-ce.repo" \
|
||||
-o /etc/yum.repos.d/docker-ce.repo \
|
||||
|| die "couldn't add Docker's $up repository — install Docker manually and re-run."
|
||||
pkg_install docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# The distro IDs get.docker.com installs directly: it matches $ID against this
|
||||
# exact set (ignoring ID_LIKE) and aborts on anything else. Mirrors the dispatch
|
||||
# in get.docker.com, including its fedora-asahi-remix -> fedora alias.
|
||||
get_docker_com_supports() {
|
||||
case "$1" in
|
||||
ubuntu|debian|raspbian|centos|fedora|rhel|rocky|sles|fedora-asahi-remix) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
install_docker() {
|
||||
case "$PKG" in
|
||||
apt|dnf|yum)
|
||||
# Decide up front which installer applies, rather than treating every
|
||||
# get.docker.com failure as "unsupported distro": for an ID it knows,
|
||||
# let it run and surface any real failure (network, apt lock, EOL) via
|
||||
# die instead of masking it with the repo path. Only unrecognized
|
||||
# derivatives (Nobara, Mint, …) — which it would just abort on — skip
|
||||
# straight to adding Docker's repo ourselves.
|
||||
if [ -n "$OS_ID" ] && ! get_docker_com_supports "$OS_ID"; then
|
||||
info "get.docker.com doesn't support '$OS_ID' — using Docker's official repository directly."
|
||||
install_docker_ce_repo
|
||||
else
|
||||
info "Installing Docker via the official get.docker.com script"
|
||||
curl -fsSL https://get.docker.com | $SUDO sh \
|
||||
|| die "get.docker.com failed to install Docker (see the output above). Fix the issue and re-run — the script resumes."
|
||||
fi
|
||||
;;
|
||||
info "Installing Docker via the official get.docker.com script"
|
||||
curl -fsSL https://get.docker.com | $SUDO sh ;;
|
||||
pacman)
|
||||
pkg_install docker docker-compose ;;
|
||||
esac
|
||||
@@ -445,15 +366,12 @@ ${GREEN}${BOLD}Turnstone is running${RESET} (${NODE_COUNT} node$([ "$NODE_COUNT"
|
||||
${DIM}cd $INSTALL_DIR && $DOCKER compose exec caddy cat /data/caddy/pki/authorities/local/root.crt${RESET}
|
||||
|
||||
Finish setup
|
||||
1. Open ${BOLD}${url}${RESET} and create the admin account when prompted —
|
||||
the first user created there gets full admin access.
|
||||
2. Log in, then add a model backend in the ${BOLD}Models${RESET} tab —
|
||||
1. Create the first admin user:
|
||||
${DIM}cd $INSTALL_DIR && $DOCKER compose exec node-1 turnstone-admin create-user --username admin --name "Admin"${RESET}
|
||||
2. Open ${url}, log in, and add a model backend in the ${BOLD}Models${RESET} tab —
|
||||
a local server (vLLM / llama.cpp) or an OpenAI / Anthropic / Gemini key.
|
||||
Nodes boot without a model and pick it up live; no restart needed.
|
||||
|
||||
${DIM}No browser? Create the admin from the CLI instead:
|
||||
cd $INSTALL_DIR && $DOCKER compose exec node-1 turnstone-admin create-admin --username admin --name "Admin"${RESET}
|
||||
|
||||
Scale Running ${scale}
|
||||
|
||||
Manage ${DIM}cd $INSTALL_DIR${RESET}
|
||||
|
||||
+14
-519
@@ -45,22 +45,6 @@ Shell harness (?split=): right (default) · down · three · none — boots the
|
||||
document.title stamps SPLIT-READY-<visible cells> on success and
|
||||
SPLIT-FAILED-<reason> when a driven split was denied — judge the focused
|
||||
cell's top accent bar, the separators, and the .shown tab marker.
|
||||
Proxy-brand harness (/proxybrand/livepass.html): back-to-console from a
|
||||
PROXIED node view, driven end to end. An iframe hosts a node page built
|
||||
from the REAL shell.js rail plus the REAL _JS_PROXY_SHIM (read out of
|
||||
turnstone/console/server.py by text, never imported -- scripts/ has no
|
||||
sys.path guard, so an import would silently pick up site-packages). The
|
||||
host clicks the brand's child span and, because the shim navigates the
|
||||
FRAME away, reads the frame's post-navigation location from the surviving
|
||||
top page. document.title stamps PROXYBRAND-READY, or
|
||||
PROXYBRAND-FAILED-<reason>: sub-not-repointed-server (nothing wired),
|
||||
showhome-also-ran (shell.js won the click), nav-<path> (went somewhere
|
||||
other than the console root), sub-not-console-<text>, aria-not-repointed,
|
||||
no-navigation, no-brand, no-sub. Needs --virtual-time-budget=9000;
|
||||
there is nothing to screenshot. Read the verdict from <title> --
|
||||
both literals also appear in the host page's inline script, so a bare
|
||||
grep over --dump-dom output false-positives.
|
||||
|
||||
Attachments harness (/attachments/livepass.html): the composer attachment
|
||||
chips + the sent-message attachment pills, both driven through the REAL
|
||||
code paths — createAttachmentController.rehydrate() builds the chips and
|
||||
@@ -95,30 +79,6 @@ Task-agent harness (/taskagent/livepass.html): the task_agent card — a task
|
||||
success, TASKAGENT-FAILED-... / TASKAGENT-ERROR when routing breaks, so a
|
||||
broken card can't screenshot green.
|
||||
|
||||
Copy harness (/copy/livepass.html): the copy-to-clipboard affordances — the
|
||||
per-bubble copy button in .msg-actions and the floating block-copy button
|
||||
over hovered fences / mermaid diagrams / tables (pointer-only; keyboard
|
||||
copies with Enter on the focused block) — driven through the REAL
|
||||
InteractivePane (replayHistory plus a live handleEvent stream turn, so the
|
||||
retry-holder buttons coexist with the persistent copy buttons on the last
|
||||
bubble; the turn ends idle, matching the affordances' idle-only gate).
|
||||
navigator.clipboard is stubbed to a recorder, hover/focus/keys are
|
||||
dispatched synthetically, and every copied payload is compared byte-exact
|
||||
against the SOURCE (fences, pipes, mermaid text, the bubble's raw
|
||||
markdown). + &theme=light. document.title stamps
|
||||
COPY-READY-<bubbles>-<blocks> only when every probe copied exact source;
|
||||
COPY-FAILED-<reason> otherwise. &kbd=1 probes the KEYBOARD path: focus a
|
||||
block, dispatch Enter — the block's source lands on the clipboard, the
|
||||
block carries the outcome flash class, and the floating button stays out
|
||||
of it — stamps COPY-KBD-READY / COPY-KBD-FAILED-<step>.
|
||||
Screenshot states: &flash=1 (visual-only
|
||||
run — no probes; floating button + ✓ state on the fence, holder bar
|
||||
revealed via focus) and &bare=1 (single hover, no decoration). Known
|
||||
capture artifact: the DARK-theme &flash=1 shot can omit the floating
|
||||
button's pixels (headless software compositor; the DOM state is correct
|
||||
and light theme paints) — judge the dark floating button from &bare=1
|
||||
and the ✓ state from the light shot. &stepmax=N bisects a paint
|
||||
regression to the interaction that triggers it.
|
||||
Perf harness (/perf/livepass.html): long-session performance baseline for the
|
||||
interactive pane — mounts the REAL InteractivePane at real scroll geometry
|
||||
(fixed-height mount, production CSS chain) and drives production-shaped
|
||||
@@ -183,24 +143,6 @@ def extract_admin_fragment() -> str:
|
||||
return html[start:end]
|
||||
|
||||
|
||||
def extract_proxy_shim(prefix: str = "/node/livepass-node") -> str:
|
||||
"""Pull ``_JS_PROXY_SHIM`` out of console/server.py BY TEXT, not import.
|
||||
|
||||
``scripts/`` has no ``sys.path`` guard, so ``import turnstone`` from here
|
||||
resolves to whatever is installed in site-packages rather than this
|
||||
checkout -- silently building the page from a DIFFERENT version of the
|
||||
shim than the one you are trying to verify. Read the source instead.
|
||||
"""
|
||||
src = (ROOT / "turnstone/console/server.py").read_text(encoding="utf-8")
|
||||
m = re.search(r'^_JS_PROXY_SHIM = """\\\n(.*?)^"""', src, re.S | re.M)
|
||||
if not m:
|
||||
raise SystemExit(
|
||||
"livepass: could not find _JS_PROXY_SHIM in turnstone/console/server.py "
|
||||
"-- the constant was renamed or reshaped; update extract_proxy_shim()."
|
||||
)
|
||||
return m.group(1).replace('"PREFIX_PLACEHOLDER"', json.dumps(prefix))
|
||||
|
||||
|
||||
def inject(template: str, marker: str, payload: str) -> str:
|
||||
begin = template.index(f"<!-- {marker}:BEGIN -->") + len(f"<!-- {marker}:BEGIN -->")
|
||||
end = template.index(f"<!-- {marker}:END -->")
|
||||
@@ -405,28 +347,6 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
<div id="toast" role="status" aria-live="polite"></div>
|
||||
<script>
|
||||
(function () {
|
||||
// Freeze window.fetch BEFORE the module scripts evaluate: auth.js
|
||||
// fires a boot-time whoami at import, and a non-OK answer from the
|
||||
// fixture server would CLEAR the permissions grant seeded below
|
||||
// mid-pass. A never-settling fetch keeps the seed authoritative;
|
||||
// everything the passes drive flows through the authFetch fixture
|
||||
// (reinstated after auth.js's window bridge runs — see the load
|
||||
// handler).
|
||||
window.fetch = function () {
|
||||
return new Promise(function () {});
|
||||
};
|
||||
// Grant the operator scopes admin.js gates on: _modelAuthEditable()
|
||||
// reads this exact key THROUGH the real auth.js hasPermission
|
||||
// (loaded below, before admin.js) — without the grant, or without
|
||||
// auth.js supplying window.hasPermission, the auth-constraints
|
||||
// stub below is dead code: _fetchModelAuthConstraints returns
|
||||
// before authFetch and every pass renders the Backend-auth section
|
||||
// in its read-only degraded state. The headless profile is fresh
|
||||
// per pass, so nothing else seeds it.
|
||||
sessionStorage.setItem(
|
||||
"turnstone_permissions",
|
||||
"admin.models,admin.mcp",
|
||||
);
|
||||
function reply(data) {
|
||||
return Promise.resolve({
|
||||
ok: true,
|
||||
@@ -452,15 +372,9 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
enabled: true, temperature: null, max_tokens: null,
|
||||
reasoning_effort: null, surface_persisted_reasoning: true,
|
||||
replay_reasoning_to_model: false,
|
||||
auth_mode: "static", obo_audience: "", obo_scopes: "",
|
||||
};
|
||||
window.__putCount = 0;
|
||||
// Held under a private name too: auth.js's legacy window bridge
|
||||
// (Object.assign(window, {authFetch})) runs at module-import time
|
||||
// and clobbers the plain window.authFetch assigned here — the load
|
||||
// handler reinstates the fixture from this name after the modules
|
||||
// have evaluated.
|
||||
window.__consoleAuthFetch = window.authFetch = function (url, opts) {
|
||||
window.authFetch = function (url, opts) {
|
||||
var method = (opts && opts.method) || "GET";
|
||||
if (method === "PUT" && url.indexOf("/model-definitions/def1") >= 0) {
|
||||
window.__putCount++;
|
||||
@@ -485,31 +399,13 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
known: true,
|
||||
capabilities: {
|
||||
context_window: 200000, supports_tools: true,
|
||||
supports_vision: true,
|
||||
supports_streaming: true, supports_vision: true,
|
||||
supports_web_search: true, supports_temperature: true,
|
||||
supports_effort: true,
|
||||
},
|
||||
});
|
||||
if (url.indexOf("/model-definitions/auth-constraints") >= 0)
|
||||
// Fetched by the shelf ON OPEN (showCreateModelModal /
|
||||
// showEditModelModal), so this stub is exercised by any pass that
|
||||
// opens the model editor — no tab-switch plumbing needed. Omitting
|
||||
// it would render the Backend-auth block in its degraded
|
||||
// no-suggestions state and quietly stop exercising the section.
|
||||
return reply({
|
||||
auth_audience_allowlist: ["api://example-gateway"],
|
||||
auth_grant_profile: "entra",
|
||||
dynamic_auth_modes: ["entra_app", "entra_obo", "rfc8693_obo"],
|
||||
scopes_auth_modes: ["rfc8693_obo"],
|
||||
app_identity_auth_modes: ["entra_app"],
|
||||
auth_mode_profiles: {
|
||||
entra_app: "entra", entra_obo: "entra",
|
||||
rfc8693_obo: "rfc8693",
|
||||
},
|
||||
});
|
||||
if (url.indexOf("/model-definitions/def1") >= 0) return reply(MODEL);
|
||||
if (url.indexOf("/model-definitions") >= 0)
|
||||
return reply({ models: [], default_alias: "fable-5" });
|
||||
if (url.indexOf("/model-definitions") >= 0) return reply({ models: [] });
|
||||
if (url.indexOf("/api/models") >= 0)
|
||||
return reply({ models: [
|
||||
{ alias: "fable-5", model: "claude-fable-5" },
|
||||
@@ -536,22 +432,10 @@ CONSOLE_TEMPLATE = """<!doctype html>
|
||||
</script>
|
||||
<script type="module" src="shared/utils.js"></script>
|
||||
<script type="module" src="shared/hatch.js"></script>
|
||||
<!-- The REAL auth.js, loaded (and therefore parsed) before admin.js's
|
||||
permission shims run any pass: it owns the sessionStorage parse
|
||||
contract and assigns the window.hasPermission /
|
||||
window.whenPermissionsReady globals the shims probe at call time.
|
||||
Without it the seeded permissions grant is never READ, the
|
||||
Backend-auth section renders read-only/hidden, and the
|
||||
auth-constraints stub above is dead code in every pass. -->
|
||||
<script type="module" src="shared/auth.js"></script>
|
||||
<script src="console-static/admin.js"></script>
|
||||
<script src="console-static/governance.js"></script>
|
||||
<script>
|
||||
window.addEventListener("load", function () {
|
||||
// Reinstate the fixture fetch now the modules (and auth.js's
|
||||
// window bridge) have evaluated — passes run after load, so every
|
||||
// shelf-open fetch flows through the fixture, not the bridge.
|
||||
window.authFetch = window.__consoleAuthFetch;
|
||||
var q = new URLSearchParams(location.search);
|
||||
if (q.get("theme") === "light")
|
||||
document.documentElement.dataset.theme = "light";
|
||||
@@ -774,108 +658,6 @@ SHELL_TEMPLATE = """<!doctype html>
|
||||
# call the same window.buildAttachmentPreview). The page frame is harness-only
|
||||
# chrome and not under review; the chips row and the pill row are.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
# The PROXIED NODE page: the real L-shell (so the rail brand is the real
|
||||
# element, with the real shell.js click listener on it) plus the real proxy
|
||||
# shim injected exactly where proxy_index puts it -- first thing inside
|
||||
# <body>, ahead of the deferred shell.js module. caps mirror a NODE, not
|
||||
# the console: brandSub "server" is what the shim has to overwrite, and
|
||||
# leaving it "console" would make the host's /console/i check vacuous.
|
||||
PROXYBRAND_FRAME_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>proxied node</title>
|
||||
<link rel="stylesheet" href="shared/base.css" />
|
||||
<link rel="stylesheet" href="shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="static/style.css" />
|
||||
<link rel="stylesheet" href="shared/shell.css" />
|
||||
</head>
|
||||
<body>
|
||||
<!-- SHIM:BEGIN -->
|
||||
<!-- SHIM:END -->
|
||||
<div id="header" style="display: none"><div id="status-bar"></div></div>
|
||||
<div id="main" style="padding: 18px">
|
||||
<h2 style="margin: 0 0 8px">Node dashboard</h2>
|
||||
</div>
|
||||
<div id="view-admin" style="display: none"></div>
|
||||
<script>
|
||||
window.TURNSTONE_SHELL_CAPS = { cluster: false, brandSub: "server" };
|
||||
window.TS_APP = {
|
||||
boot() {},
|
||||
getClusterState() { return { nodes: {} }; },
|
||||
onRender() {},
|
||||
};
|
||||
window.TS_ADMIN = {};
|
||||
// Record on the PARENT, which survives the frame's navigation.
|
||||
// A flag on the frame's own window dies with the document, so the
|
||||
// host would read undefined and pass -- a check that cannot fail.
|
||||
window.showHome = function () {
|
||||
try { window.parent.__showHomeRan = true; } catch (e) {}
|
||||
};
|
||||
</script>
|
||||
<script type="module" src="shared/shell.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
# The HOST page. The shim navigates the FRAME to "/", which would destroy
|
||||
# any verdict stamped inside it -- so the surviving top page reads the
|
||||
# frame's post-navigation location and stamps its own title instead. No
|
||||
# landing page at "/" is needed (the harness root serves a directory
|
||||
# listing) and no CDP client either.
|
||||
PROXYBRAND_HOST_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<title>proxybrand livepass</title>
|
||||
<style>
|
||||
html, body { margin: 0; height: 100%; }
|
||||
iframe { width: 100%; height: 100%; border: 0; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<iframe id="frame" src="frame.html"></iframe>
|
||||
<script>
|
||||
const frame = document.getElementById("frame");
|
||||
let phase = 0;
|
||||
const fail = (r) => { phase = 9; document.title = "PROXYBRAND-FAILED-" + r; };
|
||||
|
||||
frame.addEventListener("load", () => {
|
||||
if (phase === 9) return;
|
||||
if (phase === 0) {
|
||||
const doc = frame.contentDocument;
|
||||
const brand = doc.querySelector(".rail-brand .brand-home");
|
||||
if (!brand) return fail("no-brand");
|
||||
const sub = brand.querySelector(".brand-sub");
|
||||
if (!sub) return fail("no-sub");
|
||||
const text = sub.textContent.trim();
|
||||
if (text === "server") return fail("sub-not-repointed-server");
|
||||
if (!/console/i.test(text)) return fail("sub-not-console-" + text);
|
||||
if (brand.getAttribute("aria-label") !== "Back to console")
|
||||
return fail("aria-not-repointed");
|
||||
phase = 1;
|
||||
// Click the CHILD span, as a real user does: the shim must match
|
||||
// via contains(), not target identity.
|
||||
sub.click();
|
||||
setTimeout(() => { if (phase === 1) fail("no-navigation"); }, 2000);
|
||||
return;
|
||||
}
|
||||
const path = frame.contentWindow.location.pathname;
|
||||
const ranShowHome = !!window.__showHomeRan;
|
||||
phase = 2;
|
||||
if (path !== "/") return fail("nav-" + path);
|
||||
if (ranShowHome) return fail("showhome-also-ran");
|
||||
// Sticky, mirroring fail(): a third load must not re-stamp.
|
||||
phase = 9;
|
||||
document.title = "PROXYBRAND-READY";
|
||||
});
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
ATTACH_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
@@ -1039,24 +821,7 @@ ATTACH_TEMPLATE = """<!doctype html>
|
||||
# is exercised, not just the leaf builders. The page frame is harness-only
|
||||
# chrome; the .conv-batch / task_agent card is what's under review.
|
||||
# --------------------------------------------------------------------------
|
||||
# The host seams a mounted InteractivePane provides, stubbed once for every
|
||||
# harness that drives the REAL pane (taskagent, copy). A new required seam
|
||||
# gets added HERE — a harness left with a stale stub set does not fail at
|
||||
# review time, it throws HARNESS ERROR at run time.
|
||||
PANE_STUB_JS = """\
|
||||
// Drive the REAL pane; stub only the host seams a mounted pane provides.
|
||||
const pane = new InteractivePane("demo-ws");
|
||||
pane.messagesEl = messages;
|
||||
pane.inputEl = document.createElement("textarea");
|
||||
pane.sendBtn = document.createElement("button");
|
||||
pane.isNearBottom = () => false;
|
||||
pane.scrollToBottom = () => {};
|
||||
pane.removeEmptyState = () => {};
|
||||
pane.removeThinkingIndicator = () => {};
|
||||
pane.setBusy = () => {};"""
|
||||
|
||||
TASKAGENT_TEMPLATE = (
|
||||
"""<!doctype html>
|
||||
TASKAGENT_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
@@ -1106,9 +871,16 @@ TASKAGENT_TEMPLATE = (
|
||||
|
||||
const messages = document.getElementById("messages");
|
||||
try {
|
||||
"""
|
||||
+ PANE_STUB_JS
|
||||
+ """
|
||||
// Drive the REAL pane; stub only the host seams a mounted pane provides.
|
||||
const pane = new InteractivePane("demo-ws");
|
||||
pane.messagesEl = messages;
|
||||
pane.inputEl = document.createElement("textarea");
|
||||
pane.sendBtn = document.createElement("button");
|
||||
pane.isNearBottom = () => false;
|
||||
pane.scrollToBottom = () => {};
|
||||
pane.removeEmptyState = () => {};
|
||||
pane.removeThinkingIndicator = () => {};
|
||||
pane.setBusy = () => {};
|
||||
const ev = (e) => pane.handleEvent(e);
|
||||
|
||||
// ?recall=1: exercise the RECALL path — replayHistory rebuilding the
|
||||
@@ -1257,266 +1029,6 @@ TASKAGENT_TEMPLATE = (
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Copy harness — the copy-to-clipboard affordances over the REAL pane. The
|
||||
# bubbles come from the REAL replayHistory / handleEvent paths so the copy
|
||||
# sources are the ones production stashes (_copySource, the mermaid / table
|
||||
# data attributes), and the probes drive the REAL buttons and key path and
|
||||
# compare what landed on the (stubbed) clipboard byte-exact against the
|
||||
# source.
|
||||
# --------------------------------------------------------------------------
|
||||
COPY_TEMPLATE = (
|
||||
"""<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
||||
<title>copy livepass</title>
|
||||
<link rel="stylesheet" href="shared/base.css" />
|
||||
<link rel="stylesheet" href="shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="shared/chat.css" />
|
||||
<link rel="stylesheet" href="shared/conversation.css" />
|
||||
<link rel="stylesheet" href="shared/cards.css" />
|
||||
<link rel="stylesheet" href="shared/interactive.css" />
|
||||
<style>
|
||||
/* Harness-only framing (NOT under review) — a plausible pane context. */
|
||||
body {
|
||||
padding: 24px; margin: 0; background: var(--bg); color: var(--ink);
|
||||
font-family: var(--font-sans, system-ui, sans-serif);
|
||||
}
|
||||
.demo-frame { max-width: 720px; margin: 0 auto; }
|
||||
.demo-label {
|
||||
font: 11px var(--font-mono, monospace); color: var(--ink-3);
|
||||
text-transform: uppercase; letter-spacing: 0.08em; margin: 0 0 8px;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="demo-frame">
|
||||
<div class="demo-label">conversation — copy affordances (real InteractivePane)</div>
|
||||
<div class="messages" id="messages"></div>
|
||||
</div>
|
||||
<script>
|
||||
window.toast = { error: function (m) { console.log("toast:", m); } };
|
||||
window.authFetch = function () {
|
||||
return Promise.resolve({
|
||||
ok: true,
|
||||
json: function () { return Promise.resolve({}); },
|
||||
text: function () { return Promise.resolve(""); },
|
||||
});
|
||||
};
|
||||
// Deterministic clipboard: record instead of writing. localhost is a
|
||||
// secure context so copyTextToClipboard takes the async-API branch and
|
||||
// hits this stub; force isSecureContext for any odd serving setup.
|
||||
window.__copied = [];
|
||||
try {
|
||||
Object.defineProperty(window, "isSecureContext", { value: true });
|
||||
} catch (e) { /* already true */ }
|
||||
try {
|
||||
Object.defineProperty(navigator, "clipboard", {
|
||||
value: {
|
||||
writeText: function (t) {
|
||||
window.__copied.push(t);
|
||||
return Promise.resolve();
|
||||
},
|
||||
},
|
||||
configurable: true,
|
||||
});
|
||||
} catch (e) {
|
||||
document.title = "COPY-FAILED-clipboard-stub";
|
||||
}
|
||||
</script>
|
||||
<script type="module">
|
||||
import { InteractivePane } from "./shared/interactive.js";
|
||||
const q = new URLSearchParams(location.search);
|
||||
if (q.get("theme") === "light")
|
||||
document.documentElement.dataset.theme = "light";
|
||||
|
||||
const FENCE_SRC = 'def stash(depth):\\n total = 0\\n for k in range(depth):\\n total += k\\n return total';
|
||||
const TABLE_SRC = '| node | state |\\n|---|:--:|\\n| flat | idle |\\n| blck | busy |';
|
||||
const MERMAID_SRC = 'graph TD\\n A --> B\\n B --> C';
|
||||
const MD_ONE =
|
||||
'First reply with a fence and a table.\\n\\n' +
|
||||
'```python\\n' + FENCE_SRC + '\\n```\\n\\n' +
|
||||
TABLE_SRC + '\\n\\nTrailing prose under the table.';
|
||||
const MD_TWO =
|
||||
'Second reply with a diagram.\\n\\n' +
|
||||
'```mermaid\\n' + MERMAID_SRC + '\\n```\\n\\n' +
|
||||
'And `inline code` after it.';
|
||||
const MD_LIVE =
|
||||
'Streamed reply: the **live** turn, so the retry holder lands here.';
|
||||
|
||||
const messages = document.getElementById("messages");
|
||||
const fail = (r) => { document.title = "COPY-FAILED-" + r; };
|
||||
try {
|
||||
"""
|
||||
+ PANE_STUB_JS
|
||||
+ """
|
||||
|
||||
pane.replayHistory([
|
||||
{ role: "user", content: "Show me the stash helper and the node table." },
|
||||
{ role: "assistant", content: MD_ONE },
|
||||
{ role: "user", content: "Now the flow as a diagram, please." },
|
||||
{ role: "assistant", content: MD_TWO },
|
||||
]);
|
||||
// A live streamed turn on top — the retry holder must land on this
|
||||
// bubble WITHOUT stripping its (or any) copy button.
|
||||
pane.handleEvent({ type: "state_change", state: "running" });
|
||||
for (let k = 0; k < MD_LIVE.length; k += 16)
|
||||
pane.handleEvent({ type: "content", text: MD_LIVE.slice(k, k + 16) });
|
||||
pane.handleEvent({ type: "stream_end" });
|
||||
pane.handleEvent({ type: "state_change", state: "idle" });
|
||||
|
||||
const hover = (el) =>
|
||||
el.dispatchEvent(new MouseEvent("mouseover", { bubbles: true }));
|
||||
const fabEl = () => document.querySelector(".block-copy-btn");
|
||||
|
||||
// Let the streamed bubble's rAF render + retry attach settle.
|
||||
setTimeout(async () => {
|
||||
try {
|
||||
const bubbles = messages.querySelectorAll(".msg.assistant");
|
||||
const bars = messages.querySelectorAll(
|
||||
".msg.assistant .msg-actions .msg-copy-btn",
|
||||
);
|
||||
if (bubbles.length !== 3) return fail("bubbles" + bubbles.length);
|
||||
if (bars.length !== 3) return fail("bars" + bars.length);
|
||||
const last = bubbles[bubbles.length - 1];
|
||||
if (!last.querySelector(".msg-retry-btn"))
|
||||
return fail("no-retry-on-holder");
|
||||
if (!last.querySelector(".msg-copy-btn"))
|
||||
return fail("holder-lost-copy");
|
||||
|
||||
// Block probes: hover reveals the floating button; a click must
|
||||
// land the byte-exact SOURCE on the clipboard.
|
||||
const probes = [
|
||||
[messages.querySelector(".msg.assistant pre"), FENCE_SRC, "fence"],
|
||||
[messages.querySelector(".table-wrap"), TABLE_SRC, "table"],
|
||||
[messages.querySelector(".mermaid-container"), MERMAID_SRC, "mermaid"],
|
||||
];
|
||||
// &bare=1 — diagnostic state: no probe clicks, no repositioning;
|
||||
// one hover on the fence and stop. Splits "the probe cycle
|
||||
// corrupts the button's paint" from "it never paints here".
|
||||
if (q.get("bare") === "1") {
|
||||
hover(probes[0][0]);
|
||||
document.title = "COPY-BARE";
|
||||
return;
|
||||
}
|
||||
|
||||
// &kbd=1 — the keyboard path: Enter on a FOCUSED block copies
|
||||
// that block's source directly. Blocks are focusable (tabindex=0
|
||||
// from the fence / table / mermaid renders), the outcome flashes
|
||||
// on the block itself, and the floating button — pointer-only —
|
||||
// must stay out of it entirely (never created, never revealed).
|
||||
if (q.get("kbd") === "1") {
|
||||
const tw = messages.querySelector(".table-wrap");
|
||||
if (!tw) return fail("kbd-no-block");
|
||||
tw.focus();
|
||||
const focused = document.activeElement === tw;
|
||||
tw.dispatchEvent(
|
||||
new KeyboardEvent("keydown", { key: "Enter", bubbles: true }),
|
||||
);
|
||||
await new Promise((r) => setTimeout(r, 0));
|
||||
const copied =
|
||||
window.__copied[window.__copied.length - 1] === TABLE_SRC;
|
||||
const flashed = tw.classList.contains("is-copied");
|
||||
const fabStaysOut =
|
||||
!fabEl() || !fabEl().classList.contains("is-visible");
|
||||
document.title =
|
||||
focused && copied && flashed && fabStaysOut
|
||||
? "COPY-KBD-READY"
|
||||
: "COPY-KBD-FAILED-" +
|
||||
[
|
||||
focused ? "" : "focus",
|
||||
copied ? "" : "copy",
|
||||
flashed ? "" : "flash",
|
||||
fabStaysOut ? "" : "fab",
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join("-");
|
||||
return;
|
||||
}
|
||||
|
||||
// &flash=1 — the VISUAL state, screenshot-only: skip the probes so
|
||||
// the fence hover is the floating button's FIRST show. Returning
|
||||
// the button to an already-visited position stops it PAINTING in
|
||||
// headless captures (visible + hit-testable, no pixels — a stale
|
||||
// compositor tile; bisected via &stepmax). Function and pixels
|
||||
// are therefore split: the probe run (no flash) is the verdict,
|
||||
// this state is the picture.
|
||||
if (q.get("flash") === "1") {
|
||||
bars[bars.length - 1].focus();
|
||||
hover(probes[0][0]);
|
||||
const fab = fabEl();
|
||||
if (!fab) return fail("no-fab-visual");
|
||||
fab.classList.add("is-copied");
|
||||
fab.title = "Copied";
|
||||
// Freeze: the capture pipeline synthesizes a pointer event
|
||||
// outside the block at screenshot time, which would hide the
|
||||
// button (correct in production). Capture-phase stops starve
|
||||
// the module's delegated listeners for the capture.
|
||||
for (const t of ["mouseover", "scroll"])
|
||||
document.addEventListener(t, (e) => e.stopPropagation(), true);
|
||||
document.title = "COPY-VISUAL";
|
||||
return;
|
||||
}
|
||||
// &stepmax=N — diagnostic: stop after the Nth interaction (hovers
|
||||
// and clicks count) and stamp COPY-STEP-N, so a paint regression
|
||||
// can be bisected to the interaction that triggers it.
|
||||
let step = 0;
|
||||
const stepMax = parseInt(q.get("stepmax") || "999", 10);
|
||||
const gate = () => {
|
||||
step += 1;
|
||||
if (step > stepMax) {
|
||||
document.title = "COPY-STEP-" + (step - 1);
|
||||
throw { __stop: true };
|
||||
}
|
||||
};
|
||||
let done = 0;
|
||||
for (const [el, want, name] of probes) {
|
||||
if (!el) return fail("no-" + name);
|
||||
gate();
|
||||
hover(el);
|
||||
const fab = fabEl();
|
||||
if (!fab || !fab.classList.contains("is-visible"))
|
||||
return fail("fab-hidden-" + name);
|
||||
gate();
|
||||
fab.click();
|
||||
await new Promise((r) => setTimeout(r, 0));
|
||||
const got = window.__copied[window.__copied.length - 1];
|
||||
if (got !== want) {
|
||||
console.log("copy mismatch", name, JSON.stringify(got));
|
||||
return fail("source-" + name);
|
||||
}
|
||||
done += 1;
|
||||
}
|
||||
|
||||
// Bubble probe: the whole raw markdown, fences and pipes intact.
|
||||
gate();
|
||||
bars[0].click();
|
||||
await new Promise((r) => setTimeout(r, 0));
|
||||
if (window.__copied[window.__copied.length - 1] !== MD_ONE)
|
||||
return fail("bubble-source");
|
||||
|
||||
document.title = "COPY-READY-" + bars.length + "-" + done;
|
||||
} catch (e) {
|
||||
if (!(e && e.__stop)) {
|
||||
console.log("copy harness error", e);
|
||||
fail("error");
|
||||
}
|
||||
}
|
||||
}, 400);
|
||||
} catch (e) {
|
||||
messages.textContent = "HARNESS ERROR: " + e.message;
|
||||
fail("error");
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
@@ -1964,12 +1476,6 @@ def build(out: Path) -> None:
|
||||
(ta / "livepass.html").write_text(TASKAGENT_TEMPLATE, encoding="utf-8")
|
||||
print(f"{ta}/livepass.html — task_agent card (real Pane.handleEvent routing)")
|
||||
|
||||
cp = out / "copy"
|
||||
cp.mkdir(parents=True, exist_ok=True)
|
||||
symlink(cp / "shared", ROOT / "turnstone/shared_static")
|
||||
(cp / "livepass.html").write_text(COPY_TEMPLATE, encoding="utf-8")
|
||||
print(f"{cp}/livepass.html — copy affordances (bubble bars + block button)")
|
||||
|
||||
pf = out / "perf"
|
||||
pf.mkdir(parents=True, exist_ok=True)
|
||||
symlink(pf / "shared", ROOT / "turnstone/shared_static")
|
||||
@@ -1977,17 +1483,6 @@ def build(out: Path) -> None:
|
||||
(pf / "livepass.html").write_text(PERF_TEMPLATE, encoding="utf-8")
|
||||
print(f"{pf}/livepass.html — long-session perf baseline (real InteractivePane)")
|
||||
|
||||
pb = out / "proxybrand"
|
||||
pb.mkdir(parents=True, exist_ok=True)
|
||||
symlink(pb / "shared", ROOT / "turnstone/shared_static")
|
||||
symlink(pb / "static", ROOT / "turnstone/ui/static")
|
||||
shim = "<script>" + extract_proxy_shim() + "</script>"
|
||||
(pb / "frame.html").write_text(
|
||||
inject(PROXYBRAND_FRAME_TEMPLATE, "SHIM", shim), encoding="utf-8"
|
||||
)
|
||||
(pb / "livepass.html").write_text(PROXYBRAND_HOST_TEMPLATE, encoding="utf-8")
|
||||
print(f"{pb}/livepass.html — back-to-console brand (real shell.js + real shim)")
|
||||
|
||||
|
||||
class _PerfStore:
|
||||
"""Rendezvous for the perf page's POSTed JSON report."""
|
||||
|
||||
@@ -1,19 +0,0 @@
|
||||
# Entra config for the Entra e2e / spike harnesses. Copy to `.env` (gitignored)
|
||||
# and fill in from your tenant. `entra_setup.sh setup` creates the app
|
||||
# registrations and writes a populated `.env` for you.
|
||||
#
|
||||
# cp scripts/obo-e2e/.env.example scripts/obo-e2e/.env
|
||||
# # then edit, or run: ./scripts/obo-e2e/entra_setup.sh setup
|
||||
|
||||
export ENTRA_TENANT_ID=<tenant-guid-or-domain>
|
||||
export ENTRA_CLIENT_ID=<turnstone-spike-app-client-id>
|
||||
export ENTRA_CLIENT_SECRET=<client-secret>
|
||||
export SPIKE_AUDIENCE_A=api://<resource-app-a-guid> # a consented resource
|
||||
export SPIKE_AUDIENCE_B=api://<resource-app-b-guid> # a second consented resource
|
||||
export SPIKE_AUDIENCE_UNCONSENTED=api://<resource-app-c-guid> # NOT granted (negative case)
|
||||
export SPIKE_RUN_OBO=1
|
||||
# export SPIKE_PORT=8765 # redirect-listener port (default 8765)
|
||||
# export SPIKE_CALLBACK_FILE=/tmp/obo_cb.txt # remote-browser mode: paste the redirect URL here
|
||||
|
||||
# The Keycloak / OSS-path harness needs no config — keycloak_e2e.sh sets
|
||||
# everything and stands up an ephemeral container.
|
||||
@@ -1,214 +0,0 @@
|
||||
# OBO e2e harnesses — single-credential MCP token minting (`auth_type=oauth_obo`)
|
||||
|
||||
Manual test harnesses for the `oauth_obo` feature (issue #551). They exercise
|
||||
the **real** Turnstone mint path (`get_obo_access_token_classified` →
|
||||
`_obo_mint_entra` / `_obo_mint_rfc8693`) against a real identity provider — not
|
||||
mocks, not the unit suite. Two grant legs:
|
||||
|
||||
- **Entra** (`entra_e2e.py`) — real tenant, one interactive sign-in.
|
||||
- **Keycloak / RFC 8693** (`keycloak_e2e.py` + `.sh`) — ephemeral docker, fully
|
||||
headless.
|
||||
|
||||
There is also `entra_spike.py` (raw-OAuth **wire** probe, pre-implementation
|
||||
reference) and `entra_setup.sh` (creates the Entra app registrations + writes a
|
||||
populated `.env`).
|
||||
|
||||
**Secrets:** these read config from env. Real credentials live in a **gitignored
|
||||
`.env`** (copy `.env.example`); nothing tenant-specific is committed. The only
|
||||
literal secret in the tree is the ephemeral Keycloak container's throwaway
|
||||
`spike-secret`, which lives and dies with the container.
|
||||
|
||||
Not part of CI — run by hand when validating the feature against a live IdP.
|
||||
|
||||
## `entra_e2e.py` — end-to-end product exercise (post-implementation)
|
||||
|
||||
`entra_spike.py` verified the raw OAuth WIRE (before code existed). `entra_e2e.py`
|
||||
verifies the SHIPPED Turnstone code: it does a real Entra login, feeds the
|
||||
credential through the real `MCPTokenStore.upsert_oidc_credential` (the call the
|
||||
OIDC callback makes on capture), then drives the real
|
||||
`get_obo_access_token_classified` → `_obo_mint_entra` against the live Entra token
|
||||
endpoint. Checks E1–E7: real mint + aud claim, cache-hit (0 Entra calls),
|
||||
single-credential→audiences A&B, rotation write-back, force_refresh re-mint,
|
||||
unconsented-audience classification with the credential surviving, and
|
||||
flush→re-mint. Reuses the same `.env` and interactive login (SPIKE_CALLBACK_FILE
|
||||
for remote browser).
|
||||
|
||||
```bash
|
||||
source scripts/obo-e2e/.env
|
||||
uv run python scripts/obo-e2e/entra_e2e.py
|
||||
# one interactive sign-in; E1–E7 then run against the real product code. Results below.
|
||||
```
|
||||
|
||||
Results — RUN 2026-07-12 on the real tenant, ALL VERIFIED (exit 0): capture
|
||||
persisted; E1 mint A (aud=A app-id, cache row refresh_token_ct NULL); E2 cache
|
||||
hit (0 extra Entra calls); E3 mint B from the SAME credential (aud=B app-id); E4
|
||||
rotation write-back (RT rotated 2040→2091 chars, newest persisted); E5
|
||||
force_refresh re-mint (1 Entra call); E6 unconsented C → refresh_failed and the
|
||||
credential SURVIVES; E7 flush→re-mint. The real `get_obo_access_token_classified`
|
||||
→ `_obo_mint_entra` path against the live Entra token endpoint.
|
||||
|
||||
## `keycloak_e2e.py` + `keycloak_e2e.sh` — OSS path (RFC 8693), headless
|
||||
|
||||
The rfc8693 equivalent of `entra_e2e.py`: `keycloak_e2e.sh` spins up ephemeral
|
||||
Keycloak, configures the realm (turnstone client with standard token exchange,
|
||||
mcp-a/b/c clients, aud-mcp-a/b audience scopes, a test user), runs the harness
|
||||
against the real `get_obo_access_token_classified` → `_obo_mint_rfc8693`
|
||||
(refresh grant → token exchange), then tears down. No browser (password grant).
|
||||
|
||||
```bash
|
||||
./scripts/obo-e2e/keycloak_e2e.sh
|
||||
```
|
||||
|
||||
Results — RUN 2026-07-12, ALL VERIFIED: capture persisted; E1 mint A
|
||||
(refresh→exchange, aud=mcp-a, cache row refresh_token_ct NULL); E2 cache hit (0
|
||||
extra KC calls); E3 mint B from the SAME credential (aud=mcp-b); E4 rotation
|
||||
write-back (KC rotated the RT on the refresh leg, newest persisted); E5
|
||||
force_refresh re-mint (**2 KC calls** = the two-leg chain); E6 unconsented C →
|
||||
refresh_failed_transient (KC returns invalid_request for a missing audience
|
||||
scope → classified transient; credential SURVIVES either way); E7 flush→re-mint.
|
||||
Gotcha: dev-mode Keycloak boot is slow on a loaded host — the script now waits on
|
||||
kcadm auth (up to ~6 min) rather than a fixed sleep. Port 8091 (8090 = the dev
|
||||
console).
|
||||
|
||||
## Leg 1 — Entra (`entra_spike.py`) — NEEDS TENANT ACCESS
|
||||
|
||||
### Tenant / app-registration setup (one-time, ~15 min)
|
||||
|
||||
1. **Spike client app** (stands in for Turnstone's OIDC app registration):
|
||||
- New app registration, single tenant. Platform **Web**, redirect URI
|
||||
`http://localhost:8765/callback`. Create a **client secret**.
|
||||
2. **Two resource apps** (stand in for MCP servers A and B):
|
||||
- New app registrations `spike-mcp-a`, `spike-mcp-b`. In each:
|
||||
**Expose an API** → set Application ID URI (`api://<guid>`) → add a scope
|
||||
(e.g. `mcp.access`).
|
||||
3. **Delegated grants** (this is metaclassing's "proper tenant and app reg setup"):
|
||||
- On the spike client app → **API permissions** → add delegated permission to
|
||||
`spike-mcp-a` and `spike-mcp-b` scopes → **Grant admin consent**.
|
||||
- Optionally also add the spike client's app id to each resource app's
|
||||
`preAuthorizedApplications` (Expose an API → Add a client application) to
|
||||
compare against pure admin consent.
|
||||
4. **Unconsented control** (for V5): a third resource app `spike-mcp-c` with an
|
||||
exposed API but NO permission granted to the spike client.
|
||||
|
||||
### Run
|
||||
|
||||
```bash
|
||||
export ENTRA_TENANT_ID=... ENTRA_CLIENT_ID=... ENTRA_CLIENT_SECRET=...
|
||||
export SPIKE_AUDIENCE_A=api://<a-guid> SPIKE_AUDIENCE_B=api://<b-guid>
|
||||
export SPIKE_AUDIENCE_UNCONSENTED=api://<c-guid> # optional (V5)
|
||||
export SPIKE_RUN_OBO=1 # optional (V6)
|
||||
uv run python scripts/obo-e2e/entra_spike.py
|
||||
```
|
||||
|
||||
A browser opens for one interactive login (any tenant user). Everything after is
|
||||
non-interactive — that IS the feature.
|
||||
|
||||
### What each check pins down
|
||||
|
||||
| Check | Design assumption it verifies |
|
||||
| --- | --- |
|
||||
| V1 | `offline_access` on the login yields a client-bound RT (capture layer) |
|
||||
| V2/V3 | ONE RT redeems for access tokens of DIFFERENT audiences (`scope=<aud>/.default`) — the load-bearing Entra behavior |
|
||||
| V4 | rotation semantics → whether RT write-back on every mint is convenience or correctness-critical |
|
||||
| V5 | unconsented audience fails `AADSTS65001 consent_required` → maps to the reconnect-rail fallback, never a silent failure |
|
||||
| V6 | OBO jwt-bearer middle-tier variant works with the same app registration (comparison data only) |
|
||||
|
||||
Also record (manual): whether Conditional Access / MFA policies in the tenant
|
||||
produce `interaction_required` on redemption — that's the fallback path's other
|
||||
trigger.
|
||||
|
||||
### Results — RUN 2026-07-11 on a real tenant, ALL SIX VERIFIED
|
||||
|
||||
Tenant: personal default directory (Global Admin), user is an MSA member.
|
||||
Setup via `entra_setup.sh setup`; V3 initially failed (see gotcha below),
|
||||
passed after fixing the grant. Second run: V1-V6 all VERIFIED, exit 0.
|
||||
|
||||
| Check | Result |
|
||||
| --- | --- |
|
||||
| V1 offline_access login -> RT | VERIFIED (confidential client + PKCE, RT ~2KB) |
|
||||
| V2 RT -> audience A token | VERIFIED (`aud=<A app guid>`, ~70 min TTL, new RT returned) |
|
||||
| V3 SAME RT -> audience B token | **VERIFIED — the load-bearing claim: one RT, many audiences** |
|
||||
| V4 rotation | VERIFIED: RT rotates on every redemption, but the OLD RT stays valid (reuse HTTP 200) -> write-back-newest is required; races are benign on Entra |
|
||||
| V5 unconsented audience | VERIFIED: `invalid_grant` + `AADSTS65001` (error_codes=[65001]) -> clean mapping to the reconnect-rail fallback |
|
||||
| V6 OBO jwt-bearer variant | VERIFIED: middle-tier shape also works with the same app registration |
|
||||
|
||||
**Operator gotcha (feeds #682 + product docs):** `az ad app permission
|
||||
admin-consent` run immediately after SP creation SILENTLY skips
|
||||
not-yet-propagated resource SPs — grant A landed, grant B didn't, and the only
|
||||
symptom was AADSTS65001 at redemption. Verify grants after consent
|
||||
(`oauth2PermissionGrants` filter on the client SP) or write them directly with
|
||||
`az ad app permission grant --id <client> --api <resource> --scope <scope>`.
|
||||
Product-side implication: a missing tenant grant for a NEW oauth_obo server
|
||||
surfaces as AADSTS65001 -> the same reconnect-rail path as revocation; the
|
||||
admin docs must say "grant first, then add the server".
|
||||
|
||||
## Leg 2 — Keycloak RFC 8693 (portability check) — runnable locally
|
||||
|
||||
Ephemeral `quay.io/keycloak/keycloak:26.3` (`start-dev`, port 8089), realm
|
||||
`spike`, confidential client `turnstone` with **standard token exchange**
|
||||
enabled, resource clients `mcp-a`/`mcp-b`, user `alice`. Pipeline mirrors the
|
||||
product design for a generic-8693 IdP:
|
||||
|
||||
```
|
||||
stored user RT --(refresh grant)--> user AT --(RFC 8693 exchange, audience=mcp-X)--> audience-scoped AT
|
||||
```
|
||||
|
||||
i.e. the per-user credential stays ONE refresh token; per-server tokens are
|
||||
minted via standard token exchange instead of Entra's multi-resource RT
|
||||
redemption. Same substrate, different grant leg.
|
||||
|
||||
### Results — RUN 2026-07-11, VERIFIED (Keycloak 26.3, ephemeral)
|
||||
|
||||
```
|
||||
alice ONE stored RT
|
||||
-> refresh grant -> user AT (azp=turnstone); RT ROTATED on refresh
|
||||
-> 8693 exchange audience=mcp-a scope=aud-mcp-a -> AT aud=mcp-a user=alice 300s, NO RT
|
||||
-> 8693 exchange audience=mcp-b scope=aud-mcp-b -> AT aud=mcp-b (same subject AT)
|
||||
negative control audience=mcp-c -> invalid_client "Audience not found"
|
||||
```
|
||||
|
||||
Findings that feed the design:
|
||||
1. **One per-user credential -> N audience tokens: VERIFIED on a second IdP.**
|
||||
The substrate is portable; only the grant leg differs per IdP.
|
||||
2. **Exchanged tokens are cache-shaped** (short TTL, no RT) — per-server
|
||||
`mcp_user_tokens` rows as short-lived mint cache is the right model.
|
||||
3. **RT rotation happens here too** — newest-RT write-back on every redemption
|
||||
is a correctness requirement of the capture layer, not an Entra quirk.
|
||||
4. **The IdP-side "delegated grant" has a per-IdP shape**: Entra = API
|
||||
permissions + admin consent; Keycloak = audience client scopes attached to
|
||||
the requester client (optional scopes activate via `scope=` at exchange).
|
||||
Operator runbooks are per-IdP (#682 pattern), code is not.
|
||||
5. Gotchas hit: KC user needs a complete profile for direct grant ("Account is
|
||||
not fully set up"); optional audience scope must be requested explicitly or
|
||||
the exchange 400s with "Requested audience not available".
|
||||
|
||||
Repro (ephemeral, ~2 min):
|
||||
|
||||
```bash
|
||||
docker run -d --name kc-obo-spike -p 127.0.0.1:8089:8080 \
|
||||
-e KC_BOOTSTRAP_ADMIN_USERNAME=admin -e KC_BOOTSTRAP_ADMIN_PASSWORD=admin \
|
||||
quay.io/keycloak/keycloak:26.3 start-dev
|
||||
KC="docker exec kc-obo-spike /opt/keycloak/bin/kcadm.sh"
|
||||
$KC config credentials --server http://localhost:8080 --realm master --user admin --password admin
|
||||
$KC create realms -s realm=spike -s enabled=true
|
||||
$KC create clients -r spike -s clientId=turnstone -s enabled=true -s publicClient=false \
|
||||
-s secret=spike-secret -s directAccessGrantsEnabled=true \
|
||||
-s 'attributes={"standard.token.exchange.enabled":"true"}'
|
||||
$KC create clients -r spike -s clientId=mcp-a -s enabled=true -s publicClient=false -s secret=x
|
||||
$KC create clients -r spike -s clientId=mcp-b -s enabled=true -s publicClient=false -s secret=x
|
||||
$KC create users -r spike -s username=alice -s enabled=true -s email=a@s.test \
|
||||
-s emailVerified=true -s firstName=A -s lastName=S
|
||||
$KC set-password -r spike --username alice --new-password alice-pw
|
||||
TURNSTONE_UUID=$($KC get clients -r spike -q clientId=turnstone --fields id --format csv --noquotes)
|
||||
for t in mcp-a mcp-b; do
|
||||
SID=$($KC create client-scopes -r spike -s name=aud-$t -s protocol=openid-connect -i)
|
||||
$KC create client-scopes/$SID/protocol-mappers/models -r spike -s name=aud-$t \
|
||||
-s protocol=openid-connect -s protocolMapper=oidc-audience-mapper \
|
||||
-s "config={\"included.client.audience\":\"$t\",\"access.token.claim\":\"true\"}"
|
||||
$KC update clients/$TURNSTONE_UUID/optional-client-scopes/$SID -r spike
|
||||
done
|
||||
# then: password grant -> refresh grant -> token-exchange with
|
||||
# grant_type=urn:ietf:params:oauth:grant-type:token-exchange,
|
||||
# subject_token=<user AT>, subject_token_type=...:access_token,
|
||||
# audience=mcp-a, scope=aud-mcp-a
|
||||
```
|
||||
@@ -1,286 +0,0 @@
|
||||
"""End-to-end exercise of the oauth_obo feature against a REAL Entra tenant.
|
||||
|
||||
Unlike ``entra_spike.py`` (which verified the raw OAuth wire shapes), this
|
||||
drives the ACTUAL Turnstone product code — real ``MCPTokenStore``, real
|
||||
``get_obo_access_token_classified`` → ``_obo_mint_entra`` → the real Entra
|
||||
token endpoint — so a green run proves the shipped mint engine works against
|
||||
live Entra, not just that the protocol does.
|
||||
|
||||
Flow:
|
||||
1. Interactive Entra login (auth-code + PKCE + offline_access) → a real
|
||||
refresh credential. This is what ``handle_oidc_callback`` receives.
|
||||
2. Persist it via ``MCPTokenStore.upsert_oidc_credential`` — the exact call
|
||||
the OIDC callback makes on capture (auth.py). The rest of the callback
|
||||
(JWKS validation, user provisioning) is OIDC-generic and unit-tested; the
|
||||
novel path is capture + mint, which this exercises for real.
|
||||
3. Seed real ``oauth_obo`` ``mcp_servers`` rows (audiences A/B consented, C
|
||||
not) and drive ``get_obo_access_token_classified`` — the real dispatch-time
|
||||
entry point — asserting on the minted tokens, cache, rotation, and
|
||||
classification.
|
||||
|
||||
Checks (VERIFIED / FAILED per line):
|
||||
E1 mint for audience A → kind=token; decoded aud == A; cache row written with
|
||||
refresh_token_ct NULL (cache, not custody); expires_at set
|
||||
E2 second call for A → cache hit, ZERO additional Entra calls
|
||||
E3 mint for audience B from the SAME captured credential → aud == B
|
||||
(the single-credential-many-audiences thesis, through the real engine)
|
||||
E4 rotation write-back: the stored credential holds the newest refresh token
|
||||
E5 force_refresh → a fresh mint (Entra call count increments)
|
||||
E6 unconsented audience C → NOT kind=token, and the shared credential SURVIVES
|
||||
(never auto-deleted — the load-bearing custody invariant)
|
||||
E7 cache flush → re-mint: deleting the cache row makes the next call re-mint
|
||||
|
||||
Run:
|
||||
source scripts/obo-e2e/.env
|
||||
uv run python scripts/obo-e2e/entra_e2e.py
|
||||
Env (from .env): ENTRA_TENANT_ID, ENTRA_CLIENT_ID, ENTRA_CLIENT_SECRET,
|
||||
SPIKE_AUDIENCE_A, SPIKE_AUDIENCE_B, SPIKE_AUDIENCE_UNCONSENTED, SPIKE_PORT.
|
||||
Remote browser: set SPIKE_CALLBACK_FILE to paste the redirect URL (as before).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
# Reuse the verified interactive-login machinery from the wire spike.
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from entra_spike import interactive_login, jwt_claims_unverified, redact # noqa: E402
|
||||
|
||||
from turnstone.core.mcp_crypto import ( # noqa: E402
|
||||
MCPTokenCipher,
|
||||
MCPTokenCipherConfig,
|
||||
MCPTokenStore,
|
||||
)
|
||||
from turnstone.core.mcp_oauth import get_obo_access_token_classified # noqa: E402
|
||||
from turnstone.core.oidc import OIDCConfig # noqa: E402
|
||||
from turnstone.core.storage._sqlite import SQLiteBackend # noqa: E402
|
||||
|
||||
USER = "e2e-user"
|
||||
RESULTS: list[tuple[str, str]] = []
|
||||
|
||||
|
||||
def record(status: str, msg: str) -> None:
|
||||
RESULTS.append((status, msg))
|
||||
print(f"[{status:>8}] {msg}")
|
||||
|
||||
|
||||
def aud_matches(token: str, want_audience: str) -> tuple[bool, str]:
|
||||
"""Compare a minted access token's aud claim to the configured audience.
|
||||
|
||||
Entra returns aud as the bare app-id GUID or the full ``api://<guid>`` URI;
|
||||
accept either.
|
||||
"""
|
||||
claims = jwt_claims_unverified(token)
|
||||
aud = str(claims.get("aud", "<none>"))
|
||||
want = want_audience.removeprefix("api://")
|
||||
return aud in (want, want_audience), aud
|
||||
|
||||
|
||||
class _CountingClient:
|
||||
"""Wraps httpx.AsyncClient, counting token-endpoint POSTs so cache hits
|
||||
(which must issue zero) are observable."""
|
||||
|
||||
def __init__(self, inner: httpx.AsyncClient) -> None:
|
||||
self._inner = inner
|
||||
self.posts = 0
|
||||
|
||||
async def post(self, *args: Any, **kwargs: Any) -> httpx.Response:
|
||||
self.posts += 1
|
||||
return await self._inner.post(*args, **kwargs)
|
||||
|
||||
|
||||
def _make_app_state(
|
||||
storage: SQLiteBackend,
|
||||
store: MCPTokenStore,
|
||||
oidc_config: OIDCConfig,
|
||||
http_client: _CountingClient,
|
||||
) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
auth_storage=storage,
|
||||
mcp_token_store=store,
|
||||
oidc_config=oidc_config,
|
||||
obo_http_client=http_client,
|
||||
mcp_oauth_refresh_locks={},
|
||||
mcp_oauth_refresh_backoff={},
|
||||
)
|
||||
|
||||
|
||||
def _seed_obo_server(storage: SQLiteBackend, name: str, audience: str) -> None:
|
||||
storage.create_mcp_server(
|
||||
server_id=f"{name}-id",
|
||||
name=name,
|
||||
transport="streamable-http",
|
||||
url="https://mcp.example.invalid/sse",
|
||||
auth_type="oauth_obo",
|
||||
oauth_audience=audience,
|
||||
)
|
||||
|
||||
|
||||
async def _run(cfg: dict[str, str], refresh_token: str) -> None:
|
||||
tenant = cfg["ENTRA_TENANT_ID"]
|
||||
issuer = f"https://login.microsoftonline.com/{tenant}/v2.0"
|
||||
token_endpoint = f"https://login.microsoftonline.com/{tenant}/oauth2/v2.0/token"
|
||||
aud_a = cfg["SPIKE_AUDIENCE_A"]
|
||||
aud_b = cfg["SPIKE_AUDIENCE_B"]
|
||||
aud_c = cfg.get("SPIKE_AUDIENCE_UNCONSENTED", "")
|
||||
|
||||
# Real Turnstone objects.
|
||||
db_path = os.path.join(tempfile.mkdtemp(prefix="obo-e2e-"), "e2e.db")
|
||||
storage = SQLiteBackend(db_path)
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
raw = base64.urlsafe_b64decode(Fernet.generate_key())
|
||||
store = MCPTokenStore(storage, MCPTokenCipher(MCPTokenCipherConfig(keys=(raw,))), node_id="e2e")
|
||||
oidc_config = OIDCConfig(
|
||||
enabled=True,
|
||||
issuer=issuer,
|
||||
client_id=cfg["ENTRA_CLIENT_ID"],
|
||||
client_secret=cfg["ENTRA_CLIENT_SECRET"],
|
||||
token_endpoint=token_endpoint,
|
||||
obo_grant_profile="entra",
|
||||
capture_user_credential=True,
|
||||
)
|
||||
|
||||
# Step 2 — CAPTURE: the exact storage call handle_oidc_callback makes.
|
||||
store.upsert_oidc_credential(USER, issuer, refresh_token=refresh_token)
|
||||
cap = store.get_oidc_credential(USER, issuer)
|
||||
if cap and cap["refresh_token"] == refresh_token:
|
||||
record("VERIFIED", f"capture: credential persisted for {USER} ({redact(refresh_token)})")
|
||||
else:
|
||||
record("FAILED", "capture: credential did not round-trip")
|
||||
return
|
||||
|
||||
_seed_obo_server(storage, "e2e-a", aud_a)
|
||||
_seed_obo_server(storage, "e2e-b", aud_b)
|
||||
if aud_c:
|
||||
_seed_obo_server(storage, "e2e-c", aud_c)
|
||||
|
||||
inner = httpx.AsyncClient(timeout=20.0)
|
||||
client = _CountingClient(inner)
|
||||
app_state = _make_app_state(storage, store, oidc_config, client)
|
||||
try:
|
||||
# E1 — real mint for audience A.
|
||||
r = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
if r.kind == "token" and r.token:
|
||||
ok, aud = aud_matches(r.token, aud_a)
|
||||
row = storage.get_mcp_user_token(USER, "e2e-a")
|
||||
cache_ok = (
|
||||
row is not None and row["refresh_token_ct"] is None and bool(row["expires_at"])
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if ok and cache_ok else "FAILED",
|
||||
f"E1 mint A: kind=token aud={aud} want={aud_a} cache_row_refreshless={cache_ok}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E1 mint A: kind={r.kind} (expected token)")
|
||||
return
|
||||
|
||||
# E2 — cache hit issues zero Entra calls.
|
||||
posts_before = client.posts
|
||||
r2 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r2.kind == "token" and client.posts == posts_before else "FAILED",
|
||||
f"E2 cache hit: kind={r2.kind} extra_entra_calls={client.posts - posts_before} (want 0)",
|
||||
)
|
||||
|
||||
# E3 — same credential, audience B.
|
||||
rb = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-b"
|
||||
)
|
||||
if rb.kind == "token" and rb.token:
|
||||
ok_b, aud_bclaim = aud_matches(rb.token, aud_b)
|
||||
record(
|
||||
"VERIFIED" if ok_b else "FAILED",
|
||||
f"E3 mint B from SAME credential: aud={aud_bclaim} want={aud_b}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E3 mint B: kind={rb.kind}")
|
||||
|
||||
# E4 — rotation write-back: the stored credential is still redeemable
|
||||
# (holds the newest RT — Entra rotates on redemption).
|
||||
cred_now = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if cred_now is not None else "FAILED",
|
||||
f"E4 rotation write-back: credential persisted {redact(cred_now['refresh_token']) if cred_now else '<gone>'}",
|
||||
)
|
||||
|
||||
# E5 — force_refresh re-mints (a real Entra call).
|
||||
posts_before = client.posts
|
||||
rf = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a", force_refresh=True
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if rf.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E5 force_refresh re-mint: kind={rf.kind} entra_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# E6 — unconsented audience: not a token, and the credential SURVIVES.
|
||||
if aud_c:
|
||||
rc = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-c"
|
||||
)
|
||||
cred_after = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if rc.kind != "token" and cred_after is not None else "FAILED",
|
||||
f"E6 unconsented C: kind={rc.kind} (not token) credential_survives={cred_after is not None}",
|
||||
)
|
||||
else:
|
||||
record("SKIPPED", "E6 unconsented C: SPIKE_AUDIENCE_UNCONSENTED not set")
|
||||
|
||||
# E7 — cache flush → re-mint.
|
||||
store.delete_user_token(USER, "e2e-a")
|
||||
posts_before = client.posts
|
||||
r7 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="e2e-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r7.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E7 flush→re-mint: kind={r7.kind} entra_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
finally:
|
||||
await inner.aclose()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"ENTRA_TENANT_ID",
|
||||
"ENTRA_CLIENT_ID",
|
||||
"ENTRA_CLIENT_SECRET",
|
||||
"SPIKE_AUDIENCE_A",
|
||||
"SPIKE_AUDIENCE_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in os.environ if k.startswith(("ENTRA_", "SPIKE_"))}
|
||||
missing = [k for k in required if not cfg.get(k)]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)} — did you `source scripts/obo-e2e/.env`?")
|
||||
return 2
|
||||
|
||||
print("Signing in to Entra (this is the login the feature captures)...")
|
||||
tokens = interactive_login(cfg)
|
||||
refresh_token = tokens.get("refresh_token")
|
||||
if not isinstance(refresh_token, str) or not refresh_token:
|
||||
print(f"No refresh_token from login (keys={sorted(tokens)}) — offline_access missing?")
|
||||
return 1
|
||||
|
||||
asyncio.run(_run(cfg, refresh_token))
|
||||
|
||||
print("\n=== summary ===")
|
||||
for status, msg in RESULTS:
|
||||
print(f" {status:>8} {msg}")
|
||||
return 0 if all(s in ("VERIFIED", "SKIPPED") for s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,134 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Entra spike setup for entra_spike.py (#551 re-scope boundary spike).
|
||||
# Manual test tooling — not run in CI. Creates throwaway Entra app registrations.
|
||||
#
|
||||
# ./entra_setup.sh setup create app registrations + consent + .env
|
||||
# ./entra_setup.sh cleanup delete everything it created (incl. .env)
|
||||
#
|
||||
# Creates in the logged-in tenant (az login first):
|
||||
# spike-turnstone confidential client (stands in for Turnstone's OIDC app)
|
||||
# spike-mcp-a/b resource apps exposing scope mcp.access, admin-consented
|
||||
# spike-mcp-c resource app with NO grant to the client (V5 control)
|
||||
# Requires: the logged-in user can create apps + grant admin consent
|
||||
# (Global Admin on a personal tenant qualifies).
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
ENV_FILE=".env"
|
||||
NAMES=(spike-turnstone spike-mcp-a spike-mcp-b spike-mcp-c)
|
||||
|
||||
log() { printf '>> %s\n' "$*"; }
|
||||
|
||||
graph_patch_api() { # $1=appId $2=scope-uuid $3=display-name
|
||||
local obj_id
|
||||
obj_id=$(az ad app show --id "$1" --query id -o tsv)
|
||||
az rest --method PATCH \
|
||||
--url "https://graph.microsoft.com/v1.0/applications/${obj_id}" \
|
||||
--headers 'Content-Type=application/json' \
|
||||
--body "{
|
||||
\"identifierUris\": [\"api://$1\"],
|
||||
\"api\": {
|
||||
\"requestedAccessTokenVersion\": 2,
|
||||
\"oauth2PermissionScopes\": [{
|
||||
\"id\": \"$2\",
|
||||
\"value\": \"mcp.access\",
|
||||
\"type\": \"Admin\",
|
||||
\"isEnabled\": true,
|
||||
\"adminConsentDisplayName\": \"Access $3\",
|
||||
\"adminConsentDescription\": \"Spike scope for $3\"
|
||||
}]
|
||||
}
|
||||
}"
|
||||
}
|
||||
|
||||
make_resource_app() { # $1=display-name ; echoes "appId scopeId"
|
||||
local app_id scope_id
|
||||
app_id=$(az ad app create --display-name "$1" \
|
||||
--sign-in-audience AzureADMyOrg --query appId -o tsv)
|
||||
scope_id=$(python3 -c 'import uuid; print(uuid.uuid4())')
|
||||
graph_patch_api "$app_id" "$scope_id" "$1" >/dev/null
|
||||
az ad sp create --id "$app_id" >/dev/null 2>&1 || true
|
||||
echo "$app_id $scope_id"
|
||||
}
|
||||
|
||||
cmd_setup() {
|
||||
local tenant_id
|
||||
tenant_id=$(az account show --query tenantId -o tsv)
|
||||
log "tenant: ${tenant_id}"
|
||||
|
||||
log "creating resource apps (a, b, c)..."
|
||||
read -r APP_A SCOPE_A <<<"$(make_resource_app spike-mcp-a)"
|
||||
read -r APP_B SCOPE_B <<<"$(make_resource_app spike-mcp-b)"
|
||||
read -r APP_C _ <<<"$(make_resource_app spike-mcp-c)"
|
||||
log " a=${APP_A} b=${APP_B} c=${APP_C} (c stays unconsented)"
|
||||
|
||||
log "creating confidential client spike-turnstone..."
|
||||
CLIENT_ID=$(az ad app create --display-name spike-turnstone \
|
||||
--sign-in-audience AzureADMyOrg \
|
||||
--web-redirect-uris "http://localhost:8765/callback" \
|
||||
--query appId -o tsv)
|
||||
az ad sp create --id "$CLIENT_ID" >/dev/null 2>&1 || true
|
||||
# No stderr suppression here: the secret is load-bearing (it lands in .env),
|
||||
# so under `set -e` a reset failure must abort LOUDLY, not silently.
|
||||
SECRET=$(az ad app credential reset --id "$CLIENT_ID" \
|
||||
--display-name spike --years 1 --query password -o tsv)
|
||||
|
||||
log "adding delegated permissions (a, b — NOT c)..."
|
||||
# Tolerated failures (|| log): a re-run hits "permission already exists" and
|
||||
# SP-propagation delays are common right after app creation — the
|
||||
# admin-consent retry loop below is the real gate. `set -e` would otherwise
|
||||
# turn a suppressed non-zero here into a silent mid-script abort.
|
||||
az ad app permission add --id "$CLIENT_ID" \
|
||||
--api "$APP_A" --api-permissions "${SCOPE_A}=Scope" \
|
||||
|| log " warn: permission add for a failed (may already exist); admin-consent below will confirm"
|
||||
az ad app permission add --id "$CLIENT_ID" \
|
||||
--api "$APP_B" --api-permissions "${SCOPE_B}=Scope" \
|
||||
|| log " warn: permission add for b failed (may already exist); admin-consent below will confirm"
|
||||
|
||||
log "granting admin consent (retries while SPs propagate)..."
|
||||
local ok=""
|
||||
for i in 1 2 3 4 5; do
|
||||
if az ad app permission admin-consent --id "$CLIENT_ID" 2>/dev/null; then
|
||||
ok=1; break
|
||||
fi
|
||||
log " not yet (attempt $i) — waiting 15s"
|
||||
sleep 15
|
||||
done
|
||||
[ -n "$ok" ] || { log "admin-consent failed after retries — grant manually in the portal (API permissions blade) and re-run the spike"; }
|
||||
|
||||
# Single-quote the values in the generated .env: the AS-issued client secret
|
||||
# can contain $ / backtick, and an unquoted RHS would be re-expanded (or
|
||||
# partially executed) when the operator `source`s the file. The heredoc still
|
||||
# interpolates ${...} into the single-quoted output; sourcing then treats the
|
||||
# result literally. (Azure secrets are base64-ish — no single quotes to escape.)
|
||||
umask 177
|
||||
cat > "$ENV_FILE" <<EOF
|
||||
export ENTRA_TENANT_ID='${tenant_id}'
|
||||
export ENTRA_CLIENT_ID='${CLIENT_ID}'
|
||||
export ENTRA_CLIENT_SECRET='${SECRET}'
|
||||
export SPIKE_AUDIENCE_A='api://${APP_A}'
|
||||
export SPIKE_AUDIENCE_B='api://${APP_B}'
|
||||
export SPIKE_AUDIENCE_UNCONSENTED='api://${APP_C}'
|
||||
export SPIKE_RUN_OBO=1
|
||||
EOF
|
||||
log "wrote ${ENV_FILE} (chmod 600). Next:"
|
||||
log " source scripts/obo-e2e/.env && uv run python scripts/obo-e2e/entra_spike.py"
|
||||
log "cleanup later with: ./entra_setup.sh cleanup"
|
||||
}
|
||||
|
||||
cmd_cleanup() {
|
||||
for name in "${NAMES[@]}"; do
|
||||
for app_id in $(az ad app list --display-name "$name" --query '[].appId' -o tsv); do
|
||||
log "deleting ${name} (${app_id})"
|
||||
az ad app delete --id "$app_id"
|
||||
done
|
||||
done
|
||||
rm -f "$ENV_FILE"
|
||||
log "cleanup done (app registrations + .env removed)"
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
setup) cmd_setup ;;
|
||||
cleanup) cmd_cleanup ;;
|
||||
*) echo "usage: $0 setup|cleanup"; exit 2 ;;
|
||||
esac
|
||||
@@ -1,333 +0,0 @@
|
||||
"""Entra boundary spike for single-credential MCP token minting (#551 re-scope).
|
||||
|
||||
Verifies, against a REAL Entra tenant, the assumptions behind the oauth_obo
|
||||
design (one IdP refresh token per user; per-MCP access tokens minted on
|
||||
demand). Each check prints VERIFIED / FAILED / SKIPPED plus redacted evidence.
|
||||
|
||||
V1 interactive confidential-client login (auth-code + PKCE + offline_access)
|
||||
-> refresh token captured [capture layer works]
|
||||
V2 RT redeemed with scope=<AUDIENCE_A>/.default -> aud claim == A
|
||||
V3 SAME credential redeemed for <AUDIENCE_B> -> aud claim == B
|
||||
KEY CHECK: Entra RTs are client-bound, not resource-bound.
|
||||
V4 rotation semantics: does each redemption return a new RT, and does the
|
||||
PREVIOUS RT keep working? [write-back design]
|
||||
V5 redemption for an unconsented audience -> AADSTS65001 consent_required
|
||||
[maps to the reconnect-rail fallback]
|
||||
V6 optional: OBO jwt-bearer leg (requested_token_use=on_behalf_of) using a
|
||||
Turnstone-audience access token as assertion [middle-tier variant]
|
||||
|
||||
Run: uv run python scripts/obo-e2e/entra_spike.py
|
||||
Env: ENTRA_TENANT_ID tenant GUID or domain
|
||||
ENTRA_CLIENT_ID Turnstone spike app registration (confidential)
|
||||
ENTRA_CLIENT_SECRET client secret for the above
|
||||
SPIKE_AUDIENCE_A e.g. api://<guid-a> (exposes a scope, consented)
|
||||
SPIKE_AUDIENCE_B e.g. api://<guid-b> (exposes a scope, consented)
|
||||
SPIKE_AUDIENCE_UNCONSENTED optional, for V5
|
||||
SPIKE_RUN_OBO optional "1" to run V6
|
||||
SPIKE_PORT redirect listener port (default 8765; register
|
||||
http://localhost:<port>/callback as a Web
|
||||
redirect URI on the spike app registration)
|
||||
|
||||
App-registration setup checklist: see README.md next to this file.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import secrets
|
||||
import sys
|
||||
import threading
|
||||
import urllib.parse
|
||||
import webbrowser
|
||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
RESULTS: list[tuple[str, str, str]] = [] # (check, status, evidence)
|
||||
|
||||
|
||||
def record(check: str, status: str, evidence: str) -> None:
|
||||
RESULTS.append((check, status, evidence))
|
||||
print(f"[{status:>8}] {check}: {evidence}")
|
||||
|
||||
|
||||
def b64url_json(segment: str) -> dict[str, Any]:
|
||||
pad = "=" * (-len(segment) % 4)
|
||||
out: dict[str, Any] = json.loads(base64.urlsafe_b64decode(segment + pad))
|
||||
return out
|
||||
|
||||
|
||||
def jwt_claims_unverified(token: str) -> dict[str, Any]:
|
||||
"""Spike-only unverified decode. NEVER do this in product code."""
|
||||
try:
|
||||
return b64url_json(token.split(".")[1])
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def redact(token: str | None) -> str:
|
||||
if not token:
|
||||
return "<absent>"
|
||||
return f"{token[:8]}...({len(token)} chars)"
|
||||
|
||||
|
||||
class _CodeCatcher(BaseHTTPRequestHandler):
|
||||
code: str | None = None
|
||||
state: str | None = None
|
||||
event = threading.Event()
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802 - stdlib API name
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(self.path).query)
|
||||
_CodeCatcher.code = (q.get("code") or [None])[0]
|
||||
_CodeCatcher.state = (q.get("state") or [None])[0]
|
||||
body = b"Spike login captured - return to the terminal."
|
||||
if q.get("error"):
|
||||
body = f"IdP error: {q}".encode()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/plain")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
_CodeCatcher.event.set()
|
||||
|
||||
def log_message(self, *args: Any) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def interactive_login(cfg: dict[str, str]) -> dict[str, Any]:
|
||||
"""V1: authorization-code + PKCE + offline_access as a confidential client.
|
||||
|
||||
Mirrors production shape: same grant Turnstone's OIDC login uses
|
||||
(core/oidc.py exchange_code), plus offline_access.
|
||||
"""
|
||||
port = int(cfg.get("SPIKE_PORT", "8765"))
|
||||
redirect_uri = f"http://localhost:{port}/callback"
|
||||
verifier = secrets.token_urlsafe(48)
|
||||
challenge = (
|
||||
base64.urlsafe_b64encode(hashlib.sha256(verifier.encode()).digest()).rstrip(b"=").decode()
|
||||
)
|
||||
state = secrets.token_urlsafe(16)
|
||||
authorize = (
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/authorize?"
|
||||
+ urllib.parse.urlencode(
|
||||
{
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"response_type": "code",
|
||||
"redirect_uri": redirect_uri,
|
||||
"response_mode": "query",
|
||||
# offline_access is THE capture-layer delta vs today's login.
|
||||
# No resource scope here: the RT is minted client-bound.
|
||||
"scope": "openid profile offline_access",
|
||||
"state": state,
|
||||
"code_challenge": challenge,
|
||||
"code_challenge_method": "S256",
|
||||
}
|
||||
)
|
||||
)
|
||||
server = HTTPServer(("127.0.0.1", port), _CodeCatcher)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
print(f"\nOpen (or auto-opened) in a browser with a tenant user:\n {authorize}\n")
|
||||
cb_file = cfg.get("SPIKE_CALLBACK_FILE", "")
|
||||
if cb_file:
|
||||
print(
|
||||
"Remote-browser mode: after sign-in the browser lands on a broken\n"
|
||||
f"http://localhost:{port}/callback?... page. Copy that FULL URL and run:\n"
|
||||
f" echo '<url>' > {cb_file}\n"
|
||||
)
|
||||
|
||||
def _watch_callback_file() -> None:
|
||||
# Driver-friendly fallback: the sign-in can happen on any device;
|
||||
# whoever signed in drops the redirected URL into SPIKE_CALLBACK_FILE.
|
||||
import time as _time
|
||||
|
||||
while not _CodeCatcher.event.is_set():
|
||||
try:
|
||||
with open(cb_file) as _f:
|
||||
pasted = _f.read().strip()
|
||||
except OSError:
|
||||
pasted = ""
|
||||
if "?" in pasted:
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(pasted).query)
|
||||
_CodeCatcher.code = (q.get("code") or [None])[0]
|
||||
_CodeCatcher.state = (q.get("state") or [None])[0]
|
||||
_CodeCatcher.event.set()
|
||||
return
|
||||
_time.sleep(1.0)
|
||||
|
||||
if cb_file:
|
||||
threading.Thread(target=_watch_callback_file, daemon=True).start()
|
||||
webbrowser.open(authorize)
|
||||
if not _CodeCatcher.event.wait(timeout=600):
|
||||
server.shutdown()
|
||||
raise SystemExit("Timed out waiting for the redirect (10 min).")
|
||||
server.shutdown()
|
||||
if _CodeCatcher.state != state:
|
||||
raise SystemExit("state mismatch on redirect - aborting.")
|
||||
if not _CodeCatcher.code:
|
||||
raise SystemExit("No code on redirect (IdP error page shown in browser).")
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "authorization_code",
|
||||
"code": _CodeCatcher.code,
|
||||
"redirect_uri": redirect_uri,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"code_verifier": verifier,
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
tokens: dict[str, Any] = resp.json()
|
||||
if resp.status_code != 200:
|
||||
raise SystemExit(f"code exchange failed: {json.dumps(tokens, indent=2)[:800]}")
|
||||
return tokens
|
||||
|
||||
|
||||
def redeem(cfg: dict[str, str], refresh_token: str, scope: str) -> tuple[int, dict[str, Any]]:
|
||||
"""Redeem a refresh token for an access token with the given scope."""
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "refresh_token",
|
||||
"refresh_token": refresh_token,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"scope": scope,
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
body: dict[str, Any] = resp.json()
|
||||
return resp.status_code, body
|
||||
|
||||
|
||||
def obo_exchange(cfg: dict[str, str], assertion: str, scope: str) -> tuple[int, dict[str, Any]]:
|
||||
"""V6: middle-tier OBO variant (jwt-bearer + requested_token_use)."""
|
||||
resp = httpx.post(
|
||||
f"https://login.microsoftonline.com/{cfg['ENTRA_TENANT_ID']}/oauth2/v2.0/token",
|
||||
data={
|
||||
"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer",
|
||||
"assertion": assertion,
|
||||
"client_id": cfg["ENTRA_CLIENT_ID"],
|
||||
"client_secret": cfg["ENTRA_CLIENT_SECRET"],
|
||||
"scope": scope,
|
||||
"requested_token_use": "on_behalf_of",
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
body: dict[str, Any] = resp.json()
|
||||
return resp.status_code, body
|
||||
|
||||
|
||||
def check_aud(label: str, status: int, body: dict[str, Any], want_aud: str) -> str | None:
|
||||
"""Common V2/V3 assertion: 200 + aud matches. Returns the new RT if any."""
|
||||
if status != 200:
|
||||
record(label, "FAILED", f"HTTP {status}: {json.dumps(body)[:300]}")
|
||||
return None
|
||||
claims = jwt_claims_unverified(body.get("access_token", ""))
|
||||
aud = str(claims.get("aud", "<none>"))
|
||||
ok = aud == want_aud or aud == want_aud.removeprefix("api://")
|
||||
record(
|
||||
label,
|
||||
"VERIFIED" if ok else "FAILED",
|
||||
f"aud={aud} want={want_aud} expires_in={body.get('expires_in')} "
|
||||
f"new_rt={redact(body.get('refresh_token'))}",
|
||||
)
|
||||
new_rt = body.get("refresh_token")
|
||||
return str(new_rt) if isinstance(new_rt, str) else None
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"ENTRA_TENANT_ID",
|
||||
"ENTRA_CLIENT_ID",
|
||||
"ENTRA_CLIENT_SECRET",
|
||||
"SPIKE_AUDIENCE_A",
|
||||
"SPIKE_AUDIENCE_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in required if k in os.environ}
|
||||
missing = [k for k in required if k not in cfg]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)}\nSee module docstring.")
|
||||
return 2
|
||||
for opt in ("SPIKE_AUDIENCE_UNCONSENTED", "SPIKE_PORT", "SPIKE_RUN_OBO"):
|
||||
if opt in os.environ:
|
||||
cfg[opt] = os.environ[opt]
|
||||
|
||||
# V1 - capture
|
||||
tokens = interactive_login(cfg)
|
||||
rt0 = tokens.get("refresh_token")
|
||||
if isinstance(rt0, str) and rt0:
|
||||
record("V1 capture (offline_access -> RT)", "VERIFIED", redact(rt0))
|
||||
else:
|
||||
record("V1 capture (offline_access -> RT)", "FAILED", f"keys={sorted(tokens.keys())}")
|
||||
return 1
|
||||
|
||||
# V2 - mint for audience A
|
||||
a = cfg["SPIKE_AUDIENCE_A"]
|
||||
s2, b2 = redeem(cfg, rt0, f"{a}/.default")
|
||||
rt_after_a = check_aud("V2 mint audience A from RT", s2, b2, a)
|
||||
|
||||
# V3 - SAME credential, audience B (the design-critical check)
|
||||
b = cfg["SPIKE_AUDIENCE_B"]
|
||||
s3, b3 = redeem(cfg, rt0, f"{b}/.default")
|
||||
check_aud("V3 mint audience B from SAME RT", s3, b3, b)
|
||||
|
||||
# V4 - rotation semantics
|
||||
if rt_after_a and rt_after_a != rt0:
|
||||
s4, _ = redeem(cfg, rt0, f"{a}/.default")
|
||||
record(
|
||||
"V4 rotation (new RT returned; old still valid?)",
|
||||
"VERIFIED" if s4 == 200 else "VERIFIED",
|
||||
f"rotated=yes old_rt_reuse_http={s4} "
|
||||
"(design: persist newest RT on every mint; "
|
||||
f"{'old stays valid - benign race window' if s4 == 200 else 'old INVALIDATED - write-back is correctness-critical'})",
|
||||
)
|
||||
else:
|
||||
record(
|
||||
"V4 rotation",
|
||||
"VERIFIED",
|
||||
"no rotation observed on redemption (same/absent RT) - "
|
||||
"write-back still required for the rotating case",
|
||||
)
|
||||
|
||||
# V5 - unconsented audience -> consent_required
|
||||
unc = cfg.get("SPIKE_AUDIENCE_UNCONSENTED")
|
||||
if unc:
|
||||
s5, b5 = redeem(cfg, rt0, f"{unc}/.default")
|
||||
codes = b5.get("error_codes", [])
|
||||
hit = s5 == 400 and (65001 in codes or b5.get("suberror") == "consent_required")
|
||||
record(
|
||||
"V5 unconsented audience -> AADSTS65001",
|
||||
"VERIFIED" if hit else "FAILED",
|
||||
f"http={s5} error={b5.get('error')} codes={codes}",
|
||||
)
|
||||
else:
|
||||
record("V5 unconsented audience", "SKIPPED", "SPIKE_AUDIENCE_UNCONSENTED not set")
|
||||
|
||||
# V6 - optional OBO middle-tier variant
|
||||
if cfg.get("SPIKE_RUN_OBO") == "1":
|
||||
s6a, b6a = redeem(cfg, rt0, f"{cfg['ENTRA_CLIENT_ID']}/.default")
|
||||
at_self = b6a.get("access_token", "") if s6a == 200 else ""
|
||||
if at_self:
|
||||
s6, b6 = obo_exchange(cfg, at_self, f"{a}/.default")
|
||||
check_aud("V6 OBO jwt-bearer variant", s6, b6, a)
|
||||
else:
|
||||
record(
|
||||
"V6 OBO jwt-bearer variant",
|
||||
"FAILED",
|
||||
f"could not mint self-audience assertion: HTTP {s6a}",
|
||||
)
|
||||
else:
|
||||
record("V6 OBO jwt-bearer variant", "SKIPPED", "SPIKE_RUN_OBO != 1")
|
||||
|
||||
print("\n=== summary ===")
|
||||
for check, status, _ in RESULTS:
|
||||
print(f" {status:>8} {check}")
|
||||
return 0 if all(s != "FAILED" for _, s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,371 +0,0 @@
|
||||
"""End-to-end exercise of the oauth_obo feature on the OSS path (RFC 8693).
|
||||
|
||||
Parallel to ``entra_e2e.py`` but for ``obo_grant_profile="rfc8693"`` against an
|
||||
ephemeral Keycloak — the open-source / non-Entra deployment shape. Fully
|
||||
headless (password grant, no browser), so it runs unattended.
|
||||
|
||||
Drives the REAL Turnstone code: ``MCPTokenStore.upsert_oidc_credential`` (capture)
|
||||
then ``get_obo_access_token_classified`` → ``_obo_mint_rfc8693`` (refresh grant →
|
||||
RFC 8693 token exchange) against the live Keycloak token endpoint.
|
||||
|
||||
Checks E1–E7 mirror the Entra harness:
|
||||
E1 mint audience A → token, aud claim carries A, cache row refresh_token_ct NULL
|
||||
E2 second call → cache hit, ZERO extra Keycloak calls
|
||||
E3 audience B from the SAME captured credential → aud carries B
|
||||
E4 rotation write-back (KC rotates the RT on the refresh leg)
|
||||
E5 force_refresh → re-mint (Keycloak call count increments)
|
||||
E6 unconsented audience C → NOT token, credential SURVIVES
|
||||
E7 cache flush → re-mint
|
||||
|
||||
M1-M3 drive the MODEL-backend mint (``mint_obo_access_token``, #898/#955) on
|
||||
the same captured credential — the path an ``auth_mode=rfc8693_obo`` model
|
||||
alias takes, distinct from the classified MCP path above:
|
||||
M1 model mint audience A with the alias's exchange scopes → token carries A
|
||||
(the #955 fix: model definitions now carry per-row ``obo_scopes``, so
|
||||
the exchange leg requests the audience's scope exactly as MCP rows do)
|
||||
M2 warm re-mint serves the synthetic ``__model_obo__`` cache row —
|
||||
identity-keyed on the owning alias, audience + scopes in the row's
|
||||
own columns — with zero IdP calls
|
||||
M3 an entra-leg mode (``entra_obo``) on this rfc8693 deployment refuses
|
||||
BEFORE any IdP traffic, recording cause=grant_profile_mismatch — the
|
||||
mode/profile pairing that replaced the pre-#955 overload
|
||||
|
||||
Env (set by keycloak_e2e.sh):
|
||||
KC_TOKEN_ENDPOINT, KC_ISSUER, KC_CLIENT_ID, KC_CLIENT_SECRET,
|
||||
KC_USER, KC_PASSWORD, AUD_A, SCOPE_A, AUD_B, SCOPE_B, AUD_C
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from turnstone.core.mcp_crypto import (
|
||||
MCPTokenCipher,
|
||||
MCPTokenCipherConfig,
|
||||
MCPTokenStore,
|
||||
)
|
||||
from turnstone.core.mcp_oauth import (
|
||||
get_obo_access_token_classified,
|
||||
mint_obo_access_token,
|
||||
model_mint_refusal_cause,
|
||||
model_obo_cache_server,
|
||||
model_obo_cause_key,
|
||||
)
|
||||
from turnstone.core.oidc import OIDCConfig
|
||||
from turnstone.core.storage._sqlite import SQLiteBackend
|
||||
|
||||
USER = "e2e-user"
|
||||
RESULTS: list[tuple[str, str]] = []
|
||||
|
||||
|
||||
def record(status: str, msg: str) -> None:
|
||||
RESULTS.append((status, msg))
|
||||
print(f"[{status:>8}] {msg}")
|
||||
|
||||
|
||||
def redact(token: str | None) -> str:
|
||||
return f"{token[:8]}...({len(token)} chars)" if token else "<absent>"
|
||||
|
||||
|
||||
def jwt_claims(token: str) -> dict[str, Any]:
|
||||
seg = token.split(".")[1]
|
||||
pad = "=" * (-len(seg) % 4)
|
||||
out: dict[str, Any] = json.loads(base64.urlsafe_b64decode(seg + pad))
|
||||
return out
|
||||
|
||||
|
||||
def aud_carries(token: str, want: str) -> tuple[bool, str]:
|
||||
"""KC puts the exchanged audience in the aud claim (str or list)."""
|
||||
aud = jwt_claims(token).get("aud", [])
|
||||
auds = aud if isinstance(aud, list) else [aud]
|
||||
return want in auds, str(aud)
|
||||
|
||||
|
||||
class _CountingClient:
|
||||
def __init__(self, inner: httpx.AsyncClient) -> None:
|
||||
self._inner = inner
|
||||
self.posts = 0
|
||||
|
||||
async def post(self, *args: Any, **kwargs: Any) -> httpx.Response:
|
||||
self.posts += 1
|
||||
return await self._inner.post(*args, **kwargs)
|
||||
|
||||
|
||||
def _password_login(cfg: dict[str, str]) -> str:
|
||||
"""Headless direct-access grant → a real refresh token for the user."""
|
||||
resp = httpx.post(
|
||||
cfg["KC_TOKEN_ENDPOINT"],
|
||||
data={
|
||||
"grant_type": "password",
|
||||
"client_id": cfg["KC_CLIENT_ID"],
|
||||
"client_secret": cfg["KC_CLIENT_SECRET"],
|
||||
"username": cfg["KC_USER"],
|
||||
"password": cfg["KC_PASSWORD"],
|
||||
"scope": "openid",
|
||||
},
|
||||
timeout=15.0,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return str(resp.json()["refresh_token"])
|
||||
|
||||
|
||||
def _seed(storage: SQLiteBackend, name: str, audience: str, scopes: str | None) -> None:
|
||||
storage.create_mcp_server(
|
||||
server_id=f"{name}-id",
|
||||
name=name,
|
||||
transport="streamable-http",
|
||||
url="https://mcp.example.invalid/sse",
|
||||
auth_type="oauth_obo",
|
||||
oauth_audience=audience,
|
||||
oauth_scopes=scopes,
|
||||
)
|
||||
|
||||
|
||||
async def _run(cfg: dict[str, str], refresh_token: str) -> None:
|
||||
issuer = cfg["KC_ISSUER"]
|
||||
db_path = os.path.join(tempfile.mkdtemp(prefix="obo-kc-e2e-"), "e2e.db")
|
||||
storage = SQLiteBackend(db_path)
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
raw = base64.urlsafe_b64decode(Fernet.generate_key())
|
||||
store = MCPTokenStore(storage, MCPTokenCipher(MCPTokenCipherConfig(keys=(raw,))), node_id="e2e")
|
||||
oidc_config = OIDCConfig(
|
||||
enabled=True,
|
||||
issuer=issuer,
|
||||
client_id=cfg["KC_CLIENT_ID"],
|
||||
client_secret=cfg["KC_CLIENT_SECRET"],
|
||||
token_endpoint=cfg["KC_TOKEN_ENDPOINT"],
|
||||
obo_grant_profile="rfc8693",
|
||||
capture_user_credential=True,
|
||||
)
|
||||
|
||||
store.upsert_oidc_credential(USER, issuer, refresh_token=refresh_token)
|
||||
cap = store.get_oidc_credential(USER, issuer)
|
||||
if cap and cap["refresh_token"] == refresh_token:
|
||||
record("VERIFIED", f"capture: credential persisted ({redact(refresh_token)})")
|
||||
else:
|
||||
record("FAILED", "capture: credential did not round-trip")
|
||||
return
|
||||
|
||||
_seed(storage, "kc-a", cfg["AUD_A"], cfg.get("SCOPE_A"))
|
||||
_seed(storage, "kc-b", cfg["AUD_B"], cfg.get("SCOPE_B"))
|
||||
if cfg.get("AUD_C"):
|
||||
_seed(storage, "kc-c", cfg["AUD_C"], None) # no audience scope → unconsented
|
||||
|
||||
inner = httpx.AsyncClient(timeout=20.0)
|
||||
client = _CountingClient(inner)
|
||||
app_state = SimpleNamespace(
|
||||
auth_storage=storage,
|
||||
mcp_token_store=store,
|
||||
oidc_config=oidc_config,
|
||||
obo_http_client=client,
|
||||
mcp_oauth_refresh_locks={},
|
||||
mcp_oauth_refresh_backoff={},
|
||||
)
|
||||
try:
|
||||
# E1 — rfc8693 mint (refresh grant → token exchange) for audience A.
|
||||
r = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
if r.kind == "token" and r.token:
|
||||
ok, aud = aud_carries(r.token, cfg["AUD_A"])
|
||||
row = storage.get_mcp_user_token(USER, "kc-a")
|
||||
cache_ok = row is not None and row["refresh_token_ct"] is None
|
||||
record(
|
||||
"VERIFIED" if ok and cache_ok else "FAILED",
|
||||
f"E1 mint A (refresh→exchange): kind=token aud={aud} want={cfg['AUD_A']} "
|
||||
f"cache_row_refreshless={cache_ok}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E1 mint A: kind={r.kind} (expected token)")
|
||||
return
|
||||
|
||||
# E2 — cache hit.
|
||||
posts_before = client.posts
|
||||
r2 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r2.kind == "token" and client.posts == posts_before else "FAILED",
|
||||
f"E2 cache hit: kind={r2.kind} extra_kc_calls={client.posts - posts_before} (want 0)",
|
||||
)
|
||||
|
||||
# E3 — audience B from the SAME credential.
|
||||
rb = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-b"
|
||||
)
|
||||
if rb.kind == "token" and rb.token:
|
||||
ok_b, aud_b = aud_carries(rb.token, cfg["AUD_B"])
|
||||
record(
|
||||
"VERIFIED" if ok_b else "FAILED",
|
||||
f"E3 mint B from SAME credential: aud={aud_b} want={cfg['AUD_B']}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", f"E3 mint B: kind={rb.kind}")
|
||||
|
||||
# E4 — rotation write-back (KC rotates the RT on the refresh leg).
|
||||
cred_now = store.get_oidc_credential(USER, issuer)
|
||||
rotated = cred_now is not None and cred_now["refresh_token"] != refresh_token
|
||||
record(
|
||||
"VERIFIED" if cred_now is not None else "FAILED",
|
||||
f"E4 rotation write-back: persisted={redact(cred_now['refresh_token']) if cred_now else '<gone>'} "
|
||||
f"rotated_from_initial={rotated}",
|
||||
)
|
||||
|
||||
# E5 — force_refresh re-mints.
|
||||
posts_before = client.posts
|
||||
rf = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a", force_refresh=True
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if rf.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E5 force_refresh re-mint: kind={rf.kind} kc_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# E6 — unconsented audience: not a token, credential survives.
|
||||
if cfg.get("AUD_C"):
|
||||
rc = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-c"
|
||||
)
|
||||
cred_after = store.get_oidc_credential(USER, issuer)
|
||||
record(
|
||||
"VERIFIED" if rc.kind != "token" and cred_after is not None else "FAILED",
|
||||
f"E6 unconsented C: kind={rc.kind} (not token) credential_survives={cred_after is not None}",
|
||||
)
|
||||
else:
|
||||
record("SKIPPED", "E6 unconsented C: AUD_C not set")
|
||||
|
||||
# E7 — cache flush → re-mint.
|
||||
store.delete_user_token(USER, "kc-a")
|
||||
posts_before = client.posts
|
||||
r7 = await get_obo_access_token_classified(
|
||||
app_state=app_state, user_id=USER, server_name="kc-a"
|
||||
)
|
||||
record(
|
||||
"VERIFIED" if r7.kind == "token" and client.posts > posts_before else "FAILED",
|
||||
f"E7 flush→re-mint: kind={r7.kind} kc_calls={client.posts - posts_before} (want >=1)",
|
||||
)
|
||||
|
||||
# M1-M3 — MODEL backend mint on the rfc8693 profile: same captured
|
||||
# credential and legs as E1-E7, but through mint_obo_access_token —
|
||||
# the path an auth_mode=rfc8693_obo alias takes, carrying the
|
||||
# per-alias exchange scopes MCP rows always had (#955). The mint's
|
||||
# cache and cause records are identity-keyed on the owning alias, so
|
||||
# the harness names one per mode-variant exactly as a deployment
|
||||
# would define separate rows.
|
||||
posts_before = client.posts
|
||||
m1 = await mint_obo_access_token(
|
||||
app_state=app_state,
|
||||
user_id=USER,
|
||||
alias="model-a",
|
||||
audience=cfg["AUD_A"],
|
||||
scopes=cfg.get("SCOPE_A", ""),
|
||||
grant_leg="rfc8693",
|
||||
)
|
||||
m1_kc_calls = client.posts - posts_before
|
||||
if m1:
|
||||
ok1, why1 = aud_carries(m1, cfg["AUD_A"])
|
||||
record(
|
||||
"VERIFIED" if ok1 and m1_kc_calls > 0 else "FAILED",
|
||||
f"M1 model mint (rfc8693_obo, scoped exchange): token={redact(m1)} "
|
||||
f"aud_ok={ok1} ({why1}) kc_calls={m1_kc_calls} (want >=1)",
|
||||
)
|
||||
else:
|
||||
record(
|
||||
"FAILED",
|
||||
f"M1 model mint (rfc8693_obo): no token (kc_calls={m1_kc_calls}) — "
|
||||
"the #955 scope wire-through should mint here",
|
||||
)
|
||||
|
||||
# M2 — warm re-mint serves the synthetic __model_obo__ cache row —
|
||||
# identity-keyed on the owning alias, audience + scopes in the row's
|
||||
# own columns — with zero IdP calls, and the row is named so
|
||||
# deprovisioning can find it by prefix.
|
||||
posts_before = client.posts
|
||||
m2 = await mint_obo_access_token(
|
||||
app_state=app_state,
|
||||
user_id=USER,
|
||||
alias="model-a",
|
||||
audience=cfg["AUD_A"],
|
||||
scopes=cfg.get("SCOPE_A", ""),
|
||||
grant_leg="rfc8693",
|
||||
)
|
||||
cache_row = storage.get_mcp_user_token(USER, model_obo_cache_server("model-a"))
|
||||
if m1:
|
||||
record(
|
||||
"VERIFIED"
|
||||
if m2 and client.posts == posts_before and cache_row is not None
|
||||
else "FAILED",
|
||||
f"M2 model cache-hit: token={redact(m2)} kc_calls="
|
||||
f"{client.posts - posts_before} (want 0) synthetic_row="
|
||||
f"{'present' if cache_row is not None else 'MISSING'}",
|
||||
)
|
||||
else:
|
||||
record("FAILED", "M2 model cache-hit: blocked behind M1 — M1 failed, see above")
|
||||
|
||||
# M3 — the mode/profile pairing refusal that replaced the pre-#955
|
||||
# overload: an entra-leg mode on this rfc8693 deployment must yield
|
||||
# None with ZERO IdP calls and record the grant_profile_mismatch
|
||||
# cause the session heartbeat reads (under its own alias — a
|
||||
# deployment defines the entra-mode variant as its own row).
|
||||
posts_before = client.posts
|
||||
m3 = await mint_obo_access_token(
|
||||
app_state=app_state,
|
||||
user_id=USER,
|
||||
alias="model-a-entra",
|
||||
audience=cfg["AUD_A"],
|
||||
grant_leg="entra",
|
||||
)
|
||||
m3_cause = model_mint_refusal_cause(
|
||||
"model_obo", model_obo_cause_key("model-a-entra", grant_leg="entra"), USER
|
||||
)
|
||||
record(
|
||||
"VERIFIED"
|
||||
if m3 is None and client.posts == posts_before and m3_cause == "grant_profile_mismatch"
|
||||
else "FAILED",
|
||||
f"M3 mode/profile mismatch refusal: token={redact(m3)} (want absent) "
|
||||
f"kc_calls={client.posts - posts_before} (want 0) cause={m3_cause!r}",
|
||||
)
|
||||
finally:
|
||||
await inner.aclose()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = [
|
||||
"KC_TOKEN_ENDPOINT",
|
||||
"KC_ISSUER",
|
||||
"KC_CLIENT_ID",
|
||||
"KC_CLIENT_SECRET",
|
||||
"KC_USER",
|
||||
"KC_PASSWORD",
|
||||
"AUD_A",
|
||||
"AUD_B",
|
||||
]
|
||||
cfg = {k: os.environ[k] for k in os.environ if k.startswith(("KC_", "AUD_", "SCOPE_"))}
|
||||
missing = [k for k in required if not cfg.get(k)]
|
||||
if missing:
|
||||
print(f"Missing env: {', '.join(missing)} — run via keycloak_e2e.sh")
|
||||
return 2
|
||||
|
||||
print("Headless password login to Keycloak (the credential the feature captures)...")
|
||||
refresh_token = _password_login(cfg)
|
||||
|
||||
asyncio.run(_run(cfg, refresh_token))
|
||||
|
||||
print("\n=== summary ===")
|
||||
for status, msg in RESULTS:
|
||||
print(f" {status:>8} {msg}")
|
||||
return 0 if all(s in ("VERIFIED", "SKIPPED") for s, _ in RESULTS) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,65 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# OSS-path (RFC 8693) end-to-end: spin up ephemeral Keycloak, configure the
|
||||
# realm, run keycloak_e2e.py against the REAL Turnstone mint engine, tear down.
|
||||
# Fully headless — no browser. Manual test tooling, not run in CI.
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/../.." # repo root (uv run needs it)
|
||||
|
||||
CONTAINER=kc-obo-e2e
|
||||
PORT=8091
|
||||
KC="docker exec $CONTAINER /opt/keycloak/bin/kcadm.sh"
|
||||
|
||||
cleanup() { docker rm -f "$CONTAINER" >/dev/null 2>&1 || true; }
|
||||
trap cleanup EXIT
|
||||
cleanup
|
||||
|
||||
echo ">> starting Keycloak 26.3 (ephemeral)..."
|
||||
docker run -d --name "$CONTAINER" -p "127.0.0.1:${PORT}:8080" \
|
||||
-e KC_BOOTSTRAP_ADMIN_USERNAME=admin -e KC_BOOTSTRAP_ADMIN_PASSWORD=admin \
|
||||
quay.io/keycloak/keycloak:26.3 start-dev >/dev/null
|
||||
|
||||
echo ">> waiting for Keycloak (dev-mode boot can take a few minutes on a loaded host)..."
|
||||
# Wait on kcadm auth succeeding directly — more reliable than the host HTTP port,
|
||||
# and generous enough for a resource-starved boot (up to ~6 min).
|
||||
ready=""
|
||||
for _ in $(seq 1 90); do
|
||||
if $KC config credentials --server http://localhost:8080 --realm master \
|
||||
--user admin --password admin >/dev/null 2>&1; then
|
||||
ready=1
|
||||
break
|
||||
fi
|
||||
sleep 4
|
||||
done
|
||||
[ -n "$ready" ] || { echo "Keycloak did not become ready in time"; docker logs "$CONTAINER" 2>&1 | tail -15; exit 1; }
|
||||
|
||||
echo ">> configuring realm 'spike'..."
|
||||
$KC create realms -s realm=spike -s enabled=true >/dev/null
|
||||
# Confidential client with standard token exchange (the RFC 8693 leg) + direct
|
||||
# access grant (headless password login to fetch the user's refresh token).
|
||||
$KC create clients -r spike -s clientId=turnstone -s enabled=true -s publicClient=false \
|
||||
-s secret=spike-secret -s directAccessGrantsEnabled=true \
|
||||
-s 'attributes={"standard.token.exchange.enabled":"true"}' >/dev/null
|
||||
for t in mcp-a mcp-b mcp-c; do
|
||||
$KC create clients -r spike -s clientId=$t -s enabled=true -s publicClient=false -s secret=x >/dev/null
|
||||
done
|
||||
$KC create users -r spike -s username=e2e-user -s enabled=true -s email=e2e@spike.test \
|
||||
-s emailVerified=true -s firstName=E2E -s lastName=User >/dev/null
|
||||
$KC set-password -r spike --username e2e-user --new-password e2e-pw >/dev/null
|
||||
|
||||
TURNSTONE_UUID=$($KC get clients -r spike -q clientId=turnstone --fields id --format csv --noquotes)
|
||||
# Audience client scopes for mcp-a and mcp-b ONLY (mcp-c stays unconsented → E6).
|
||||
for t in mcp-a mcp-b; do
|
||||
SID=$($KC create client-scopes -r spike -s name=aud-$t -s protocol=openid-connect -i)
|
||||
$KC create "client-scopes/$SID/protocol-mappers/models" -r spike -s name=aud-$t \
|
||||
-s protocol=openid-connect -s protocolMapper=oidc-audience-mapper \
|
||||
-s "config={\"included.client.audience\":\"$t\",\"access.token.claim\":\"true\"}" >/dev/null
|
||||
$KC update "clients/$TURNSTONE_UUID/optional-client-scopes/$SID" -r spike >/dev/null
|
||||
done
|
||||
|
||||
echo ">> running the product e2e harness..."
|
||||
export KC_TOKEN_ENDPOINT="http://127.0.0.1:${PORT}/realms/spike/protocol/openid-connect/token"
|
||||
export KC_ISSUER="http://127.0.0.1:${PORT}/realms/spike"
|
||||
export KC_CLIENT_ID=turnstone KC_CLIENT_SECRET=spike-secret
|
||||
export KC_USER=e2e-user KC_PASSWORD=e2e-pw
|
||||
export AUD_A=mcp-a SCOPE_A=aud-mcp-a AUD_B=mcp-b SCOPE_B=aud-mcp-b AUD_C=mcp-c
|
||||
uv run python scripts/obo-e2e/keycloak_e2e.py
|
||||
File diff suppressed because it is too large
Load Diff
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Console API",
|
||||
"version": "1.8.0a5",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Cluster-wide visibility and control across all turnstone nodes."
|
||||
},
|
||||
"paths": {
|
||||
@@ -3723,7 +3723,7 @@
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelDefinitionWriteResponse"
|
||||
"$ref": "#/components/schemas/ModelDefinitionInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3738,16 +3738,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "Error 403",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
@@ -3757,47 +3747,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/admin/model-definitions/auth-constraints": {
|
||||
"get": {
|
||||
"summary": "Dynamic-auth affordance data for the model editor (requires admin.mcp)",
|
||||
"operationId": "v1_api_admin_model-definitions_auth-constraints_get",
|
||||
"tags": [
|
||||
"Admin"
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelAuthConstraintsResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "Error 403",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3895,7 +3844,7 @@
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelDefinitionWriteResponse"
|
||||
"$ref": "#/components/schemas/ModelDefinitionInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3910,16 +3859,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "Error 403",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Error 404",
|
||||
"content": {
|
||||
@@ -3939,16 +3878,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -3970,14 +3899,7 @@
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/DeleteModelDefinitionResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
"description": "Success"
|
||||
},
|
||||
"404": {
|
||||
"description": "Error 404",
|
||||
@@ -4070,26 +3992,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"500": {
|
||||
"description": "Error 500",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7889,12 +7791,6 @@
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"description": "Project to attach the workstream to (validated against membership, empty = none)",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"resume_ws": {
|
||||
"default": "",
|
||||
"description": "Workstream ID to resume (loads previous conversation)",
|
||||
@@ -8452,7 +8348,7 @@
|
||||
"type": "string"
|
||||
},
|
||||
"status": {
|
||||
"description": "One of: pending / in_progress / done / blocked / needs_user. ``blocked`` is waiting on a dependency the coordinator may clear itself; ``needs_user`` is waiting on a decision only the operator can make.",
|
||||
"description": "One of: pending / in_progress / done / blocked.",
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
@@ -8461,12 +8357,6 @@
|
||||
"title": "Child Ws Id",
|
||||
"type": "string"
|
||||
},
|
||||
"note": {
|
||||
"default": "",
|
||||
"description": "Optional one-sentence note, typically what the coordinator needs from the operator on a ``needs_user`` task. Absent from the stored record when unset (there is no backfill for rows written before the field existed), so it defaults to the empty string here.",
|
||||
"title": "Note",
|
||||
"type": "string"
|
||||
},
|
||||
"created": {
|
||||
"title": "Created",
|
||||
"type": "string"
|
||||
@@ -8606,18 +8496,6 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"description": "Persona slug (empty = kind default)",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"description": "Project to attach the workstream to",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"description": "Notification targets on completion (channel_type + channel_id/user_id)",
|
||||
"items": {
|
||||
@@ -8781,30 +8659,6 @@
|
||||
"default": null,
|
||||
"title": "Skill"
|
||||
},
|
||||
"persona": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Persona"
|
||||
},
|
||||
"project_id": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"notify_targets": {
|
||||
"anyOf": [
|
||||
{
|
||||
@@ -8900,16 +8754,6 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"project_id": {
|
||||
"default": "",
|
||||
"title": "Project Id",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"items": {
|
||||
"additionalProperties": {
|
||||
@@ -10922,21 +10766,6 @@
|
||||
"title": "Replay Reasoning To Model",
|
||||
"type": "boolean"
|
||||
},
|
||||
"auth_mode": {
|
||||
"default": "static",
|
||||
"title": "Auth Mode",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_audience": {
|
||||
"default": "",
|
||||
"title": "Obo Audience",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"default": "",
|
||||
"title": "Obo Scopes",
|
||||
"type": "string"
|
||||
},
|
||||
"source": {
|
||||
"default": "",
|
||||
"title": "Source",
|
||||
@@ -10966,147 +10795,6 @@
|
||||
"title": "ModelDefinitionInfo",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelDefinitionWriteResponse": {
|
||||
"description": "Create/update response: the stored row plus an optional caveat.\n\n``registry_warning`` is present only when the DB write succeeded but\nTHIS console's live coordinator registry refused to adopt it (keyless\nhost with dynamic-auth rows): the row is saved, yet running sessions\nkeep the previous config until the deployment fault is remedied.\nClients should surface it as a warning beside the success, never as a\nfailure. Absent on a clean save (SkipJsonSchema: the server omits the\nkey rather than sending null).",
|
||||
"properties": {
|
||||
"definition_id": {
|
||||
"title": "Definition Id",
|
||||
"type": "string"
|
||||
},
|
||||
"alias": {
|
||||
"title": "Alias",
|
||||
"type": "string"
|
||||
},
|
||||
"model": {
|
||||
"title": "Model",
|
||||
"type": "string"
|
||||
},
|
||||
"provider": {
|
||||
"default": "openai",
|
||||
"title": "Provider",
|
||||
"type": "string"
|
||||
},
|
||||
"base_url": {
|
||||
"default": "",
|
||||
"title": "Base Url",
|
||||
"type": "string"
|
||||
},
|
||||
"api_key": {
|
||||
"default": "",
|
||||
"title": "Api Key",
|
||||
"type": "string"
|
||||
},
|
||||
"context_window": {
|
||||
"default": 32768,
|
||||
"title": "Context Window",
|
||||
"type": "integer"
|
||||
},
|
||||
"capabilities": {
|
||||
"default": "{}",
|
||||
"title": "Capabilities",
|
||||
"type": "string"
|
||||
},
|
||||
"enabled": {
|
||||
"default": true,
|
||||
"title": "Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"temperature": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "number"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Temperature"
|
||||
},
|
||||
"max_tokens": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Max Tokens"
|
||||
},
|
||||
"reasoning_effort": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Reasoning Effort"
|
||||
},
|
||||
"surface_persisted_reasoning": {
|
||||
"default": true,
|
||||
"title": "Surface Persisted Reasoning",
|
||||
"type": "boolean"
|
||||
},
|
||||
"replay_reasoning_to_model": {
|
||||
"default": false,
|
||||
"title": "Replay Reasoning To Model",
|
||||
"type": "boolean"
|
||||
},
|
||||
"auth_mode": {
|
||||
"default": "static",
|
||||
"title": "Auth Mode",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_audience": {
|
||||
"default": "",
|
||||
"title": "Obo Audience",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"default": "",
|
||||
"title": "Obo Scopes",
|
||||
"type": "string"
|
||||
},
|
||||
"source": {
|
||||
"default": "",
|
||||
"title": "Source",
|
||||
"type": "string"
|
||||
},
|
||||
"created_by": {
|
||||
"default": "",
|
||||
"title": "Created By",
|
||||
"type": "string"
|
||||
},
|
||||
"created": {
|
||||
"default": "",
|
||||
"title": "Created",
|
||||
"type": "string"
|
||||
},
|
||||
"updated": {
|
||||
"default": "",
|
||||
"title": "Updated",
|
||||
"type": "string"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the save landed but this console's live registry refused the swap (e.g. dynamic auth configured without the startup encryption key); carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"definition_id",
|
||||
"alias",
|
||||
"model"
|
||||
],
|
||||
"title": "ModelDefinitionWriteResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"CreateModelDefinitionRequest": {
|
||||
"properties": {
|
||||
"alias": {
|
||||
@@ -11192,21 +10880,6 @@
|
||||
"default": false,
|
||||
"title": "Replay Reasoning To Model",
|
||||
"type": "boolean"
|
||||
},
|
||||
"auth_mode": {
|
||||
"default": "static",
|
||||
"title": "Auth Mode",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_audience": {
|
||||
"default": "",
|
||||
"title": "Obo Audience",
|
||||
"type": "string"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"default": "",
|
||||
"title": "Obo Scopes",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -11291,11 +10964,17 @@
|
||||
"title": "Context Window"
|
||||
},
|
||||
"capabilities": {
|
||||
"additionalProperties": true,
|
||||
"anyOf": [
|
||||
{
|
||||
"additionalProperties": true,
|
||||
"type": "object"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Full replacement capabilities object. Omit to leave the stored value unchanged; JSON null is refused (400).",
|
||||
"title": "Capabilities",
|
||||
"type": "object"
|
||||
"title": "Capabilities"
|
||||
},
|
||||
"enabled": {
|
||||
"anyOf": [
|
||||
@@ -11368,42 +11047,6 @@
|
||||
],
|
||||
"default": null,
|
||||
"title": "Replay Reasoning To Model"
|
||||
},
|
||||
"auth_mode": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Auth Mode"
|
||||
},
|
||||
"obo_audience": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Obo Audience"
|
||||
},
|
||||
"obo_scopes": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Obo Scopes"
|
||||
}
|
||||
},
|
||||
"title": "UpdateModelDefinitionRequest",
|
||||
@@ -11417,80 +11060,14 @@
|
||||
},
|
||||
"title": "Models",
|
||||
"type": "array"
|
||||
},
|
||||
"default_alias": {
|
||||
"description": "Effective default alias after the config/enabled-list fallbacks",
|
||||
"title": "Default Alias",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"models",
|
||||
"default_alias"
|
||||
"models"
|
||||
],
|
||||
"title": "ListModelDefinitionsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelAuthConstraintsResponse": {
|
||||
"description": "Affordance data for the model shelf's Backend-auth section.\n\nSuggestions and labels only \u2014 never a gate. The write validator is the\nauthority; a client that fails to fetch this must degrade to free-text\ninput with server-side validation, not to a refusal.",
|
||||
"properties": {
|
||||
"auth_audience_allowlist": {
|
||||
"description": "Exact gateway audiences a definition may use with entra_obo / entra_app, rendered as input suggestions. Empty means none are registered yet; writes are refused until an operator populates model.auth_audience_allowlist.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Auth Audience Allowlist",
|
||||
"type": "array"
|
||||
},
|
||||
"auth_grant_profile": {
|
||||
"description": "Deployment [oidc] obo_grant_profile, or empty when single sign-on is not configured. Each dynamic auth_mode pairs with exactly one profile (see auth_mode_profiles); the write validator refuses a new pairing that contradicts it. A transient discovery outage reports the configured profile, not empty.",
|
||||
"title": "Auth Grant Profile",
|
||||
"type": "string"
|
||||
},
|
||||
"dynamic_auth_modes": {
|
||||
"description": "auth_mode values that mint per-call backend credentials, derived server-side from the registry's mode classification so the shelf's affordances (audience enable/require, section visibility) track it by data. Clients keep a hand-listed fallback only for a missing or failed constraints fetch.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Dynamic Auth Modes",
|
||||
"type": "array"
|
||||
},
|
||||
"scopes_auth_modes": {
|
||||
"description": "auth_mode values whose mint reads obo_scopes (the token-exchange scope request), same server-derived contract as dynamic_auth_modes; drives the scopes input's visibility.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Scopes Auth Modes",
|
||||
"type": "array"
|
||||
},
|
||||
"app_identity_auth_modes": {
|
||||
"description": "auth_mode values that mint a shared app/deployment identity rather than a per-user one, same server-derived contract as dynamic_auth_modes; drives the model list's auth badge wording.",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "App Identity Auth Modes",
|
||||
"type": "array"
|
||||
},
|
||||
"auth_mode_profiles": {
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
},
|
||||
"description": "Required [oidc] obo_grant_profile per dynamic auth_mode. Affordance for greying options that cannot validate under this deployment's profile; the write validator remains the authority.",
|
||||
"title": "Auth Mode Profiles",
|
||||
"type": "object"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"auth_audience_allowlist",
|
||||
"auth_grant_profile",
|
||||
"dynamic_auth_modes",
|
||||
"scopes_auth_modes",
|
||||
"app_identity_auth_modes",
|
||||
"auth_mode_profiles"
|
||||
],
|
||||
"title": "ModelAuthConstraintsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"PersonaInfo": {
|
||||
"description": "Full persona row \u2014 the authoring shape (contrast PersonaChoice, the\npicker's display-only projection on the server surface).",
|
||||
"properties": {
|
||||
@@ -11841,42 +11418,11 @@
|
||||
"additionalProperties": true,
|
||||
"title": "Results",
|
||||
"type": "object"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the node fan-out ran but THIS console's live registry refused the swap (e.g. dynamic auth configured without the startup encryption key); carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"title": "ModelReloadResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"DeleteModelDefinitionResponse": {
|
||||
"description": "Delete response: the removed row id plus an optional caveat.\n\n``registry_warning`` mirrors ModelDefinitionWriteResponse: the DB row\nis gone, but a keyless console's live registry refused the swap and\nkeeps SERVING the deleted alias to running and new coordinator\nsessions until the deployment fault is remedied.",
|
||||
"properties": {
|
||||
"status": {
|
||||
"default": "ok",
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
"definition_id": {
|
||||
"title": "Definition Id",
|
||||
"type": "string"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the delete landed but this console's live registry refused the swap and keeps serving the deleted alias; carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"definition_id"
|
||||
],
|
||||
"title": "DeleteModelDefinitionResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"DetectModelRequest": {
|
||||
"properties": {
|
||||
"provider": {
|
||||
@@ -12043,12 +11589,6 @@
|
||||
"default": "",
|
||||
"title": "Error",
|
||||
"type": "string"
|
||||
},
|
||||
"registry_warning": {
|
||||
"default": null,
|
||||
"description": "Set when the calibration was stored but this console's live registry refused the swap; carries the operator-facing remediation text.",
|
||||
"title": "Registry Warning",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"title": "CalibrateModelResponse",
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Server API",
|
||||
"version": "1.8.0a5",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Single-node workstream management, chat interaction, and real-time streaming."
|
||||
},
|
||||
"paths": {
|
||||
@@ -228,16 +228,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -400,26 +390,6 @@
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"409": {
|
||||
"description": "Error 409",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"503": {
|
||||
"description": "Error 503",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -639,7 +609,7 @@
|
||||
"tags": [
|
||||
"Streaming"
|
||||
],
|
||||
"description": "Server-Sent Events stream for node-level state broadcasts. Emits a node_snapshot event on connect (workstreams, health, aggregate), followed by real-time delta events (ws_state, ws_activity, ws_created, ws_closed, ws_rename, health_changed, aggregate). Pass ?expected_node_id=X for identity verification (returns 409 on mismatch). Every event's SSE id is an opaque '{boot_epoch}-{counter}' string; presenting it on reconnect (Last-Event-ID header or ?last_event_id=) replays missed events, or emits a replay_truncated event (reason: ring_evicted with lost_count + earliest_available_id, or boot_epoch when the cursor predates this server process) followed by a fresh node_snapshot. Treat the id as opaque \u2014 its format may change.",
|
||||
"description": "Server-Sent Events stream for node-level state broadcasts. Emits a node_snapshot event on connect (workstreams, health, aggregate), followed by real-time delta events (ws_state, ws_activity, ws_created, ws_closed, ws_rename, health_changed, aggregate). Pass ?expected_node_id=X for identity verification (returns 409 on mismatch).",
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success"
|
||||
@@ -2290,23 +2260,16 @@
|
||||
"SendResponse": {
|
||||
"properties": {
|
||||
"status": {
|
||||
"description": "'ok' (fresh turn dispatched), 'queued' (folded into the live turn's interjection queue, or \u2014 when `deferred` is true \u2014 parked for dispatch after the current command window), 'queue_full', 'attachments_busy' (attachments can't ride a queued turn; retry when idle), or 'cross_user_interjection' (another participant's turn is in flight; carried on the 409 body).",
|
||||
"description": "'ok', 'busy', 'queued', or 'queue_full'",
|
||||
"examples": [
|
||||
"ok",
|
||||
"busy",
|
||||
"queued",
|
||||
"queue_full",
|
||||
"attachments_busy",
|
||||
"cross_user_interjection"
|
||||
"queue_full"
|
||||
],
|
||||
"title": "Status",
|
||||
"type": "string"
|
||||
},
|
||||
"deferred": {
|
||||
"default": false,
|
||||
"description": "Set on `queued` responses: the message is parked on the workstream's deferred-send list (a slash-command window holds the worker slot, or earlier deferred sends are still pending) and dispatches as an ordinary full-fidelity send afterwards \u2014 it is NOT in a live turn's interjection queue. `DELETE .../send` retracts it until dispatch. Node-local and in-memory: a node restart before dispatch drops it (at-most-once intake).",
|
||||
"title": "Deferred",
|
||||
"type": "boolean"
|
||||
},
|
||||
"attached_ids": {
|
||||
"description": "Attachment ids actually attached to this turn. Subset of the request's `attachment_ids` (or the auto-consumed pending set). Empty when the send carries no attachments.",
|
||||
"items": {
|
||||
|
||||
Generated
+173
-534
File diff suppressed because it is too large
Load Diff
@@ -32,7 +32,7 @@
|
||||
],
|
||||
"license": "Apache-2.0",
|
||||
"devDependencies": {
|
||||
"typescript": "^7.0.0",
|
||||
"typescript": "^6.0.0",
|
||||
"vitest": "^4.1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -157,49 +157,6 @@ export interface CancelledEvent {
|
||||
type: "cancelled";
|
||||
}
|
||||
|
||||
/**
|
||||
* Context-compaction lifecycle. `start` carries `trigger` ("manual"/"auto";
|
||||
* auto adds `where` + `pct`); `progress` carries chunked-summarization
|
||||
* `part`/`total`/`depth` (or `retry_in`/`error` for a retry wait); `end`
|
||||
* carries `ok` plus either `before_tokens`/`after_tokens`/`summary` or the
|
||||
* failure `reason`/`message`. The successful end's summary also replays from
|
||||
* `/history` as a `role: "system"`, `source: "compaction"` entry.
|
||||
*/
|
||||
export interface CompactionEvent {
|
||||
type: "compaction";
|
||||
phase: "start" | "progress" | "end";
|
||||
/** Correlates every event of one compaction run (0 from legacy emitters). */
|
||||
compaction_id?: number;
|
||||
/**
|
||||
* End events only: true marks a force-abandoned compaction retiring
|
||||
* after a successor generation took over — skip failure notices for
|
||||
* those (an OK end's result still stands; the history swap happened).
|
||||
*/
|
||||
superseded?: boolean;
|
||||
/**
|
||||
* Failed ends only: the emitter-computed display verdict — show
|
||||
* `message` only when true, instead of re-deriving suppression from
|
||||
* reason/trigger/superseded client-side.
|
||||
*/
|
||||
notice?: boolean;
|
||||
/** Present on start and on every end (ok or failed). */
|
||||
trigger?: "manual" | "auto";
|
||||
where?: string;
|
||||
pct?: number;
|
||||
part?: number;
|
||||
total?: number;
|
||||
depth?: number;
|
||||
retry_in?: number;
|
||||
error?: string;
|
||||
warning?: string;
|
||||
ok?: boolean;
|
||||
reason?: string;
|
||||
message?: string;
|
||||
before_tokens?: number;
|
||||
after_tokens?: number;
|
||||
summary?: string;
|
||||
}
|
||||
|
||||
// Global events
|
||||
|
||||
export interface WsStateEvent {
|
||||
@@ -255,7 +212,6 @@ export type ServerEvent =
|
||||
| BusyErrorEvent
|
||||
| ClearUiEvent
|
||||
| CancelledEvent
|
||||
| CompactionEvent
|
||||
| WsStateEvent
|
||||
| WsActivityEvent
|
||||
| WsRenameEvent
|
||||
|
||||
@@ -74,7 +74,7 @@ class _FakeConfigStore:
|
||||
def _fake_registry() -> MagicMock:
|
||||
"""MagicMock whose ``.resolve()`` succeeds so the 503 gate passes."""
|
||||
reg = MagicMock()
|
||||
reg.resolve.return_value = (MagicMock(), "gpt-4", MagicMock(), 0)
|
||||
reg.resolve.return_value = (MagicMock(), "gpt-4", MagicMock())
|
||||
return reg
|
||||
|
||||
|
||||
|
||||
@@ -1,45 +0,0 @@
|
||||
"""Shared helpers for the Python-driven node harnesses that evaluate the
|
||||
``shared_static`` ES modules with script semantics."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import shutil
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def has_node() -> bool:
|
||||
return shutil.which("node") is not None
|
||||
|
||||
|
||||
# Module-level ``pytestmark = node_skip`` in each harness suite — the node
|
||||
# detection lives here once, so a future change (version floor, env
|
||||
# override) cannot land in one suite and silently miss another.
|
||||
node_skip = pytest.mark.skipif(not has_node(), reason="node not available")
|
||||
|
||||
|
||||
def demodulize(path: Path) -> str:
|
||||
"""Strip ES-module syntax so ``vm.runInThisContext`` (script semantics)
|
||||
can evaluate the file: imports drop (the harness loads the whole
|
||||
dependency set into one shared context, so cross-file bindings resolve
|
||||
as context globals, exactly like the pre-module classic scripts), and
|
||||
``export`` keywords peel off their declarations.
|
||||
|
||||
Single-sourced here for every JS harness: a new module syntax form
|
||||
(``export default``, re-exports) must be handled once, not per suite —
|
||||
a divergence between per-file copies surfaces as a confusing
|
||||
``vm.runInThisContext`` SyntaxError in whichever suite lagged.
|
||||
"""
|
||||
src = path.read_text(encoding="utf-8")
|
||||
src = re.sub(r"^import\s+\{[\s\S]*?\}\s+from\s+\"[^\"]+\";\s*$", "", src, flags=re.M)
|
||||
src = re.sub(r"^import\s+[^;\n]+;\s*$", "", src, flags=re.M)
|
||||
src = re.sub(
|
||||
r"^export\s+(?=(?:async\s+)?(?:function|const|let|var|class)\b)", "", src, flags=re.M
|
||||
)
|
||||
src = re.sub(r"^export\s*\{[^}]*\};\s*$", "", src, flags=re.M)
|
||||
return src
|
||||
@@ -1,57 +0,0 @@
|
||||
"""Shared OIDC posture builder for the model-auth / OBO test surface.
|
||||
|
||||
One construction site for the posture the mint and write-validator suites
|
||||
read, built as a REAL (frozen) ``OIDCConfig`` so an override for a field
|
||||
the dataclass does not carry raises at the call site. Named with a leading
|
||||
underscore so pytest does not collect it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from turnstone.core.oidc import OIDCConfig
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
# The issuer / token-endpoint pair the mint suites route their mock
|
||||
# transports on.
|
||||
ISSUER = "https://idp.test"
|
||||
TOKEN_ENDPOINT = "https://idp.test/token"
|
||||
|
||||
|
||||
def make_oidc_config(**overrides: Any) -> OIDCConfig:
|
||||
"""A full, mintable OIDC posture; tests override the field under test,
|
||||
everything else rides the dataclass defaults."""
|
||||
defaults: dict[str, Any] = {
|
||||
"enabled": True,
|
||||
"issuer": ISSUER,
|
||||
"client_id": "cid",
|
||||
"client_secret": "csecret",
|
||||
"token_endpoint": TOKEN_ENDPOINT,
|
||||
}
|
||||
defaults.update(overrides)
|
||||
return OIDCConfig(**defaults)
|
||||
|
||||
|
||||
def keyed_app_state() -> SimpleNamespace:
|
||||
"""App-state stub satisfying ``ModelRegistry.reload``'s dynamic-auth key
|
||||
guard, for suites exercising reload mechanics rather than key policy."""
|
||||
return SimpleNamespace(mcp_token_store=object())
|
||||
|
||||
|
||||
def mint_warn_state_reset() -> Iterator[None]:
|
||||
"""Reset generator behind the mint suites' autouse fixtures: empties the
|
||||
process-global mint warn/dedup/cause state before AND after each test,
|
||||
so warn-dedup assertions are not order-dependent. Modules install it as
|
||||
``yield from mint_warn_state_reset()`` in an autouse fixture.
|
||||
"""
|
||||
# Lazy import: non-mint consumers of this helper module (the write-
|
||||
# validator suites) shouldn't pay the mcp_oauth import.
|
||||
from turnstone.core.mcp_oauth import reset_model_mint_warn_state_for_tests
|
||||
|
||||
reset_model_mint_warn_state_for_tests()
|
||||
yield
|
||||
reset_model_mint_warn_state_for_tests()
|
||||
@@ -1,230 +0,0 @@
|
||||
"""#832 replay-parity harness: scenario table + runner.
|
||||
|
||||
The audit is controller determinism: with the plant's chunk sequence held
|
||||
fixed, the streaming phase must produce an identical UI event sequence
|
||||
and an identical committed message — modulo the RULED behavior changes
|
||||
restated in full on the transforms in ``test_832_parity.py``. This
|
||||
module is the shared half: the scenario scripts (one row per
|
||||
chunk-field→UI translation the consumer performs) and the runner that
|
||||
drives one through the streaming seam, recording everything the turn
|
||||
observably produced.
|
||||
|
||||
Baselines are captured from the PRE-FOLD path (``UPDATE_832_PARITY=1``,
|
||||
run at a tree where ``session.py`` is byte-identical to pre-fold main)
|
||||
into ``tests/data/parity_832/``. The runner adapts to EITHER world by
|
||||
signature, so a recapture at an old tree records real old-world
|
||||
behavior, and capture mode refuses to write a record whose failure is
|
||||
the harness's own call shape. Assert mode replays the same scripts
|
||||
through the current tree and compares against the baseline, applying the
|
||||
ruled transforms; a mismatch outside a ruled transform is a regression.
|
||||
|
||||
The provider fake arms ``cancel_ref`` EAGERLY (a closeable sentinel
|
||||
appended inside ``create_streaming``, before the iterator is returned),
|
||||
mirroring every real adapter: the wrapper classifies
|
||||
creation-vs-midstream failures by that arming, so a fake that skipped it
|
||||
would exercise only the creation arm.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import inspect
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from tests._session_helpers import RecordingUI, make_session, scripted_provider
|
||||
from turnstone.core.providers._protocol import StreamChunk, ToolCallDelta, UsageInfo
|
||||
from turnstone.core.trajectory import Turn
|
||||
|
||||
FIXTURE_DIR = Path(__file__).parent / "data" / "parity_832"
|
||||
UPDATE = os.environ.get("UPDATE_832_PARITY") == "1"
|
||||
|
||||
|
||||
def _tc(index: int, call_id: str, name: str = "", args: str = "") -> ToolCallDelta:
|
||||
return ToolCallDelta(index=index, id=call_id, name=name, arguments_delta=args)
|
||||
|
||||
|
||||
_USAGE_A = UsageInfo(prompt_tokens=11, completion_tokens=0, total_tokens=11)
|
||||
_USAGE_B = UsageInfo(prompt_tokens=11, completion_tokens=7, total_tokens=18)
|
||||
|
||||
# Scenario table — the V11 grid, one script per row. Scripts are chunk
|
||||
# LISTS; the runner re-iterates a fresh iterator per attempt.
|
||||
SCENARIOS: dict[str, list[StreamChunk]] = {
|
||||
"content_only": [
|
||||
StreamChunk(content_delta="Hello "),
|
||||
StreamChunk(content_delta="world."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"reasoning_then_content": [
|
||||
StreamChunk(reasoning_delta="think a", usage=_USAGE_A),
|
||||
StreamChunk(reasoning_delta=" think b"),
|
||||
StreamChunk(content_delta="Answer."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"tools_simple": [
|
||||
StreamChunk(content_delta="Calling."),
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "call_1", "get_weather", '{"city": ')]),
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "", "", '"Paris"}')]),
|
||||
StreamChunk(finish_reason="tool_calls", usage=_USAGE_B),
|
||||
],
|
||||
"combined_content_tools_finish": [
|
||||
StreamChunk(content_delta="Before "),
|
||||
StreamChunk(
|
||||
content_delta="tools",
|
||||
tool_call_deltas=[_tc(0, "call_1", "get_weather", '{"city": "Nice"}')],
|
||||
finish_reason="tool_calls",
|
||||
),
|
||||
StreamChunk(usage=_USAGE_B),
|
||||
],
|
||||
"info_prefinish": [
|
||||
StreamChunk(info_delta="[Searching: pinniped taxonomy]"),
|
||||
StreamChunk(content_delta="Seals are pinnipeds."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"info_postfinish_footer": [
|
||||
StreamChunk(content_delta="Answer with sources."),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
StreamChunk(info_delta="Sources:\n- example.com/page"),
|
||||
],
|
||||
"think_tags_split_across_chunks": [
|
||||
StreamChunk(content_delta="<thi"),
|
||||
StreamChunk(content_delta="nk>plan</think>\n\nAnswer"),
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"blank_id_tools": [
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "", "get_weather", '{"city": "Oslo"}')]),
|
||||
StreamChunk(finish_reason="tool_calls", usage=_USAGE_B),
|
||||
],
|
||||
"length_with_tools": [
|
||||
StreamChunk(content_delta="Partial answer"),
|
||||
StreamChunk(tool_call_deltas=[_tc(0, "call_1", "get_weather", '{"city": "Par')]),
|
||||
StreamChunk(finish_reason="length", usage=_USAGE_B),
|
||||
],
|
||||
"content_filter": [
|
||||
StreamChunk(content_delta="Redac"),
|
||||
StreamChunk(finish_reason="content_filter", usage=_USAGE_B),
|
||||
],
|
||||
"no_finish_clean_exhaust": [
|
||||
StreamChunk(content_delta="Half an ans"),
|
||||
StreamChunk(usage=_USAGE_A),
|
||||
],
|
||||
"finish_only_no_content": [
|
||||
StreamChunk(finish_reason="stop", usage=_USAGE_B),
|
||||
],
|
||||
"provider_blocks_on_terminal": [
|
||||
StreamChunk(content_delta="Blocked."),
|
||||
StreamChunk(
|
||||
finish_reason="stop",
|
||||
usage=_USAGE_B,
|
||||
provider_blocks=[{"type": "reasoning_text", "text": "captured"}],
|
||||
),
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
_SYNTH_ID = re.compile(r"^call_[0-9a-f]{32}$")
|
||||
|
||||
|
||||
def _mask_synth_ids(record: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Replace uuid-backfilled tool-call ids with stable placeholders.
|
||||
|
||||
The blank-id repair mints ``call_<uuid4hex>`` per run — real
|
||||
nondeterminism inside the seam, but not behavior: mask ONLY that exact
|
||||
shape (never a scripted provider id) with an index-stable token so
|
||||
captures compare across runs. Applied to the committed projection;
|
||||
UI events never carry call ids in this harness.
|
||||
"""
|
||||
result = record.get("result")
|
||||
if not result:
|
||||
return record
|
||||
for i, tc in enumerate(result.get("tool_calls") or []):
|
||||
if _SYNTH_ID.match(tc.get("id", "")):
|
||||
tc["id"] = f"synth-id-{i}"
|
||||
for i, block in enumerate(result.get("provider_content") or []):
|
||||
if isinstance(block, dict) and _SYNTH_ID.match(str(block.get("id", ""))):
|
||||
block["id"] = f"synth-id-{i}"
|
||||
return record
|
||||
|
||||
|
||||
def run_scenario(name: str) -> dict[str, Any]:
|
||||
"""Drive one scenario through the streaming seam; return the record.
|
||||
|
||||
The record is everything the streaming phase observably produced: the
|
||||
ordered UI events, the committed-message projection, the mid-stream
|
||||
usage slot, and the exception class if the seam raised. Deliberately
|
||||
seam-level at ``_stream_response`` — full ``send()`` scenarios ride
|
||||
the ported ladder suites instead.
|
||||
|
||||
Signature-adaptive so ``UPDATE_832_PARITY=1`` at a PRE-fold tree
|
||||
records real old-world behavior: the pre-fold seam was
|
||||
``_stream_response(msgs, my_generation) -> dict``, the post-fold one
|
||||
is ``_stream_response(my_generation) -> ModelTurnResult`` (wire
|
||||
prepared inside). A harness-shape failure must never be recorded as
|
||||
behavior — ``write_fixture`` refuses one.
|
||||
"""
|
||||
ui = RecordingUI()
|
||||
session = make_session(ui=ui)
|
||||
# Zero the ladder backoff: a scenario that reaches the mid-stream
|
||||
# re-issue ladder (no_finish_clean_exhaust) must not sleep real
|
||||
# exponential delays in a unit run. The retry-notice transform in
|
||||
# test_832_parity hardcodes the matching "0s" wording.
|
||||
session._RETRY_BASE_DELAY = 0
|
||||
session._provider = scripted_provider(SCENARIOS[name])
|
||||
|
||||
pre_fold = "msgs" in inspect.signature(type(session)._stream_response).parameters
|
||||
record: dict[str, Any] = {"scenario": name}
|
||||
try:
|
||||
if pre_fold:
|
||||
# Splatted: the pre-fold seam took (msgs, my_generation), and a
|
||||
# literal two-argument call reads as an arity error against the
|
||||
# signature this tree actually has.
|
||||
pre_fold_args: tuple[Any, ...] = ([{"role": "user", "content": "hi"}], 0)
|
||||
msg = session._stream_response(*pre_fold_args)
|
||||
msg.pop("_wire_msgs", None)
|
||||
record["result"] = {
|
||||
"content": msg.get("content", ""),
|
||||
"tool_calls": msg.get("tool_calls"),
|
||||
"provider_content": msg.get("_provider_content"),
|
||||
}
|
||||
else:
|
||||
session.messages.append(Turn.user("hi"))
|
||||
result = session._stream_response(0)
|
||||
record["result"] = {
|
||||
"content": result.content,
|
||||
"tool_calls": result.tool_calls or None,
|
||||
"provider_content": (
|
||||
[dict(b) for b in result.turn.native.blocks] if result.turn.native else None
|
||||
),
|
||||
}
|
||||
record["raised"] = None
|
||||
except BaseException as exc: # noqa: BLE001 — the record IS the observation
|
||||
record["result"] = None
|
||||
record["raised"] = type(exc).__name__
|
||||
record["ui_events"] = [[k, d] for k, d in ui.events]
|
||||
record["last_usage"] = session._last_usage
|
||||
record["cancelled_partial"] = session._cancelled_partial_msg
|
||||
return _mask_synth_ids(record)
|
||||
|
||||
|
||||
def fixture_path(name: str) -> Path:
|
||||
return FIXTURE_DIR / f"{name}.json"
|
||||
|
||||
|
||||
def load_fixture(name: str) -> dict[str, Any]:
|
||||
return json.loads(fixture_path(name).read_text())
|
||||
|
||||
|
||||
def write_fixture(name: str, record: dict[str, Any]) -> None:
|
||||
# A TypeError before ANY UI event is the harness's own call-shape
|
||||
# failure (run_scenario's signature adapter no longer matches this
|
||||
# tree's seam), not old-world behavior — refuse to destroy the
|
||||
# baseline with it.
|
||||
if record.get("raised") == "TypeError" and not record.get("ui_events"):
|
||||
raise AssertionError(
|
||||
f"parity capture for {name!r} died calling the seam (TypeError before "
|
||||
f"any UI event) — fix run_scenario's signature adapter; do not record"
|
||||
)
|
||||
FIXTURE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
fixture_path(name).write_text(json.dumps(record, indent=2, sort_keys=True) + "\n")
|
||||
@@ -1,43 +0,0 @@
|
||||
"""Shared process/polling helpers for the bash + background-shell suites.
|
||||
|
||||
One copy instead of three: ``test_bash_tool_background_hang``,
|
||||
``test_background_shells`` and ``test_bash_background_tool`` all assert on
|
||||
process liveness and poll for asynchronous state. Leading underscore so
|
||||
pytest doesn't collect it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import os
|
||||
import signal
|
||||
import time
|
||||
|
||||
|
||||
def pid_alive(pid: int) -> bool:
|
||||
try:
|
||||
os.kill(pid, 0)
|
||||
except ProcessLookupError:
|
||||
return False
|
||||
except PermissionError:
|
||||
return True
|
||||
return True
|
||||
|
||||
|
||||
def kill_pid(pid: int) -> None:
|
||||
with contextlib.suppress(OSError):
|
||||
os.kill(pid, signal.SIGKILL)
|
||||
|
||||
|
||||
def poll_until(predicate, timeout=10.0, interval=0.05):
|
||||
"""Poll ``predicate`` until truthy or ``timeout``; RETURNS the last value
|
||||
(falsy on timeout — assert at the call site). Deliberately named apart
|
||||
from ``tests/_helpers.wait_until``, which RAISES on timeout: two
|
||||
same-named helpers with opposite failure semantics invite silently-green
|
||||
tests."""
|
||||
deadline = time.monotonic() + timeout
|
||||
value = predicate()
|
||||
while not value and time.monotonic() < deadline:
|
||||
time.sleep(interval)
|
||||
value = predicate()
|
||||
return value
|
||||
@@ -1,171 +0,0 @@
|
||||
"""Inline-reasoning dialect conformance catalog.
|
||||
|
||||
Passthrough servers (parserless vLLM/llama.cpp, LM Studio, bare gateways)
|
||||
emit model reasoning inline as ``<think>``/``<reasoning>`` blocks inside the
|
||||
content stream — a *dialect* of model output. This module is that dialect's
|
||||
executable specification for ``split_inline_reasoning``: each case maps an
|
||||
utterance to the exact ``(content, reasoning)`` lanes the one-shot must
|
||||
produce.
|
||||
|
||||
Consumers: the one-shot conformance and one-shot≡streaming property suites
|
||||
in tests/test_think_tag_split.py. Lane suites (session, judge,
|
||||
output-guard, optimizer, drain-stream) pin their lanes with suite-local
|
||||
utterances through their own fakes — adding a case HERE extends the
|
||||
semantics spec, not automatically any lane suite.
|
||||
|
||||
The split is RAW (residue whitespace stays; ``drain_stream`` owns the one
|
||||
trim over its joined runs). ``passthrough`` marks cases the split must
|
||||
return BYTE-IDENTICAL: tag-free text, and text whose only tags are orphan
|
||||
CLOSE tags. The latter is a
|
||||
review ruling, not an accident: a close tag whose open never arrived is
|
||||
indistinguishable from prose QUOTING the tag, and drained lanes routinely
|
||||
quote third-party text (web-fetch answers citing pages about reasoning
|
||||
models, guard verdicts echoing judged content) — any reclassification
|
||||
would let quoted text destroy real results. Display lanes wanting
|
||||
stricter cosmetic peeling (the title) own that locally as formatting.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DialectCase:
|
||||
id: str
|
||||
utterance: str
|
||||
content: str
|
||||
reasoning: str
|
||||
# The split returns the utterance byte-identical (no tag consumed):
|
||||
# tag-free text, or orphan-close-only text (quoted-tag safety).
|
||||
passthrough: bool = False
|
||||
|
||||
|
||||
CASES: tuple[DialectCase, ...] = (
|
||||
DialectCase(
|
||||
id="no_tag_byte_identity",
|
||||
utterance="Just an answer.",
|
||||
content="Just an answer.",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
# The fast path must not strip: unconsumed content is
|
||||
# byte-identical, whitespace included.
|
||||
id="no_tag_preserves_whitespace",
|
||||
utterance=" spaced \n",
|
||||
content=" spaced \n",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
id="leading_block",
|
||||
utterance="<think>plan</think>Answer",
|
||||
content="Answer",
|
||||
reasoning="plan",
|
||||
),
|
||||
DialectCase(
|
||||
# The split is RAW — tag residue stays; drain_stream owns the ONE
|
||||
# blank-edge-line trim over its joined runs (pinned there), which
|
||||
# preserves the code block's first-line indentation.
|
||||
id="indented_code_block_raw_residue",
|
||||
utterance="<think>plan</think>\n\n print(1)\n more()",
|
||||
content="\n\n print(1)\n more()",
|
||||
reasoning="plan",
|
||||
),
|
||||
DialectCase(
|
||||
id="leading_block_raw_residue",
|
||||
utterance="<think>plan</think>\n\nAnswer\n",
|
||||
content="\n\nAnswer\n",
|
||||
reasoning="plan",
|
||||
),
|
||||
DialectCase(
|
||||
id="interleaved_blocks_both_vocabularies",
|
||||
utterance="Intro <think>a</think>mid <reasoning>b</reasoning>end",
|
||||
content="Intro mid end",
|
||||
reasoning="ab",
|
||||
),
|
||||
DialectCase(
|
||||
id="unterminated_open_tail_is_reasoning",
|
||||
utterance="Answer part<think>never closed",
|
||||
content="Answer part",
|
||||
reasoning="never closed",
|
||||
),
|
||||
DialectCase(
|
||||
# QUOTED-CLOSE SAFETY (review ruling): an orphan close is
|
||||
# indistinguishable from a quoted tag — everything passes through.
|
||||
# A malicious page embedding the literal string must not be able
|
||||
# to wipe the extraction that quotes it.
|
||||
id="orphan_close_passes_through",
|
||||
utterance="The page says templates emit </think> after the preamble. Answer: 42.",
|
||||
content="The page says templates emit </think> after the preamble. Answer: 42.",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
# Template-pre-injected shape ("reasoning</think>answer"): the seam
|
||||
# deliberately passes it through — segregating it would require
|
||||
# treating every quoted close as a boundary. Post-#831 every lane
|
||||
# streams, and known streaming surfaces strip the orphan close
|
||||
# server-side; display lanes peel cosmetically on their own.
|
||||
id="preinject_shape_passes_through",
|
||||
utterance="plan text</think>\n\nAnswer",
|
||||
content="plan text</think>\n\nAnswer",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
id="immediate_close_passes_through",
|
||||
utterance="</think>Answer",
|
||||
content="</think>Answer",
|
||||
reasoning="",
|
||||
passthrough=True,
|
||||
),
|
||||
DialectCase(
|
||||
# Any close tag closes any open block (splitter semantics; the old
|
||||
# pairwise per-caller strip treated this as unterminated).
|
||||
id="cross_vocabulary_close",
|
||||
utterance="<think>x</reasoning>Answer",
|
||||
content="Answer",
|
||||
reasoning="x",
|
||||
),
|
||||
DialectCase(
|
||||
id="think_only",
|
||||
utterance="<think>all reasoning</think>",
|
||||
content="",
|
||||
reasoning="all reasoning",
|
||||
),
|
||||
DialectCase(
|
||||
id="think_only_unterminated",
|
||||
utterance="<think>everything",
|
||||
content="",
|
||||
reasoning="everything",
|
||||
),
|
||||
DialectCase(
|
||||
# A balanced block followed by a stray close: the block is
|
||||
# consumed, the stray close stays in content (quoted-tag safety),
|
||||
# and the consumed-tag strip applies.
|
||||
id="balanced_block_then_stray_close",
|
||||
utterance="<think>a</think>b</think>c",
|
||||
content="b</think>c",
|
||||
reasoning="a",
|
||||
),
|
||||
DialectCase(
|
||||
id="multiple_blocks_accumulate",
|
||||
utterance="<think>one</think>mid<think>two</think>tail",
|
||||
content="midtail",
|
||||
reasoning="onetwo",
|
||||
),
|
||||
DialectCase(
|
||||
# ACCEPTED RESIDUAL (R2): the split is content-blind, so a literal
|
||||
# OPEN tag in legitimate prose misroutes the remainder — the same
|
||||
# false positive the interactive splitter has carried in the
|
||||
# field. This pin makes any future fix a conscious change.
|
||||
# Scope note: R2 applies only where the scan runs — a backend
|
||||
# declaring ``server_parses_reasoning`` turns the scan off and
|
||||
# this utterance passes through byte-identical (pinned in
|
||||
# test_scan_tags_off_returns_every_utterance_byte_identical).
|
||||
id="literal_open_tag_false_positive_r2",
|
||||
utterance="The `<think>` tag opens a block.",
|
||||
content="The `",
|
||||
reasoning="` tag opens a block.",
|
||||
),
|
||||
)
|
||||
+7
-563
@@ -1,13 +1,12 @@
|
||||
"""Shared session-test helpers.
|
||||
|
||||
The minimal ``ChatSession`` factory, the ``SessionUIBase`` no-op/recording
|
||||
subclasses, and the tree's standard streaming provider fakes
|
||||
(``make_result`` / ``arm_session`` / ``scripted_provider`` /
|
||||
``ArmedHandle``, at the bottom): every suite driving the streaming seam
|
||||
imports them from here, so the eager-arming contract lives in one place.
|
||||
The one deliberate exception, ``test_model_registry.py``'s
|
||||
``_make_session``, takes a different signature (registry / model_alias /
|
||||
reasoning_effort + ``_FakeUI``) and is NOT a candidate for sharing.
|
||||
Two reasoning-test modules (``test_session_replay_reasoning.py`` and
|
||||
``test_session_synth_reasoning_block.py``) need the same minimal
|
||||
``ChatSession`` factory + a ``SessionUIBase`` no-op subclass. Hoisting
|
||||
keeps a future third caller from drifting on the defaults — the third
|
||||
existing ``_make_session`` (``test_model_registry.py``) deliberately
|
||||
takes a different signature (registry / model_alias / reasoning_effort
|
||||
+ ``_FakeUI``) and is NOT a candidate for sharing this helper.
|
||||
|
||||
Module is named with a leading underscore so pytest doesn't try to
|
||||
collect it as a test file — it's an importable utility, not a test.
|
||||
@@ -15,16 +14,11 @@ collect it as a test file — it's an importable utility, not a test.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from turnstone.core.model_turn import ModelTurnResult
|
||||
from turnstone.core.providers import ModelCapabilities, StreamChunk, ToolCallDelta, UsageInfo
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.trajectory import ProviderNative, ToolCall, Turn
|
||||
|
||||
|
||||
class NullUI(SessionUIBase):
|
||||
@@ -49,553 +43,3 @@ def make_session(**kwargs: Any) -> ChatSession:
|
||||
}
|
||||
defaults.update(kwargs)
|
||||
return ChatSession(**defaults)
|
||||
|
||||
|
||||
def mock_completion_result(
|
||||
content: str = "",
|
||||
tool_calls: list[dict[str, Any]] | None = None,
|
||||
) -> MagicMock:
|
||||
"""A provider result shaped like ``CompletionResult``.
|
||||
|
||||
Callers that route through ``model_turn`` (judges, task agents, and
|
||||
every lane #827 migrates) hit its re-ingest, which iterates
|
||||
``tool_calls``/``provider_blocks`` and joins ``reasoning`` — a bare
|
||||
MagicMock attribute would TypeError deep inside the seam, so every
|
||||
field the re-ingest reads is pinned to a real value here. ONE shared
|
||||
definition: when the re-ingest starts reading a new CompletionResult
|
||||
field, add it here and every suite moves together.
|
||||
"""
|
||||
result = MagicMock()
|
||||
result.content = content
|
||||
result.tool_calls = tool_calls
|
||||
result.finish_reason = "stop"
|
||||
result.usage = None
|
||||
result.provider_blocks = []
|
||||
result.reasoning = ""
|
||||
return result
|
||||
|
||||
|
||||
def fake_chat_stream(
|
||||
*,
|
||||
content: str | None = None,
|
||||
tool_calls: list[dict[str, str]] | None = None,
|
||||
finish_reason: str = "stop",
|
||||
prompt_tokens: int = 10,
|
||||
completion_tokens: int = 5,
|
||||
reasoning_content: str | None = None,
|
||||
reasoning: str | None = None,
|
||||
) -> list[Any]:
|
||||
"""Fake OpenAI Chat Completions SSE chunks for driving the REAL
|
||||
``OpenAIChatCompletionsProvider`` through a fake SDK client::
|
||||
|
||||
client.chat.completions.create = lambda **kw: fake_chat_stream(...)
|
||||
|
||||
Exercises the adapter's ``_iter_stream`` plus ``drain_stream`` end to
|
||||
end (the highest-fidelity fake lane), unlike ``as_stream`` which fakes
|
||||
at the provider boundary. ``tool_calls`` entries are
|
||||
``{"id", "name", "arguments"}`` dicts. ``SimpleNamespace`` (not
|
||||
``MagicMock``) so absent SDK fields read as real ``None`` — an
|
||||
auto-created mock attribute would leak into ``len()``/string paths.
|
||||
|
||||
Emits the realistic three-phase shape: data chunk(s), a finish-reason
|
||||
chunk, then the ``stream_options.include_usage`` usage-only chunk with
|
||||
empty ``choices``.
|
||||
"""
|
||||
|
||||
def _delta(
|
||||
content_val: str | None = None,
|
||||
tcs: list[Any] | None = None,
|
||||
rc: str | None = None,
|
||||
rsn: str | None = None,
|
||||
) -> SimpleNamespace:
|
||||
return SimpleNamespace(
|
||||
content=content_val,
|
||||
tool_calls=tcs,
|
||||
reasoning=rsn,
|
||||
reasoning_content=rc,
|
||||
annotations=None,
|
||||
)
|
||||
|
||||
chunks: list[Any] = []
|
||||
if reasoning_content is not None or reasoning is not None:
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[
|
||||
SimpleNamespace(
|
||||
finish_reason=None, delta=_delta(rc=reasoning_content, rsn=reasoning)
|
||||
)
|
||||
],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
if content is not None:
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=None, delta=_delta(content))],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
if tool_calls:
|
||||
tcs = [
|
||||
SimpleNamespace(
|
||||
index=i,
|
||||
id=tc.get("id", ""),
|
||||
function=SimpleNamespace(
|
||||
name=tc.get("name", ""), arguments=tc.get("arguments", "")
|
||||
),
|
||||
)
|
||||
for i, tc in enumerate(tool_calls)
|
||||
]
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=None, delta=_delta(None, tcs))],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[SimpleNamespace(finish_reason=finish_reason, delta=_delta())],
|
||||
usage=None,
|
||||
)
|
||||
)
|
||||
chunks.append(
|
||||
SimpleNamespace(
|
||||
choices=[],
|
||||
usage=SimpleNamespace(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=None,
|
||||
input_tokens_details=None,
|
||||
),
|
||||
)
|
||||
)
|
||||
return chunks
|
||||
|
||||
|
||||
class _ScriptedClient:
|
||||
"""Callable client-method fake following a script of stream builders.
|
||||
|
||||
Call N returns the stream described by ``scripts[N]``; the last script
|
||||
repeats for any further calls. Each script is a dict of kwargs for
|
||||
the bound stream builder, or a pre-built return value. Records every
|
||||
call's kwargs on ``.calls`` — read ``len(fn.calls)`` where a test
|
||||
previously kept its own counter cell, and ``fn.calls[i]["messages"]``
|
||||
where it captured request bodies.
|
||||
"""
|
||||
|
||||
def __init__(self, scripts: tuple[Any, ...], to_stream: Any) -> None:
|
||||
self._scripts = scripts
|
||||
self._to_stream = to_stream
|
||||
self.calls: list[dict[str, Any]] = []
|
||||
|
||||
def __call__(self, **kwargs: Any) -> Any:
|
||||
self.calls.append(kwargs)
|
||||
script = self._scripts[min(len(self.calls) - 1, len(self._scripts) - 1)]
|
||||
return self._to_stream(**script) if isinstance(script, dict) else script
|
||||
|
||||
|
||||
def scripted_chat_client(*scripts: Any) -> _ScriptedClient:
|
||||
"""A scripted ``client.chat.completions.create`` — dict scripts are
|
||||
:func:`fake_chat_stream` kwargs."""
|
||||
return _ScriptedClient(scripts, fake_chat_stream)
|
||||
|
||||
|
||||
def scripted_anthropic_client(*scripts: Any) -> _ScriptedClient:
|
||||
"""A scripted ``client.messages.stream`` — dict scripts are
|
||||
:func:`fake_anthropic_stream` kwargs (``blocks`` plus optional
|
||||
``stop_reason``/``usage``)."""
|
||||
return _ScriptedClient(scripts, fake_anthropic_stream)
|
||||
|
||||
|
||||
class FakeAnthropicBlock:
|
||||
"""A full-content Anthropic content-block fake for
|
||||
:func:`fake_anthropic_stream` — plain attributes plus the
|
||||
``model_dump()`` the provider's block capture reads."""
|
||||
|
||||
def __init__(self, **fields: Any) -> None:
|
||||
self._fields = fields
|
||||
for key, value in fields.items():
|
||||
setattr(self, key, value)
|
||||
|
||||
def model_dump(self, **_kw: Any) -> dict[str, Any]:
|
||||
return dict(self._fields)
|
||||
|
||||
|
||||
def fake_anthropic_stream(
|
||||
blocks: list[Any],
|
||||
*,
|
||||
stop_reason: str | None = "end_turn",
|
||||
usage: Any = None,
|
||||
) -> Any:
|
||||
"""Fake Anthropic SDK stream context manager for tests that drive the
|
||||
REAL ``AnthropicProvider`` through a fake client::
|
||||
|
||||
client.messages.stream = lambda **kw: fake_anthropic_stream(...)
|
||||
|
||||
Accepts the same full-content block fakes the pre-#831
|
||||
``get_final_message`` fixtures used (objects with ``.type`` + fields
|
||||
and ``model_dump()``) and synthesizes the real event grammar the
|
||||
streaming iterator consumes: ``content_block_start`` carries the block
|
||||
with its text/thinking/signature EMPTIED and ``input`` as ``{}`` (the
|
||||
SDK start shape), deltas carry the content, ``content_block_stop``
|
||||
finalizes tool input, and the closing ``message_delta`` carries
|
||||
``stop_reason`` (+ optional usage object). Without the stripping, the
|
||||
provider's raw-block accumulator would double every text/thinking
|
||||
field (start capture + delta append).
|
||||
|
||||
``stop_reason=None`` omits the closing ``message_delta`` entirely —
|
||||
the terminal-signal-less lax-gateway shape ``finish_reason_optional``
|
||||
exists for (content arrives, then the stream just ends).
|
||||
"""
|
||||
events: list[Any] = []
|
||||
for idx, block in enumerate(blocks):
|
||||
d = dict(block.model_dump()) if hasattr(block, "model_dump") else dict(vars(block))
|
||||
btype = d.get("type", "")
|
||||
start = dict(d)
|
||||
if btype == "text":
|
||||
start["text"] = ""
|
||||
elif btype == "thinking":
|
||||
start["thinking"] = ""
|
||||
start["signature"] = ""
|
||||
elif btype == "tool_use":
|
||||
start["input"] = {}
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_start", index=idx, content_block=SimpleNamespace(**start)
|
||||
)
|
||||
)
|
||||
if btype == "text" and d.get("text"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="text_delta", text=d["text"]),
|
||||
)
|
||||
)
|
||||
elif btype == "thinking":
|
||||
if d.get("thinking"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="thinking_delta", thinking=d["thinking"]),
|
||||
)
|
||||
)
|
||||
if d.get("signature"):
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(type="signature_delta", signature=d["signature"]),
|
||||
)
|
||||
)
|
||||
elif btype == "tool_use":
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="content_block_delta",
|
||||
index=idx,
|
||||
delta=SimpleNamespace(
|
||||
type="input_json_delta",
|
||||
partial_json=json.dumps(d.get("input", {})),
|
||||
),
|
||||
)
|
||||
)
|
||||
events.append(SimpleNamespace(type="content_block_stop", index=idx))
|
||||
if stop_reason is not None or usage is not None:
|
||||
events.append(
|
||||
SimpleNamespace(
|
||||
type="message_delta", usage=usage, delta=SimpleNamespace(stop_reason=stop_reason)
|
||||
)
|
||||
)
|
||||
|
||||
mgr = MagicMock()
|
||||
mgr.__enter__ = MagicMock(return_value=events)
|
||||
mgr.__exit__ = MagicMock(return_value=False)
|
||||
return mgr
|
||||
|
||||
|
||||
def as_stream(result: Any) -> list[StreamChunk]:
|
||||
"""Adapt a ``CompletionResult``-shaped fake to a ``create_streaming``
|
||||
return value (single terminal chunk).
|
||||
|
||||
The #831 transport collapse routes every single-shot lane through
|
||||
``drain_stream(provider.create_streaming(...))``, so provider fakes
|
||||
return chunk iterables now. Tests keep building result-shaped fakes
|
||||
(``mock_completion_result`` or hand-rolled) and wrap them at
|
||||
assignment: ``provider.create_streaming.return_value =
|
||||
as_stream(result)``. A list re-iterates on every call, so one
|
||||
``return_value`` serves repeated-call tests; convert AFTER mutating
|
||||
the fake's fields — the chunk snapshots them.
|
||||
|
||||
Multi-chunk accumulation semantics are exercised by the dedicated
|
||||
``drain_stream`` unit tests, not through this helper.
|
||||
"""
|
||||
deltas = [
|
||||
ToolCallDelta(
|
||||
index=i,
|
||||
id=tc.get("id", ""),
|
||||
name=tc.get("function", {}).get("name", ""),
|
||||
arguments_delta=tc.get("function", {}).get("arguments", ""),
|
||||
)
|
||||
for i, tc in enumerate(result.tool_calls or [])
|
||||
]
|
||||
return [
|
||||
StreamChunk(
|
||||
content_delta=result.content or "",
|
||||
reasoning_delta=getattr(result, "reasoning", "") or "",
|
||||
tool_call_deltas=deltas,
|
||||
usage=result.usage,
|
||||
finish_reason=result.finish_reason or "stop",
|
||||
provider_blocks=list(result.provider_blocks or []),
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
def think_tag_stream(utterance: str) -> list[StreamChunk]:
|
||||
"""``create_streaming`` return value simulating a passthrough server
|
||||
that emits *utterance* — typically think-tag-bearing — as plain
|
||||
streamed content.
|
||||
|
||||
The per-lane fixture for inline-reasoning dialect pins: lane tests
|
||||
supply their own utterances (the dialect's SEMANTICS are specified
|
||||
once, in ``tests._reasoning_dialect.CASES``, and pinned by the
|
||||
one-shot suites — lane pins assert lane behavior, not tag grammar).
|
||||
Routes through the real ``drain_stream`` seam exactly like
|
||||
``as_stream``.
|
||||
"""
|
||||
return as_stream(mock_completion_result(content=utterance))
|
||||
|
||||
|
||||
def seam_provider(utterance: str, *, provider_name: str = "openai-compatible") -> MagicMock:
|
||||
"""Provider fake whose ``create_streaming`` replays *utterance* through
|
||||
the REAL drain seam (``think_tag_stream``) — THE lane-suite seam fake.
|
||||
|
||||
One definition so the lane suites cannot drift when the provider
|
||||
surface ``model_turn`` probes grows: real ``ModelCapabilities`` for
|
||||
the clamp math, ``provider_name`` overridable per suite.
|
||||
|
||||
Assign the RETURNED fake to ``session._provider`` — never mutate the
|
||||
provider a session resolved on its own: with a MagicMock client the
|
||||
session resolves the process-wide ``create_provider(...)`` singleton,
|
||||
and writing that shared instance's ``create_streaming`` poisons every
|
||||
later session in the test run (the SSE-recovery e2e servers resolve
|
||||
the same instance).
|
||||
"""
|
||||
provider = provider_shell(provider_name)
|
||||
provider.create_streaming = MagicMock(return_value=think_tag_stream(utterance))
|
||||
return provider
|
||||
|
||||
|
||||
class RecordingUI:
|
||||
"""UI adapter recording the ordered event stream ``send()`` emits."""
|
||||
|
||||
def __init__(self):
|
||||
self.events = []
|
||||
|
||||
def _rec(self, kind, detail=""):
|
||||
self.events.append((kind, detail))
|
||||
|
||||
def on_turn_start(self):
|
||||
self._rec("turn_start")
|
||||
|
||||
def on_turn_committed(self):
|
||||
self._rec("turn_committed")
|
||||
|
||||
def on_stream_discarded(self):
|
||||
self._rec("stream_discarded")
|
||||
|
||||
def on_thinking_start(self):
|
||||
self._rec("thinking_start")
|
||||
|
||||
def on_thinking_stop(self):
|
||||
self._rec("thinking_stop")
|
||||
|
||||
def on_reasoning_token(self, text):
|
||||
self._rec("reasoning", text)
|
||||
|
||||
def on_content_token(self, text):
|
||||
self._rec("content", text)
|
||||
|
||||
def on_stream_end(self):
|
||||
self._rec("stream_end")
|
||||
|
||||
def approve_tools(self, items):
|
||||
return True, None
|
||||
|
||||
def on_tool_result(self, call_id, name, output, **kwargs):
|
||||
pass
|
||||
|
||||
def on_tool_output_chunk(self, call_id, chunk):
|
||||
pass
|
||||
|
||||
def on_status(self, usage, context_window, effort):
|
||||
pass
|
||||
|
||||
def on_info(self, message):
|
||||
self._rec("info", message)
|
||||
|
||||
def on_error(self, message):
|
||||
self._rec("error", message)
|
||||
|
||||
def on_state_change(self, state):
|
||||
self._rec("state", state)
|
||||
|
||||
def on_rename(self, name):
|
||||
pass
|
||||
|
||||
def on_output_warning(self, call_id, assessment):
|
||||
pass
|
||||
|
||||
def record_output_assessment(
|
||||
self,
|
||||
call_id,
|
||||
assessment,
|
||||
*,
|
||||
tier="heuristic",
|
||||
reasoning="",
|
||||
judge_model="",
|
||||
latency_ms=0,
|
||||
confidence=0.0,
|
||||
):
|
||||
pass
|
||||
|
||||
def kinds(self):
|
||||
return [k for k, _ in self.events]
|
||||
|
||||
def of(self, kind):
|
||||
return [d for k, d in self.events if k == kind]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Streaming provider fakes — the #832 seam contract
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def make_result(
|
||||
content: str = "",
|
||||
*,
|
||||
tool_calls: list[dict[str, Any]] | None = None,
|
||||
finish_reason: str = "stop",
|
||||
usage: UsageInfo | None = None,
|
||||
native_blocks: list[dict[str, Any]] | None = None,
|
||||
producer: str = "openai-compatible",
|
||||
wire_msgs: list[dict[str, Any]] | None = None,
|
||||
) -> ModelTurnResult:
|
||||
"""A ``ModelTurnResult`` shaped like the streaming wrapper's return,
|
||||
for tests that only need "a turn happened" and patch
|
||||
``_stream_response`` wholesale. Turn and ``tool_calls`` mirror are
|
||||
built from the same dicts, preserving the #825 pairing invariant."""
|
||||
calls = list(tool_calls or [])
|
||||
tc_tuple = tuple(
|
||||
ToolCall(
|
||||
id=tc.get("id", ""),
|
||||
name=tc.get("function", {}).get("name", ""),
|
||||
arguments=tc.get("function", {}).get("arguments", ""),
|
||||
)
|
||||
for tc in calls
|
||||
)
|
||||
native = (
|
||||
ProviderNative(producer=producer, blocks=tuple(native_blocks)) if native_blocks else None
|
||||
)
|
||||
return ModelTurnResult(
|
||||
turn=Turn.assistant(content, tool_calls=tc_tuple, native=native),
|
||||
finish_reason=finish_reason,
|
||||
usage=usage,
|
||||
tool_calls=calls,
|
||||
wire_msgs=wire_msgs,
|
||||
producer=producer,
|
||||
)
|
||||
|
||||
|
||||
class ArmedHandle:
|
||||
"""Closeable sentinel standing in for the SDK stream handle."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.closed = False
|
||||
|
||||
def close(self) -> None:
|
||||
self.closed = True
|
||||
|
||||
|
||||
def provider_shell(
|
||||
name: str = "openai-compatible",
|
||||
retryable: frozenset[str] = frozenset({"IncompleteStreamError"}),
|
||||
) -> MagicMock:
|
||||
"""The armed-provider fake skeleton every streaming fake builds on:
|
||||
provider_name / capabilities / retryable set. ONE spelling, so a new
|
||||
attribute the seam starts probing lands in every fake at once."""
|
||||
provider = MagicMock()
|
||||
provider.provider_name = name
|
||||
provider.get_capabilities.return_value = ModelCapabilities()
|
||||
provider.retryable_error_names = retryable
|
||||
return provider
|
||||
|
||||
|
||||
def arm_session(
|
||||
session: Any,
|
||||
*streams: Any,
|
||||
retryable: frozenset[str] = frozenset({"IncompleteStreamError"}),
|
||||
name: str = "openai-compatible",
|
||||
) -> MagicMock:
|
||||
"""Install a sequential multi-turn armed provider fake on *session*.
|
||||
|
||||
Each ``create_streaming`` call serves the next element of *streams*:
|
||||
an iterable/generator is armed (a closeable sentinel appended to
|
||||
``cancel_ref`` — the eager append every real adapter performs, which
|
||||
the creation-vs-midstream classifier keys on) and returned to be
|
||||
consumed once; an EXCEPTION instance is raised at create time WITHOUT
|
||||
arming, a creation-phase failure the per-lane ladder owns. Calls
|
||||
beyond the script fail loudly: the strict finish gate rejects an
|
||||
exhausted iterator rather than absorbing it as a silent empty turn,
|
||||
so an under-scripted test must say so.
|
||||
|
||||
Title generation is latched off — with a provider-LEVEL fake the
|
||||
best-effort title lane would otherwise consume the first script
|
||||
before the main loop ran.
|
||||
"""
|
||||
session._title_generated = True
|
||||
provider = provider_shell(name, retryable)
|
||||
# One handle PER CREATE (the real adapters' rule): `handles` records
|
||||
# them all, `_armed_handle` is the latest.
|
||||
provider._armed_handle = None
|
||||
provider.handles = []
|
||||
remaining = list(streams)
|
||||
|
||||
def _create(**kwargs: Any):
|
||||
assert remaining, "arm_session: script exhausted — send looped for more turns than scripted"
|
||||
nxt = remaining.pop(0)
|
||||
if isinstance(nxt, BaseException):
|
||||
raise nxt
|
||||
ref = kwargs.get("cancel_ref")
|
||||
if ref is not None:
|
||||
handle = ArmedHandle()
|
||||
provider.handles.append(handle)
|
||||
provider._armed_handle = handle
|
||||
ref.append(handle)
|
||||
return iter(nxt) if not hasattr(nxt, "__next__") else nxt
|
||||
|
||||
provider.create_streaming = MagicMock(side_effect=_create)
|
||||
session._provider = provider
|
||||
return provider
|
||||
|
||||
|
||||
def scripted_provider(chunks: list[StreamChunk]) -> MagicMock:
|
||||
"""Provider fake replaying *chunks*, arming ``cancel_ref`` eagerly.
|
||||
|
||||
Assign to ``session._provider`` (never mutate a resolved provider —
|
||||
the create_provider singleton rule above). Each call returns a FRESH
|
||||
iterator over the same script so ladder tests re-drive it; the armed
|
||||
handle is appended per call, matching the one-handle-per-create
|
||||
behavior of every real adapter.
|
||||
"""
|
||||
provider = provider_shell()
|
||||
|
||||
def _create(**kwargs: Any):
|
||||
ref = kwargs.get("cancel_ref")
|
||||
if ref is not None:
|
||||
ref.append(ArmedHandle())
|
||||
return iter(chunks)
|
||||
|
||||
provider.create_streaming = MagicMock(side_effect=_create)
|
||||
return provider
|
||||
|
||||
@@ -1,604 +0,0 @@
|
||||
"""Browser-fidelity SSE recovery harness helpers.
|
||||
|
||||
The load-bearing assembly for ``tests/test_sse_recovery_e2e.py``: a
|
||||
``BrowserlikeSSEClient`` that speaks the exact wire contract the real
|
||||
``turnstone/shared_static/interactive.js`` pane speaks, and the
|
||||
assertion helpers the scenarios share. The server boot machinery lives
|
||||
in ``_sse_recovery_server.py``.
|
||||
|
||||
Why a raw-socket SSE reader (and not ``httpx.stream``): the slow-consumer
|
||||
overflow scenario needs the consumer to STALL — stop reading the socket
|
||||
so the server's SSE generator blocks on ``await send`` and stops draining
|
||||
the per-UI listener queue, which then poisons at its cap. A faithful
|
||||
stall needs (a) precise control over when bytes are read and (b) a small
|
||||
``SO_RCVBUF`` so the in-flight backlog before poison stays bounded to
|
||||
~100 KB instead of the client kernel's multi-MB autotuned default (which
|
||||
would need tens of thousands of events to overflow). A raw socket gives
|
||||
both; httpx (used here only for the plain ``/history`` request/response)
|
||||
gives neither. This is ALSO closer to the browser: EventSource has a
|
||||
bounded receive buffer, not an unbounded one.
|
||||
|
||||
Client contract mirrored from interactive.js (line references are to
|
||||
that file on the ``fix/sse-truncated-resync`` branch):
|
||||
|
||||
- ``_last_event_id`` advances ONLY from SSE ``id:`` fields, and only
|
||||
ring-buffer events carry one — synthetic replay frames (connected /
|
||||
status / state_change / in_progress_snapshot / replay_truncated /
|
||||
stream_overflow) do not, exactly like ``EventSource.lastEventId``
|
||||
(interactive.js onmessage ~1378).
|
||||
- reconnect presents ``connectCursor = _truncatedFromCursor ??
|
||||
_lastEventId`` as ``?last_event_id=`` (manual path) or a
|
||||
``Last-Event-ID`` header (native EventSource auto-reconnect path)
|
||||
(interactive.js connectSSE ~1328).
|
||||
- on a ``replay_truncated`` envelope the client records the
|
||||
truncation-time cursor keep-oldest (``_truncatedFromCursor =
|
||||
_lastEventId`` only when null) and runs ``_loadHistoryThenConnect``
|
||||
(disconnect → /history → adopt cursor → reconnect); a FAILED
|
||||
/history leaves the record armed so the reconnect re-presents the
|
||||
truncation-time cursor and re-draws the envelope (interactive.js
|
||||
handleEvent replay_truncated ~2303, _loadHistoryThenConnect ~1604,
|
||||
_refetchHistory seedCursor ~1706).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import json
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import httpx
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
|
||||
# Small client receive buffer so a stalled consumer's in-flight backlog
|
||||
# before the server-side poison stays bounded (~100 KB) instead of the
|
||||
# multi-MB autotuned default. Paired with the server's small SO_SNDBUF
|
||||
# (see _sse_recovery_server.build_recovery_server).
|
||||
_CLIENT_RCVBUF = 2048
|
||||
|
||||
|
||||
@dataclass
|
||||
class SSEFrame:
|
||||
"""One decoded SSE frame, tagged with the connection it arrived on.
|
||||
|
||||
``event_id`` is the ``id:`` field verbatim (a stringified integer,
|
||||
or ``None`` for id-less synthetic frames — the same string domain as
|
||||
``EventSource.lastEventId``). ``etype`` is the ``type`` field of the
|
||||
JSON ``data:`` payload (the application event type), distinct from
|
||||
any SSE ``event:`` field, which the server never uses.
|
||||
"""
|
||||
|
||||
conn_index: int
|
||||
event_id: str | None
|
||||
etype: str | None
|
||||
payload: dict[str, Any] | None
|
||||
raw: str
|
||||
|
||||
@property
|
||||
def event_id_int(self) -> int | None:
|
||||
if self.event_id is None:
|
||||
return None
|
||||
try:
|
||||
return int(self.event_id)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
class BrowserlikeSSEClient:
|
||||
"""A single interactive pane's SSE + /history state machine.
|
||||
|
||||
Not thread-safe against concurrent public calls; drive it from one
|
||||
test thread. Internally a per-connection reader thread decodes the
|
||||
stream; ``stall()`` / ``resume()`` gate that thread's socket reads so
|
||||
a test can build server-side backpressure without closing the
|
||||
connection (the slow-consumer → listener-queue-poison path).
|
||||
"""
|
||||
|
||||
def __init__(self, base_url: str, ws_id: str, token: str) -> None:
|
||||
parts = urlsplit(base_url)
|
||||
self._host = parts.hostname or "127.0.0.1"
|
||||
self._port = parts.port or 80
|
||||
self._ws_id = ws_id
|
||||
self._token = token
|
||||
self._auth = {"Authorization": f"Bearer {token}"}
|
||||
self._http = httpx.Client(
|
||||
base_url=f"http://{self._host}:{self._port}", timeout=httpx.Timeout(15.0)
|
||||
)
|
||||
|
||||
# EventSource-equivalent cursor state.
|
||||
self._last_event_id: str | None = None
|
||||
self._truncated_from_cursor: str | None = None
|
||||
|
||||
# Transcript. ``_all_frames`` is the cross-connection accumulation
|
||||
# (what "the client eventually saw"); ``_conn_frames`` keeps each
|
||||
# connection's slice for per-connection assertions (contiguity).
|
||||
self._all_frames: list[SSEFrame] = []
|
||||
self._conn_frames: list[list[SSEFrame]] = []
|
||||
self._frames_lock = threading.Lock()
|
||||
|
||||
# Reader plumbing.
|
||||
self._sock: socket.socket | None = None
|
||||
self._reader: threading.Thread | None = None
|
||||
self._stop = threading.Event()
|
||||
self._read_gate = threading.Event()
|
||||
self._read_gate.set() # reading permitted by default
|
||||
self._status: int | None = None
|
||||
self._headers_done = threading.Event()
|
||||
|
||||
# -- connection lifecycle ------------------------------------------------
|
||||
|
||||
def _events_path(self, cursor: str | None) -> str:
|
||||
path = f"/v1/api/workstreams/{self._ws_id}/events"
|
||||
if cursor is not None:
|
||||
path += f"?last_event_id={cursor}"
|
||||
return path
|
||||
|
||||
def connect(self, *, native: bool = False, rcvbuf: int | None = None) -> None:
|
||||
"""Open the SSE stream, presenting the client's current cursor.
|
||||
|
||||
``native=True`` models the browser's EventSource auto-reconnect:
|
||||
the cursor rides a ``Last-Event-ID`` HEADER and never appears in
|
||||
the URL. ``native=False`` models the manual ``new EventSource(url
|
||||
+ '?last_event_id=')`` path interactive.js uses when it must
|
||||
override the live cursor (the ``connectCursor`` chokepoint).
|
||||
|
||||
``rcvbuf`` shrinks this connection's ``SO_RCVBUF`` — pass
|
||||
``_CLIENT_RCVBUF`` on a connection the test will ``stall()`` so the
|
||||
in-flight backlog before the server-side poison stays bounded.
|
||||
Leave it ``None`` (OS default) on recovery reconnects so the ring
|
||||
replay is not throttled to a crawl.
|
||||
"""
|
||||
if self._reader is not None:
|
||||
raise RuntimeError("already connected; disconnect() first")
|
||||
connect_cursor = (
|
||||
self._truncated_from_cursor
|
||||
if self._truncated_from_cursor is not None
|
||||
else self._last_event_id
|
||||
)
|
||||
header_lines = [
|
||||
f"Host: {self._host}:{self._port}",
|
||||
f"Authorization: Bearer {self._token}",
|
||||
"Accept: text/event-stream",
|
||||
"Cache-Control: no-cache",
|
||||
]
|
||||
if native:
|
||||
path = self._events_path(None)
|
||||
if connect_cursor is not None:
|
||||
header_lines.append(f"Last-Event-ID: {connect_cursor}")
|
||||
else:
|
||||
path = self._events_path(connect_cursor)
|
||||
request = f"GET {path} HTTP/1.1\r\n" + "\r\n".join(header_lines) + "\r\n\r\n"
|
||||
|
||||
sock = socket.create_connection((self._host, self._port), timeout=10)
|
||||
if rcvbuf is not None:
|
||||
sock.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf)
|
||||
sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
|
||||
sock.settimeout(None)
|
||||
sock.sendall(request.encode())
|
||||
self._sock = sock
|
||||
|
||||
self._stop.clear()
|
||||
self._read_gate.set()
|
||||
self._status = None
|
||||
self._headers_done.clear()
|
||||
conn_index = len(self._conn_frames)
|
||||
frames: list[SSEFrame] = []
|
||||
self._conn_frames.append(frames)
|
||||
self._reader = threading.Thread(
|
||||
target=self._read_loop,
|
||||
args=(sock, conn_index, frames),
|
||||
name=f"sse-reader-{self._ws_id[:6]}-{conn_index}",
|
||||
daemon=True,
|
||||
)
|
||||
self._reader.start()
|
||||
|
||||
# Surface a non-200 handshake to the caller (409 half-built UI,
|
||||
# 404 unknown ws, 401 auth) rather than silently reading nothing.
|
||||
if not self._headers_done.wait(timeout=10):
|
||||
self.disconnect()
|
||||
raise AssertionError("events connect: no HTTP response headers")
|
||||
if self._status != 200:
|
||||
status = self._status
|
||||
self.disconnect()
|
||||
raise AssertionError(f"events connect returned HTTP {status}")
|
||||
|
||||
def disconnect(self) -> None:
|
||||
"""Close the stream and join the reader (leak-guard clean)."""
|
||||
self._stop.set()
|
||||
self._read_gate.set() # release a stalled reader so it sees _stop
|
||||
sock = self._sock
|
||||
if sock is not None:
|
||||
with contextlib.suppress(OSError):
|
||||
sock.shutdown(socket.SHUT_RDWR) # interrupt a blocked recv
|
||||
reader = self._reader
|
||||
if reader is not None:
|
||||
reader.join(timeout=15)
|
||||
if reader.is_alive():
|
||||
raise AssertionError("SSE reader thread failed to stop")
|
||||
if sock is not None:
|
||||
with contextlib.suppress(OSError):
|
||||
sock.close()
|
||||
self._sock = None
|
||||
self._reader = None
|
||||
|
||||
def close(self) -> None:
|
||||
"""Full teardown: disconnect any live stream + close the HTTP client."""
|
||||
if self._reader is not None:
|
||||
self.disconnect()
|
||||
self._http.close()
|
||||
|
||||
# -- the stall gate (backpressure driver) --------------------------------
|
||||
|
||||
def stall(self) -> None:
|
||||
"""Stop reading the socket. The kernel + uvicorn send buffers fill,
|
||||
blocking the server's SSE generator on its ``await send``, so it
|
||||
stops draining the per-UI listener queue — which poisons at its cap.
|
||||
"""
|
||||
self._read_gate.clear()
|
||||
|
||||
def resume(self) -> None:
|
||||
"""Resume reading. A poisoned-and-closed stream delivers its
|
||||
``stream_overflow`` farewell frame once the backlog drains."""
|
||||
self._read_gate.set()
|
||||
|
||||
# -- reader --------------------------------------------------------------
|
||||
|
||||
def _read_loop(self, sock: socket.socket, conn_index: int, frames: list[SSEFrame]) -> None:
|
||||
raw = b"" # undecoded bytes (headers, then chunked framing)
|
||||
sse = b"" # decoded SSE byte stream
|
||||
headers_parsed = False
|
||||
chunked = False
|
||||
while not self._stop.is_set():
|
||||
# Backpressure gate: while stalled we do NOT read the socket, so
|
||||
# its receive buffer fills and TCP flow control stalls the server.
|
||||
if not self._read_gate.wait(timeout=0.1):
|
||||
continue
|
||||
if self._stop.is_set():
|
||||
break
|
||||
try:
|
||||
chunk = sock.recv(65536)
|
||||
except OSError:
|
||||
break
|
||||
if not chunk:
|
||||
break # server closed
|
||||
raw += chunk
|
||||
if not headers_parsed:
|
||||
if b"\r\n\r\n" not in raw:
|
||||
continue
|
||||
header_blob, raw = raw.split(b"\r\n\r\n", 1)
|
||||
self._parse_headers(header_blob)
|
||||
chunked = b"transfer-encoding: chunked" in header_blob.lower()
|
||||
headers_parsed = True
|
||||
self._headers_done.set()
|
||||
if chunked:
|
||||
decoded, raw = _dechunk(raw)
|
||||
sse += decoded
|
||||
else:
|
||||
sse += raw
|
||||
raw = b""
|
||||
sse = sse.replace(b"\r\n", b"\n")
|
||||
while b"\n\n" in sse:
|
||||
block, sse = sse.split(b"\n\n", 1)
|
||||
self._handle_block(block.decode("utf-8", "replace"), conn_index, frames)
|
||||
|
||||
def _parse_headers(self, header_blob: bytes) -> None:
|
||||
first_line = header_blob.split(b"\r\n", 1)[0].decode("latin-1")
|
||||
# "HTTP/1.1 200 OK"
|
||||
parts = first_line.split(" ", 2)
|
||||
if len(parts) >= 2 and parts[1].isdigit():
|
||||
self._status = int(parts[1])
|
||||
|
||||
def _handle_block(self, block_text: str, conn_index: int, frames: list[SSEFrame]) -> None:
|
||||
event_id: str | None = None
|
||||
data_parts: list[str] = []
|
||||
retry: str | None = None
|
||||
for line in block_text.split("\n"):
|
||||
if not line or line.startswith(":"):
|
||||
continue # blank or comment (ping)
|
||||
field_name, _, value = line.partition(":")
|
||||
if value.startswith(" "):
|
||||
value = value[1:] # SSE strips a single leading space
|
||||
if field_name == "id":
|
||||
event_id = value
|
||||
elif field_name == "data":
|
||||
data_parts.append(value)
|
||||
elif field_name == "retry":
|
||||
retry = value
|
||||
# EventSource semantics: an event carrying an ``id:`` sets the
|
||||
# last-event-id buffer; an event without one leaves it unchanged.
|
||||
if event_id is not None:
|
||||
self._last_event_id = event_id
|
||||
if not data_parts:
|
||||
if retry is not None:
|
||||
self._record(SSEFrame(conn_index, None, "retry", None, block_text), frames)
|
||||
return
|
||||
data_str = "\n".join(data_parts)
|
||||
payload: dict[str, Any] | None
|
||||
try:
|
||||
parsed = json.loads(data_str)
|
||||
payload = parsed if isinstance(parsed, dict) else None
|
||||
except ValueError:
|
||||
payload = None
|
||||
etype = payload.get("type") if payload is not None else None
|
||||
frame = SSEFrame(conn_index, event_id, etype, payload, data_str)
|
||||
self._record(frame, frames)
|
||||
# Mirror the pane: the FIRST replay_truncated for an unrepaired gap
|
||||
# records the truncation-time cursor (keep-oldest). Its consumer is
|
||||
# the reconnect chokepoint (see ``connect``).
|
||||
if etype == "replay_truncated" and self._truncated_from_cursor is None:
|
||||
self._truncated_from_cursor = self._last_event_id
|
||||
|
||||
def _record(self, frame: SSEFrame, frames: list[SSEFrame]) -> None:
|
||||
with self._frames_lock:
|
||||
frames.append(frame)
|
||||
self._all_frames.append(frame)
|
||||
|
||||
# -- /history + cursor flow ----------------------------------------------
|
||||
|
||||
def fetch_history(self) -> dict[str, Any]:
|
||||
"""GET /history and return the parsed JSON ({ws_id, messages, cursor})."""
|
||||
r = self._http.get(f"/v1/api/workstreams/{self._ws_id}/history", headers=self._auth)
|
||||
r.raise_for_status()
|
||||
result: dict[str, Any] = r.json()
|
||||
return result
|
||||
|
||||
def seed_from_history(self) -> dict[str, Any]:
|
||||
"""The seedCursor step: fetch /history, adopt a non-null resume
|
||||
cursor into ``_last_event_id``, and clear the truncation record on
|
||||
a successful render (replayHistory clears ``_truncatedFromCursor``).
|
||||
"""
|
||||
data = self.fetch_history()
|
||||
cursor = data.get("cursor")
|
||||
if cursor is not None:
|
||||
self._last_event_id = str(cursor)
|
||||
self._truncated_from_cursor = None # successful full render repairs the gap
|
||||
return data
|
||||
|
||||
def load_history_then_connect(
|
||||
self, *, fail_history: bool = False, native: bool = False
|
||||
) -> dict[str, Any] | None:
|
||||
"""Reproduce interactive.js ``_loadHistoryThenConnect``.
|
||||
|
||||
Disconnect first, drop the live cursor (``_last_event_id = None``)
|
||||
but KEEP ``_truncated_from_cursor`` armed, then fetch /history and
|
||||
reconnect. On success adopt the returned cursor and clear the
|
||||
truncation record; on a FAILED /history (``fail_history`` — the
|
||||
harness IS the client here, so a client-side simulated failure is
|
||||
faithful) leave the record armed so the reconnect re-presents the
|
||||
truncation-time cursor and re-draws ``replay_truncated``.
|
||||
|
||||
Returns the /history JSON, or ``None`` when the fetch failed.
|
||||
"""
|
||||
if self._reader is not None:
|
||||
self.disconnect()
|
||||
self._last_event_id = None
|
||||
data: dict[str, Any] | None
|
||||
if fail_history:
|
||||
data = None
|
||||
else:
|
||||
data = self.fetch_history()
|
||||
cursor = data.get("cursor")
|
||||
if cursor is not None:
|
||||
self._last_event_id = str(cursor)
|
||||
self._truncated_from_cursor = None
|
||||
self.connect(native=native)
|
||||
return data
|
||||
|
||||
# -- accessors + waits ---------------------------------------------------
|
||||
|
||||
@property
|
||||
def last_event_id(self) -> str | None:
|
||||
return self._last_event_id
|
||||
|
||||
@property
|
||||
def truncated_from_cursor(self) -> str | None:
|
||||
return self._truncated_from_cursor
|
||||
|
||||
def all_frames(self) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._all_frames)
|
||||
|
||||
def conn_frames(self, conn_index: int) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._conn_frames[conn_index])
|
||||
|
||||
def latest_conn_frames(self) -> list[SSEFrame]:
|
||||
with self._frames_lock:
|
||||
return list(self._conn_frames[-1]) if self._conn_frames else []
|
||||
|
||||
def num_connections(self) -> int:
|
||||
with self._frames_lock:
|
||||
return len(self._conn_frames)
|
||||
|
||||
def frames_of_type(self, etype: str) -> list[SSEFrame]:
|
||||
return [f for f in self.all_frames() if f.etype == etype]
|
||||
|
||||
def has_type(self, etype: str) -> bool:
|
||||
return any(f.etype == etype for f in self.all_frames())
|
||||
|
||||
def tool_output_by_call(self) -> dict[str, str]:
|
||||
"""Concatenate every ``tool_output_chunk`` payload per call_id, in
|
||||
arrival order — the reconstructed live stream for each call."""
|
||||
out: dict[str, str] = {}
|
||||
for f in self.all_frames():
|
||||
if f.etype == "tool_output_chunk" and f.payload is not None:
|
||||
cid = str(f.payload.get("call_id", ""))
|
||||
out[cid] = out.get(cid, "") + str(f.payload.get("chunk", ""))
|
||||
return out
|
||||
|
||||
def tool_results_by_call(self) -> dict[str, str]:
|
||||
"""The last ``tool_result`` output seen per call_id."""
|
||||
out: dict[str, str] = {}
|
||||
for f in self.all_frames():
|
||||
if f.etype == "tool_result" and f.payload is not None:
|
||||
out[str(f.payload.get("call_id", ""))] = str(f.payload.get("output", ""))
|
||||
return out
|
||||
|
||||
def wait_for_type(self, etype: str, *, timeout: float = 45.0) -> SSEFrame:
|
||||
"""Block until a frame of ``etype`` has arrived on ANY connection."""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
for f in self.all_frames():
|
||||
if f.etype == etype:
|
||||
return f
|
||||
time.sleep(0.05)
|
||||
raise AssertionError(f"timed out waiting for a {etype!r} frame")
|
||||
|
||||
def wait_for(
|
||||
self, predicate: Callable[[BrowserlikeSSEClient], bool], *, timeout: float = 45.0
|
||||
) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if predicate(self):
|
||||
return
|
||||
time.sleep(0.05)
|
||||
raise AssertionError("timed out waiting for predicate")
|
||||
|
||||
def wait_for_call_result(self, call_id: str, *, timeout: float = 45.0) -> None:
|
||||
self.wait_for(lambda c: call_id in c.tool_results_by_call(), timeout=timeout)
|
||||
|
||||
|
||||
def _dechunk(buf: bytes) -> tuple[bytes, bytes]:
|
||||
"""Incrementally decode HTTP/1.1 chunked transfer-encoding.
|
||||
|
||||
Consumes as many COMPLETE chunks from ``buf`` as possible and returns
|
||||
``(decoded_bytes, remainder)`` where ``remainder`` is the trailing
|
||||
partial chunk to carry into the next read. A zero-length chunk (stream
|
||||
end) simply stops consumption; the reader's ``recv`` EOF handles close.
|
||||
"""
|
||||
decoded = b""
|
||||
while True:
|
||||
if b"\r\n" not in buf:
|
||||
break # incomplete size line
|
||||
size_line, rest = buf.split(b"\r\n", 1)
|
||||
try:
|
||||
n = int(size_line.strip() or b"z", 16)
|
||||
except ValueError:
|
||||
break # malformed / partial — wait for more bytes
|
||||
if n == 0:
|
||||
break # last chunk marker
|
||||
if len(rest) < n + 2: # need n data bytes + trailing CRLF
|
||||
break
|
||||
decoded += rest[:n]
|
||||
buf = rest[n + 2 :]
|
||||
return decoded, buf
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Assertion helpers (shared by the scenarios).
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def assert_contiguous_ids(frames: list[SSEFrame]) -> None:
|
||||
"""Every id-bearing frame in a connection forms a gap-free, dup-free,
|
||||
strictly increasing run.
|
||||
|
||||
Holds for a connection that took no ``_seq``-filtered fresh path — a
|
||||
fresh connect made before any event (snap_seq == 0) and every
|
||||
``replay_ok`` reconnect (snap_seq == 0). The server stamps a fresh
|
||||
monotonic id per enqueue with no in-ring coalescing, so a
|
||||
non-filtered consumer sees consecutive ids.
|
||||
"""
|
||||
ids = [f.event_id_int for f in frames if f.event_id_int is not None]
|
||||
assert ids, "connection carried no id-bearing frames"
|
||||
assert len(set(ids)) == len(ids), f"duplicate SSE ids: {ids}"
|
||||
assert ids == sorted(ids), f"SSE ids not monotonic: {ids}"
|
||||
for prev, cur in zip(ids, ids[1:], strict=False):
|
||||
assert cur == prev + 1, f"gap in SSE ids between {prev} and {cur}: {ids}"
|
||||
|
||||
|
||||
def assert_ids_monotonic_no_dupes(frames: list[SSEFrame]) -> None:
|
||||
"""Weaker invariant that holds on EVERY connection (including
|
||||
``_seq``-filtered fresh/truncated paths, where gaps are legal): ids
|
||||
are strictly increasing with no duplicates."""
|
||||
ids = [f.event_id_int for f in frames if f.event_id_int is not None]
|
||||
assert len(set(ids)) == len(ids), f"duplicate SSE ids: {ids}"
|
||||
assert ids == sorted(ids), f"SSE ids not monotonic: {ids}"
|
||||
|
||||
|
||||
def assert_chunk_result_ordering(frames: list[SSEFrame]) -> None:
|
||||
"""Every ``tool_output_chunk`` for a call precedes that call's own
|
||||
``tool_result`` on the wire (the load-bearing ordering — the client
|
||||
removes the streaming <pre> when it renders the result)."""
|
||||
result_index: dict[str, int] = {}
|
||||
for i, f in enumerate(frames):
|
||||
if f.etype == "tool_result" and f.payload is not None:
|
||||
result_index[str(f.payload.get("call_id", ""))] = i
|
||||
for i, f in enumerate(frames):
|
||||
if f.etype == "tool_output_chunk" and f.payload is not None:
|
||||
cid = str(f.payload.get("call_id", ""))
|
||||
assert cid in result_index, f"chunk for call {cid} has no tool_result"
|
||||
assert i < result_index[cid], (
|
||||
f"chunk for call {cid} arrived AFTER its tool_result "
|
||||
f"(chunk idx {i} >= result idx {result_index[cid]})"
|
||||
)
|
||||
|
||||
|
||||
def assert_children_stamped(frames: list[SSEFrame], parent_call_id: str) -> None:
|
||||
"""Every sub-agent child tool event carries ``parent_call_id`` (stamped
|
||||
at the flush chokepoint). Sub-tool call_ids are minted
|
||||
``{parent}::r{run}s{step}::{provider_id}`` — the ``::`` segment is the
|
||||
identifying mark — and NONE may escape unstamped to the top level."""
|
||||
unstamped: list[tuple[str | None, str, Any]] = []
|
||||
stamped = 0
|
||||
for f in frames:
|
||||
if f.payload is None:
|
||||
continue
|
||||
items = f.payload.get("items")
|
||||
entries = items if isinstance(items, list) else [f.payload]
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
cid = str(entry.get("call_id", ""))
|
||||
if "::" not in cid:
|
||||
continue
|
||||
if entry.get("parent_call_id") == parent_call_id:
|
||||
stamped += 1
|
||||
else:
|
||||
unstamped.append((f.etype, cid, entry.get("parent_call_id")))
|
||||
assert stamped > 0, f"no child events found for parent {parent_call_id}"
|
||||
assert not unstamped, f"child events escaped unstamped (parent {parent_call_id}): {unstamped}"
|
||||
|
||||
|
||||
def history_tool_outputs(history_json: dict[str, Any]) -> dict[str, str]:
|
||||
"""Extract {call_id: output} from a /history projection, however the
|
||||
projection surfaces results (a folded ``output`` on a tool_call, or a
|
||||
trailing ``role: tool`` row keyed by ``tool_call_id``)."""
|
||||
out: dict[str, str] = {}
|
||||
for msg in history_json.get("messages", []):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
if msg.get("role") == "tool":
|
||||
cid = msg.get("tool_call_id") or msg.get("call_id")
|
||||
if cid is not None:
|
||||
out[str(cid)] = str(msg.get("content", ""))
|
||||
for tc in msg.get("tool_calls") or ():
|
||||
if not isinstance(tc, dict):
|
||||
continue
|
||||
cid = tc.get("id") or tc.get("call_id")
|
||||
if cid is not None and tc.get("output") is not None:
|
||||
out[str(cid)] = str(tc.get("output", ""))
|
||||
return out
|
||||
|
||||
|
||||
def assert_converged(client: BrowserlikeSSEClient, history_json: dict[str, Any]) -> None:
|
||||
"""Turn-level equivalence: every tool result the client assembled live
|
||||
is present, with the same output, in a fresh /history projection.
|
||||
|
||||
Compares by call_id so a reconnect that re-delivered a result can't
|
||||
hide a divergence, and asserts the /history side isn't empty (a
|
||||
silently-lost turn would leave the projection short)."""
|
||||
live = client.tool_results_by_call()
|
||||
hist = history_tool_outputs(history_json)
|
||||
assert hist, "fresh /history projected no tool results — a turn was lost"
|
||||
for call_id, output in live.items():
|
||||
assert call_id in hist, f"call {call_id} seen live but absent from /history: {sorted(hist)}"
|
||||
assert hist[call_id] == output, (
|
||||
f"call {call_id} output diverged: live={output!r} history={hist[call_id]!r}"
|
||||
)
|
||||
@@ -1,628 +0,0 @@
|
||||
"""Boot the REAL interactive Turnstone server for the SSE recovery e2e
|
||||
harness: real ``SessionManager`` + real ``ChatSession`` engine driven
|
||||
through a scripted chat-completions client at the SDK boundary, executing
|
||||
REAL bash tools, exposed over a real uvicorn socket.
|
||||
|
||||
The recipe (verified end-to-end) has four load-bearing pieces:
|
||||
|
||||
1. **Provider injection seam.** ``create_app`` takes a PRE-BUILT
|
||||
``SessionManager``, so the harness owns the ``session_factory``: it
|
||||
passes ``client=fake_client`` and OMITS the registry, so
|
||||
``ChatSession`` falls back to ``create_provider("openai-compatible")``
|
||||
== ``OpenAIChatCompletionsProvider`` — exactly what
|
||||
``tests._session_helpers.scripted_chat_client`` targets. No production
|
||||
monkeypatch of the engine.
|
||||
|
||||
2. **Auto-title suppression.** The first user message spawns a background
|
||||
``_generate_title`` LLM call that would consume the first scripted
|
||||
response (the tool call) and desync a positional script. Setting
|
||||
``session._title_generated = True`` before the first send disables it.
|
||||
|
||||
3. **Completion barrier.** ``/send`` returns immediately after spawning
|
||||
``ws.worker_thread``; joining that thread is the true "turn complete,
|
||||
every SSE event enqueued" barrier (``stream_end`` is per-LLM-call, not
|
||||
per-turn, so it is NOT a completion marker).
|
||||
|
||||
4. **Thread hygiene.** ``create_app``'s lifespan starts daemon fan-out
|
||||
threads (``_global_fanout_thread`` blocking on ``global_queue.get()``,
|
||||
``_aggregate_emitter_thread`` on a 10s loop). The harness used to
|
||||
swap them for no-ops (they had no shutdown and tripped conftest's
|
||||
leaked-thread guard), which kept the global lane dead here; #885 gave
|
||||
the lifespan a real shutdown (stop Event + a queue sentinel for the
|
||||
fanout, joined in the lifespan exit that ``stop()``'s
|
||||
``should_exit``/join drives), so the harness now runs them REAL — the
|
||||
``roster-restart`` scenario depends on a live global lane — and
|
||||
teardown stays clean with no ``allow_thread_leak``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import queue as _q
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import httpx
|
||||
import uvicorn
|
||||
|
||||
from tests._session_helpers import scripted_chat_client
|
||||
from turnstone.core.adapters.interactive_adapter import InteractiveAdapter
|
||||
from turnstone.core.auth import JWT_AUD_SERVER, create_jwt
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_manager import SessionManager
|
||||
from turnstone.core.session_ui_base import SessionUIBase
|
||||
from turnstone.core.storage import get_storage
|
||||
from turnstone.core.workstream import WorkstreamKind
|
||||
from turnstone.prompts import ClientType
|
||||
from turnstone.server import WebUI, create_app
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import MutableMapping
|
||||
|
||||
from turnstone.core.workstream import Workstream
|
||||
|
||||
_JWT_SECRET = "sse-recovery-e2e-jwt-secret-minimum-32-chars!"
|
||||
# Small server send buffer so a stalled consumer's in-flight backlog before
|
||||
# the listener-queue poison stays bounded (paired with the client's small
|
||||
# SO_RCVBUF in _sse_recovery_helpers). Harmless for prompt readers.
|
||||
_DEFAULT_SNDBUF = 8192
|
||||
|
||||
|
||||
def _fake_client(scripts: tuple[Any, ...]) -> Any:
|
||||
"""An SDK-shaped fake whose ``chat.completions.create`` follows a
|
||||
positional script (each a :func:`fake_chat_stream` kwargs dict)."""
|
||||
create_fn = scripted_chat_client(*scripts)
|
||||
client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create_fn)))
|
||||
client.calls = create_fn.calls
|
||||
return client
|
||||
|
||||
|
||||
class RecoveryServer:
|
||||
"""A booted interactive node the recovery scenarios drive."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
sndbuf: int = _DEFAULT_SNDBUF,
|
||||
listener_cap: int | None = None,
|
||||
extra_routes: list[Any] | None = None,
|
||||
port: int = 0,
|
||||
sock: socket.socket | None = None,
|
||||
) -> None:
|
||||
self._global_queue: _q.Queue[dict[str, Any]] = _q.Queue(maxsize=100000)
|
||||
self._global_listeners: list[_q.Queue[dict[str, Any]]] = []
|
||||
self._global_listeners_lock = threading.Lock()
|
||||
# Per-ws scripted client, resolved at factory-call time.
|
||||
self._pending_client: Any = _fake_client((dict(content="ok", finish_reason="stop"),))
|
||||
self._clients: dict[str, Any] = {}
|
||||
|
||||
WebUI._global_queue = self._global_queue
|
||||
|
||||
def session_factory(
|
||||
ui: Any,
|
||||
model_alias: str | None = None,
|
||||
ws_id: str | None = None,
|
||||
*,
|
||||
skill: Any = None,
|
||||
client_type: str = "",
|
||||
kind: WorkstreamKind = WorkstreamKind.INTERACTIVE,
|
||||
parent_ws_id: str | None = None,
|
||||
project_id: str = "",
|
||||
**_extra: Any,
|
||||
) -> ChatSession:
|
||||
client = self._pending_client
|
||||
if ws_id is not None:
|
||||
self._clients[ws_id] = client
|
||||
return ChatSession(
|
||||
client=client,
|
||||
model="test-model",
|
||||
ui=ui,
|
||||
instructions=None,
|
||||
temperature=None,
|
||||
max_tokens=1024,
|
||||
tool_timeout=30,
|
||||
ws_id=ws_id,
|
||||
user_id="recovery-user",
|
||||
client_type=ClientType.WEB,
|
||||
kind=kind,
|
||||
# Don't truncate large tool outputs: the harness tests
|
||||
# recovery, not the tool-result truncation budget, and a
|
||||
# truncated /history would diverge from the full live event
|
||||
# and defeat the convergence assertions.
|
||||
tool_truncation=10_000_000,
|
||||
)
|
||||
|
||||
self._adapter = InteractiveAdapter(
|
||||
global_queue=self._global_queue,
|
||||
ui_factory=lambda ws: WebUI(
|
||||
ws_id=ws.id, user_id=ws.user_id, kind=ws.kind, parent_ws_id=ws.parent_ws_id
|
||||
),
|
||||
session_factory=session_factory,
|
||||
)
|
||||
self._manager = SessionManager(
|
||||
self._adapter, storage=get_storage(), max_active=32, node_id="recovery-node"
|
||||
)
|
||||
self._adapter.attach(self._manager)
|
||||
WebUI._workstream_mgr = self._manager
|
||||
|
||||
# delay_load knob state (see the method): wraps the storage
|
||||
# singleton's load_messages; restored in stop().
|
||||
self._load_delay_ms = 0
|
||||
self._load_calls = 0
|
||||
# load_messages runs on asyncio.to_thread WORKERS, and delay_load
|
||||
# exists precisely to overlap two of them — unlike the
|
||||
# single-writer HTTP counters, this one has genuine concurrent
|
||||
# writers, so the increment takes a lock (a lost update would
|
||||
# false-FAIL G7's load_delta === 2, or mask a third load).
|
||||
self._load_calls_lock = threading.Lock()
|
||||
_storage_obj = get_storage()
|
||||
self._orig_load_messages = _storage_obj.load_messages
|
||||
|
||||
def _delayed_load(*a: Any, **k: Any) -> Any:
|
||||
with self._load_calls_lock:
|
||||
self._load_calls += 1
|
||||
result = self._orig_load_messages(*a, **k)
|
||||
# Sleep AFTER the load: the held flight must hold the data it
|
||||
# actually read (its transaction point), so a flight parked
|
||||
# across a rewind genuinely carries PRE-rewind rows — a
|
||||
# pre-load sleep would read post-rewind storage and mask a
|
||||
# wrongly-joined flight as fresh truth.
|
||||
d = self._load_delay_ms
|
||||
if d > 0:
|
||||
time.sleep(d / 1000.0)
|
||||
return result
|
||||
|
||||
_storage_obj.load_messages = _delayed_load # type: ignore[method-assign]
|
||||
self._patched_storage = _storage_obj
|
||||
|
||||
# Optional small listener-queue cap. The cap is a default arg on the
|
||||
# registration methods with no config/env override, so lower it by
|
||||
# patching their ``__defaults__`` (restored on stop). fix-3's
|
||||
# de-amplification makes a real 500-cap overflow need a pathological
|
||||
# storm; a small cap exercises the identical _ListenerOverflow ->
|
||||
# stream_overflow -> reconnect-replay path within a bounded storm.
|
||||
self._orig_defaults: list[tuple[Any, tuple[Any, ...] | None]] = []
|
||||
if listener_cap is not None:
|
||||
for meth in (
|
||||
SessionUIBase._register_listener,
|
||||
SessionUIBase.register_listener_with_in_progress_snapshot,
|
||||
SessionUIBase.register_listener_with_replay,
|
||||
):
|
||||
self._orig_defaults.append((meth, meth.__defaults__))
|
||||
meth.__defaults__ = (listener_cap,)
|
||||
|
||||
self._app = create_app(
|
||||
workstreams=self._manager,
|
||||
global_queue=self._global_queue,
|
||||
global_listeners=self._global_listeners,
|
||||
global_listeners_lock=self._global_listeners_lock,
|
||||
skip_permissions=True,
|
||||
jwt_secret=_JWT_SECRET,
|
||||
node_id="recovery-node",
|
||||
# /history + tenant checks read app.state.auth_storage.
|
||||
auth_storage=get_storage(),
|
||||
)
|
||||
# Same-origin extras (Tier 2 serves its recovery page here so the real
|
||||
# Pane's cookie auth + EventSource work without cross-origin plumbing).
|
||||
if extra_routes:
|
||||
self._app.router.routes.extend(extra_routes)
|
||||
|
||||
# Pre-bind a listening socket with a small SO_SNDBUF (accepted conns
|
||||
# inherit it), then hand it to uvicorn. ``sock`` injection: the
|
||||
# gap-free restart scenarios (roster-restart-native) bind a
|
||||
# placeholder BEFORE stopping the prior node and hand it in here —
|
||||
# a failed EventSource reconnect attempt is TERMINAL per WHATWG
|
||||
# (fail-the-connection → CLOSED, no further retries), so the
|
||||
# native-retry leg must never observe a refused-window; the
|
||||
# placeholder's listen backlog completes the TCP handshake during
|
||||
# the boot and uvicorn drains it once serving.
|
||||
self._sock = sock if sock is not None else make_listen_socket(port, sndbuf=sndbuf)
|
||||
self._port = int(self._sock.getsockname()[1])
|
||||
|
||||
# -- fault injection (public knobs below) ----------------------------
|
||||
# In-process arming: the Tier-2 runner holds this RecoveryServer and
|
||||
# arms a knob, THEN drives the browser request that consumes it.
|
||||
# Single-writer by construction — the runner never arms a knob while
|
||||
# the loop thread is mid-consume — and CPython makes each int read /
|
||||
# write atomic, so these need no lock even though the uvicorn loop
|
||||
# thread increments/decrements them while the runner thread reads.
|
||||
self.history_requests = 0
|
||||
# /history responses the PRODUCTION route answered 200 (not the
|
||||
# fault layer's injected 500s). See _fault_app for why arrival
|
||||
# counting cannot substitute.
|
||||
self.history_ok = 0
|
||||
self.rewind_requests = 0
|
||||
# Per-ws SSE connection opens (``GET …/events`` — the EventSource the
|
||||
# pane's connectSSE builds). A TRANSPORT-FREE heal (the #890 idle-edge
|
||||
# staleness backstop, a quiesced REST refetch) must leave this FLAT; a
|
||||
# reload-based backstop would bump it once per reconnect (the round-5
|
||||
# storm). Same lock-free single-writer int discipline as above.
|
||||
self.events_requests = 0
|
||||
# Global-lane SSE connection opens (``GET …/events/global`` — the
|
||||
# roster stream app.js's connectGlobalSSE builds). The
|
||||
# roster-restart scenario (#881) asserts the post-restart manual
|
||||
# reconnect actually reached the reborn node's real endpoint.
|
||||
self.global_events_requests = 0
|
||||
self._history_fail_remaining = 0
|
||||
self._history_delay_ms = 0
|
||||
# A thin pure-ASGI fault layer wrapping the REAL app (the production
|
||||
# app itself is untouched): count + optionally delay/fail
|
||||
# ``GET …/history``, count ``POST …/rewind``, count each per-ws SSE
|
||||
# connection open (``GET …/events``), forward everything else (SSE
|
||||
# bodies, /send, lifespan, static) verbatim.
|
||||
production_app = self._app
|
||||
|
||||
async def _fault_app(scope: dict[str, Any], receive: Any, send: Any) -> None:
|
||||
if scope.get("type") == "http":
|
||||
path = scope.get("path", "")
|
||||
method = scope.get("method", "")
|
||||
if path.endswith("/history") and method == "GET":
|
||||
# ARRIVAL, never move. Scenarios that hold a request open
|
||||
# use this bump as the IN-FLIGHT edge (E6/E7/G1/G7 say so
|
||||
# at their poll sites); counting on forward instead would
|
||||
# delay it past the hold and silently stop those scenarios
|
||||
# testing anything.
|
||||
self.history_requests += 1
|
||||
if self._history_delay_ms > 0:
|
||||
await asyncio.sleep(self._history_delay_ms / 1000.0)
|
||||
if self._history_fail_remaining > 0:
|
||||
self._history_fail_remaining -= 1
|
||||
await send(
|
||||
{
|
||||
"type": "http.response.start",
|
||||
"status": 500,
|
||||
"headers": [(b"content-type", b"application/json")],
|
||||
}
|
||||
)
|
||||
await send({"type": "http.response.body", "body": b'{"error": "injected"}'})
|
||||
return
|
||||
|
||||
# Successful-RESPONSE counter, distinct from the arrival
|
||||
# bump above. A scenario asserting that a render was
|
||||
# DECLINED needs to know a good payload actually existed —
|
||||
# otherwise "the client refused to render" and "there was
|
||||
# nothing to render" produce identical observables (no
|
||||
# wipe, latch held). Arrival cannot prove that, and
|
||||
# neither can an injected-fail budget: a PRODUCTION-side
|
||||
# 500/404 would slip through both. Reading the real
|
||||
# status off the response start is the only honest signal.
|
||||
async def _counting_send(message: MutableMapping[str, Any]) -> None:
|
||||
if (
|
||||
message.get("type") == "http.response.start"
|
||||
and message.get("status") == 200
|
||||
):
|
||||
self.history_ok += 1
|
||||
await send(message)
|
||||
|
||||
await production_app(scope, receive, _counting_send)
|
||||
return
|
||||
elif path.endswith("/rewind") and method == "POST":
|
||||
self.rewind_requests += 1
|
||||
elif path.endswith("/events") and method == "GET":
|
||||
# Per-ws SSE connection open — count it (readable on
|
||||
# RecoveryServer) and forward the long-lived stream
|
||||
# verbatim below. Uniquely the per-ws stream: the global
|
||||
# lane is ``…/events/global`` (ends ``/global``), and the
|
||||
# route the pane's EventSource hits is
|
||||
# ``…/workstreams/{ws_id}/events`` (session_routes).
|
||||
self.events_requests += 1
|
||||
elif path.endswith("/events/global") and method == "GET":
|
||||
self.global_events_requests += 1
|
||||
await production_app(scope, receive, send)
|
||||
|
||||
# ``timeout_graceful_shutdown``: an SSE stream that is still open
|
||||
# at ``stop()`` would otherwise park uvicorn's graceful drain
|
||||
# indefinitely (the 20s thread-join just expires and the browser
|
||||
# stays attached to the zombie server — the roster-restart-native
|
||||
# scenario is the one caller that stops a node mid-stream). A
|
||||
# bounded drain force-closes the stream after 2s and the lifespan
|
||||
# shutdown (#885's daemon-thread teardown) still runs after it.
|
||||
self._server = uvicorn.Server(
|
||||
uvicorn.Config(
|
||||
_fault_app,
|
||||
log_level="warning",
|
||||
lifespan="on",
|
||||
timeout_graceful_shutdown=2,
|
||||
)
|
||||
)
|
||||
self._thread = threading.Thread(
|
||||
target=self._serve, name=f"uvicorn-recovery-{self._port}", daemon=True
|
||||
)
|
||||
self._thread.start()
|
||||
if not _tcp_ready(self._port, 10.0):
|
||||
self.stop()
|
||||
raise AssertionError("recovery server did not accept TCP")
|
||||
|
||||
self._token = create_jwt(
|
||||
user_id="recovery-user",
|
||||
scopes=frozenset({"read", "write", "approve", "service"}),
|
||||
source="recovery",
|
||||
secret=_JWT_SECRET,
|
||||
audience=JWT_AUD_SERVER,
|
||||
)
|
||||
self._http = httpx.Client(base_url=self.base_url, timeout=httpx.Timeout(30.0))
|
||||
|
||||
def _serve(self) -> None:
|
||||
loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(loop)
|
||||
try:
|
||||
loop.run_until_complete(self._server.serve(sockets=[self._sock]))
|
||||
finally:
|
||||
pending = asyncio.all_tasks(loop)
|
||||
for task in pending:
|
||||
task.cancel()
|
||||
if pending:
|
||||
with contextlib.suppress(Exception):
|
||||
loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
|
||||
loop.close()
|
||||
|
||||
# -- properties ----------------------------------------------------------
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
return f"http://127.0.0.1:{self._port}"
|
||||
|
||||
@property
|
||||
def token(self) -> str:
|
||||
return self._token
|
||||
|
||||
@property
|
||||
def manager(self) -> SessionManager:
|
||||
return self._manager
|
||||
|
||||
# -- workstream lifecycle ------------------------------------------------
|
||||
|
||||
def create_workstream(self, *scripts: Any, name: str = "recovery-ws") -> str:
|
||||
"""Create a ws whose scripted LLM follows ``scripts`` (positional
|
||||
:func:`fake_chat_stream` kwargs). Auto-approves tools and suppresses
|
||||
the auto-title call so the positional script stays in sync."""
|
||||
self._pending_client = _fake_client(scripts)
|
||||
ws = self._manager.create(user_id="recovery-user", name=name)
|
||||
self._prime_ws(ws)
|
||||
return ws.id
|
||||
|
||||
def open_workstream(self, ws_id: str, *scripts: Any) -> None:
|
||||
"""Rehydrate a persisted ws on THIS node (the restart path). Fresh
|
||||
UI → empty ring + storage-seeded ``_event_id``."""
|
||||
if scripts:
|
||||
self._pending_client = _fake_client(scripts)
|
||||
ws = self._manager.open(ws_id)
|
||||
if ws is None:
|
||||
raise AssertionError(f"open_workstream: ws {ws_id} not resurrectable")
|
||||
self._prime_ws(ws)
|
||||
|
||||
def _prime_ws(self, ws: Workstream) -> None:
|
||||
if isinstance(ws.ui, SessionUIBase):
|
||||
ws.ui.auto_approve = True # blanket tool auto-approval
|
||||
if ws.session is not None:
|
||||
ws.session._title_generated = True # suppress the auto-title LLM call
|
||||
|
||||
def send(self, ws_id: str, message: str = "go") -> None:
|
||||
"""POST /send — spawns the worker thread and returns immediately."""
|
||||
r = self._http.post(
|
||||
f"/v1/api/workstreams/{ws_id}/send",
|
||||
headers={"Authorization": f"Bearer {self._token}"},
|
||||
json={"message": message},
|
||||
)
|
||||
r.raise_for_status()
|
||||
|
||||
def wait_turn(self, ws_id: str, *, timeout: float = 45.0) -> None:
|
||||
"""Block until the turn's worker thread finishes (the true
|
||||
turn-complete barrier) and the ws is idle."""
|
||||
deadline = time.monotonic() + timeout
|
||||
worker: threading.Thread | None = None
|
||||
while time.monotonic() < deadline:
|
||||
ws = self._manager.get(ws_id)
|
||||
worker = ws.worker_thread if ws is not None else None
|
||||
if worker is not None:
|
||||
break
|
||||
time.sleep(0.02)
|
||||
if worker is not None:
|
||||
worker.join(timeout=max(0.5, deadline - time.monotonic()))
|
||||
if worker.is_alive():
|
||||
raise AssertionError(f"turn worker for {ws_id} did not finish in {timeout}s")
|
||||
|
||||
def get_ws(self, ws_id: str) -> Workstream | None:
|
||||
return self._manager.get(ws_id)
|
||||
|
||||
def ws_state(self, ws_id: str) -> str:
|
||||
ws = self._manager.get(ws_id)
|
||||
return ws.state.value if ws is not None else ""
|
||||
|
||||
def ring_span(self, ws_id: str) -> tuple[int | None, int]:
|
||||
"""(earliest retained ring event_id or None, latest counter) — lets a
|
||||
scenario wait for the ring to evict a specific cursor."""
|
||||
ws = self._manager.get(ws_id)
|
||||
ui = ws.ui if ws is not None else None
|
||||
if not isinstance(ui, SessionUIBase):
|
||||
return None, 0
|
||||
buf = ui._event_buffer
|
||||
earliest = buf[0][0] if buf else None
|
||||
return earliest, ui._event_id
|
||||
|
||||
def listener_poisoned(self, ws_id: str) -> bool:
|
||||
"""True once any live SSE listener on the ws has poisoned (overflow)."""
|
||||
ws = self._manager.get(ws_id)
|
||||
ui = ws.ui if ws is not None else None
|
||||
if not isinstance(ui, SessionUIBase):
|
||||
return False
|
||||
return any(getattr(q, "poisoned", False) for q in list(ui._listeners))
|
||||
|
||||
def max_event_id(self, ws_id: str) -> int | None:
|
||||
"""The storage high-water ``MAX(conversations.event_id)`` — what a
|
||||
restarted node's fresh UI seeds ``_event_id`` from."""
|
||||
result: int | None = get_storage().get_max_event_id(ws_id)
|
||||
return result
|
||||
|
||||
def fetch_history(self, ws_id: str) -> dict[str, Any]:
|
||||
r = self._http.get(
|
||||
f"/v1/api/workstreams/{ws_id}/history",
|
||||
headers={"Authorization": f"Bearer {self._token}"},
|
||||
)
|
||||
r.raise_for_status()
|
||||
result: dict[str, Any] = r.json()
|
||||
return result
|
||||
|
||||
# -- fault-injection knobs -----------------------------------------------
|
||||
# Armed in-process by the Tier-2 runner (single writer at a time — see
|
||||
# __init__). A plain int is deliberate: CPython makes the loop thread's
|
||||
# increment/decrement and the runner thread's read each atomic, and the
|
||||
# arm-then-consume ordering means they never race.
|
||||
|
||||
def delay_load(self, ms: int) -> None:
|
||||
"""Hold ``storage.load_messages`` itself open for ``ms`` (0 = off).
|
||||
|
||||
``delay_history`` sleeps in the FAULT LAYER — before the route —
|
||||
so two delayed requests never overlap inside the #884 flight
|
||||
machinery (the first flight completes and pops before the second
|
||||
arrives at the route). This knob sleeps INSIDE the shared
|
||||
reconstruction's ``load_messages`` (sync, called via
|
||||
``asyncio.to_thread`` — the sleep parks only that worker), which
|
||||
is the same layer the unit tests gate, so held flights genuinely
|
||||
overlap and join/miss behavior is observable end to end via
|
||||
``load_calls``.
|
||||
"""
|
||||
self._load_delay_ms = ms
|
||||
|
||||
@property
|
||||
def load_calls(self) -> int:
|
||||
"""``load_messages`` entries (pre-sleep) — the flight-layer twin
|
||||
of ``history_requests`` (which counts HTTP arrivals): a JOINED
|
||||
request never enters ``load_messages``, so join=1 / miss=2."""
|
||||
return self._load_calls
|
||||
|
||||
def fail_history(self, count: int) -> None:
|
||||
"""Make the next ``count`` ``GET …/history`` responses a 500 — the
|
||||
failed refetch the #890 guard-before-wipe must survive."""
|
||||
self._history_fail_remaining = count
|
||||
|
||||
def delay_history(self, ms: int) -> None:
|
||||
"""Hold each ``GET …/history`` ``ms`` ms before forwarding (0
|
||||
clears). Opens the clear_ui-refetch quiesce window that the row
|
||||
affordance gate (``busy || _historyStale``) must close."""
|
||||
self._history_delay_ms = ms
|
||||
|
||||
@property
|
||||
def history_fail_remaining(self) -> int:
|
||||
"""Unconsumed forced-failure budget — 0 proves the armed failure
|
||||
actually fired (assert backend state, never scripted absence)."""
|
||||
return self._history_fail_remaining
|
||||
|
||||
# -- teardown ------------------------------------------------------------
|
||||
|
||||
def stop(self, *, hard: bool = False) -> None:
|
||||
"""Stop the node.
|
||||
|
||||
``hard=True`` skips the per-workstream ``manager.close`` sweep — a
|
||||
graceful close routes through ``cleanup_session_ui`` →
|
||||
``session.cancel()``, whose bash cancel path PERSISTS a
|
||||
synthesized "Cancelled by user" result while the old node is
|
||||
still alive, which masks crash states. A hard stop leaves any
|
||||
in-flight tool call genuinely unresulted in storage, modelling a
|
||||
SIGKILL/OOM death (the coord-orphan-rewind scenario's premise).
|
||||
The 2s graceful-shutdown timeout (uvicorn config) force-closes
|
||||
open SSE streams, and the lifespan teardown still runs, so the
|
||||
#885 daemon threads are joined on both paths.
|
||||
"""
|
||||
if not hard:
|
||||
with contextlib.suppress(Exception):
|
||||
for ws in list(self._manager.list_all()):
|
||||
with contextlib.suppress(Exception):
|
||||
self._manager.close(ws.id)
|
||||
# hard=True relies on ``timeout_graceful_shutdown=2`` (set in the
|
||||
# uvicorn config above) to force-close the pane's EventSource:
|
||||
# should_exit alone still runs the ASGI lifespan teardown, so the
|
||||
# #885 daemon threads and the sse_executor are joined either way
|
||||
# (``force_exit`` would SKIP the lifespan and leak them — the
|
||||
# fanout thread blocks on queue.get() forever). NOTE: the killed
|
||||
# workstream's in-flight tool keeps executing on this process's
|
||||
# session thread and persists its result at natural completion —
|
||||
# hard-kill scenarios must use a paced tool that outlives their
|
||||
# observation window.
|
||||
self._server.should_exit = True
|
||||
self._thread.join(timeout=20)
|
||||
with contextlib.suppress(Exception):
|
||||
self._http.close()
|
||||
with contextlib.suppress(OSError):
|
||||
self._sock.close()
|
||||
# Restore any patched cap defaults.
|
||||
for meth, defaults in self._orig_defaults:
|
||||
meth.__defaults__ = defaults
|
||||
# Restore the storage singleton's load_messages (delay_load knob).
|
||||
with contextlib.suppress(Exception):
|
||||
self._patched_storage.load_messages = self._orig_load_messages # type: ignore[method-assign]
|
||||
|
||||
|
||||
def make_listen_socket(port: int, *, sndbuf: int = _DEFAULT_SNDBUF) -> socket.socket:
|
||||
"""Bound + listening socket the way :class:`RecoveryServer` binds its own.
|
||||
|
||||
``SO_REUSEPORT`` on every listener (same process, same uid) is what
|
||||
lets a restart scenario bind the successor's socket while the prior
|
||||
node still holds the port — the seam behind the gap-free handoff
|
||||
documented at the ``sock`` parameter. ``SO_SNDBUF`` matches the
|
||||
server's small send buffer so accepted connections inherit identical
|
||||
backpressure behavior regardless of which side bound the socket.
|
||||
"""
|
||||
s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEPORT, 1)
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, sndbuf)
|
||||
s.bind(("127.0.0.1", port)) # port=0 -> ephemeral; fixed -> restart reuse
|
||||
s.listen(128)
|
||||
return s
|
||||
|
||||
|
||||
def _tcp_ready(port: int, timeout: float) -> bool:
|
||||
end = time.monotonic() + timeout
|
||||
while time.monotonic() < end:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
def bash_toolcall_script(
|
||||
call_id: str, command: str, *, finish_reason: str = "tool_calls"
|
||||
) -> dict[str, Any]:
|
||||
"""A scripted assistant turn issuing ONE bash tool call."""
|
||||
return dict(
|
||||
tool_calls=[{"id": call_id, "name": "bash", "arguments": json.dumps({"command": command})}],
|
||||
finish_reason=finish_reason,
|
||||
)
|
||||
|
||||
|
||||
def parallel_bash_script(commands: dict[str, str]) -> dict[str, Any]:
|
||||
"""A scripted assistant turn issuing SEVERAL bash tool calls at once
|
||||
(the parallel-pool storm), ``{call_id: command}``.
|
||||
|
||||
Each command is prefixed with a no-op ``: <call_id>;`` so the tool
|
||||
ARGUMENTS are distinct per call while the OUTPUT is unchanged (``:``
|
||||
ignores its args and prints nothing). Identical-argument parallel
|
||||
calls otherwise trip the session's repeat-tool-call guard, which
|
||||
appends a warning to the PERSISTED result only (not the live event) —
|
||||
an orthogonal divergence that would mask the recovery behavior the
|
||||
convergence assertions test.
|
||||
"""
|
||||
return dict(
|
||||
tool_calls=[
|
||||
{
|
||||
"id": cid,
|
||||
"name": "bash",
|
||||
"arguments": json.dumps({"command": f": {cid}; {cmd}"}),
|
||||
}
|
||||
for cid, cmd in commands.items()
|
||||
],
|
||||
finish_reason="tool_calls",
|
||||
)
|
||||
|
||||
|
||||
def final_text_script(content: str = "done") -> dict[str, Any]:
|
||||
"""The scripted assistant turn that ends the agent loop (no tools)."""
|
||||
return dict(content=content, finish_reason="stop")
|
||||
@@ -4,9 +4,6 @@ import asyncio
|
||||
import contextlib
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
@@ -219,98 +216,6 @@ def _seed_static_state(mgr: MCPClientManager, name: str, **overrides: Any) -> St
|
||||
return state
|
||||
|
||||
|
||||
def _run_on_loop(loop: asyncio.AbstractEventLoop, coro: Any, timeout: float = 10) -> Any:
|
||||
"""Submit *coro* to *loop*, wait for the result.
|
||||
|
||||
The ONE copy shared by the MCP test files — four hand-synced copies
|
||||
had already drifted on the timeout (5s hardcoded vs a 10s default).
|
||||
The timeout is an upper bound on waiting, not a behavior assertion,
|
||||
so the most generous variant won the merge.
|
||||
"""
|
||||
fut = asyncio.run_coroutine_threadsafe(coro, loop)
|
||||
return fut.result(timeout=timeout)
|
||||
|
||||
|
||||
def _drain_background(mgr: MCPClientManager, loop: asyncio.AbstractEventLoop) -> None:
|
||||
"""Deterministically await ``mgr``'s tracked background tasks.
|
||||
|
||||
Replaces fixed sleeps for synchronizing with scheduled dead-grant
|
||||
drops / spawned refreshes: exact, and immune to slow-runner flake.
|
||||
"""
|
||||
|
||||
async def _drain() -> None:
|
||||
tasks = [t for t in list(mgr._background_tasks) if not t.done()]
|
||||
if tasks:
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
|
||||
_run_on_loop(loop, _drain())
|
||||
|
||||
|
||||
def _poll_until(predicate: Callable[[], bool], timeout: float, interval: float = 0.05) -> bool:
|
||||
"""Poll *predicate* until true or *timeout* elapses — the ONE wait loop.
|
||||
|
||||
Shared by the live MCP smoke tests' condition helpers so the
|
||||
deadline/poll pattern doesn't accrete per-file hand-synced copies.
|
||||
"""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if predicate():
|
||||
return True
|
||||
time.sleep(interval)
|
||||
return False
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
"""Grab an ephemeral localhost port for a live-server subprocess.
|
||||
|
||||
Shared by the live MCP smoke tests (flaky-server, push-refresh) so
|
||||
the socket-probe helpers stay in one place instead of drifting per
|
||||
file.
|
||||
"""
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
return int(s.getsockname()[1])
|
||||
|
||||
|
||||
def _tcp_accepts(port: int) -> bool:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
return True
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def _wait_tcp_ready(port: int, timeout: float) -> bool:
|
||||
"""Poll until something accepts TCP on 127.0.0.1:*port* (live tests)."""
|
||||
return _poll_until(lambda: _tcp_accepts(port), timeout)
|
||||
|
||||
|
||||
def _wait_session_live(mgr: MCPClientManager, name: str, timeout: float) -> bool:
|
||||
"""Poll until static server *name* has a live session (live tests)."""
|
||||
|
||||
def _live() -> bool:
|
||||
state = mgr._static_servers.get(name)
|
||||
return state is not None and state.session is not None
|
||||
|
||||
return _poll_until(_live, timeout)
|
||||
|
||||
|
||||
def _popen_mcp_server(script_path: Any, port: int) -> subprocess.Popen[bytes]:
|
||||
"""Start a FastMCP live-server subprocess, streams to DEVNULL.
|
||||
|
||||
The shared spawn primitive for the live MCP smoke tests
|
||||
(flaky-server flap loop, push-refresh) — the readiness wait and the
|
||||
skip-vs-raise-on-failure policy legitimately differ per test and
|
||||
stay at the call sites. ``sys.executable`` runs the same interpreter,
|
||||
so a server-side import gap surfaces as a failed TCP wait, not here.
|
||||
"""
|
||||
return subprocess.Popen(
|
||||
[sys.executable, str(script_path), str(port)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
|
||||
|
||||
def make_oidc_test_config(**overrides: Any) -> OIDCConfig:
|
||||
"""Build a test ``OIDCConfig`` with sensible defaults.
|
||||
|
||||
@@ -430,40 +335,6 @@ def mock_openai_client():
|
||||
return client
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def make_config_store():
|
||||
"""Factory for a lightweight ConfigStore double.
|
||||
|
||||
``make_config_store(**overrides)`` returns an object whose ``.get(key)``
|
||||
yields the override when present, else the registered SettingDef default —
|
||||
mirroring the real :meth:`ConfigStore.get` fail-open (a bool setting reads
|
||||
as its ``False`` default on a miss, never ``None``). Shared by the
|
||||
``server.require_project`` gate / advisory tests.
|
||||
"""
|
||||
|
||||
_unset = object()
|
||||
|
||||
def _make(**overrides: Any) -> Any:
|
||||
from turnstone.core.settings_registry import SETTINGS
|
||||
|
||||
class _ConfigStoreDouble:
|
||||
def get(self, key: str, default: Any = _unset) -> Any:
|
||||
# Mirror ConfigStore.get precedence exactly: cache (overrides)
|
||||
# first, then a caller-supplied default, then the registry
|
||||
# default, then None — so a reused caller passing an explicit
|
||||
# default for an unset key gets the same value production would.
|
||||
if key in overrides:
|
||||
return overrides[key]
|
||||
if default is not _unset:
|
||||
return default
|
||||
defn = SETTINGS.get(key)
|
||||
return defn.default if defn else None
|
||||
|
||||
return _ConfigStoreDouble()
|
||||
|
||||
return _make
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_policy_cache():
|
||||
"""Drop the in-process tool-policy cache between tests.
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "",
|
||||
"provider_content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": "{\"city\": \"Oslo\"}",
|
||||
"name": "get_weather"
|
||||
},
|
||||
"id": "synth-id-0",
|
||||
"type": "function"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scenario": "blank_id_tools",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Before tools",
|
||||
"provider_content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": "{\"city\": \"Nice\"}",
|
||||
"name": "get_weather"
|
||||
},
|
||||
"id": "call_1",
|
||||
"type": "function"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scenario": "combined_content_tools_finish",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Before tools"
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,35 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Redac",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "content_filter",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Redac"
|
||||
],
|
||||
[
|
||||
"error",
|
||||
"Warning: response blocked by content filter."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Hello world.",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "content_only",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Hello world."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "finish_only_no_content",
|
||||
"ui_events": [
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Answer with sources.",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "info_postfinish_footer",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Answer w"
|
||||
],
|
||||
[
|
||||
"info",
|
||||
"Sources:\n- example.com/page"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"ith sources."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Seals are pinnipeds.",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "info_prefinish",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"info",
|
||||
"[Searching: pinniped taxonomy]"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Seals ar"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"e pinnipeds."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,43 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Partial answer",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "length_with_tools",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Pa"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"rtial answer"
|
||||
],
|
||||
[
|
||||
"error",
|
||||
"Warning: response truncated (hit 4096 token limit). Use --max-tokens to increase, or /compact to free context."
|
||||
],
|
||||
[
|
||||
"error",
|
||||
"Discarding partial tool calls from truncated response."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 0,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 11
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Half an ans",
|
||||
"provider_content": null,
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "no_finish_clean_exhaust",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Half an ans"
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,36 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Blocked.",
|
||||
"provider_content": [
|
||||
{
|
||||
"text": "captured",
|
||||
"type": "reasoning_text"
|
||||
}
|
||||
],
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "provider_blocks_on_terminal",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Blocked."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,44 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Answer.",
|
||||
"provider_content": [
|
||||
{
|
||||
"text": "think a think b",
|
||||
"type": "reasoning_text"
|
||||
}
|
||||
],
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "reasoning_then_content",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"reasoning",
|
||||
"think a"
|
||||
],
|
||||
[
|
||||
"reasoning",
|
||||
" think b"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Answer."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "\n\nAnswer",
|
||||
"provider_content": [
|
||||
{
|
||||
"text": "plan",
|
||||
"type": "reasoning_text"
|
||||
}
|
||||
],
|
||||
"tool_calls": null
|
||||
},
|
||||
"scenario": "think_tags_split_across_chunks",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"reasoning",
|
||||
"plan"
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"\n\nAnswer"
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
{
|
||||
"cancelled_partial": null,
|
||||
"last_usage": {
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"completion_tokens": 7,
|
||||
"prompt_tokens": 11,
|
||||
"total_tokens": 18
|
||||
},
|
||||
"raised": null,
|
||||
"result": {
|
||||
"content": "Calling.",
|
||||
"provider_content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": "{\"city\": \"Paris\"}",
|
||||
"name": "get_weather"
|
||||
},
|
||||
"id": "call_1",
|
||||
"type": "function"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scenario": "tools_simple",
|
||||
"ui_events": [
|
||||
[
|
||||
"thinking_stop",
|
||||
""
|
||||
],
|
||||
[
|
||||
"content",
|
||||
"Calling."
|
||||
],
|
||||
[
|
||||
"stream_end",
|
||||
""
|
||||
]
|
||||
]
|
||||
}
|
||||
@@ -57,6 +57,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -28,5 +28,6 @@
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b"
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -49,6 +49,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -42,6 +42,7 @@
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -28,5 +28,6 @@
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b"
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -40,6 +40,7 @@
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
|
||||
@@ -51,6 +51,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,6 +43,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -35,6 +35,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-sonnet-4-6",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"thinking": {
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -51,6 +51,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,6 +43,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -39,6 +39,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
}
|
||||
],
|
||||
"model": "claude-opus-4-8",
|
||||
"output_config": {
|
||||
"effort": "medium"
|
||||
},
|
||||
"thinking": {
|
||||
"display": "summarized",
|
||||
"type": "adaptive"
|
||||
|
||||
@@ -43,10 +43,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
@@ -18,8 +18,10 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
}
|
||||
},
|
||||
"temperature": 0.5
|
||||
}
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
},
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"function": {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user