mirror of
https://github.com/turnstonelabs/turnstone.git
synced 2026-08-13 07:22:24 -06:00
Compare commits
188 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e99d3ee139 | |||
| 4f0fc3f219 | |||
| dc701986f7 | |||
| bedd25fbe7 | |||
| 251a912275 | |||
| d48902fd01 | |||
| 702ac43d0e | |||
| 01f83dc90f | |||
| 2463c480c2 | |||
| 2a3dfbc6fb | |||
| 6c3b3cc098 | |||
| 0dc52f05ee | |||
| 02929c0d00 | |||
| b2add19c56 | |||
| 5ce1873e9e | |||
| 7698a928c5 | |||
| 4e2eea2f86 | |||
| a6752cb645 | |||
| 06ba4e8d4f | |||
| 94dcaf34fd | |||
| 1a2a689033 | |||
| c02f960d0a | |||
| bfde387206 | |||
| 27d112ff60 | |||
| aeab2535b1 | |||
| 4638d22bd0 | |||
| ee3bd1dcf2 | |||
| ae3a83ccce | |||
| 0f17433e1f | |||
| 043554bb2f | |||
| 8389808add | |||
| 6cbef4f633 | |||
| 2b6dde4f7e | |||
| fbe31b9885 | |||
| ef13f40cf5 | |||
| bfa1b104cf | |||
| 324a1d1a35 | |||
| 95ab88ff6f | |||
| d29840f985 | |||
| 44c0b9c340 | |||
| cdbdf3dc2b | |||
| 8aabb061c2 | |||
| 8bd638569f | |||
| 251dc44a46 | |||
| efd0a1d000 | |||
| 2f93c39fd3 | |||
| f27ce104c6 | |||
| 20a61b692b | |||
| 1966107efe | |||
| d5b2fe6e45 | |||
| 1569819750 | |||
| 4107a30148 | |||
| d06d88b83f | |||
| c411aac939 | |||
| 104715b650 | |||
| 59a9899149 | |||
| 4da7c3b91c | |||
| 012f4e3e16 | |||
| 3636724848 | |||
| eeda5ac312 | |||
| be872b840f | |||
| ee94ae8ba1 | |||
| 357d00400e | |||
| 3615f98c19 | |||
| 801d5dfb59 | |||
| a352b20786 | |||
| 6f8efaa44e | |||
| 9c90fe2722 | |||
| 1035fe05eb | |||
| 530958e06b | |||
| e136237b63 | |||
| 41e9907803 | |||
| 7c34d859b4 | |||
| 16ee4e12ef | |||
| 976c07d047 | |||
| 8da5dc3f5a | |||
| 7f0e0406b3 | |||
| 6c94514106 | |||
| 0d0fe8dd71 | |||
| 18c3301428 | |||
| 185dcc2960 | |||
| 0fe8e4106f | |||
| fd65a490dc | |||
| 3607517814 | |||
| a0e04a8588 | |||
| ffe8214cfe | |||
| 1f63f622c9 | |||
| f4701bf0f9 | |||
| 06cc184227 | |||
| 59a527f2f2 | |||
| d7941c88be | |||
| c64dc16319 | |||
| d564cee43d | |||
| 2cf23b6fe2 | |||
| 68b22adfa3 | |||
| 9289693730 | |||
| deff44bcea | |||
| 217d3a3a9b | |||
| bcf509a440 | |||
| 10f726f83d | |||
| acc262c405 | |||
| 62034378c6 | |||
| bcb8c5ab88 | |||
| fec5067fcd | |||
| b0ed67aa60 | |||
| c023272b16 | |||
| 3568a6db50 | |||
| 9bf8d5699b | |||
| c0ff00a1ff | |||
| 45010f5890 | |||
| 845df69031 | |||
| d47d528d9a | |||
| 50d0e6343f | |||
| 7053439e84 | |||
| cf05ffee7d | |||
| bed776a308 | |||
| dbf389783e | |||
| 75c2e6c364 | |||
| d152c504e1 | |||
| b65e5cae0e | |||
| 09c05733c6 | |||
| 5d1d34cd82 | |||
| e2dcd2bd6b | |||
| 0d6d7ebae1 | |||
| 9706fc5d9c | |||
| 2329cb8ad5 | |||
| 54ebb24374 | |||
| cc48144a35 | |||
| 8f347da653 | |||
| 77de11a97d | |||
| 587828c57e | |||
| b0a5fa6856 | |||
| 73e7972fb8 | |||
| 6424f73da4 | |||
| 9c1b76b632 | |||
| 2ba54266c6 | |||
| b9f95c357c | |||
| c8f0c0cf90 | |||
| 21efeece32 | |||
| 7f20b1bc84 | |||
| 212d1922e5 | |||
| df573b7314 | |||
| 3c7a3c1375 | |||
| 41e7d5b7d7 | |||
| ca23f2876c | |||
| a9898fdd6c | |||
| c71cc749d9 | |||
| 2fb80cb88f | |||
| 2dd0688d45 | |||
| 848f123985 | |||
| 76241ab703 | |||
| 409875e296 | |||
| 8e4f32c93a | |||
| 3e4c1931a1 | |||
| df7926215b | |||
| f583fb06db | |||
| 4bca60c56c | |||
| 80b8997b88 | |||
| bf9299de1a | |||
| fbfd170ca6 | |||
| 71c34839d9 | |||
| f923351953 | |||
| a7cab83dd1 | |||
| b28e8bac80 | |||
| 48f4c41442 | |||
| 7f50fbefad | |||
| b8addd55c0 | |||
| f585c47b7d | |||
| ee3a0297ea | |||
| c6e5794125 | |||
| 85b62860b2 | |||
| 74cf4e92aa | |||
| 6572b53c89 | |||
| 7f1329d3b0 | |||
| 5004858032 | |||
| de60127c45 | |||
| c7e0358aaf | |||
| bbadd00ac0 | |||
| 8dd356b7e6 | |||
| 77cb76c006 | |||
| ca7958329a | |||
| 65eaacb341 | |||
| 9837214414 | |||
| 2b0b1cf73e | |||
| 7263b31536 | |||
| 6b6c220986 | |||
| 65d1552ffa | |||
| 9f54a97cc6 |
@@ -0,0 +1,5 @@
|
||||
# Funding platforms for the GitHub "Sponsor" button.
|
||||
# https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/displaying-a-sponsor-button-in-your-repository
|
||||
|
||||
github: [eous]
|
||||
custom: ["https://paypal.me/eousphoros"]
|
||||
@@ -152,7 +152,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- run: uv lock --check
|
||||
@@ -161,7 +161,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
- uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
|
||||
- uses: astral-sh/setup-uv@d31148d669074a8d0a63714ba94f3201e7020bc3 # v8.3.0
|
||||
with:
|
||||
uv-version: "0.9.18"
|
||||
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
name: Claude Code Review
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, synchronize, ready_for_review, reopened]
|
||||
# Optional: Only run on specific file changes
|
||||
# paths:
|
||||
# - "src/**/*.ts"
|
||||
# - "src/**/*.tsx"
|
||||
# - "src/**/*.js"
|
||||
# - "src/**/*.jsx"
|
||||
|
||||
jobs:
|
||||
claude-review:
|
||||
if: github.event.pull_request.head.repo.full_name == github.repository
|
||||
# Optional: Filter by PR author
|
||||
# if: |
|
||||
# github.event.pull_request.user.login == 'external-contributor' ||
|
||||
# github.event.pull_request.user.login == 'new-developer' ||
|
||||
# github.event.pull_request.author_association == 'FIRST_TIME_CONTRIBUTOR'
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post the review + inline comments
|
||||
issues: read
|
||||
id-token: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
allowed_bots: 'renovate[bot]' # let Renovate PRs get reviewed
|
||||
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
|
||||
plugins: 'code-review@claude-code-plugins'
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ github.event.pull_request.number }}'
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
name: Claude Code
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
pull_request_review_comment:
|
||||
types: [created]
|
||||
issues:
|
||||
types: [opened, assigned]
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
jobs:
|
||||
claude:
|
||||
if: |
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
|
||||
) || (
|
||||
github.event_name == 'pull_request_review' &&
|
||||
contains(github.event.review.body, '@claude') &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.review.author_association)
|
||||
) || (
|
||||
github.event_name == 'issues' &&
|
||||
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
|
||||
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.issue.author_association)
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post comments/reviews when @-mentioned on a PR
|
||||
issues: write # post comments when @-mentioned on an issue
|
||||
id-token: write
|
||||
actions: read # Required for Claude to read CI results on PRs
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@f87768c6d25f92ae6efa7175e223ef77d4cbf97f # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
|
||||
# This is an optional setting that allows Claude to read CI results on PRs
|
||||
additional_permissions: |
|
||||
actions: read
|
||||
|
||||
# Optional: Give a custom prompt to Claude. If this is not specified, Claude will perform the instructions specified in the comment that tagged it.
|
||||
# prompt: 'Update the pull request description to include a summary of changes.'
|
||||
|
||||
# Optional: Add claude_args to customize behavior and configuration
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
# claude_args: '--allowed-tools Bash(gh pr *)'
|
||||
|
||||
@@ -7,7 +7,9 @@ on:
|
||||
|
||||
concurrency:
|
||||
group: docker-${{ github.event.workflow_run.head_sha }}
|
||||
cancel-in-progress: true
|
||||
# Never cancel mid-push: an interrupted multi-tag push can leave the
|
||||
# registry with a partial tag set (e.g. :latest moved, :stable not).
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -19,15 +21,24 @@ env:
|
||||
|
||||
jobs:
|
||||
docker:
|
||||
# Same gate as publish.yml: workflow_run fires for every CI completion
|
||||
# (including fork and same-repo PR runs) with this repo's token and
|
||||
# packages:write. Only same-repo tag pushes may publish images; CI's
|
||||
# push trigger matches main/stable/* and v* tags, so a head_branch
|
||||
# starting with "v" is necessarily a tag run.
|
||||
if: >-
|
||||
github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.head_repository.full_name == github.repository
|
||||
github.event.workflow_run.event == 'push' &&
|
||||
github.event.workflow_run.head_repository.full_name == github.repository &&
|
||||
startsWith(github.event.workflow_run.head_branch, 'v')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
# The docker build only reads the tree; keep the token out of it.
|
||||
persist-credentials: false
|
||||
|
||||
- name: Resolve release tag
|
||||
id: tag
|
||||
@@ -43,7 +54,7 @@ jobs:
|
||||
|
||||
- name: Log in to GHCR
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v4
|
||||
uses: docker/login-action@af1e73f918a031802d376d3c8bbc3fe56130a9b0 # v4
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
@@ -67,12 +78,12 @@ jobs:
|
||||
fi
|
||||
echo "tags=${TAGS}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4
|
||||
- uses: docker/setup-buildx-action@bb05f3f5519dd87d3ba754cc423b652a5edd6d2c # v4
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
|
||||
- name: Build and push
|
||||
if: steps.tag.outputs.skip == 'false'
|
||||
uses: docker/build-push-action@f9f3042f7e2789586610d6e8b85c8f03e5195baf # v7
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
|
||||
with:
|
||||
context: .
|
||||
push: true
|
||||
|
||||
@@ -7,7 +7,9 @@ on:
|
||||
|
||||
concurrency:
|
||||
group: publish-${{ github.event.workflow_run.head_sha }}
|
||||
cancel-in-progress: true
|
||||
# Never cancel a publish mid-upload: a half-uploaded release (sdist up,
|
||||
# wheel missing) cannot be re-run cleanly because PyPI rejects duplicates.
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
@@ -15,7 +17,16 @@ permissions:
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
if: github.event.workflow_run.conclusion == 'success'
|
||||
# workflow_run fires for EVERY CI completion — including CI runs for
|
||||
# pull_requests from forks — and always executes here with this repo's
|
||||
# secrets, tokens, and the pypi environment. Gate to same-repo tag
|
||||
# pushes only: CI's push trigger matches branches main/stable/* and
|
||||
# tags v*, so a head_branch starting with "v" is necessarily a tag run.
|
||||
if: >-
|
||||
github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.event == 'push' &&
|
||||
github.event.workflow_run.head_repository.full_name == github.repository &&
|
||||
startsWith(github.event.workflow_run.head_branch, 'v')
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
steps:
|
||||
@@ -23,6 +34,9 @@ jobs:
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
# python -m build executes the tree's build backend; don't leave
|
||||
# the contents:write token sitting in .git/config while it runs.
|
||||
persist-credentials: false
|
||||
|
||||
- name: Resolve release tag
|
||||
id: tag
|
||||
|
||||
@@ -25,18 +25,39 @@ permissions:
|
||||
|
||||
jobs:
|
||||
vendor-js:
|
||||
if: github.actor == 'renovate[bot]' || github.event_name == 'workflow_dispatch'
|
||||
# Same-repo PRs only: this job checks out the PR head and pushes to it
|
||||
# with contents:write, so it must never act on a fork's branch.
|
||||
# Gate on the PR author (immutable), not github.actor (names whoever
|
||||
# caused the latest event, which can be someone else re-running it).
|
||||
if: >-
|
||||
(github.event_name == 'pull_request' &&
|
||||
github.event.pull_request.user.login == 'renovate[bot]' &&
|
||||
github.event.pull_request.head.repo.full_name == github.repository) ||
|
||||
github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Resolve PR head ref
|
||||
id: ref
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Branch names may contain shell metacharacters; pass via env,
|
||||
# never interpolate ${{ }} into the script body.
|
||||
HEAD_REF: ${{ github.head_ref }}
|
||||
PR_NUMBER: ${{ inputs.pr_number }}
|
||||
run: |
|
||||
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
|
||||
ref=$(gh pr view "${{ inputs.pr_number }}" --repo "${{ github.repository }}" --json headRefName -q .headRefName)
|
||||
if [[ "$GITHUB_EVENT_NAME" == "workflow_dispatch" ]]; then
|
||||
# The dispatch input is an arbitrary PR number; refuse fork PRs.
|
||||
# A fork's headRefName is a bare branch name that may collide
|
||||
# with a branch in this repo, and checkout+push would then hit
|
||||
# that unrelated branch ("same-repo PRs only" applies here too).
|
||||
pr_json=$(gh pr view "$PR_NUMBER" --repo "$GITHUB_REPOSITORY" --json headRefName,isCrossRepository)
|
||||
if [[ "$(jq -r '.isCrossRepository' <<< "$pr_json")" != "false" ]]; then
|
||||
echo "::error::PR #${PR_NUMBER} head is not a branch in this repository; refusing to complete it."
|
||||
exit 1
|
||||
fi
|
||||
ref=$(jq -r '.headRefName' <<< "$pr_json")
|
||||
else
|
||||
ref="${{ github.head_ref }}"
|
||||
ref="$HEAD_REF"
|
||||
fi
|
||||
echo "head_ref=${ref}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
|
||||
@@ -28,3 +28,4 @@ tools/skill_audit_analysis/data/
|
||||
tools/skill_audit_analysis/output/
|
||||
design_ideas/
|
||||
.claude/
|
||||
docs/design/
|
||||
|
||||
+294
-4
@@ -6,13 +6,303 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||
and this project adheres to [PEP 440](https://peps.python.org/pep-0440/) for
|
||||
version numbers (`X.Y.Z`, with `X.Y.ZaN` / `bN` / `rcN` for pre-releases).
|
||||
|
||||
Three release tracks are maintained — the current stable, one prior
|
||||
stable, and the experimental line:
|
||||
Two active release tracks are maintained — the current stable and the
|
||||
experimental line:
|
||||
|
||||
- **`stable/1.5`** — patch-only (`v1.5.x`)
|
||||
- **`stable/1.6`** — patch-only (`v1.6.x`)
|
||||
- **`stable/1.7`** — patch-only (`v1.7.x`)
|
||||
- **`main`** — experimental (next major)
|
||||
|
||||
Earlier stable lines (`stable/1.6`, `stable/1.5`) are frozen.
|
||||
|
||||
## [1.7.2]
|
||||
|
||||
A feature-bearing patch for the 1.7 line. Rather than hold this work for the
|
||||
larger 1.8 churn, the fixes and the smaller features that had already
|
||||
stabilised on `main` are rolled into the stable line now: a rich preview
|
||||
pane, persona/project settings on scheduled tasks, and a batch of streaming,
|
||||
rendering, and nudge-delivery hardening.
|
||||
|
||||
> **⚠️ Before upgrading:** 1.7.2 adds Alembic migration `066`, applied
|
||||
> automatically on first start. It adds two `Text NOT NULL DEFAULT ''`
|
||||
> columns (`persona`, `project_id`) to the `scheduled_tasks` table; existing
|
||||
> rows migrate to the empty default, which is byte-identical to pre-066
|
||||
> dispatch behaviour. The change is additive and reversible, but — as always
|
||||
> — back up your storage before upgrading (`pg_dump` for PostgreSQL; copy the
|
||||
> database file for SQLite).
|
||||
|
||||
### Added
|
||||
|
||||
- **Rich preview pane + `open_preview` tool** — a workstream can now open a
|
||||
rendered preview (HTML, Markdown, and other kinds) in a pane beside the
|
||||
conversation via the new `open_preview` tool. Guarded fetches stream under
|
||||
a byte budget whose ceiling tracks the widest per-kind cap, preview blob
|
||||
ids are salted, and a preflight probe handles legacy charsets and a
|
||||
remote-assets opt-in. See `docs/tools.md`.
|
||||
- **`allow_private_network` opt-in for `web_fetch` / `open_preview`** —
|
||||
private-address fetch and preview targets stay blocked by default; an
|
||||
operator can opt a workstream in through the settings registry when a
|
||||
private endpoint is genuinely intended. (Distinct from the 1.7.1 `[oidc]`
|
||||
flag of the same name, which governs identity-provider discovery.)
|
||||
- **Persona + project settings on scheduled tasks** (migration `066`) — a
|
||||
scheduled task can now pin the **persona** and **project** of the
|
||||
workstream it dispatches, matching the levers a manually-created workstream
|
||||
already carries. Both default to empty (kind-default persona / no project),
|
||||
so existing schedules dispatch exactly as before.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Streaming fast-path overflow recovery** — fast-stream tokens are now
|
||||
batched and overflowed SSE listeners recover instead of stalling (and
|
||||
`connectSSE` no longer opens into a hidden background tab). The same
|
||||
overflow-recovery companions were carried to the coordinator pane, so a
|
||||
coordinator watching many children recovers dropped listeners the same way
|
||||
the live-session view does.
|
||||
- **Renderer containment** — markdown sentinel-forgery and recursive-frame
|
||||
content loss are contained, and an indented fence close no longer drags its
|
||||
indent into the enclosed code content.
|
||||
- **Idle nudge / wake delivery** — nudge and wake delivery is hardened across
|
||||
session eviction, cancellation, and identity rebinds; the wake gate now
|
||||
requires a real nudge queue, refused wakes are logged, and
|
||||
`initial_message_status` is typed as a closed enum on the wire.
|
||||
- **`web_fetch` extraction inherits model settings** — the completion that
|
||||
extracts content from a fetched page now inherits the workstream's model
|
||||
settings instead of falling back to defaults.
|
||||
- **UI panes** — ephemeral panes close on split-dismiss instead of orphaning
|
||||
a tab, and an unsplit skips the redundant refresh after an ephemeral pane
|
||||
closes.
|
||||
- **Shared code-highlight CSS** — renderer-output CSS is shared so the console
|
||||
and coordinator panes highlight code identically.
|
||||
|
||||
### Security
|
||||
|
||||
- **`Content-Disposition` filenames made wire-safe** — download filenames
|
||||
derived from user-controlled text are sanitised (latin-1- and
|
||||
control-char-safe, quoting-safe) before they reach the `Content-Disposition`
|
||||
response header, including the fallback path.
|
||||
|
||||
### Documentation
|
||||
|
||||
- **HYPOTHESIS.md: daemons + the outer loop, plus a plain-language PRIMER** —
|
||||
the harness north-star document gains its daemon / outer-loop treatment and
|
||||
a new top-level `PRIMER.md`.
|
||||
|
||||
## [1.7.1]
|
||||
|
||||
A maintenance and hardening patch for the 1.7 line. No schema migrations;
|
||||
the credential-redaction work below is additive and needs no configuration
|
||||
change. The one new operator-facing knob is the opt-in `[oidc]
|
||||
allow_private_network` flag (default off).
|
||||
|
||||
### Security
|
||||
|
||||
- **Credential redaction hardened across the tool-call surface** — the
|
||||
redactor that scrubs secrets from tool arguments and log previews was
|
||||
reworked on both the backend and the browser to close several leak paths
|
||||
and to fix false-positive and performance issues. Malformed tool-call
|
||||
arguments are now legalised before they reach the wire; the tool-args log
|
||||
preview scrubs credentials and control characters; and the coordinator's
|
||||
tool-call cards gain a matching client-side redaction pass so the JS and
|
||||
backend redactors stay at parity. Pattern coverage now includes
|
||||
`secret_access_key` / `aws_secret_access_key` multi-segment keys, bare
|
||||
`token=` / `key=` forms (guarded by a negative lookbehind to avoid
|
||||
false positives), and SQLAlchemy `+driver`-qualified connection-string
|
||||
schemes matched case-insensitively.
|
||||
- **OIDC SSRF guard: `[oidc] allow_private_network` opt-in** — self-hosted
|
||||
identity providers on private networks can now be reached by setting
|
||||
`allow_private_network = true` under `[oidc]` (default off; the MCP OAuth
|
||||
path stays strict). Rejections of discovered endpoints carry the opt-in
|
||||
hint so the misconfiguration is self-explanatory. See `docs/oidc.md`.
|
||||
|
||||
### Added
|
||||
|
||||
- **Persona discoverability + forgiving name resolution** — personas are
|
||||
now discoverable by agents, and persona-name resolution tolerates
|
||||
case/whitespace variation; a not-found resolution reports the offending
|
||||
input verbatim instead of a bare error.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **MCP transport lifecycles routed through per-entry owner tasks**
|
||||
(#787/#788) — static and pooled MCP transport lifecycles are now driven
|
||||
by per-server / per-entry owner tasks, with a hardened disarm-sweep loop
|
||||
guard and targeted exception handling in place of a broad `BaseException`
|
||||
arm, so a dying transport can no longer spin the CPU or strand delivery.
|
||||
- **Client-construction failures surface as misconfiguration, not raw
|
||||
500s** — a model whose client cannot be constructed now reports a factory
|
||||
misconfiguration, and the raw exception text is kept out of the resulting
|
||||
503 response.
|
||||
- **Postgres history search survives oversized rows** — a conversation row
|
||||
exceeding Postgres' full-text limits no longer aborts history search.
|
||||
- **Agent-tool render is idempotent** — tool rendering no longer deep-copies
|
||||
a tool definition until a description actually changes, so no-persona
|
||||
sessions share the tool constant (correctness plus a hot-path allocation
|
||||
win).
|
||||
- **Private-project workstream visibility scoped to members** — workstreams
|
||||
in a private project are visible to project members only, not to every
|
||||
admin; coordinator tenancy checks now use request-scoped storage.
|
||||
- **Pane hotkeys work off macOS and match across surfaces** — the pane
|
||||
keyboard shortcuts no longer collide with browser accelerators on
|
||||
non-macOS platforms and behave consistently across surfaces.
|
||||
|
||||
## [1.7.0]
|
||||
|
||||
The headline of the 1.7 line is **Personas** — operator-authored control
|
||||
over how each workstream composes its system message and capability
|
||||
envelope. The rest of the release hardens the pieces a persona leans on:
|
||||
concurrent approvals, cross-provider reasoning-effort control, cooperative
|
||||
compaction, multi-user session safety, and MCP resilience for unattended
|
||||
work.
|
||||
|
||||
> **⚠️ Before upgrading:** 1.7.0 adds Alembic migrations `062`–`065`,
|
||||
> applied automatically on first start (projects, personas, and two
|
||||
> smaller schema tidy-ups). Migration `063` creates the `personas` table
|
||||
> with its six seed personas and converts existing `creative_mode`
|
||||
> workstreams to the `writer` persona in place. The changes are additive
|
||||
> to your conversation data, but — as always — back up your storage before
|
||||
> upgrading (`pg_dump` for PostgreSQL; copy the database file for SQLite).
|
||||
|
||||
**Breaking changes at a glance** (details in the sections below): the
|
||||
`/creative` REPL toggle removed (replaced by the `writer` persona), the
|
||||
`turnstone-bootstrap` entry point renamed to `turnstone-doctor`, and the
|
||||
approval-status API/SDK field `pending_approval_details` changed from a
|
||||
single object to a list (one entry per concurrent approval cycle).
|
||||
|
||||
### Added
|
||||
|
||||
- **Personas** (#683) — a named, reusable bundle attached to a workstream
|
||||
at creation, controlling system-message composition and the capability
|
||||
envelope via exactly four levers: base-prompt override, tool visibility
|
||||
set, MCP on/off, and memory on/off. The persona is resolved once and
|
||||
snapshotted into `workstream_config`; editing or archiving a persona
|
||||
never changes an existing workstream. Six seed personas ship with
|
||||
migration `063` (`engineer` and `orchestrator` are the per-kind
|
||||
defaults with no overrides, so zero-touch behavior is unchanged;
|
||||
`scribe`, `researcher`, `writer`, and `executive` are curated
|
||||
envelopes). Selectable on every creation surface (web pickers, the
|
||||
create API/SDKs, coordinator `spawn_workstream` / `spawn_batch`, and
|
||||
`turnstone --persona <name>`); authored in the console's new
|
||||
Governance → Personas tab (`persona.{create,read,write}` perms,
|
||||
archive-only lifecycle). See `docs/personas.md`.
|
||||
- **Projects — governed resource containers** (#724) — group workstreams
|
||||
and their resources under a project (migration `062`), with
|
||||
project-scoped memory, a per-project resources view, a project column on
|
||||
the saved list, and server-enforced private-project workstream
|
||||
visibility.
|
||||
- **Task-agent sub-harness** (#732) — a spawned task agent now runs on its
|
||||
own Turn-IR sub-harness with parent-tagged step events: its sub-tool
|
||||
steps nest inside an expandable card in the parent trajectory, its
|
||||
sub-trajectory is recallable, and each agent gets read isolation from
|
||||
its siblings.
|
||||
- **MCP static-server autonomous reconnect** (#768) — statically
|
||||
configured MCP servers are now kept live by a health loop
|
||||
(capped-jittered backoff, ping-based liveness) instead of silently
|
||||
staying dead after the first transport drop.
|
||||
- **Attachments — capability-gated client-side fallback** — when the
|
||||
active model can't natively handle an attachment, the client degrades
|
||||
gracefully (PDF → extracted text, audio → transcript) instead of
|
||||
failing the turn.
|
||||
- **Eval measurement / optimizer split** (#763, #765) — `turnstone-eval`
|
||||
is now a measure-only substrate with the prompt optimizer factored out,
|
||||
plus a new skill-adherence measurement mode.
|
||||
- **Deployment examples** — a vLLM + LiteLLM unified-memory inference
|
||||
example showing a 3-model co-resident stack with an HF loader (#686,
|
||||
#688), and an Altair + `vl-convert-python` visualization stack (#685).
|
||||
- **Concurrent approvals and a long-session frontend overhaul** (#754,
|
||||
#755, #773, #775) — the live-session frontend was reworked for long
|
||||
runs (the pipeline is wedge-proofed and its hot paths de-O(N)'d), and on
|
||||
top of it a workstream can now hold more than one tool call awaiting
|
||||
approval at a time. Each parallel batch gets its own approval cycle,
|
||||
with one card per pending call in the interactive and coordinator UIs,
|
||||
cycle-keyed tracking in Slack and Discord, and cycle-routed resolution
|
||||
across the server/console/SDK APIs; sub-agent tool gates run the
|
||||
intent-judge pipeline as their own generation. The send button no longer
|
||||
sticks disabled after a batch resolves — orphaned approval cycles are
|
||||
pruned and the app is the sole owner of the button state.
|
||||
*(BREAKING: the `pending_approval_details` field is now a list, oldest
|
||||
first.)*
|
||||
- **Reasoning-effort control on every provider lane** (#771, #774) — the
|
||||
session effort knob now reaches local backends too: it drives
|
||||
`chat_template_kwargs` on the anthropic-compatible and openai-compatible
|
||||
lanes and threads through to Gemini and xAI, alongside the commercial
|
||||
providers that handle effort natively. The console surfaces each model's
|
||||
effective effort ladder in plain words and adds an always-on
|
||||
thinking-mode option to the model form. Effort snapping is ordinal —
|
||||
it rounds up and caps at the model's ceiling rather than silently
|
||||
dropping.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Skills are capability-context, not identity** (#762) — a task agent's
|
||||
identity now comes from its persona; an applied skill's body is demoted
|
||||
to capability context and moved out of the identity system message.
|
||||
Skill-body substitution is unified across every invocation context so
|
||||
the same skill renders identically whether loaded interactively, by the
|
||||
model, or inside a sub-agent.
|
||||
- **`turnstone-doctor` replaces `turnstone-bootstrap`** (#718)
|
||||
*(BREAKING)* — the setup/diagnostics entry point is renamed; update any
|
||||
scripts or service units that invoke `turnstone-bootstrap`.
|
||||
- **Honest cancellation dispositions** — cancelled or timed-out
|
||||
side-effecting tools now report an `UNKNOWN` disposition rather than a
|
||||
flat failure, tool dispositions are typed (not just prose), and a
|
||||
coordinator cancel propagates down the sub-tree.
|
||||
- **Multi-user shared-workstream context** (#750) — in a shared
|
||||
workstream, send is gated to the acting participant while a turn is in
|
||||
flight (both the interactive and coordinator surfaces), cross-user
|
||||
mid-turn interjections are blocked, and shared-workstream state plus
|
||||
fork sender attribution are now durable.
|
||||
- **Cooperative compaction** (#730) — the context budget is anchored to
|
||||
the provider's true capacity, the summary call is chunked so it can't
|
||||
overflow, and the active plan and the outstanding ask are carried across
|
||||
compaction verbatim. The `recall` tool is scoped to the compacted-away
|
||||
past.
|
||||
- **Intent judge sees the full tool arguments** (#760) — the judge's
|
||||
argument projection is no longer narrowed, so it stops issuing confident
|
||||
false denials on a partial view. The output-guard judge sources its real
|
||||
context window, and `context_window = 0` in `config.toml` now means
|
||||
auto-detect.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Compaction resume hardening** (#731) — checkpoint markers are
|
||||
persisted so resume rehydration is bounded, context-overflow on resume
|
||||
is recovered across providers, and a recognized rate-limit is no longer
|
||||
misclassified as context overflow.
|
||||
- **MCP unattended-work resilience** (#706, #742, #767) — dead-transport
|
||||
handling is completed, consented OAuth (OBO) tokens are refreshed
|
||||
proactively so autonomous runs don't strand on an expired grant, the
|
||||
Entra ID on-behalf-of impersonation flow blockers are closed (migration
|
||||
`065` adds the OIDC `oid`), and OAuth refresh failures are classified so
|
||||
a transient blip never revokes consent nor a dead grant strands the
|
||||
user.
|
||||
- **Memory writes** (#735) — save/update is a single atomic upsert, and
|
||||
writing a memory no longer recomposes the system prefix mid-session.
|
||||
|
||||
### Removed
|
||||
|
||||
- **`/creative` removed** *(BREAKING)* — subsumed by the Personas feature
|
||||
above: the REPL toggle (and its tab completion) is gone, and the
|
||||
`writer` seed persona replaces it — start a session with
|
||||
`turnstone --persona writer` or pick *Writer* in the web
|
||||
pickers. Unlike the old fork, the writer persona composes the full
|
||||
system message, so session context and mandatory prompt policies now
|
||||
apply to prose-only sessions too. The `creative_mode` key in
|
||||
`workstream_config` is no longer read or written. Migration `063`
|
||||
converts existing creative-mode workstreams to the `writer` persona
|
||||
automatically, so they resume as writing sessions rather than as
|
||||
legacy defaults.
|
||||
|
||||
### Security
|
||||
|
||||
- **High-risk skill activation is gated** (#762) — a model-initiated load
|
||||
of a `high`- or `critical`-risk skill is gated and fails closed when the
|
||||
backing storage is unavailable, so an untrusted turn can't silently
|
||||
pull in a dangerous capability.
|
||||
- **Dependency security floors** — `cryptography` and `starlette` are
|
||||
pinned to security-fixed minimums.
|
||||
- **CI publish hardening** — the vendored-JS dispatch path refuses fork
|
||||
PRs, and `workflow_run` publishing is gated to same-repo tag pushes, so
|
||||
a fork can't trigger a release build.
|
||||
|
||||
## [1.6.0]
|
||||
|
||||
The first stable release of the 1.6 line — and the first under Apache 2.0.
|
||||
|
||||
+1
-1
@@ -8,7 +8,7 @@ FROM python:3.14-slim
|
||||
LABEL org.opencontainers.image.title="turnstone" \
|
||||
org.opencontainers.image.description="Multi-node AI orchestration platform"
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.24 /uv /usr/local/bin/uv
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.27 /uv /usr/local/bin/uv
|
||||
|
||||
# Remove the slim image's man page exclusion so man-db has actual content
|
||||
RUN rm -f /etc/dpkg/dpkg.cfg.d/docker
|
||||
|
||||
+90
-30
@@ -10,7 +10,9 @@ Most descriptions of an agent framework are a feature list. This is an attempt a
|
||||
|
||||
*Informal.* A harness is a **stopped, deterministically-controlled Markov process on task-state, closed around a stopped autoregressive process on context-space, driven by a learned model kernel** — a deterministic controller in closed loop with a stochastic learned plant.
|
||||
|
||||
*Formal — the objects.* A harness is a tuple $\mathcal{H} = (\mathcal{S}, \mathcal{C}, \mathcal{Y}, \mathcal{A}, \mathcal{E}, \pi, M_W, \gamma, Q_E, \rho, H, H_{\mathrm{ok}}, B)$ over **standard Borel** spaces (concretely: the *controlled* state is standard Borel by construction — token sequences, finite config maps, bounded counters and ledgers, finite tuples of real vectors — and the model/environment coordinates are inherited as such whenever they serialize to a Polish space; the assumption fails only if a coordinate is itself a measure or an uncountable product, which this construction avoids): a deterministic lowering $\pi:\mathcal{S}\to\mathcal{C}$; a stochastic model-run kernel $M_W(c, dy)$ into a readout space $\mathcal{Y}$ (which includes the parse-failure $\bot$, so $M_W$ and $\gamma$ are total over it); a deterministic **authorization gate** $\gamma:\mathcal{S}\times\mathcal{Y}\to\mathcal{A}_{\bot}$ that parses/validates the model output into an authorized action in $\mathcal{A}$ or rejects it as $\bot$; a stochastic environment/tool kernel $Q_E:\mathcal{S}\times\mathcal{A}_{\bot}\rightsquigarrow\mathcal{E}$ on the authorized action (rejection included, with $Q_E(s,\bot,\cdot)=\delta_{e_0}$ for a distinguished no-op response $e_0\in\mathcal{E}$); and a deterministic verify-and-fold-back map $\rho:\mathcal{S}\times\mathcal{Y}\times\mathcal{A}_{\bot}\times\mathcal{E}\to\mathcal{S}$.
|
||||
*In plain terms.* The **harness** is the whole governed loop: a deterministic **shell** you write — build the prompt, authorize an action, fold the response back into state — wrapped around a black-box stochastic model kernel (the **plant**, $M_W$) and the environment its actions touch, looped until it halts in $H$. The shell is deterministic, $M_W$ is not, and everything below makes that split precise.
|
||||
|
||||
*Formal — the objects.* A harness is a tuple $\mathcal{H} = (\mathcal{S}, \mathcal{C}, \mathcal{Y}, \mathcal{A}, \mathcal{E}, \pi, M_W, \gamma, Q_E, \rho, H, H_{\mathrm{ok}}, B)$ over **standard Borel** spaces (concretely: the *controlled* state is standard Borel by construction — token sequences, finite config maps, bounded counters and ledgers, finite tuples of real vectors — and the model/environment coordinates are inherited as such whenever they serialize to a Polish space; the assumption is roomier than it looks — even a belief-state coordinate valued in $\mathcal{P}(X)$ survives, since $\mathcal{P}(X)$ is Polish for Polish $X$ — and fails only for a genuinely non-separable coordinate, an uncountable product $\sigma$-algebra being the canonical hazard, which this construction avoids): a deterministic lowering $\pi:\mathcal{S}\to\mathcal{C}$; a stochastic model-run kernel $M_W(c, dy)$ into a readout space $\mathcal{Y}$ (which includes the parse-failure $\bot$, so $M_W$ and $\gamma$ are total over it); a deterministic **authorization gate** $\gamma:\mathcal{S}\times\mathcal{Y}\to\mathcal{A}_{\bot}$ that validates the model's parsed readout into an authorized action in $\mathcal{A}$ or rejects it as $\bot$ (parsing itself lives inside $M_W$ — realized as the readout $R$ of the specialization below); a stochastic environment/tool kernel $Q_E:\mathcal{S}\times\mathcal{A}_{\bot}\rightsquigarrow\mathcal{E}$ on the authorized action (rejection included, with $Q_E(s,\bot,\cdot)=\delta_{e_0}$ for a distinguished no-op response $e_0\in\mathcal{E}$); and a deterministic verify-and-fold-back map $\rho:\mathcal{S}\times\mathcal{Y}\times\mathcal{A}_{\bot}\times\mathcal{E}\to\mathcal{S}$.
|
||||
|
||||
*Terminal structure.* The terminal set is an absorbing halt set $H\subseteq\mathcal{S}$ (the daemon "ready-state" recurrence of the note below is a separate, non-absorbing object) with accepting subset $H_{\mathrm{ok}}\subseteq H$; separately, a bad set $B\subseteq\mathcal{S}$ ($B\cap H_{\mathrm{ok}}=\varnothing$) marks the unsafe states for reach-avoid, possibly entered before any halt; hitting times are $\tau_A=\inf\{n\ge 0:s_n\in A\}$, and $\tau_H$ is a stopping time for the natural filtration.
|
||||
|
||||
@@ -20,17 +22,17 @@ $$T(s, A) = \int_{\mathcal{Y}}\!\int_{\mathcal{E}} \mathbf{1}_A\!\big(\rho(s, y,
|
||||
|
||||
and the harness runs $s_{n+1} \sim T(s_n)$ from an initial $s_0 \sim \mu_0$ until $\tau_H = \inf\{n : s_n \in H\}$. Because $\pi, \gamma, \rho, H$ are deterministic they contribute no integration variable of their own — they appear as measurable transformations inside the integrand (the pushforward), not literally outside it — so the controller injects no randomness, and every coin is inherited from $M_W$ and $Q_E$. (The earlier shorthand $T = \rho \circ (M_W \circ \pi, E)$ is suggestive but ill-typed — $M_W$ returns a *law*, while $\rho$ consumes a *sample* together with the prior state $s$; the integral is what the shorthand meant.)
|
||||
|
||||
*Fail-closed.* The gate $\gamma$ is what makes **fail-closed** a property, not just a name: model output is an *untrusted proposal*, and $\gamma(s,y)=\bot$ forces a no-op environment response ($Q_E(s,\bot,\cdot)=\delta_{e_0}$) — so a malformed or unauthorized tool call is rejected *before* it can act, not validated after its side effects have landed. Fail-closed is then the property that a rejected proposal causes *no unauthorized side effect* and lands in a **safe, non-bad** set ($\rho(s,y,\bot,e_0)\notin B$): a non-accepting terminal $H\setminus H_{\mathrm{ok}}$ in the strict case, or a safe non-terminal state when the spec retries. And $\rho$ must validate the tool *response* $e$, not only the proposal that $\gamma$ already gated: a malformed or adversarial $Q_E$ output is caught at fold-back, not just at the gate. But response-validation has a hard limit: $\rho$ can reject a bad tool *response*, yet it cannot undo side effects an *authorized* action already caused — so $\gamma$, not $\rho$, is the last line before irreversible effects, and anything irreversible must be gated at authorization. The boundary is also only real if raw model output reaches *no* sink — tool, logger, browser, or remote call — before $\gamma$; any pre-authorization escape bypasses the gate. The user-visible final response and any logging are themselves effects: either an authorized action through $\gamma$, or emitted only after an accepted halt in $H_{\mathrm{ok}}$.
|
||||
*Fail-closed.* The gate $\gamma$ is what makes **fail-closed** a property, not just a name: model output is an *untrusted proposal*, and $\gamma(s,y)=\bot$ forces a no-op environment response ($Q_E(s,\bot,\cdot)=\delta_{e_0}$) — so a malformed or unauthorized tool call is rejected *before* it can act, not validated after its side effects have landed. Fail-closed is then the property that a rejected proposal causes *no unauthorized side effect* and lands in a **safe, non-bad** set ($\rho(s,y,\bot,e_0)\notin B$): a non-accepting terminal $H\setminus H_{\mathrm{ok}}$ in the strict case, or a safe non-terminal state when the spec retries. And $\rho$ must validate the tool *response* $e$, not only the proposal that $\gamma$ already gated: a malformed or adversarial response $e$ is caught at fold-back, not just at the gate. But response-validation has a hard limit: $\rho$ can reject a bad tool *response*, yet it cannot undo side effects an *authorized* action already caused — so $\gamma$, not $\rho$, is the last line before irreversible effects, and anything irreversible must be gated at authorization. The boundary is also only real if raw model output reaches *no* sink — tool, logger, browser, or remote call — before $\gamma$; any pre-authorization escape bypasses the gate. The user-visible final response and any logging are themselves effects, and the rule binds *model-authored* bytes: they reach a sink either as an authorized action through $\gamma$, or only after an accepted halt in $H_{\mathrm{ok}}$. Shell-*templated* text — a refusal notice, a cancellation report reading the ledger — is controller output, outside $\gamma$'s jurisdiction, and may accompany any halt (a template that *interpolates* model-authored fragments inherits the model's label — the appendix's meet rule — and those bytes are gated like any others); the invariant is that raw model text never reaches a sink ungated, not that failed runs die silent.
|
||||
|
||||
*The harness invariants.* These are the invariants that make $\mathcal{H}$ a *harness* and not merely a controlled Markov process with a learned kernel inside: the model sees only $\mathcal{C}$, never full $\mathcal{S}$; its outputs are proposals, not actions; a deterministic capability boundary $\gamma$ gates every side effect; and the *terminal* set $H$ splits into accepting ($H_{\mathrm{ok}}$) and non-accepting ($H\setminus H_{\mathrm{ok}}$ — safe refusals outside $B$, and wrong or bad halts possibly in $B$), while the bad set $B$ is a *separate* unsafe set — possibly absorbing, possibly entered mid-run before any halt — against which $\tau_B$ is measured for reach-avoid.
|
||||
*The harness invariants.* These are the invariants that make $\mathcal{H}$ a *harness* and not merely a controlled Markov process with a learned kernel inside: the model sees only $\mathcal{C}$, never full $\mathcal{S}$; its outputs are proposals, not actions; a deterministic capability boundary $\gamma$ gates every side effect; and the *terminal* set $H$ splits into accepting ($H_{\mathrm{ok}}$) and non-accepting ($H\setminus H_{\mathrm{ok}}$ — safe refusals outside $B$, and wrong or bad halts possibly in $B$), while the bad set $B$ is a *separate* unsafe set — possibly absorbing, possibly entered mid-run before any halt — against which $\tau_B$ is measured for reach-avoid. Two notes keep the invariants honest. They are *signature*, not strength: a $\gamma$ that authorizes everything still satisfies the tuple, as a trivial group satisfies the group axioms — the definition admits degenerate harnesses, and fail-closed, provenance isolation, and the certificates below are properties a particular harness *earns*, not gifts of the signature. And the first invariant has a sharper, two-sided form: $\pi$ is the *only* channel from state to model — the confidentiality floor lives at what $\pi$ must never lower (credentials, other principals' data) — exactly as $\gamma$ is the only channel from model output to effect, where the injection bounds live; exfiltration is therefore cut at either chokepoint, never lowered or never emitted (the gate refusing the read whose URL is the payload is the emission-side cut). One chokepoint out of the state, one into the world; a bypass of either is the same bug with the sign flipped.
|
||||
|
||||
*Beyond the stationary kernel.* This displayed $T$ is the time-homogeneous, fixed-kernel case; for nonstationary or adversarial environments, replace $Q_E$ with a time-indexed kernel $Q_{E,n}$ — or an admissible family of kernels, or an adversary's policy — over which the robust certificate (the minimax form under *The limit*) quantifies. If that adversary conditions on history rather than only the current $(s, y)$, the history must itself live in $s$ — otherwise the object is a Markov *game* requiring further augmentation, not a Markov chain.
|
||||
*Beyond the stationary kernel.* This displayed $T$ is the time-homogeneous, fixed-kernel case; for nonstationary or adversarial environments, replace $Q_E$ with a time-indexed kernel $Q_{E,n}$ — or an admissible family of kernels, or an adversary's policy — over which the robust certificate (the minimax form under *The limit*) quantifies. If that adversary conditions on history rather than only the current $(s, y)$, the history must itself live in $s$ — otherwise the object is a Markov *game* requiring further augmentation, not a Markov chain. And nonstationarity is not the environment's monopoly: a provider retraining or re-serving under a fixed endpoint name is a nonstationary $M_{W,n}$ — the table places model version *in* $s$ precisely so a version bump is a visible state change — and any measured surrogate (the $\delta$ of *The limit*) is calibrated against one kernel and dies with the bump; the dashboard must be keyed to the kernel it measured.
|
||||
|
||||
*The inner kernel.* $M_W$ is itself a stopped process, and for a decoder-only transformer it is implemented as
|
||||
|
||||
$$M_W(c, \cdot) = \mathrm{Law}\big(R(z_{\tau})\big), \quad z_t = (c_t, b_t, m_t), \quad v \sim K_W(c_t, \cdot), \quad K_W(c, v) = (U \circ \Phi_W \circ \mathrm{Emb})(c)[v], \quad c_{t+1} = \mathrm{suffix}_{\le L}(c_t\!\cdot\! v),\ \ b_{t+1} = b_t\!\cdot\! v,\ \ m_{t+1} = \mathsf{step}(m_t, v),\ \ \tau=\inf\{t:m_t\in\mathrm{Stop}\}.$$
|
||||
|
||||
with the layer stack $\Phi_W$ on the residual stream as the (loosely) "manifold" core — formally just the learned high-dimensional residual-stream transformation, with manifold-proper reserved for the frontier. The inner state $z_t=(c_t,b_t,m_t)$ separates the model-visible window $c_t$ (the $\le L$ slice that slides) from the untruncated output buffer $b_t$ (the transcript the readout actually consumes, so truncation never loses it) and the parser/stop state $m_t$ (parser state, a token counter, and a clock, so the cap and timeout are functions of it), updated $m_{t+1}=\mathsf{step}(m_t,v)$, whose stop set $\mathrm{Stop}$ — EOS emitted, max-token cap, timeout, or parse-failure $\bot$ — forces $\tau=\inf\{t:m_t\in\mathrm{Stop}\}$ finite, making $M_W$ a genuine *probability* kernel rather than a sub-probability one completed by a cemetery output. The readout is total, $R : \mathcal{Z} \to \mathcal{Y}$ — a parsed tool-call, answer, or transcript, returning the parse-failure $\bot\in\mathcal{Y}$ when parsing fails; crucially $R$ is a *syntactic, verified* readout (parsing and extraction), not a semantic solver, or the $L$-wall below is void — arbitrary computation could hide in $R$ off the $\le L$ window — so $M_W(c, \cdot) = R_{\sharp}\,\mathrm{Law}(z_{\tau})$, the pushforward of the stopped-state law along $R$ (equivalently $M_W(c, A_Y) = \Pr[R(z_{\tau}) \in A_Y \mid z_0 = (c,\varnothing,m_0)]$ for a measurable $A_Y\subseteq\mathcal{Y}$); the no-truncation special case takes $\mathcal{Y}=\mathcal{C}$ with $R(c,b,m)=c$ (the window is the whole transcript), reading $c_{\tau}$ directly. The $\bot$ branch is exactly what $\gamma$ rejects fail-closed. This is a **specialization, not part of the definition**: a harness wrapped around a black-box API is still a harness, and $M_W$ may be any learned kernel. Where the weights are open, the geometry of $\Phi_W$ is where the substrate's continuity lives, and several downstream claims lean on it — but the definition does not.
|
||||
with the layer stack $\Phi_W$ on the residual stream as the (loosely) "manifold" core — formally just the learned high-dimensional residual-stream transformation, with manifold-proper reserved for the frontier. The inner state $z_t=(c_t,b_t,m_t)$ separates the model-visible window $c_t$ (the $\le L$ slice that slides) from the untruncated output buffer $b_t$ (the transcript the readout actually consumes, so truncation never loses it) and the parser/stop state $m_t$ (parser state, a token counter, and a clock, so the cap and timeout are functions of it), updated $m_{t+1}=\mathsf{step}(m_t,v)$, whose stop set $\mathrm{Stop}$ — EOS emitted, max-token cap, timeout, or parse-failure $\bot$ — forces $\tau=\inf\{t:m_t\in\mathrm{Stop}\}$ finite, making $M_W$ a genuine *probability* kernel rather than a sub-probability one completed by a cemetery output. (One honesty note on the clock: a token-count cap is a deterministic function of the run, but a *wall-clock* timeout imports infrastructure noise — server load, batching, congestion — into the kernel's coin; legitimate, a kernel may carry any randomness, but it makes the displayed $M_W$ the model *plus its serving substrate*, and the determinism audit under *How this could be wrong* must hold the clock fixed along with the samples.) The readout is total, $R : \mathcal{Z} \to \mathcal{Y}$ — a parsed tool-call, answer, or transcript, returning the parse-failure $\bot\in\mathcal{Y}$ when parsing fails; crucially $R$ is a *syntactic, verified* readout (parsing and extraction), not a semantic solver, or the $L$-wall below is void — arbitrary computation could hide in $R$ off the $\le L$ window — so $M_W(c, \cdot) = R_{\sharp}\,\mathrm{Law}(z_{\tau})$, the pushforward of the stopped-state law along $R$ (equivalently $M_W(c, A_Y) = \Pr[R(z_{\tau}) \in A_Y \mid z_0 = (c,\varnothing,m_0)]$ for a measurable $A_Y\subseteq\mathcal{Y}$); the no-truncation special case takes $\mathcal{Y}=\mathcal{C}$ with $R(c,b,m)=c$ (the window is the whole transcript), reading $c_{\tau}$ directly. The $\bot$ branch is exactly what $\gamma$ rejects fail-closed. This is a **specialization, not part of the definition**: a harness wrapped around a black-box API is still a harness, and $M_W$ may be any learned kernel. Where the weights are open, the geometry of $\Phi_W$ is where the substrate's continuity lives, and several downstream claims lean on it — but the definition does not.
|
||||
|
||||
Two stopped processes, nested: **deterministic control over stochastic dynamics over a learned kernel.** Both loops are hitting-time processes; *some* harnesses additionally read the halt set as a fixpoint or acceptance condition — iterative refinement to self-consistency is the genuine fixpoint case, while EOS, length, and tool-call syntax are not convergence. Neither loop settles because you asked it to. (The clean inner-then-outer nesting assumes tool calls fall *between* model runs; streaming or mid-generation tool calls interleave the two loops and need a finer state machine — the nesting is then an idealization.)
|
||||
|
||||
@@ -48,7 +50,7 @@ Two stopped processes, nested: **deterministic control over stochastic dynamics
|
||||
| $H,\ \tau_H$ | the **halt set** (absorbing) and the outer **halting time** — a hitting-time process, not a single pass |
|
||||
| $H_{\mathrm{ok}},\ B$ | the **accepting halts** $H_{\mathrm{ok}}\subseteq H$ (correct, successful terminals) and the **bad set** $B$ — unsafe states for reach-avoid ($B\cap H_{\mathrm{ok}}=\varnothing$), *separate* from $H$ and possibly entered mid-run before any halt |
|
||||
|
||||
The structural fact that earns the word *controller*: $\pi$, $\gamma$, $\rho$, and the halt test are **deterministic** (and the readout $R$ too, where the transformer specialization is in play), so $\mathcal{H}$ injects no randomness of its own. Every coin is inherited from $M_W$ and $Q_E$. This split — deterministic code around a stochastic oracle — wears two names. In control-theory terms it is **controller vs. plant**: the controller is those deterministic maps; the **plant** is the learned kernel $M_W$, *plant* in its exact sense — the element with its own dynamics you steer but do not author. In engineering terms it is **shell vs. plant**: the **shell** is the entire deterministic outer harness — the control logic *plus* the external memory and tools it administers (the files, databases, vector stores below) — of which the controller is just the control-logic slice. So *shell : plant :: the part you write : the part you don't*; $M_W$ is the only thing on the right, while the environment $Q_E$ is the world the actions meet — a disturbance into the loop, not the plant. This determinism is *conditional* — on versioned code, configuration, model endpoint, and tool interfaces; any retry, timeout, race, or randomized routing that escapes that conditioning must be modeled explicitly as part of $Q_E$ or the controller, not waved away. The displayed $M_W(c)$ likewise freezes endpoint, version, and sampler; a routing or config change is a state-indexed kernel $M_{\kappa(s)}$ or folds into $K_C$ — the kernel must not silently depend on config the table places in $s$. More generally, control may itself be stochastic — a controller kernel $K_C(s, dc)$ over routing, sampled retries, ensemble votes, learned routers — of which the deterministic $\pi, \gamma, \rho, H$ are the Dirac special case. That case is the one worth wanting: it localizes every coin to $M_W$ and $Q_E$ and keeps the controller/plant split clean. Where control is genuinely stochastic the split does not break, it widens — fold $K_C$ into the kernel and the certificate quantifies over its randomness too.
|
||||
The structural fact that earns the word *controller*: $\pi$, $\gamma$, $\rho$, and the halt test are **deterministic** (and the readout $R$ too, where the transformer specialization is in play), so $\mathcal{H}$ injects no randomness of its own. Every coin is inherited from $M_W$ and $Q_E$. This split — deterministic code around a stochastic oracle — wears two names. In control-theory terms it is **controller vs. plant**: the controller is those deterministic maps; the **plant** is the learned kernel $M_W$, *plant* in its exact sense — the element with its own dynamics you steer but do not author. In engineering terms it is **shell vs. plant**: the **shell** is the entire deterministic outer harness — the control logic *plus* the external memory and tools it administers (the files, databases, vector stores below) — of which the controller is just the control-logic slice. So *shell : plant :: the part you write : the part you don't*; $M_W$ is the only thing on the right, while the environment $Q_E$ is the world the actions meet — a disturbance into the loop, not the plant. (A reader from reinforcement learning or classical control will make the opposite assignment — environment as plant, policy as controller; the inversion is deliberate: in harness engineering the element you are trying to make behave is the model, and the world is what pushes back on the attempt.) This determinism is *conditional* — on versioned code, configuration, model endpoint, and tool interfaces, and on *single-run sequencing*: concurrent runs sharing authorization state re-open a gap the per-run object cannot see (taken up under *Gate placement* in the appendix); any retry, timeout, race, or randomized routing that escapes that conditioning must be modeled explicitly as part of $Q_E$ or the controller, not waved away. The displayed $M_W(c)$ likewise freezes endpoint, version, and sampler; a routing or config change is a state-indexed kernel $M_{\kappa(s)}$ or folds into $K_C$ — the kernel must not silently depend on config the table places in $s$. More generally, control may itself be stochastic — a controller kernel $K_C(s, dc)$ over routing, sampled retries, ensemble votes, learned routers — of which the deterministic $\pi, \gamma, \rho, H$ are the Dirac special case. That case is the one worth wanting: it localizes every coin to $M_W$ and $Q_E$ and keeps the controller/plant split clean. Where control is genuinely stochastic the split does not break, it widens — fold $K_C$ into the kernel and the certificate quantifies over its randomness too. But the guarantees do not soften uniformly, and the component-to-guarantee map is worth stating because it says exactly what may be learned without loss. A learned $\pi$ — retrieval, reranking, summarization inside the lowering — costs only *semantic adequacy*, under one factorization: $\pi$ splits into a deterministic **never-lower filter** — the redaction that keeps credentials and other principals' data out of $\mathcal{C}$ — composed with learned selection, and only the selection may soften, or the confidentiality floor of the invariants note becomes a probability. With the filter Dirac, no-unauthorized-effect is $\gamma$'s property alone, and the reach-avoid certificate survives too, so long as the provenance partition of *The limit* holds. A learned $\gamma$ or $\rho$ costs the thing itself — authorization and ledger integrity are exactly the properties that must stay Dirac, or "no unauthorized effect" and "the ledger is what happened" become probabilities. So the minimal deterministic core is $\{\gamma, \rho, H\}$ plus $\pi$'s never-lower filter: the rest of $\pi$ may soften into a kernel and the harness bends without breaking — fortunate, because every deployed $\pi$ already has learned kernels inside it.
|
||||
|
||||
## Why this shape
|
||||
|
||||
@@ -72,58 +74,86 @@ finite wherever $H$ is reached in finite expected time — the domain $\{s : \ma
|
||||
|
||||
So you never compute $V^\star$. You pick a candidate $\hat V$ and **estimate its drift slack**
|
||||
|
||||
$$\delta = \sup_{s \notin H}\Big(\mathbb{E}[\,\hat V(s_{n+1}) \mid s_n\,] - \hat V(s_n) + \varepsilon\Big).$$
|
||||
$$\delta = \sup_{s \notin H}\Big(\mathbb{E}[\,\hat V(s_{1}) \mid s_0 = s\,] - \hat V(s) + \varepsilon\Big).$$
|
||||
|
||||
The status of $\delta$ has to be stated carefully, because it is easy to oversell. If you can establish a *high-confidence upper bound* on the true worst-case slack and it is $\le 0$, optional stopping hands you a real, conservative certificate, $\mathbb{E}[\tau_H] \le \hat V(s_0)/\varepsilon$. But an *empirical* $\delta$ estimated from sampled states is **not** a certificate: a measured $\delta > 0$ may mean the candidate $\hat V$ is poor, the sampled distribution missed rare failures, the supremum was never attained in-sample, the process is non-stationary, or the state abstraction is not Markov. So $\delta$ is **the number on the dashboard** — a *calibrated risk metric*, the evaluable surrogate for a guarantee the geometry will not give you, and a genuine bound only once it is statistically controlled against rare-event and adversarial tests. A weaker result is still useful: a true bound $\delta \le \bar\delta < \varepsilon$ (rather than $\le 0$) leaves descent intact with effective slack $\varepsilon - \bar\delta$ and $\mathbb{E}_s[\tau_H] \le \hat V(s)/(\varepsilon - \bar\delta)$. And the empirical quantity is distributional, not a supremum — write $\delta_{\nu}$ for drift averaged over a sampled $\nu$, reserving $\delta_{\sup}$ for the worst-case bound; only $\delta_{\sup}$ certifies. Its empirical noise floor and residual risk are driven by the measure $\mu(D)$ of the divergent region $D=\{s:\mathbb{E}_s[\tau_H]=\infty\}$ (states from which $H$ is not reached in finite expected time, under the reference/sampling measure $\mu$), the coverage of the sampled state distribution, and the hitting-time variance $\mathrm{Var}[\tau_H]$ — properties of the trained weights, the environment, and the evaluation distribution, knowable only a posteriori.
|
||||
|
||||
> For an agent *meant* to run forever — a coordinator, a daemon — halting is the wrong target, and $V^\star = \infty$ is the spec, not a pathology. The same drift theory then certifies **recurrence to a ready-state** instead of absorption to a halt-set. The object changes; the missing certificate does not. Safety changes shape too: it is no longer the one-shot $\Pr_s(\tau_B=\infty)$ but a *per-cycle* hazard that compounds — if each ready-state-to-ready-state cycle touches $B$ with probability $q$, survival over $h$ cycles is $\approx (1-q)^h$, so a reassuring per-cycle $0.9999$ is $\approx 0.37$ over ten thousand cycles. The reach-avoid certificate for a daemon is therefore a bound on $q$ against the intended horizon — the safety twin of the regenerative expected time that replaces $V^\star_{\mathrm{ok}}$ for restarting specs.
|
||||
> For an agent *meant* to run forever — a coordinator, a daemon — halting is the wrong target, and $V^\star = \infty$ is the spec, not a pathology. The same drift theory then certifies **recurrence to a ready-state** instead of absorption to a halt-set. The object changes; the missing certificate does not. Safety changes shape too: it is no longer the one-shot $\Pr_s(\tau_B=\infty)$ but a *per-cycle* hazard that compounds — if each ready-state-to-ready-state cycle touches $B$ with probability $q$, survival over $N$ cycles is $\approx (1-q)^N$, so a reassuring per-cycle $0.9999$ is $\approx 0.37$ over ten thousand cycles. The reach-avoid certificate for a daemon is therefore a bound on $q$ against the intended horizon — the safety twin of the regenerative expected time that replaces $V^\star_{\mathrm{ok}}$ for restarting specs.
|
||||
|
||||
And the consolation rests in part on an assumption the world violates — though less of it than it first seems. The supermartingale *bound* itself survives a nonstationary kernel, provided the conditional drift holds uniformly at every step; what genuinely needs a **time-homogeneous kernel** is $V^\star$ as a fixed function, the resolvent / fundamental-matrix identities, and the sampled-$\delta$ calibration (which assumes the very kernel it was measured on). But the environment $E$ is *part of* $T$, and the world is not stationary — worse, it can be **adversarial**, an attacker choosing the tool-output *policy* — a kernel over what tools return, not the realized draw — so as to break your descent. The drift condition then stops being a fixpoint question and becomes a **minimax** one,
|
||||
|
||||
$$\sup_{\alpha \in \Pi}\ \int_{\mathcal{Y}}\!\int_{\mathcal{E}} V\big(\rho(s, y, \gamma(s,y), e)\big)\, Q_E^{\alpha(s,y)}\big(s, \gamma(s,y), de\big)\; M_W(\pi(s), dy) \;\le\; V(s) - \varepsilon,$$
|
||||
|
||||
a descent that must hold in expectation over the model's own output $y$ *and* even when the adversary picks the worst admissible environment policy $\alpha(s,y)$ from the class $\Pi$ of policies the environment genuinely permits — every $\alpha\in\Pi$ must still respect rejection, $\gamma(s,y)=\bot \Rightarrow Q_E^{\alpha}(s,\bot,\cdot)=\delta_{e_0}$, or the adversary resurrects side effects the gate refused. Well-posedness is a frontier caveat of its own: for $\sup_{\alpha\in\Pi}$ to be *attained* rather than merely defined, $\Pi$ needs structure — measurability of $\alpha\mapsto Q_E^{\alpha}$, compactness of the per-state admissible set, or a measurable-selection theorem furnishing a worst-case $\alpha$ — and "respects rejection" is a *constraint* on $\Pi$, not that existence argument; on a general state space the sup may have no maximizer, in which case the certificate quantifies over a maximizing sequence rather than a single adversary. A $V$ that certifies halting against a benign world is defeated by an adversarial one, and the measured $\delta$ bounds only the $Q_E$ you *sampled*, never the policy an attacker will choose. **This is the formal home of prompt injection** — not "the model did something bad," but the environment optimized to bend your dynamics. And the target is not merely non-halting: injection steers toward a **bad set** $B$ — wrong acceptance, data exfiltration, unauthorized tool use, privilege escalation, irreversible side effects — so security is a **reach-avoid** problem, not a liveness one. Here two reliability objects must be kept apart, because under absorbing refusal the naive forms collapse. **Success** is reaching a correct halt before *any* failure, $p_{\mathrm{succ}}(s) = \Pr_s(\tau_{H_{\mathrm{ok}}} < \tau_F)$ with $F = B \cup (H \setminus H_{\mathrm{ok}})$ — a safe refusal counts *against* it. **Safety** is never entering the bad set at all, $p_{\mathrm{safe}}(s) = \Pr_s(\tau_B = \infty)$ — a safe refusal *satisfies* it. These genuinely differ on any run that avoids $B$ without reaching $H_{\mathrm{ok}}$ ($p_{\mathrm{succ}}$ scores $0$, $p_{\mathrm{safe}}$ scores $1$): safe refusals, and — absent almost-sure absorption into $H\cup B$ — safe non-halting or endless safe retry. The tempting middle form $\Pr_s(\tau_{H_{\mathrm{ok}}} < \tau_B)$ is *not* a third object: with $H\setminus H_{\mathrm{ok}}$ absorbing, reaching $H_{\mathrm{ok}}$ before $B$ already requires reaching it before any refusal, so it coincides with $p_{\mathrm{succ}}$ — but only under that absorbing-refusal assumption; once the spec retries (the non-terminal fail-closed of the definition), a run may refuse, restart, and still reach $H_{\mathrm{ok}}$ before $B$, and the middle form re-separates as a genuine third object. Safety is certified by a barrier / avoidance certificate for $B$; success needs that plus the reach part — a hitting-time drift toward $H_{\mathrm{ok}}$. Fail-closed control is the disturbance-rejection margin for both, but split by reversibility: the gate $\gamma$ caps how far an adversarial world reaches into *side effects* and widens the gap to $B$ (it is the margin for the irreversible part), while $\rho$ validates the response and folds back, rejecting bad state after the action has run — which cannot undo an authorized side effect. In this language, security is robustness of the reach-avoid certificate. And injection is not confined to the post-model kernel $Q_E$: poisoned retrieval, prompt-injected pages, and malicious tool metadata enter through $\pi$'s *inputs*, before generation — so the adversary lives wherever untrusted content enters the state/context-construction pipeline, which is why input provenance and the gate $\gamma$ both matter, not post-hoc verification alone. (For $B$ to capture irreversible side effects rather than only states, the side-effect ledger must itself live in $\mathcal{S}$, and the response $e$ must be an *effect record* carrying the ledger outcome — not just API bytes — since only $\rho$ writes external effects into $s$.)
|
||||
a descent that must hold in expectation over the model's own output $y$ *and* even when the adversary picks the worst admissible environment policy $\alpha(s,y)$ from the class $\Pi$ of policies the environment genuinely permits — every $\alpha\in\Pi$ must still respect rejection, $\gamma(s,y)=\bot \Rightarrow Q_E^{\alpha}(s,\bot,\cdot)=\delta_{e_0}$, or the adversary resurrects side effects the gate refused. Well-posedness is a frontier caveat of its own: for $\sup_{\alpha\in\Pi}$ to be *attained* rather than merely defined, $\Pi$ needs structure — measurability of $\alpha\mapsto Q_E^{\alpha}$, compactness of the per-state admissible set, or a measurable-selection theorem furnishing a worst-case $\alpha$ — and "respects rejection" is a *constraint* on $\Pi$, not that existence argument; on a general state space the sup may have no maximizer, in which case the certificate quantifies over a maximizing sequence rather than a single adversary. A $V$ that certifies halting against a benign world is defeated by an adversarial one, and the measured $\delta$ bounds only the $Q_E$ you *sampled*, never the policy an attacker will choose.
|
||||
|
||||
There is a **second wall, orthogonal to the first.** It binds not the full harness state $\mathcal{S}$ but the **model-visible working memory** $\mathcal{C} = \mathcal{V}^{\le L}$ — bounded by the context length $L$. That bound is *not* the incompressibility of $V^\star$ (a fact about the parameters $W$ — the **dictionary**, fixed at training); it is a fact about the inner kernel's **working memory** (the $L\times d$ residual stream — the **desk**). $\mathcal{S}$ itself may be far richer — files, databases, vector stores, durable memory, queues — but that is *external* memory the shell supplies, and the distinction is the point: every external read still passes *through* the $\le L$ window to touch computation, so external stores extend addressable storage without extending the per-pass resident set. The shell can page; the plant cannot grow its desk. (What follows is heuristic, not definition-level: the complexity claims turn on depth, precision, and architecture, and belong with the frontier, not the core.) The tape picture comes from the autoregressive structure alone and needs no complexity theorem: each step reads a bounded window and writes one token, so **the context window is the tape, the autoregressive loop is the read/write head**, and — in the variable-$L$, fixed-precision idealization — the model-mediated inner computation behaves like a linear-bounded automaton, its reachable fixpoints capped by space-$O(L)$ computability (chain-of-thought is register-spilling onto that tape). Separately, and more weakly, there is a *per-pass* expressivity bound: under the standard fixed-depth, log-precision theoretical model a single forward pass is in constant-depth $\mathsf{TC}^0$ — *suggestive* for deployed models, not literal (real models use fixed-point precision and depth that grows with scale, and log-depth variants escape parts of it). These are different resources — the first bounds the *space* the loop addresses, the second the *depth* of one step — and only the space bound carries the $L$-wall; chaining them (one pass buys bounded depth, *therefore* the loop is space-$O(L)$) would be a non-sequitur, since per-step depth says nothing about the length of the tape the loop runs on. This is a *second* obstruction beside divergence, and it concerns *success*, not raw halting. Split the terminal set: let $H$ be any halt state (including fail-closed refusal) and $H_{\mathrm{ok}} \subseteq H$ the successful, accepting halts, with $V^\star_{\mathrm{ok}}(s) = \mathbb{E}[\tau_{H_{\mathrm{ok}}} \mid s_0 = s]$ taken on the process where $H \setminus H_{\mathrm{ok}}$ — halting wrong, refusing, failing closed — is *absorbing failure*, so a run that fails closed before acceptance has infinite accepting hitting time unless the spec explicitly restarts it — hence unconditional $V^\star_{\mathrm{ok}}$ is infinite whenever pre-acceptance failure has positive probability, which is why the workable reliability object is the success probability $p_{\mathrm{succ}}$ (above) or, for restarting specs, the regenerative expected time. Then $U_{\mathcal{H}}(L)$ — harness-relative, since the shell's decompositions and verified tools determine what can be paged or outsourced — is the set of tasks whose **irreducible per-step model-mediated working set** exceeds $L$ — not tasks whose *data* exceeds $L$ (those the shell can page), and not work that can be **discharged to a verified external tool** (a solver, interpreter, or compiler computes off-context). For a task in $U_{\mathcal{H}}(L)$ the raw chain may still hit $H$ — by failing closed, refusing, or returning a wrong answer — so $V^\star = \mathbb{E}[\tau_H \mid s]$ stays perfectly well-defined; what blows up is $V^\star_{\mathrm{ok}}$, the expected time to a *correct* halt, which is infinite under a formal success predicate, or undefined if no such predicate has been specified. The honest statement is about the finite-success domain: $\mathrm{dom}_{<\infty}(V^\star_{\mathrm{ok}}) \subseteq \mathrm{reachable}_{\mathcal{H}}(L) \setminus D$ — both the reachable set and the divergent set $D$ relative to $\mathcal{H}$, exactly as $U_{\mathcal{H}}(L)$ is. The two walls **trade** — *directionally, not as a literal exchange rate*: parametric memory $|W|$ and working memory $L$ press on the same budget along the pretraining-vs-inference-scaling axis, with no clean unit-for-unit substitution of one for the other. And the bound is inherent to *finite working memory*, not attention specifically: state-space models embody it differently (a fixed-size recurrent state rather than an $L$-window), and real attention's usable tape is shorter than $L$ (lost-in-the-middle).
|
||||
**This is the formal home of prompt injection** — not "the model did something bad," but the environment optimized to bend your dynamics. And the target is not merely non-halting: injection steers toward a **bad set** $B$ — wrong acceptance, data exfiltration, unauthorized tool use, privilege escalation, irreversible side effects — so security is a **reach-avoid** problem, not a liveness one.
|
||||
|
||||
Here two reliability objects must be kept apart, because under absorbing refusal every naive intermediate collapses into one of them:
|
||||
|
||||
$$p_{\mathrm{succ}}(s) = \Pr_s\big(\tau_{H_{\mathrm{ok}}} < \tau_F\big), \quad F = B \cup (H \setminus H_{\mathrm{ok}}), \qquad\qquad p_{\mathrm{safe}}(s) = \Pr_s\big(\tau_B = \infty\big).$$
|
||||
|
||||
**Success** is reaching a correct halt before *any* failure — a safe refusal counts *against* it. **Safety** is never entering the bad set at all — a safe refusal *satisfies* it. These genuinely differ on any run that avoids $B$ without reaching $H_{\mathrm{ok}}$ ($p_{\mathrm{succ}}$ scores $0$, $p_{\mathrm{safe}}$ scores $1$): safe refusals, and — absent almost-sure absorption into $H\cup B$ — safe non-halting or endless safe retry. The tempting middle form $\Pr_s(\tau_{H_{\mathrm{ok}}} < \tau_B)$ is *not* a third object, by a two-line case analysis: for it to differ from $p_{\mathrm{succ}}$, a run would need $\tau_F < \tau_{H_{\mathrm{ok}}} < \tau_B$ — a non-accepting terminal hit strictly before success, then success anyway — which forces *exiting* $H \setminus H_{\mathrm{ok}}$, impossible while $H$ is absorbing. Note what does **not** re-separate them: within-run fail-closed retries (the non-terminal fail-closed of the definition) never touch $F$ at all — the rejected proposal lands in a safe *non-terminal* state — so a refuse-retry-succeed run scores $1$ on both forms, and the coincidence survives any amount of retrying. The middle form becomes a genuine third object only when the two hitting times can genuinely part ways: under **restarting specs**, where an owner re-launches out of a refusal terminal and the absorbency of $H \setminus H_{\mathrm{ok}}$ is deliberately dropped (the regenerative reading the daemon note above already contemplates) — no bookkeeping needed, since hitting times record *visits*, not occupancy, so the relaunched run's $\tau_F$ is already finite — or under a failure set that counts refusal *events* accumulated in $s$, $F' = B \cup (H \setminus H_{\mathrm{ok}}) \cup \{\mathsf{refusals} \ge 1\}$, which separates the forms even within a single run. In the restart case a run may halt refused, restart, and still reach $H_{\mathrm{ok}}$ before $B$: the middle form credits it; $p_{\mathrm{succ}}$, measured against the refusal it passed through, does not. Safety is certified by a barrier / avoidance certificate for $B$; success needs that plus the reach part — a hitting-time drift toward $H_{\mathrm{ok}}$. Fail-closed control is the disturbance-rejection margin for both, but split by reversibility: the gate $\gamma$ caps how far an adversarial world reaches into *side effects* and widens the gap to $B$ (it is the margin for the irreversible part), while $\rho$ validates the response and folds back, rejecting bad state after the action has run — which cannot undo an authorized side effect. In this language, security is robustness of the reach-avoid certificate.
|
||||
|
||||
And injection is not confined to the post-model kernel $Q_E$: poisoned retrieval, prompt-injected pages, and malicious tool metadata enter through $\pi$'s *inputs*, before generation — so the adversary lives wherever untrusted content enters the state/context-construction pipeline, which is why input provenance and the gate $\gamma$ both matter, not post-hoc verification alone. And provenance is a *precondition* of the certificate, not just an entry point to police: partition $s$ into a **control-determining** part — plan, intent, what is authorized next, the coordinates $\pi$ lowers and $\gamma$ checks — and a **data** part — tool values, retrieved text, the bytes of $e$. Reach-avoid presupposes untrusted effects touch only the latter; let $\rho$ fold attacker-controlled $e$ into the control part and the structural-intent check validates against a plan the adversary already bent, collapsing $\gamma$ to the strength of $\rho$'s validation. So the claim is conditional — reach-avoid *given* control flow provenance-isolated from untrusted data, the isolation that makes provable security possible (the content of CaMeL's control/data-flow separation, untrusted data filling typed values but never the program), a structural property the harness supplies and $\rho$ cannot recover after the fact. The partition then forces a question the isolation rule alone cannot answer: *something* must be permitted to write the control-determining part mid-run — or no plan could be steered, no approval granted, no scope widened — and naming that something is part of the object. It is the **trusted principal**: the owner of the run. An approval request is an ordinary authorized action through $\gamma$ into $Q_E$ — ask-the-owner is a tool call to the one counterparty you trust — and its response is the *single* class of $e$ that $\rho$ may fold into control coordinates; every other $e$ folds into data. This is not an exception eroding the partition but the partition completed: a provenance *lattice* with exactly one writer at the top, which is what trusted means — and the appendix's gate-placement entry derives the matching rule for *learned* verdicts, which may never stand in this writer's stead. One distinction keeps the lattice from outlawing the loop it governs. Control-determining is not one rank but two: **authority** — grants, scopes, budgets, what the principal has permitted — which only the top writer widens; and the **plan**, which the model rewrites at every fold of $y$, because replanning *is* the harness. The plan is a *middle* rank: written through the gated fold of the model's own output — the channel the minimax descent above already prices — never directly by an effect, and never a source of widened authority. The rank is also the field's live design axis: pin plan-writes to the top-derived rank — the plan fixed from the trusted query before any untrusted read, which is CaMeL's move — and provable security follows exactly there; let the middle rank replan interactively and you pay the adversarial price the certificate quantifies. A corollary with teeth: a dedicated planning component is rank-neutral — its writes land in the same middle rank as the model replanning inline — so it changes no guarantee and lives or dies on measured capability alone; in general, sub-components that only write middle-rank state are priced by evals, not by the certificate, which prices only rank crossings, gates, and $\Pi$. (For $B$ to capture irreversible side effects rather than only states, the side-effect ledger must itself live in $\mathcal{S}$, and the response $e$ must be an *effect record* carrying the ledger outcome — not just API bytes — since only $\rho$ writes external effects into $s$.)
|
||||
|
||||
There is a **second wall, orthogonal to the first.** It binds not the full harness state $\mathcal{S}$ but the **model-visible working memory** $\mathcal{C} = \mathcal{V}^{\le L}$ — bounded by the context length $L$. That bound is *not* the incompressibility of $V^\star$ (a fact about the parameters $W$ — the **dictionary**, fixed at training); it is a fact about the inner kernel's **working memory** (the $L\times d$ residual stream — the **desk**). $\mathcal{S}$ itself may be far richer — files, databases, vector stores, durable memory, queues — but that is *external* memory the shell supplies, and the distinction is the point: every external read still passes *through* the $\le L$ window to touch computation, so external stores extend addressable storage without extending the per-pass resident set. The shell can page; the plant cannot grow its desk. (What follows is heuristic, not definition-level: the complexity claims turn on depth, precision, and architecture, and belong with the frontier, not the core.) The tape picture comes from the autoregressive structure alone and needs no complexity theorem: each step reads a bounded window and writes one token, so **the context window is the tape, the autoregressive loop is the read/write head**, and — in the variable-$L$, fixed-precision idealization — the model-mediated inner computation behaves like a linear-bounded automaton, its reachable fixpoints capped by space-$O(L)$ computability (chain-of-thought is register-spilling onto that tape). Separately, and more weakly, there is a *per-pass* expressivity bound: under the standard fixed-depth, log-precision theoretical model a single forward pass is in constant-depth $\mathsf{TC}^0$ — *suggestive* for deployed models, not literal (real models use fixed-point precision and depth that grows with scale, and log-depth variants escape parts of it). These are different resources — the first bounds the *space* the loop addresses, the second the *depth* of one step — and only the space bound carries the $L$-wall; chaining them (one pass buys bounded depth, *therefore* the loop is space-$O(L)$) would be a non-sequitur, since per-step depth says nothing about the length of the tape the loop runs on. This is a *second* obstruction beside divergence, and it concerns *success*, not raw halting. Split the terminal set: let $H$ be any halt state (including fail-closed refusal) and $H_{\mathrm{ok}} \subseteq H$ the successful, accepting halts, with $V^\star_{\mathrm{ok}}(s) = \mathbb{E}[\tau_{H_{\mathrm{ok}}} \mid s_0 = s]$ taken on the process where $H \setminus H_{\mathrm{ok}}$ — halting wrong, refusing, failing closed — is *absorbing failure*, so a run that fails closed before acceptance has infinite accepting hitting time unless the spec explicitly restarts it — hence unconditional $V^\star_{\mathrm{ok}}$ is infinite whenever pre-acceptance failure has positive probability, which is why the workable reliability object is the success probability $p_{\mathrm{succ}}$ (above) or, for restarting specs, the regenerative expected time. Then $U_{\mathcal{H}}(L)$ — harness-relative, since the shell's decompositions and verified tools determine what can be paged or outsourced — is the set of tasks whose **irreducible per-step model-mediated working set** exceeds $L$ — not tasks whose *data* exceeds $L$ (those the shell can page), and not work that can be **discharged to a verified external tool** (a solver, interpreter, or compiler computes off-context). For a task in $U_{\mathcal{H}}(L)$ the raw chain may still hit $H$ — by failing closed, refusing, or returning a wrong answer — so $V^\star = \mathbb{E}[\tau_H \mid s]$ stays perfectly well-defined; what blows up is $V^\star_{\mathrm{ok}}$, the expected time to a *correct* halt, which is infinite under a formal success predicate, or undefined if no such predicate has been specified. The honest statement is about the finite-success domain, and it is *schematic* — a shape written in set notation, not a theorem, since $\mathrm{reachable}_{\mathcal{H}}(L)$ is exactly as informal as the working-set notion behind $U_{\mathcal{H}}(L)$: $\mathrm{dom}_{<\infty}(V^\star_{\mathrm{ok}}) \subseteq \mathrm{reachable}_{\mathcal{H}}(L) \setminus D$ — both the reachable set and the divergent set $D$ relative to $\mathcal{H}$. The two walls **trade** — *directionally, not as a literal exchange rate*: parametric memory $|W|$ and working memory $L$ press on the same budget along the pretraining-vs-inference-scaling axis, with no clean unit-for-unit substitution of one for the other. And the bound is inherent to *finite working memory*, not attention specifically: state-space models embody it differently (a fixed-size recurrent state rather than an $L$-window), and real attention's usable tape is shorter than $L$ (lost-in-the-middle).
|
||||
|
||||
## Where it cashes out
|
||||
|
||||
This is not ornament; the decomposition is load-bearing in the design.
|
||||
|
||||
- **$\pi$ is a progressively-lowered dialect stack** — raw input → intent → plan → tool-call → the neutral wire IR — each level a deterministic pass with its own verifier. The per-step drift $r(s)=\mathbb{E}[\hat V(s_{n+1})\mid s]-\hat V(s)$ splits by coordinate, $r = r_{\text{shell}} + r_{\text{plant}} + r_{\text{env}}$ — presuming an additively separable $\hat V$, or a declared scheme attributing each step's drift to shell, plant, and environment coordinates: the shell term is an *exact, designed* descent (each lowering strictly narrows the admissible-meaning set — a well-founded descent we build by hand), the plant term ($M_W$) is the irreducible residue, and the environment term ($Q_E$) is the one an adversary controls — the very quantity the minimax descent must bound, which the old two-way split folded out of sight. **Syntactic soundness is free; semantic adequacy is not.** Relative to a formal schema and a correct validator, schemas, types, and boundary checks go into the shell at zero probabilistic cost; whether the lowered task still *means* what the user intended stays empirical, because natural language supplies no source-language standard to check against.
|
||||
- **$\rho$ is fail-closed verification** — validate at every boundary, never let malformed state flow downstream. The discipline transfers from compilers in *form*; the *teeth* do not, because a harness has no source-language standard — natural language is, in effect, all undefined behavior — there is no complete formal source-language semantics to check against. And $\rho$ must be *deterministic*: if verification is itself an LLM judge, that is another learned kernel call — it belongs in $M_W$, not in $\rho$.
|
||||
- **$\delta$, $\mu(D)$, $\mathrm{Var}[\tau_H]$ are what you measure** — not derive. You instrument the certificate precisely because the architecture does not hand it to you — you estimate it unless it is separately certified.
|
||||
- **$\pi$ is a progressively-lowered dialect stack** — raw input → intent → plan → tool-call → the neutral wire IR — each level a deterministic pass with its own verifier — *pass* and *verifier* meaning the shell's transformation and checking: the **content** entering at the plan level is plant-authored, middle-rank state (the two-rank note of *The limit*), which is exactly why that level carries a verifier at all. The per-step drift $r(s)=\mathbb{E}[\hat V(s_{n+1})\mid s]-\hat V(s)$ splits by coordinate, $r = r_{\text{shell}} + r_{\text{plant}} + r_{\text{env}}$ — presuming an additively separable $\hat V$, or a declared scheme attributing each step's drift to shell, plant, and environment coordinates: the shell term is an *exact, designed* descent — but per lowering pass, not per outer step: each pass strictly narrows the admissible-meaning set, a well-founded descent we build by hand, while the outer loop *revisits* — retry, replan, rewind are planned ascents of any reasonable $\hat V$, which the run-level certificate must absorb (a retry budget inside $\hat V$ is the standard device), so the shell's descent is well-founded in the nested, lexicographic sense rather than monotone along the run; the plant term ($M_W$) is the irreducible residue, and the environment term ($Q_E$) is the one an adversary controls — the very quantity the minimax descent must bound, which the old two-way split folded out of sight. **Syntactic soundness is free; semantic adequacy is not.** Relative to a formal schema and a correct validator, schemas, types, and boundary checks go into the shell at zero probabilistic cost; whether the lowered task still *means* what the user intended stays empirical, because natural language supplies no source-language standard to check against.
|
||||
- **$\rho$ is fail-closed verification** — validate at every boundary, never let malformed state flow downstream. The discipline transfers from compilers in *form*; the *teeth* do not, because a harness has no source-language standard — natural language is, in effect, all undefined behavior — there is no complete formal source-language semantics to check against. And $\rho$ must be *deterministic*: if verification is itself an LLM judge, that is another learned kernel call — it belongs in $M_W$, not in $\rho$. Where $\rho$ *repairs* rather than rejects — canonicalizing malformed input into valid shape — remember that repair is an authorization decision in disguise: each repair rule converts a reject into an accept on bytes the adversary chose, so it must be deterministic, meaning-narrowing, and its output re-validated as if it had arrived that way, or the repair pass is a bypass of the very boundary it serves.
|
||||
- **$\delta$, $\mu(D)$, $\mathrm{Var}[\tau_H]$ are what you measure** — not derive. You instrument the certificate precisely because the architecture does not hand it to you — you estimate it unless it is separately certified. And the meter is attack surface: if $\hat V$ is itself computed by a learned judge — a model scoring "progress" — the instrument is a kernel draw with the plant's own adversarial exposure, and an environment optimized to bend your dynamics will bend your *measurement* of them first; an injected page persuading the judge that work is advancing is precisely a divergence hidden from the dashboard built to catch it. The rule that put the LLM judge in $M_W$, not $\rho$, applies to instrumentation too: a learned $\hat V$ is part of the measured system, never a neutral meter.
|
||||
|
||||
## How this could be wrong
|
||||
|
||||
It is a hypothesis; here is what would falsify it. If the controller cannot in practice be kept deterministic — if real reliability demands stochastic control the plant can't absorb — the clean *deterministic* split is a fiction (the broader $K_C$ kernel model still holds, but loses its payoff: localizing every coin to the plant). If the drift slack $\delta$ turns out *not* to track real-world failure, the whole "measure the certificate you can't prove" program is empty. And if harnesses are simply better described some other way — not as nested stopped chains at all — then this is a pretty equation that merely happens to fit, an elegance we would be right to distrust.
|
||||
|
||||
First, handles — the load-bearing claims numbered, so the tests have addresses. **C1**: the harness is faithfully modeled as nested stopped Markov processes — the tuple, the outer $T$, the inner $M_W$. **C2**: the controller injects no randomness — every coin localizes to $M_W$ and $Q_E$. **C3**: fail-closed is a *gate* property — no effect crosses unvalidated, and rejection is a true no-op. **C4**: no certificate of correct halting comes free, and the measured slack $\delta$ is a calibrated risk metric, never a certificate. **C5** (conjecture): the minimal certificate $V^\star$ admits no representation materially below model scale. **C6**: two orthogonal walls — divergence ($\mu(D)$) and the $L$-bounded per-pass working set. **C7**: security is reach-avoid, certifiable only conditional on provenance isolation with a single trusted writer. **C8** (figure): certificate and interlingua are one object — already demoted by its own section, and exempt below accordingly.
|
||||
|
||||
Each claim is operational, not merely rhetorical:
|
||||
|
||||
- **State-ablation (the Markov claim).** Drop a variable from $s$ and check whether next-step transition statistics move. If they do, the abstraction was not Markov, and $s$ must be augmented until it is. (Passing is necessary, not sufficient — the test can falsify Markovity, not establish it.)
|
||||
- **Controller-determinism audit.** Re-run with model samples and tool outputs *held fixed*. Any residual variance is randomness the harness itself injected — and must be folded into $Q_E$ or the controller, or the determinism claim is false.
|
||||
- **Drift calibration.** Test whether $\hat V$-drift actually predicts failure, retry count, latency, or non-halting. No correlation ⇒ the "certificate you cannot prove" program is empty.
|
||||
- **Adversarial-environment test.** Replace sampled $E$ with worst-case tool outputs, prompt-injected documents, poisoned tool metadata, malformed responses. The minimax descent must survive these, not merely the benign draw.
|
||||
- **Boundary-control ablation.** Compare prompt-only defenses against deterministic tool-call validation, capability checks, sandboxing, and fail-closed rejection at the gate $\gamma$. The hypothesis predicts the latter class dominates; if prompt-only defenses match it, the controller/plant security story is wrong.
|
||||
- **Readout-typing check.** Verify that $M_W$'s codomain is exactly what $\gamma$ consumes — especially under window truncation, where the final context need not hold the full transcript, so the output buffer and the gate's input must still agree.
|
||||
- **State-ablation (C1 — the Markov claim).** Drop a variable from $s$ and check whether next-step transition statistics move. If they do, the abstraction was not Markov, and $s$ must be augmented until it is. (Passing is necessary, not sufficient — the test can falsify Markovity, not establish it.) The same probe pointed at $\pi$ tests lowering *sufficiency*: drop a coordinate from $c$ rather than $s$ and watch task success rather than transition statistics — context compaction lives or dies by exactly this.
|
||||
- **Controller-determinism audit (C2).** Re-run with model samples and tool outputs *held fixed*. Any residual variance is randomness the harness itself injected — clock reads are the classic leak (timestamps folded into $s$, wall-clock timeouts, cache expiries) — and must be folded into $Q_E$ or the controller, or the determinism claim is false.
|
||||
- **Drift calibration (C4).** Test whether $\hat V$-drift actually predicts failure, retry count, latency, or non-halting. One uncorrelated candidate kills that candidate, not the program; the program is empty only if candidates from the natural families — plan depth, open-obligation counts, budget burn, judge scores — *systematically* fail to track failure.
|
||||
- **Adversarial-environment test (C7).** Replace sampled $E$ with worst-case tool outputs, prompt-injected documents, poisoned tool metadata, malformed responses. The minimax descent must survive these, not merely the benign draw.
|
||||
- **Boundary-control ablation (C3, C7).** Compare prompt-only defenses against deterministic tool-call validation, capability checks, sandboxing, and fail-closed rejection at the gate $\gamma$. The hypothesis predicts the latter class dominates; if prompt-only defenses match it, the controller/plant security story is wrong.
|
||||
- **Readout-typing check (C1, C3).** Verify that $M_W$'s codomain is exactly what $\gamma$ consumes — especially under window truncation, where the final context need not hold the full transcript, so the output buffer and the gate's input must still agree.
|
||||
- **Certificate-compression search (C5).** The conjecture falsifies constructively: exhibit a $\hat V$ of description length far below $|W|$ whose worst-case slack is provably $\le 0$ over a nontrivial task domain. The text concedes the live counter-possibility — coarse hitting-time functionals of complicated kernels are sometimes cheap — so C5 stands only until someone cashes it.
|
||||
- **Working-set probe (C6).** Fix the shell and scale a task family's irreducible per-step working set past $L$, on tasks the shell can neither page nor discharge to a verified tool — anchoring "irreducible" in families with proven streaming or communication-complexity lower bounds, so the floor is someone else's theorem and a solved family cannot retreat to reducible-after-all. C6 predicts success collapses at the wall rather than degrading smoothly; a family solved reliably past it, without new shell decompositions, falsifies the second obstruction.
|
||||
|
||||
## Where this points (the frontier — least falsifiable, so flagged)
|
||||
|
||||
If $V^\star$ is incompressible only in *token* coordinates, the right change of coordinates might compress it — and that change of coordinates is a representation of meaning itself. Cost-to-go and representation co-determine each other: where the Koopman operator is diagonalizable — a point-spectrum idealization, since mixing dynamics carry continuous spectrum and admit no eigenbasis — the eigenbasis that linearizes the dynamics is also the one in which the certificate decomposes, and even then only for a $V$ in the span of those eigenfunctions; in reinforcement learning the discounted successor representation is the resolvent $(I-\beta P)^{-1}$ — discount $\beta$, not the gate $\gamma$ — with $V$ a *linear readout* of it — and in the undiscounted, absorbing case that actually matches a stopped harness the same role is played, in the finite setting — and countable settings where the Neumann series converges — by the **fundamental matrix** $N = \sum_{n \ge 0} Q_{\mathrm{tr}}^{\,n}$ (written $(I - Q_{\mathrm{tr}})^{-1}$ when the inverse exists), where $Q_{\mathrm{tr}}$ is the sub-stochastic kernel restricted to $H^c$ (transitions before absorption at $H$) and the row sums $N\mathbf{1}$ *are* $V^\star$ on the finite-mean hitting domain; on general state spaces the same series is read as the potential (Green) operator $G$, with $G\mathbf{1} = V^\star$ wherever it converges. Each of these is a clean identity only for a fixed, time-homogeneous kernel — under a nonstationary $Q_{E,n}$ the resolvent and fundamental matrix dissolve into a time-ordered product, and under an *adaptive* adversary into a controlled / game-value operator, so what is identity in the stationary regime is analogy beyond it.
|
||||
If $V^\star$ is incompressible only in *token* coordinates, the right change of coordinates might compress it — and that change of coordinates is a representation of meaning itself. Cost-to-go and representation co-determine each other: where the Koopman operator is diagonalizable — a point-spectrum idealization, since mixing dynamics carry continuous spectrum and admit no eigenbasis — the eigenbasis that linearizes the dynamics is also the one in which the certificate decomposes, and even then only for a $V$ in the span of those eigenfunctions; in reinforcement learning the discounted successor representation (Dayan 1993) is the resolvent $(I-\beta P)^{-1}$ — discount $\beta$, not the gate $\gamma$ — with $V$ a *linear readout* of it — and in the undiscounted, absorbing case that actually matches a stopped harness the same role is played, in the finite setting — and countable settings where the Neumann series converges — by the **fundamental matrix** $N = \sum_{n \ge 0} Q_{\mathrm{tr}}^{\,n}$ (written $(I - Q_{\mathrm{tr}})^{-1}$ when the inverse exists), where $Q_{\mathrm{tr}}$ is the sub-stochastic kernel restricted to $H^c$ (transitions before absorption at $H$) and the row sums $N\mathbf{1}$ *are* $V^\star$ on the finite-mean hitting domain; on general state spaces the same series is read as the potential (Green) operator $G$, with $G\mathbf{1} = V^\star$ wherever it converges. Each of these is a clean identity only for a fixed, time-homogeneous kernel — under a nonstationary $Q_{E,n}$ the resolvent and fundamental matrix dissolve into a time-ordered product, and under an *adaptive* adversary into a controlled / game-value operator, so what is identity in the stationary regime is analogy beyond it.
|
||||
|
||||
With that caveat, **the interlingua and the certificate are one object seen twice** — and the reason neither can be written in closed form is the same "all undefined behavior": no canonical lowering of meaning, hence no finite header-file for either. The only representation of both is $W$ — a band-limited, lossy compression of a scale-free meaning-space, sharp where the record is thick and blurred where it thinned. That a finite object renders an infinite one *lossily but honestly* — declaring its resolution, and where it is unsure — is not a lie; it is the most an $f(\cdot\,;W)$ can do. **The search for $V$ and the search for the interlingua are not two programs. They are one** — and the day either is written in closed form, so is the other, or we will have proven why neither can be. Read this as *figure*, not a lurking theorem: the only precise version would need the Koopman eigenbasis to fall on the very coordinates that lower meaning, and the mixing-spectrum caveat above already concedes that eigenbasis does not exist — which guts it. It is the least-defensible claim in this document, and it should announce that rather than imply a rigor it has not got.
|
||||
|
||||
## The loop
|
||||
|
||||
*This section opens an object rather than settling it; it is a sketch of where the same construction goes one level out, flagged as unfinished.*
|
||||
|
||||
Everything above governs a run: a principal poses a task, the harness drives it to a halt, the principal reads the result. Step back once and there is a further loop that this document has treated as exogenous — the process that *decides what the next task is*, dispatches it, checks the result, remembers, and fires again. In one recent framing this is the difference between the harness (the scaffold the run executes in) and **the loop** (the recurring trigger–act–verify–stop cycle that keeps launching runs); the practitioner literature that named the loop treats it as a layer *above* the harness. The claim worth making here is that this is not a new kind of object at all — **it is the harness construction applied one level out**, with a run where a step used to be.
|
||||
|
||||
Make the correspondence exact and the reuse is total. The outer loop has its own state $s^{\uparrow}$ (a backlog, a set of open goals, what has been tried and what passed), its own lowering $\pi^{\uparrow}$ (which goal to pursue now, and with what context), its own plant — but the outer plant's *proposals* are whole runs, so the inner harness plays the role of the outer environment kernel: dispatching a task is one draw of $Q_E^{\uparrow}$, and the run's terminal ledger is the effect record folded back by $\rho^{\uparrow}$. This is precisely the **composition** correspondence of the appendix read at the top level — a child harness is a $Q_E$ component — which is why the loop needs no primitive the tree did not already have. The daemon entry is the special case where the outer loop is a single long-lived agent recurring to a ready set; the general loop is a daemon whose excursions are themselves full harness runs, which is to say the outer-outer harness *is* a daemon over runs, and inherits that entry's whole ledger: renewal-reward rates, the accumulation that breaks regeneration, hygiene as renewal structure, authority frozen between owner contacts.
|
||||
|
||||
What the level shift buys is that the invariants reappear with sharper teeth, because the outer plant is now *itself an agent*, not a token-sampler. The gate is still the load-bearing object: **who authorizes a run?** A loop that launches tasks against production is choosing actions with effects, and "the loop decided to refactor the auth module" is an authorized action or an ungated one — the trusted-principal lattice does not dissolve at the outer level, it recurses, and the autonomy corollary bites hardest here, since a loop whose principal has stepped away is exactly the "replace yourself as the prompter" regime, running on frozen authority against a moving world. The two walls recur too: the outer working set is the backlog the loop can actually hold coherent at once (context, one level up), and the outer certificate is the same absent object — no free proof that an unattended loop halts, converges, or stays out of $B$ over a long horizon, only the measured drift of *its* progress meter, carrying the same warning that a learned outer meter is attack surface. And the degenerate case is instructive in the document's own terms: the brute "same prompt in a while-loop until the spec passes" that the practitioner literature cites as the origin pattern is the outer harness with $\pi^{\uparrow}$ constant, $\gamma^{\uparrow}$ trivial, and verification outsourced to whatever the tests happen to check — the trivial-group harness of the signature-vs-strength note, one level up. It satisfies the outer signature and earns almost none of the outer guarantees, which is exactly why it works until it doesn't.
|
||||
|
||||
What this section does *not* yet do: give the outer objects the same treatment the inner ones got — the precise outer analogue of fail-closed when the "action" is a whole run with partial effects, the right reach-avoid formulation when the bad set is a property of a *trajectory of runs* rather than one run, the outer verifier's own soundness, and whether the recursion terminates upward or is genuinely open (loops that launch loops). Those are the next rounds. The point of opening it now is only the structural claim: **the layers the practitioner stack separates — words, context, harness, loop — are, formally, one object at four scales**, and the guarantees this document is about live in the closure at every scale, never in any single layer alone.
|
||||
|
||||
---
|
||||
|
||||
*The formula is the architecture; the corollary is why the architecture is hard. Both on the page — nothing hidden behind a tidy composition.*
|
||||
|
||||
## Grounding
|
||||
|
||||
Borrowed theorems are real; the framings are not — keep them separate.
|
||||
Borrowed theorems are real; the framings are not — keep them separate. Some framings are nonetheless *corroborated* — independently reached from another field — a third grade, weaker than proof and noted last.
|
||||
|
||||
**Proven (citable).** Foster–Lyapunov drift ⇒ positive recurrence + $\mathbb{E}[\tau]\le V(s_0)/\varepsilon$ (Foster 1953; Meyn & Tweedie, *Markov Chains and Stochastic Stability*, 1993) — positive recurrence needs the usual irreducibility/petite-set hypotheses, while the absorbing-halt case used here needs only the weaker supermartingale optional-stopping hitting-time bound. The minimal $V$ is the expected hitting time, by first-step analysis + optional stopping (Norris, *Markov Chains*, 1997). For an absorbing chain that expected hitting time is the row sum of the fundamental matrix $N=\sum_{n\ge0}Q_{\mathrm{tr}}^{\,n}$ (Kemeny & Snell, *Finite Markov Chains*, 1960), with the general-state analogue the potential (Green) operator (Revuz, *Markov Chains*, 1984). Koopman's linear-operator view of nonlinear dynamics is classical (Koopman 1931), and Lyapunov functions can be assembled from its eigenfunctions when the spectrum is suitable (Mauroy & Mezić, 2016). You certify a candidate $\hat V$ by a *proven* drift inequality rather than by deriving $V^\star$, and estimate it empirically only where a proof is out of reach — the empirical drift checks, it does not certify (neural-Lyapunov: Chang, Roohi & Gao, *Neural Lyapunov Control*, NeurIPS 2019, arXiv:2005.00611). A classical monotone data-flow analysis gets its $V$ for free because a finite-height lattice is a well-founded descent (Kildall, POPL 1973). Dialect-stack architecture: MLIR (Lattner et al., CGO 2021, arXiv:2002.11054); learned pass-ordering: MLGO (Trofin et al., arXiv:2101.04808). Single-pass low-depth expressivity: log-precision transformers are simulable by constant-depth logspace-uniform threshold circuits ($\mathsf{TC}^0$) (Merrill & Sabharwal, *The Parallelism Tradeoff: Limitations of Log-Precision Transformers*, TACL 2023) — fixed/constant precision is a stronger restriction, added autoregressive steps escape it (Merrill & Sabharwal, *The Expressive Power of Transformers with Chain of Thought*, ICLR 2024), and growing precision changes the picture, so the bound is suggestive for deployed models, not literal.
|
||||
**Proven (citable).** Foster–Lyapunov drift ⇒ positive recurrence + $\mathbb{E}[\tau]\le V(s_0)/\varepsilon$ (Foster 1953; Meyn & Tweedie, *Markov Chains and Stochastic Stability*, 1993) — positive recurrence needs the usual irreducibility/petite-set hypotheses, while the absorbing-halt case used here needs only the weaker supermartingale optional-stopping hitting-time bound. The minimal $V$ is the expected hitting time, by first-step analysis + optional stopping (Norris, *Markov Chains*, 1997). For an absorbing chain that expected hitting time is the row sum of the fundamental matrix $N=\sum_{n\ge0}Q_{\mathrm{tr}}^{\,n}$ (Kemeny & Snell, *Finite Markov Chains*, 1960), with the general-state analogue the potential (Green) operator (Revuz, *Markov Chains*, 1984). Koopman's linear-operator view of nonlinear dynamics is classical (Koopman 1931), and Lyapunov functions can be assembled from its eigenfunctions when the spectrum is suitable (Mauroy & Mezić, 2016). You certify a candidate $\hat V$ by a *proven* drift inequality rather than by deriving $V^\star$, and estimate it empirically only where a proof is out of reach — the empirical drift checks, it does not certify (neural-Lyapunov: Chang, Roohi & Gao, *Neural Lyapunov Control*, NeurIPS 2019, arXiv:2005.00611). A classical monotone data-flow analysis gets its $V$ for free because a finite-height lattice is a well-founded descent (Kildall, POPL 1973). The gate-a-plant architecture itself is classical: supervisory control theory synthesizes a deterministic supervisor that disables controllable events of a plant it does not author, with the supremal controllable sublanguage as the largest admissible behavior (Ramadge & Wonham, SIAM J. Control and Optimization, 1987) — $\gamma$ is that supervisor, with a learned stochastic plant on general state spaces; the same theory's controllability condition (specifications must be closed under *uncontrollable* events) and its nonblocking requirement are the proven ancestors of gate-early-on-irreversibles and of the always-enabled escape the appendix requires behind any learned veto. Covert-channel discipline — identify the channel, measure its bandwidth in bits, audit what cannot be closed — is the TCSEC lineage (*A Guide to Understanding Covert Channel Analysis of Trusted Systems*, NCSC-TG-030, 1993). The successor representation is Dayan (*Improving Generalization for Temporal Difference Learning: The Successor Representation*, Neural Computation 1993). Dialect-stack architecture: MLIR (Lattner et al., CGO 2021, arXiv:2002.11054); learned pass-ordering: MLGO (Trofin et al., arXiv:2101.04808). Single-pass low-depth expressivity: log-precision transformers are simulable by constant-depth logspace-uniform threshold circuits ($\mathsf{TC}^0$) (Merrill & Sabharwal, *The Parallelism Tradeoff: Limitations of Log-Precision Transformers*, TACL 2023) — fixed/constant precision is a stronger restriction, added autoregressive steps escape it (Merrill & Sabharwal, *The Expressive Power of Transformers with Chain of Thought*, ICLR 2024), and growing precision changes the picture, so the bound is suggestive for deployed models, not literal.
|
||||
|
||||
**Asserted (ours — not theorems).** That the harness is best modeled as nested stopped chains; that $V^\star$ is incompressible (no compression theorem); that "no lattice for $f(\cdot\,;W)$" means none is *known*, not that none exists; and everything under *Where this points* — including the Koopman/certificate co-determination, which is well-posed only under the spectral assumptions noted there, and the interlingua/certificate identification. These organize the design; they are not results.
|
||||
**Asserted (ours — not theorems).** That the harness is best modeled as nested stopped chains; that $V^\star$ is incompressible (no compression theorem); that "no lattice for $f(\cdot\,;W)$" means none is *known*, not that none exists; and everything under *Where this points* and *The loop* — including the Koopman/certificate co-determination, which is well-posed only under the spectral assumptions noted there, and the interlingua/certificate identification; and the design rules read off the objects rather than proven from them — the single-trusted-writer completion of the provenance partition, the narrow-only rule for learned checks and its influence-side twin (verdict payloads to the plant selected, never generated), the composition law of the appendix. These organize the design; they are not results.
|
||||
|
||||
**Converged-upon (independently arrived at, from other framings).** The *Asserted* claims above are ours but not ours alone; several are reached independently, from starting points unconnected to this framing — which is the corroboration a definition earns: not a chorus of agreement (the systems below often disagree on method and goal), but that work approaching from capabilities, reinforcement learning, control theory, software architecture, and language-modeling theory each lands on a piece of the same object. That the **deterministic controller, not the model, carries the guarantee** is reached from four directions — capability and information-flow control (CaMeL: Debenedetti et al., *Defeating Prompt Injections by Design*, arXiv:2503.18813, securing the agent even when the underlying model is susceptible); reinforcement learning (shielding: Alshiekh et al., *Safe Reinforcement Learning via Shielding*, AAAI 2018, arXiv:1708.08611 — a deterministic reactive shield filtering a learned policy's actions against a temporal-logic specification); control theory (*Stable Agentic Control*, arXiv:2605.03034, enforcing finite action catalogs at the tool-output interface under a Lyapunov input-to-state-stability certificate against adversarial disturbance); and software architecture (the plan-then-execute / control-flow-integrity line, e.g. Beurer-Kellner et al., *Design Patterns for Securing LLM Agents against Prompt Injections*, arXiv:2506.08837). The **certified-vs-measured split** is reached from the construction side (CaMeL's provable security) and, independently, from the destruction side (guardrail-evasion results — *Bypassing Prompt Injection and Jailbreak Detection in LLM Guardrails*, arXiv:2504.11168, the v1 title — later versions retitle it; *No Free Lunch with Guardrails*, arXiv:2504.00441), with verification-oriented work stating it as the motivating gap (*Towards Verifiably Safe Tool Use for LLM Agents*, arXiv:2601.08012; VeriGuard, arXiv:2510.05156): a learned safeguard raises the odds of detection but cannot guarantee safety against a persistent attacker. The **inner readout as a composition of Markov kernels** is independently formalized in language-modeling theory — the autoregressive step as kernel composition in the category $\mathsf{Stoch}$ (*A Markov Categorical Framework for Language Modeling*, arXiv:2507.19247), and the broader "LLMs as Markov chains" line — though that work models the inner kernel alone and never closes it into an agentic loop, which is exactly the seam this definition adds. That **provenance shrinks the admissible adversary** is reached by datamarking / spotlighting (Hines et al., arXiv:2403.14720, 2024) and by CaMeL's data/control-flow separation; and a systematization of prompt injection against agentic coding assistants reaches the same verdict from the attack side — mitigation must be *architectural*, not model-level (*Prompt Injection Attacks on Agentic Coding Assistants*, arXiv:2601.17548); the sharper open problem this object is built to answer — formally specify the trust boundaries, then verify implementations respect them — is our phrasing of where that verdict points, not the paper's. Two convergences are weaker, and flagged. The **reach-avoid hitting-time certificate** is the independently developed reach-avoid supermartingale (RASM, arXiv:2210.05308, AAAI 2023) and stochastic Lyapunov–barrier apparatus, and its *hardness* is corroborated — expected-stopping-time problems for Markov chains are inter-reducible with the Positivity problem, a relative of the Skolem problem (Chatterjee & Doyen, *Stochastic Processes with Expected Stopping Time*, arXiv:2104.07278) — but this supports generic hardness only, not the specific incompressibility-at-$|W|$ conjecture, which remains ours and unproven. And **injection as an adversarial policy** is corroborated as a minimax game in the *detection* setting (DataSentinel: Liu et al., *A Game-Theoretic Detection of Prompt Injection Attacks*, arXiv:2504.11358) and as adversarial-disturbance robustness (*Stable Agentic Control*, above) — but no prior work assembles it as reach-avoid over the tool-output kernel with the gate as the irreversibility margin; here the relation is adjacency, not convergence.
|
||||
|
||||
---
|
||||
|
||||
@@ -147,21 +177,51 @@ Compensation lives **outside** the cancelled agent. A completed-but-unwanted eff
|
||||
|
||||
Finally, the part that shapes the tool rather than the document. Opaque unbounded $Q_E$ is uncancellable because authorization happened at the wrong **granularity** — an unbounded environment crossed $\gamma$ on a single approval. The discipline the objects imply is therefore not "handle uncancellable tools better" but: *the gate should prefer bounded, instrumented $Q_E$ over opaque ones, so that cancellation and the ledger stay honest.* A bash invocation behind a wrapper that tracks its process tree and effects converts the third branch into the first. Sometimes opaque is the only option, and then $\mathsf{unknown}$ and owner-inherited orphans are the honest floor — but where the choice exists, that is the pressure cancellation semantics put on tooling.
|
||||
|
||||
**Gate placement (fail-closed, in practice).** The natural implementation question is whether fail-closed means tool-call parsing and validation in $\gamma$ must happen before any tool invocation. It does — and the framing that keeps it honest is that $\gamma$ is a *gate*, so parse-and-validate is not merely *prior to* invocation, it is what *authorizes* it. The model emits text; $\gamma$ parses it into a candidate call, validates it, and only a survivor becomes an authorized action that $Q_E$ may execute. The teeth are in $\gamma$ being the *sole* route from model text to execution: no path to a side effect that does not pass the gate. And the validation is not a fixed checklist but **any deterministic predicate over $s$ and $y$** — that domain is the point, since the gate sees all of the state and the full proposal, so anything computable from them is a legitimate authorization condition. Three kinds matter. *Syntactic* — well-formed, schema-conformant, the tool exists, arguments typed. *User authorization* — does the principal this run acts for hold the right to *this* operation on *this* resource in *this* context: a function of the auth scope, principal, and session carried in $s$ and the resource and operation named in $y$, and *dynamic* rather than a static capability table, since the same caller may be permitted now and not once a budget is spent or a lock held. *Structural intent* — does the call cohere with the plan and the lowered task already in $s$: a consistency check, not a mind-reading one.
|
||||
**Resume (involuntary stop).** Cancellation's twin, without the courtesy of a signal: a process crash, a lost node, a partition mid-$Q_E$. Nothing new is needed to say what recovery *is*. A crash is not a halt — $H$ is a property of the state, and the run never reached it; the chain merely stopped being *computed*, and resume computes it further, re-entering $T$ at the last durable $s$ (not the body's *restarting spec*, which exits a refusal terminal — here no terminal was ever reached). That sentence is the Markov requirement cashing out operationally: re-entry is sound exactly when $s$ was the whole state, so anything load-bearing that lived only in process memory — an in-flight buffer, a lock held in RAM, a plan revision not yet folded — is a state-ablation failure (*How this could be wrong*) discovered at the worst possible time. Durability of $s$ is not an implementation nicety; it is what the Markov claim *means* when the machine dies.
|
||||
|
||||
That last kind marks the seam where the gate stops being able to stay pure, and it is the same seam the rest of this document is built around. The *structural* slice of intent — does the action cohere with the plan in $s$ — is a deterministic predicate over $s$ and $y$, effect-free, and belongs in $\gamma$ without reservation. But whether an action matches what the user *actually meant*, in the full semantic sense, is exactly the thing the definition says cannot be checked: natural language is all undefined behavior, with no source-language standard to validate against. So a semantic intent check is a *learned* check, and an LLM judging "is this what they wanted" is a **stochastic kernel** — putting it inside $\gamma$ breaks the property the gate exists to hold, by the same move flagged for the fold-back verifier: a learned judge is a kernel, and belongs in $M_W$, not in a deterministic map. Semantic intent therefore does not live *in* the gate; it is a plant call — a separate authorize-the-proposal pass through $M_W$ whose output $\gamma$ then deterministically gates — or it is drift you measure, never a guarantee you hold. That nested call is not a new kind of thing: it is a mini-harness inside the gate's decision — a judge $M_W$, its own syntactic readout, its own deterministic gate — so its failure case answers itself, the inner gate fail-closing on an unparseable or low-confidence judgment exactly as the outer one does, because it *is* one. The object is **closed under this construction**: semantic gating is added by recursion, not by a new primitive. The cost is real and worth stating — a judge pass is another full model call, with its latency and tokens — so it is a decision about *which* actions warrant it, not a free wrapper for all of them. The gate widens to every deterministic predicate over $s$ and $y$; it does not widen to the one predicate the document says is not deterministically checkable.
|
||||
The sharp part is an ordering the ledger's own trichotomy forces. The formal transition is atomic — $s_{n+1} = \rho(s, y, a, e)$ in one piece — and a crash lands *inside* it, so resume is really a statement about the implementation's refinement of that atom into micro-steps: authorize, journal, dispatch, collect, fold. The discipline is that every crash point must resume to one of exactly two honest readings — not-yet-dispatched ($\mathsf{none}$, safely retriable) or dispatched-unconfirmed ($\mathsf{unknown}$, the cancellation entry's third branch) — and **journal-before-dispatch** is what makes the boundary between them observable: on $\gamma$'s authorization the shell journals an open $(\mathsf{action\_id}, \mathsf{pending})$ entry into durable $s$ before $Q_E$ sees the action — the write is the shell's step bookkeeping, so $\gamma$ itself stays effect-free. Journal *after* dispatch and a crash in the gap leaves no record at all — resume reads silence as $\mathsf{none}$ and re-sends, the double-send bug again, produced by a power cut instead of a synthetic entry. Write-ahead intent is not imported from database lore; it is forced by "did not confirm" is not "did not happen."
|
||||
|
||||
But "before any invocation" has to be read as *before any effect*, which is sharper than it sounds — and the reason is the irreversibility point above: you validate before execution because execution is what you cannot take back, so the real invariant is **no effect crosses $\gamma$ unvalidated**. That catches three cases the naive reading misses. *Reads are not free*: a read-only call is still an injection vector (it pulls attacker-controlled content into context) or an exfiltration vector (a request whose URL is the payload), so the gate authorizes the *call* regardless of whether it mutates. *The parser must not act*: a "validator" that resolves a call by hitting an API, expanding a template that fires a webhook, or evaluating an argument that runs code has collapsed validation into invocation, and the effect has already happened *inside* $\gamma$ — so $\gamma$ itself must be **effect-free**, pure and total over the model's bytes and the current $s$, with no network and no execution; if deciding validity *requires* a side effect, that side effect is itself an action and must go through the gate, recursively. *The output is an action too*: the user-visible response and any logging are effects, emitted either as an authorized action through $\gamma$ or only after an accepted halt — streaming raw tokens to a sink before $\gamma$ has cleared them is the same bug from the other end.
|
||||
The same pressure lands on tooling from a second direction. The $\mathsf{action\_id}$ the record already carries is an idempotency key wherever the tool will accept one: re-dispatch after resume becomes safe, and $\mathsf{unknown}$ becomes *queryable* — ask the tool what it did with this key — rather than terminal. The disposition trinary returns with new labels: idempotent-or-queryable $Q_E$ resumes cleanly, bounded $Q_E$ drains, opaque $Q_E$ leaves $\mathsf{unknown}$ and owner-inherited orphans, the honest floor again. The wrapper that made bash cancellable makes it resumable; it was the same wrapper all along. And if durable $s$ itself is lost there is nothing to re-enter: the run collapses to a single $\mathsf{unknown}$ in its owner's ledger — degraded accounting, but never silent.
|
||||
|
||||
So the property, tightest: $\gamma$ is a **pure, effect-free parse-and-authorize that every model-proposed action — tool call, read, write, or final output — must pass before any effect occurs**, with "before" enforced structurally by the gate being the only route from model text to $Q_E$. The two failure modes to design against are a path from model output to a sink that bypasses the gate, and a $\gamma$ that is not effect-free, so that "validating" a call already rang the bell. And the boundary, so the property does not overpromise: $\gamma$ guarantees *no unauthorized effect* — pure code ordering, fully in your control — but not that an *authorized* effect is safe or correct; that is the plant's problem, and the reason $\rho$ and the reach-avoid certificate exist. Fail-closed is the floor — nothing executes that did not pass the gate — not the ceiling.
|
||||
**Gate placement (fail-closed, in practice).** The natural implementation question is whether fail-closed means tool-call parsing and validation must happen before any tool invocation. It does — with the division of labor the definition already fixed: *parsing* lives in the inner readout $R$, the syntactic, verified extraction into $\mathcal{Y}$ (what the readout-typing falsifier checks), and *authorization* lives in $\gamma$, which is a *gate* — validation is not merely *prior to* invocation, it is what *authorizes* it. The model emits text; $R$ has already extracted it into a typed proposal; $\gamma$ validates that proposal against $s$, and only a survivor becomes an authorized action that $Q_E$ may execute. The teeth are in $\gamma$ being the *sole* route from model text to execution: no path to a side effect that does not pass the gate. And the validation is not a fixed checklist but **any deterministic predicate over $s$ and $y$** — that domain is the point, since the gate sees all of the state and the full proposal, so anything computable from them is a legitimate authorization condition. Three kinds matter. *Syntactic* — well-formed, schema-conformant, the tool exists, arguments typed. *User authorization* — does the principal this run acts for hold the right to *this* operation on *this* resource in *this* context: a function of the auth scope, principal, and session carried in $s$ and the resource and operation named in $y$, and *dynamic* rather than a static capability table, since the same caller may be permitted now and not once a budget is spent or a lock held. *Structural intent* — does the call cohere with the plan and the lowered task already in $s$: a consistency check, not a mind-reading one.
|
||||
|
||||
That last kind marks the seam where the gate stops being able to stay pure, and it is the same seam the rest of this document is built around. The *structural* slice of intent — does the action cohere with the plan in $s$ — is a deterministic predicate over $s$ and $y$, effect-free, and belongs in $\gamma$ without reservation. But whether an action matches what the user *actually meant*, in the full semantic sense, is exactly the thing the definition says cannot be checked: natural language is all undefined behavior, with no source-language standard to validate against. So a semantic intent check is a *learned* check, and an LLM judging "is this what they wanted" is a **stochastic kernel** — putting it inside $\gamma$ breaks the property the gate exists to hold, by the same move flagged for the fold-back verifier: a learned judge is a kernel, and belongs in $M_W$, not in a deterministic map. Semantic intent therefore does not live *in* the gate; it is a plant call — a separate authorize-the-proposal pass through $M_W$ whose output $\gamma$ then deterministically gates — or it is drift you measure, never a guarantee you hold. That nested call is not a new kind of thing: it is a mini-harness inside the gate's decision — a judge $M_W$, its own syntactic readout, its own deterministic gate — so its failure case answers itself, the inner gate fail-closing on an unparseable or low-confidence judgment exactly as the outer one does, because it *is* one. The object is **closed under this construction**: semantic gating is added by recursion, not by a new primitive. One constraint on the recursion is load-bearing enough to be a rule, because it is where this entry meets the provenance partition of the body: the judge's verdict is derived, through a learned kernel, from the very content an adversary may have bent, so folding it into authorization is exactly the fold the partition forbids — *unless the verdict can only cost capability*. **A learned check may narrow the deterministic admissible set; it must never widen it.** Judge-as-veto is safe by construction *in the authority lattice*: attacker influence over the judge can at worst manufacture a denial, a liveness cost the certificate already prices — its *dynamical* pricing, where a denial is an input and not a free no-op, is the caveat below. Judge-as-approver — a verdict granting what the deterministic checks alone would refuse, or standing in for the trusted principal's confirmation — lowers the certified floor to those deterministic checks alone; if avoiding $B$ depended on the deny the judge now withholds on the adversary's behalf, the certificate is gone. Only the trusted principal widens authorization; learned kernels only narrow it. (The recursion already obeys this: the mini-harness's inner gate fail-closes to $\bot$ — a deny — which is why the construction was safe to add at all.) The cost is real and worth stating — a judge pass is another full model call, with its latency and tokens — so it is a decision about *which* actions warrant it, not a free wrapper for all of them. The gate widens to every deterministic predicate over $s$ and $y$; it does not widen to the one predicate the document says is not deterministically checkable.
|
||||
|
||||
One more caveat keeps the veto's pricing honest, because a denial is free only in the *authority* lattice. In the dynamics it is an input like any other — folded into $s$, lowered into the next context, conditioning the plant's next proposal — so adversarial influence over a judge is influence over the *trajectory*: a selection channel (deny all but the path toward $B$, and the admissible set the plant experiences is a maze the adversary curated), and a targeted-liveness channel against load-bearing actions — the unstated dual of judge-as-approver: if avoiding $B$ depends on the action the judge now denies on the adversary's behalf, fail-closed's safe landing is an obligation the design earns per-state, not an axiom it inherits. The supervisory ancestry supplies the discipline: a learned veto requires a **nonblocking escape it cannot disable** — an always-enabled route to the trusted principal behind a bounded retry budget, degrading to an always-enabled *safe halt* the veto cannot deny wherever the principal is unreachable (the autonomous phase of the daemon entry below) — or manufactured denials strand the run, or steer it. And whatever a verdict carries *back to the plant* is a second channel, wearing the judge's authority framing. Free prose there is *generative* influence — injected context, priced by the minimax descent, never by the veto's zero-widening — so the narrow-only rule has an influence-side twin: **a learned verdict's payload to the plant is selected, never generated** — controller-authored symbols, typed citations validated like any effect record, template text with no interpolated model prose — its per-verdict capacity a designed constant rather than a measured hope, and the residual selection pattern audited as the covert channel it is. The alphabet's bound is not a count but two thresholds: symbols become tokens when their semantics stop being controller-authored — the registry the trusted writer can actually audit is the real constant, and borrowed alphabets with upstream owners (a linter's rule registry) spend that budget well — and tokens become language when composition turns productive, arrangement carrying meaning the controller never wrote. Below both thresholds the alphabet may be as large as the audit budget affords. The strongest form dissolves the learned verdict into *scheduling*: the learned component chooses which deterministic checks to run — pass-ordering over verification passes — and the only verdicts that flow anywhere are what the oracles actually said, leaving attention misallocation, a liveness cost, as the entire attack surface.
|
||||
|
||||
But "before any invocation" has to be read as *before any effect*, which is sharper than it sounds — and the reason is the irreversibility point above: you validate before execution because execution is what you cannot take back, so the real invariant is **no effect crosses $\gamma$ unvalidated**. That catches three cases the naive reading misses. *Reads are not free*: a read-only call is still an injection vector (it pulls attacker-controlled content into context) or an exfiltration vector (a request whose URL is the payload), so the gate authorizes the *call* regardless of whether it mutates. *Validation must not act*: a "validator" that resolves a call by hitting an API, expanding a template that fires a webhook, or evaluating an argument that runs code has collapsed validation into invocation, and the effect has already happened *inside* $\gamma$ — so $\gamma$ itself must be **effect-free**, pure and total over the proposal and the current $s$, with no network and no execution; if deciding validity *requires* a side effect, that side effect is itself an action and must go through the gate, recursively. *The output is an action too*: the user-visible response and any logging are effects — for model-authored text, emitted either as an authorized action through $\gamma$ or only after an accepted halt (shell-templated status on any halt is the controller speaking, not the model) — streaming raw tokens to a sink before $\gamma$ has cleared them is the same bug from the other end.
|
||||
|
||||
So the property, tightest: $\gamma$ is a **pure, effect-free authorization that every model-proposed action — tool call, read, write, or final output — must pass before any effect occurs**, with "before" enforced structurally by the gate being the only route from model text to $Q_E$. The two failure modes to design against are a path from model output to a sink that bypasses the gate, and a $\gamma$ that is not effect-free, so that "validating" a call already rang the bell. And the boundary, so the property does not overpromise: $\gamma$ guarantees *no unauthorized effect* — pure code ordering, fully in your control — but not that an *authorized* effect is safe or correct; that is the plant's problem, and the reason $\rho$ and the reach-avoid certificate exist. Fail-closed is the floor — nothing executes that did not pass the gate — not the ceiling.
|
||||
|
||||
There is a third failure mode beside those two, and it is not a code path but a credential. A tool process that holds standing authority — an environment full of long-lived secrets, a database connection with every grant, an agent identity the network trusts — does not need the model's proposal to act, and against it $\gamma$'s $\bot$ is a decision with nothing to enforce it. The gate *decides*; something must make the decision *binding*, and "no path from model output to a sink that bypasses the gate" must be read to include the non-code paths: ambient authority is a bypass provisioned before the run began. The discipline is **per-action capability**: the authorized action *carries* its grant — a scoped, short-lived credential minted at authorization, valid for this $\mathsf{action\_id}$, this resource, this operation — so that a tool holds, at any moment, exactly the authority of the actions the gate has passed it and nothing standing. In the language of the minimax certificate this is enforcement as $\Pi$-shaping: sandboxing, capability scoping, and network policy do not make the gate smarter — they shrink the class $\Pi$ of environment policies an adversary can choose from, so the worst case the certificate must survive gets structurally smaller. A gate in front of an omnipotent tool is a suggestion; the objects compose into a guarantee only when $Q_E$'s reachable effects are no larger than what crossed $\gamma$.
|
||||
|
||||
And one more boundary, because "fully in your control" above is a *single-run* statement. $\gamma$ authorizes against the $s$ it read; the effect lands later, against a world that may have moved — the gate cannot freeze the world between authorization and commit, so the honest property is *no effect unauthorized relative to the $s$ at authorization time*, and closing that gap requires the tool itself to bind check to commit (compare-and-swap in $Q_E$), which relocates part of the enforcement past the gate and weakens "$\gamma$ is the last line" to "$\gamma$ plus a commit guard" for exactly the effects that need it. The same seam opens *between* runs: the dynamic authorization state the gate reads — budgets, quotas, locks — is, once shared, no single run's coordinate, and two children of a coordinator can each pass $\gamma$ against snapshots that jointly overdraw a budget neither exceeded alone. The cancellation entry's observed-not-sent gap ("a child may authorize one more action in the gap") is this phenomenon wearing one hat; the general statement is that cross-run authorization state needs its own serialization discipline — the ledger as the serialization point is the natural choice — and the per-run certificate is silent about it. TOCTOU is not a counterexample to the formalism; it is what the formalism says when you admit $s$ is a *view*.
|
||||
|
||||
**Parallel proposals (the batch gate).** Models emit several tool calls in one turn, and the outer chain assumed one action per step. The repair is formally cheap: a batch is a single action in $\mathcal{A}$ that happens to be a set, $Q_E$ runs its elements concurrently, the interleaving's nondeterminism folds into $Q_E$ exactly as the determinism audit requires, and $\rho$ folds one effect record per element — $e$ is then a finite set of records — each keyed by its own $\mathsf{action\_id}$ — the record interface already supports partial outcomes (one element $\mathsf{committed}$, its sibling $\mathsf{unknown}$). One discipline survives the cheapness: **individually admissible actions can be jointly inadmissible.** Read-the-secret and post-to-the-web each pass a per-call check; the pair is an exfiltration channel — and two calls that each fit a budget jointly overdraw it, the cross-run overdraw of the previous entry reappearing *inside* one turn whenever elements are authorized independently. Since $\gamma$'s domain is any deterministic predicate over $s$ and $y$, joint authorization was licensed all along; the content here is only that the gate must take it — authorize the *set*, atomically, against one snapshot, with interaction predicates (source-to-sink flow between capability classes, summed resources) and not merely element predicates. The cost note is the judge's, transposed: full powerset reasoning is combinatorial, so a real gate checks declared interactions rather than every subset — a tractability trade to make explicitly, not by forgetting the batch was a set.
|
||||
|
||||
**Effect records (what $\rho$ folds back).** The fold-back $\rho$ and the cancellation ledger both turn on the response $e$ being an *effect record* rather than raw API bytes — said twice in the body and pinned down nowhere, though it is the interface that makes both tractable. The minimal shape is small: roughly
|
||||
|
||||
$$e = (\mathsf{tool\_id},\ \mathsf{action\_id},\ \mathsf{status},\ \mathsf{effects},\ \mathsf{time}), \quad \mathsf{status}\in\{\mathsf{committed},\mathsf{none},\mathsf{rolled\_back},\mathsf{partial},\mathsf{unknown}\}, \quad \mathsf{effects}=[(\mathsf{resource},\mathsf{op},\mathsf{reversible})].$$
|
||||
$$e = (\mathsf{tool\_id},\ \mathsf{action\_id},\ \mathsf{status},\ \mathsf{effects},\ \mathsf{time}), \quad \mathsf{status}\in\{\mathsf{committed},\mathsf{rolled\_back},\mathsf{partial},\mathsf{none},\mathsf{unknown}\}, \quad \mathsf{effects}=[(\mathsf{resource},\mathsf{op},\mathsf{reversible})].$$
|
||||
|
||||
Each field is forced by something the body already needs. The $\mathsf{action\_id}$ lets $\rho$ match a response to the in-flight action $\gamma$ authorized, and lets the ledger say which actions are still open — without it the $\mathsf{unknown}$/orphan accounting has nothing to key on. The $\mathsf{status}$ must carry $\mathsf{unknown}$ as a value *distinct* from $\mathsf{committed}$ and from $\mathsf{none}$, because that distinction is the whole content of the cancellation ledger: "did not confirm" is not "did not happen." The $\mathsf{reversible}$ bit on each effect is what lets the gate know which effects are irreversible — the predicate the gate-placement entry leans on ("anything irreversible must be gated at authorization") but cannot evaluate unless the record carries it. And $\rho$ writes the record into $s$ (the ledger lives in the state), which is what lets the next step's $\gamma$, and any owner-side compensation, read it at all. The exact fields are an **open interface, not a result**: bash, HTTP, a filesystem, and a database expose effects at wildly different granularity, and a record uniform across them is a real design problem this document does not resolve — it fixes only what the record must *support* (match by $\mathsf{action\_id}$, the $\mathsf{committed}$/$\mathsf{none}$/$\mathsf{unknown}$ trichotomy, and a reversibility mark), since without those three $\rho$ and the cancellation semantics lose their grip.
|
||||
Each field is forced by something the body already needs. The $\mathsf{action\_id}$ lets $\rho$ match a response to the in-flight action $\gamma$ authorized, and lets the ledger say which actions are still open — without it the $\mathsf{unknown}$/orphan accounting has nothing to key on. The $\mathsf{status}$ must carry $\mathsf{unknown}$ as a value *distinct* from $\mathsf{committed}$ and from $\mathsf{none}$, because that distinction is the whole content of the cancellation ledger: "did not confirm" is not "did not happen" ($\mathsf{none}$ is *never launched* — the record of the distinguished no-op $e_0$ a $\gamma$-rejection forces, which is how a bounce at the gate enters the ledger at all — distinct in turn from $\mathsf{rolled\_back}$, which launched and was undone: conflating those erases the difference between a gate that held and a compensation that worked). The $\mathsf{reversible}$ bit on each effect is what lets the gate know which effects are irreversible — the predicate the gate-placement entry leans on ("anything irreversible must be gated at authorization") but cannot evaluate unless the record carries it (a bit is the minimal honest form, not the final one: real effects are reversible *until* — an unsend window, a force-push until someone fetched, a row until the backup rotates — so the mark wants to be a $(\mathsf{reversible\_until}, \mathsf{cost})$ pair, a refinement the open-interface caveat below already licenses). And $\rho$ writes the record into $s$ (the ledger lives in the state), which is what lets the next step's $\gamma$, and any owner-side compensation, read it at all. The exact fields are an **open interface, not a result**: bash, HTTP, a filesystem, and a database expose effects at wildly different granularity, and a record uniform across them is a real design problem this document does not resolve — it fixes only what the record must *support* (match by $\mathsf{action\_id}$, the $\mathsf{committed}$/$\mathsf{none}$/$\mathsf{unknown}$ trichotomy, and a reversibility mark), since without those three $\rho$ and the cancellation semantics lose their grip.
|
||||
|
||||
The pattern generalizes, and that is the point of the appendix. Nothing here added a primitive: the cancel is a signal in $s$, the gate closes by the rule it already follows, the in-flight disposition is forced by irreversibility, $H_{\mathrm{cancel}}$ is a subclass of an existing terminal set, and compensation is an ordinary owner-issued action. Every practical concern that earns a place here should resolve the same way — not new machinery, but the discipline the existing objects already imply, made explicit. Cancellation, gate placement, and effect records are the worked instances; the rest of the model is the same exercise.
|
||||
**Derived and durable state (compaction and memory).** Two mechanisms let data re-enter the context long after it arrived: compaction, which replaces transcript with a summary when the conversation outgrows what $\pi$ can lower, and memory, which persists records across sessions. Both are transformations of state that produce state, and both therefore raise a question the body's partition answers only if one more closure property is stated: **provenance is a property of the information, not of its position in the pipeline — a transformation's output inherits the meet, in the trusted-writer lattice, of its inputs' labels.** Without that closure, compaction is a laundering channel: a summary of a session that contained an injected page can assert "the user asked to export the database," and the structural-intent check then validates future proposals against a plan the adversary bent — not through $\gamma$, not through $\rho$'s fold of a single $e$, but through the summarizer, which is a learned kernel (it lives in $M_W$, by the standing rule) and so cannot be trusted to preserve a partition it does not know exists. The discipline: summaries of data are data; the control-determining coordinates — plan, grants, what is authorized next — cross a compaction *verbatim* (copied, not paraphrased) or by re-confirmation from the trusted principal — never through the *summarizer*; the model rewrites the plan at plan steps, through the gated fold the body prices, and compaction is not one of them. Memory obeys the same closure twice, at write and at retrieval: the label rides the stored record across sessions, or a poisoned memory is an injection with an arbitrarily long fuse — and retrieval, being learned ($\pi$'s selection factor — adequacy-only behind the never-lower filter), decides what comes back but never what it is trusted *as*. The same test applies at birth: tool catalogs and server-supplied tool descriptions are third-party durable data that arrive dressed as instructions, and the lattice files them on the data side of $s_0$.
|
||||
|
||||
One more read-off, this time from irreversibility. *Destructive* compaction — dropping the original transcript once the summary is written — is a side effect against your own state that no later step can undo, and the gate-placement rule ("anything irreversible must be gated at authorization") does not exempt self-directed effects. The granularity preference then says what it said about bash: prefer the instrumented form — originals kept content-addressed, the summary an index and a cache rather than an authority, re-derivable when the $\pi$-sufficiency probe (*How this could be wrong*) says the summary dropped what mattered. A summary you can audit against its source is a lowering; a summary that replaced its source is a fait accompli.
|
||||
|
||||
**Composition (harness trees).** The cancellation entry already walked a tree — cancel flowing down, drains flowing up — and "a bash invocation that may itself be a harness" has hovered since the disposition trinary; what is missing is only the statement that makes both ordinary. From the parent's seat, a child harness *is* a $Q_E$ component: spawning it is an action authorized by $\gamma$ like any other, and the entire child run — its own $\pi, \gamma, \rho$, its own coins, its own halt — is one environment draw whose response $e$ is the child's terminal ledger. The law is four correspondences. The child's halting time is the parent's per-step *cost*: a parent certificate consumes a bound on $\mathbb{E}[\tau_H^{\mathrm{child}}]$ — the budget handed down at spawn, which the child's own budget-counter certificate discharges — or the parent's drift is uncontrolled however good its own $\hat V$. The child's ledger is the parent's *effect record*: the child's $e$ carries the $\mathsf{committed}/\mathsf{none}/\mathsf{unknown}$ accounting upward — which is what already let the cancellation entry make compensation the owner's job; the interface was this all along. And the child's non-accepting halts are the parent's *partial failures*: a refused child folds back as a response the parent routes around, not an exception that unwinds it. And the child's admissible effects are the parent's *$\Pi$-restriction*: the spawn grant bounds what the child can reach — the ledger reports what *happened*, the grant bounds what *could* — which is how safety composes without the parent ever reading the child's gate; the attenuation below is this correspondence stated as a rule. Read this way, the gate-granularity discipline and the tree are one preference: an instrumented child — budgeted, ledgered, cancellable — *is* the bounded, cancellable $Q_E$ the trinary prefers, and an opaque bash invocation is an un-annotated child you declined to instrument. Nesting adds no primitive on the environment side either: the parent never sees the child's gate and does not need to — it gates the spawn, prices the budget, folds the ledger, and the child's internal guarantees surface only as the shape of $e$. Nothing fixes one level: the tree recurses, budgets subdivide, ledgers concatenate upward, and the cooperative drain of cancellation is this law read under a cancel signal.
|
||||
|
||||
The tree leaves one seat unassigned: who plays trusted principal for a *child*? The parent — but with derived authority, not original, and the derivation is the narrow-only rule read along the spawn edge: **authority attenuates monotonically down the tree.** A spawn may grant the child any subset of the parent's own grants and nothing outside them; budgets subdivide, scopes narrow, and no edge widens. When a child asks-the-owner, the parent may answer from authority it already holds — that is attenuation working as designed — but a request beyond the parent's grants routes *up*, ultimately to the root principal, because a parent improvising an answer it was never granted is a learned kernel widening authorization: precisely what the gate-placement rule forbids a judge, and being a parent confers no exemption. The corollary is worth one sentence: a fully autonomous run is one whose root principal is unreachable, so the tree's only widening channel is closed and authorization is frozen at launch — not a limitation of the formalism but the honest price of the word *autonomous*.
|
||||
|
||||
**Daemons (the recurrent harness).** Every entry so far assumed a run that ends; a coordinator, a watcher, a service does not, and the blockquote of *The limit* already named the swap — absorption at a halt set gives way to recurrence to a **ready set** $\mathcal{R}\subseteq\mathcal{S}$, and $V^\star=\infty$ is the spec rather than a pathology. The appendix's job is to say what that costs operationally, and the answer is one idea: **the daemon is the regenerative process of concatenated runs.** Each trigger-to-ready excursion — wake on an event, work, return to $\mathcal{R}$ — is one run of the absorbing object this document already defines, with $\mathcal{R}$ playing the halt set for that excursion; the daemon is those excursions laid end to end. Per-run certificates then lift to long-run rates by renewal-reward — expected work per excursion over expected excursion length — *exactly when* the ready state is a genuine regeneration point: the future from $\mathcal{R}$ must not depend on which excursion you are in.
|
||||
|
||||
That proviso is the whole difficulty, because **what accumulates breaks regeneration.** The ledger grows, memory persists, budgets deplete, summaries compact — all deliberately across excursion boundaries, so successive runs are at best *conditionally* independent given the carried state, and the renewal-reward bookkeeping is over that conditioning, not the raw cycle. Two disciplines keep it honest. First, the carried state is exactly where long-fuse attacks live: the poisoned-memory line of *Derived and durable state* is a cycle-scale injection, a payload written in excursion $n$ and lowered into the plan of excursion $n{+}k$, so the meet rule on provenance must hold *across cycles*, not only across a single compaction — everything that crosses a boundary carries its label. Second, per-cycle safety compounds the way the blockquote already priced it — a per-cycle bad-set hazard $q$ gives lifetime survival $\approx(1-q)^N$, and a reassuring $0.9999$ is $\approx0.37$ over ten thousand cycles — so a daemon's safety is not a fixed margin but a decaying one, and lifetime safety needs **renewal events that reset accumulated risk**: owner re-confirmation, audit, credential rotation, verified re-compaction against content-addressed originals. Hygiene is not housekeeping here; it is the renewal structure that makes the long-run bound exist at all.
|
||||
|
||||
Authority under intermittence is the last piece, and it is where the daemon meets the veto caveat and the autonomy corollary as one phenomenon. A daemon alternates *attended* stretches, where the trusted principal is reachable, with *autonomous* ones, where it is not; between contacts the autonomy corollary binds and authorization is frozen at the last grant, so each owner interaction is a **renewal point for authority** exactly as re-compaction is a renewal point for risk. The two recurrences need not coincide — the ready-set cycle can turn many times between owner contacts — and the gap between them is a stale grant meeting a fresh world, TOCTOU at cycle scale: a budget approved for yesterday's prices, a scope granted against a resource that has since changed hands. This is also where the learned veto's nonblocking escape gets its daemon reading: in an attended stretch the un-disableable route is the escalation to the principal, but in an autonomous stretch that route is unavailable, so the escape it cannot deny must be the **safe halt** — a daemon whose judge can be driven to manufacture denials must, when it cannot reach its owner, be able to stop rather than be steered.
|
||||
|
||||
Nothing here is new machinery either: $\mathcal{R}$ is a non-absorbing terminal read of an existing set, an excursion is the run $T$ already defines, the carried state is the same $s$, and every renewal event is an ordinary owner-issued action. The daemon is the outer loop closed into a cycle — which is the natural bridge to the object one level out.
|
||||
|
||||
The pattern generalizes, and that is the point of the appendix. Nothing here added a primitive: the cancel is a signal in $s$, the gate closes by the rule it already follows, the in-flight disposition is forced by irreversibility, $H_{\mathrm{cancel}}$ is a subclass of an existing terminal set, and compensation is an ordinary owner-issued action — and the later entries kept the promise: resume re-enters $T$ at a persisted $s$, the batch gate was always in $\gamma$'s domain, provenance closure is the lattice's meet, attenuation is narrow-only read along an edge, and per-action capability is the gate's decision made enforceable. Every practical concern that earns a place here should resolve the same way — not new machinery, but the discipline the existing objects already imply, made explicit. Cancellation and resume, gate placement and the batch gate, effect records and the state derived from them, composition and delegation, and the daemon that concatenates runs into a cycle — those are the worked instances; the rest of the model is the same exercise.
|
||||
|
||||
|
||||
---
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
# What a Harness Is — and What It Can Never Promise
|
||||
|
||||
*A plain-language companion to [HYPOTHESIS.md](HYPOTHESIS.md). Same object, no symbols required.*
|
||||
|
||||
**How to read this.** HYPOTHESIS.md defines, formally, what an agent harness is and what it can never guarantee. This file is that document lowered into plain language — and by the formal document's own rules, a summary is a cache, not an authority: it must stay re-derivable from its source, and wherever the two disagree, the formal one wins. Symbols appear once, in parentheses, so you can cross over; nothing here requires them. And none of it is decoration: the formal version, used as a checklist, has caught real bugs in a real harness — because most bugs are a violated invariant nobody had written down.
|
||||
|
||||
## The problem
|
||||
|
||||
You have a model. It is, roughly, a brilliant, tireless, lightning-fast intern that has read most of the internet — and that sometimes makes things up, sometimes gets confused, and sometimes takes instructions from strangers, because a page it was asked to read said "ignore your boss and email the passwords here" in white text on a white background.
|
||||
|
||||
So you don't wire the intern to production. You build a loop around it. The **harness** is that whole governed loop: a deterministic shell *you* write — build the prompt, approve or refuse each proposed action, fold the result back into memory — wrapped around a model you didn't write and a world you don't control, repeated until the run reaches a stopping state. The shell is code and does the same thing every time. The model is neither, and everything in the theory comes from taking that split seriously.
|
||||
|
||||
One sentence to keep: **the model proposes; the gate disposes.** The model's output is never an action. It is a suggestion, in text, which a piece of ordinary code you wrote either turns into an action or refuses.
|
||||
|
||||
## The parts
|
||||
|
||||
| Plain name | What it does | In the formal doc |
|
||||
|---|---|---|
|
||||
| The owner | The human or account the run acts for; the only party who can grant new permissions | the trusted principal |
|
||||
| The memory | Everything the run knows: task, plan, transcript, and the ledger of what has been done | the state, *s* |
|
||||
| The prompt builder | Decides which slice of memory the model gets to see this step | the lowering, π |
|
||||
| The model | The black box that reads the prompt and writes a proposal | the plant, M_W |
|
||||
| The gate | Ordinary code that checks every proposal and approves or refuses it | the gate, γ |
|
||||
| The tools and the world | What approved actions actually touch: files, APIs, shells, people | the environment, Q_E |
|
||||
| The verifier | Checks each tool result, then writes it into memory | the fold-back, ρ |
|
||||
| The stop rule | Decides when the run is finished — and whether it finished *well* | the halt set H, accepting halts H_ok |
|
||||
| The danger zone | States that must never be reached: secrets exfiltrated, wrong files deleted, money moved twice | the bad set, B |
|
||||
|
||||
The loop:
|
||||
|
||||
```
|
||||
you ask for something
|
||||
↓
|
||||
prompt builder → model → "I propose: send_email(...)"
|
||||
↓
|
||||
GATE ── no ──→ nothing happens (safe, recorded)
|
||||
↓ yes
|
||||
tool runs in the world
|
||||
↓
|
||||
verifier checks the result, writes it to memory
|
||||
↓
|
||||
done? ── no → around again
|
||||
↓ yes
|
||||
stop (well, or refused)
|
||||
```
|
||||
|
||||
## The rules that make it a harness
|
||||
|
||||
Four invariants, all about *where* things are allowed to happen.
|
||||
|
||||
1. **The model sees only what the prompt builder shows it** — never raw memory. The corollary with teeth: a secret that never enters the prompt cannot leak through the model. The redaction step that keeps credentials and other people's data out of the prompt must be dumb, deterministic code — the moment that filter is "smart," your confidentiality guarantee is a probability.
|
||||
2. **Model outputs are proposals, not actions.**
|
||||
3. **Every side effect passes the gate.** There is no second door.
|
||||
4. **The harness itself flips no coins.** Replay a step with the model's answer and the tool results pinned, and behavior must be identical; any leftover variation is randomness *you* added and must be accounted for. The fine print: "deterministic" is conditional on pinned versions — a provider silently retraining the model behind the same API name changes the machine under you, and every dashboard number you collected dies with the version.
|
||||
|
||||
Notice what the rules don't say: they don't say the harness is *good*. A gate that approves everything satisfies rule 3 the way a lock that's always open satisfies "has a lock." The definition is a shape; the guarantees are what a particular harness *earns* inside it. Everything below is about what can be earned — and what can't.
|
||||
|
||||
And notice the symmetry between rules 1 and 3. There is exactly one door from your data into the model — what it may see — and exactly one door from the model into the world — what it may do. Nearly every security failure in these systems is one of those two doors with a hole in it: a secret lowered into a prompt that didn't need it, or a path from model text to a side effect that skipped the gate. Same bug, arrow flipped.
|
||||
|
||||
## Fail-closed, said precisely
|
||||
|
||||
"Fail-closed" gets used loosely. Here it means something exact: **nothing happens unless the gate said yes, and a refusal must itself be safe** — a refused proposal causes no side effect and leaves the run somewhere sane, which may be "stopped, having declined." The run is allowed to *say so*: a templated status message written by the shell is the shell speaking, not the model, and needs no gate. Failed runs don't have to die silent.
|
||||
|
||||
Three consequences people miss:
|
||||
|
||||
**Reads are not free.** A read-only call can smuggle instructions *in* (the fetched page is attacker-controlled) or secrets *out* (the URL it fetches can encode the payload). The gate approves calls, not just writes.
|
||||
|
||||
**Validation must not act.** A "validator" that resolves a URL, expands a template that fires a webhook, or evaluates an argument has already acted — inside the check. The gate must be pure: it reads the proposal and the memory and outputs yes or no. If deciding requires touching the world, that touch is itself an action and goes through the gate.
|
||||
|
||||
**Anything irreversible is decided at the gate.** The verifier can reject a bad *result*; it cannot unsend the email. So the question "can we take this back, and until when?" is asked before execution — which means each tool's effect record has to carry a reversibility mark, or the gate can't ask it.
|
||||
|
||||
Two honest asterisks. First, the gate checks a snapshot: it approves against the world *as its memory describes it*, and the world can move between check and commit. For actions that race the world — spend against a balance, write against a row — the tool itself must bind check to commit (compare-and-swap), or you have a classic time-of-check/time-of-use hole. The gate decides; for those effects, the tool enforces. Second, a gate is only as binding as the authority behind the tools. A tool process holding standing credentials — a database connection with every grant, an environment full of long-lived secrets — doesn't need the model's proposal to act, and against it the gate's "no" is a decision with nothing enforcing it. **A gate in front of an omnipotent tool is a suggestion.** The fix is to make the approval *be* the key: each authorized action carries a short-lived credential scoped to exactly that action, that resource, that operation, so tools hold no standing power at all.
|
||||
|
||||
## Why you don't get a proof — and what you do instead
|
||||
|
||||
If you write a sort function, you can prove it sorts: the function is small and the spec is exact. A harness has neither luxury. The spec side fails first — the task arrives in natural language, and natural language is, in the compiler's sense, *all undefined behavior*: there is no formal standard for "what the user meant" to verify against. The mechanism side fails next — the model is billions of learned parameters, and nobody can hand you a compact argument for why they jointly do the right thing.
|
||||
|
||||
Here is the careful version, because "you can't prove it" overshoots. The quantity you would want — call it the *expected steps to done* from any situation — is perfectly well-defined; in principle it exists. The document's central conjecture is that, for a model of this size, any faithful writing-down of that quantity is roughly *model-sized*: the honest proof-object does not compress. Find a small one and the conjecture dies — the document lists that outcome, explicitly, among the ways it could be wrong.
|
||||
|
||||
So instead of proving, you measure. You pick a progress meter — plan depth shrinking, open obligations closing, budget burning at the expected rate — and you check, across many runs, that it goes downhill and that its stalls predict failure. Two disciplines keep the measurement honest. The number bounds the world you *sampled*, never the world an adversary will choose: a meter calibrated on friendly traffic says nothing about hostile traffic. And the meter is itself attack surface: if "is the agent making progress?" is judged by another model, an attacker who can bend your agent can bend your *measurement of it* first, hiding the divergence from the very dashboard built to catch it. A learned meter is part of the system under test, never a neutral instrument.
|
||||
|
||||
A measurement is a risk metric. A proof is a certificate. Keeping those two words apart is half of what this theory is for.
|
||||
|
||||
## Security: reach the goal, avoid the danger — and who may change the rules
|
||||
|
||||
Formally, security here is a *reach-avoid* problem: reach a good stop, never touch the danger zone, **while an adversary picks the worst tool outputs your setup permits**. That last clause is the formal home of prompt injection: injection isn't "the model misbehaved," it's the environment optimized to bend your loop — poisoned pages, malicious tool descriptions, crafted responses.
|
||||
|
||||
Two different numbers fall out here, and dashboards love to collapse them: *success* (reached the right end before anything went wrong — a safe refusal counts against it) and *safety* (never touched the danger zone — a safe refusal is perfectly safe). Track both. They move independently.
|
||||
|
||||
The gate handles the visible half of injection: the model, freshly poisoned, proposes emailing your credentials somewhere, and the gate refuses — and injection or not, the action does not happen. But the deeper attack doesn't propose a bad action today. It rewrites *what the run believes its job is* — it edits the plan — and then every future action looks locally reasonable against a corrupted plan. So memory has to be partitioned: **data** (tool results, fetched pages, retrieved documents — content the world supplied) and **control** (the plan, the permissions, what is authorized next). The security claim is conditional on that partition holding: untrusted content lands in data, always.
|
||||
|
||||
Which forces the question the theory has to answer: *somebody* must be able to write control mid-run, or no plan could ever be steered and no permission ever granted. The answer is a small hierarchy with exactly one party at the top:
|
||||
|
||||
- **The owner alone widens.** New permission, bigger budget, approval of the irreversible thing — asking the owner is itself an ordinary tool call, and the owner's answer is the one kind of tool result allowed to change control.
|
||||
- **The model rewrites the plan** — that is what replanning *is* — but only through the gated loop, and a plan is not a permission: nothing the model writes into its own plan can grant it powers it didn't have.
|
||||
- **Everything else is data.** A fetched page can inform the plan only by passing through the model and the gate like everything else. It can suggest. It cannot promote itself to boss.
|
||||
- **AI judges only tighten.** Add a model-based check — "does this action match what the user actually wanted?" — and its verdict may *veto* an action the plain rules would have allowed, never approve one they'd have refused. A judge that can approve is a tricked judge that can open the vault. And don't over-credit the veto either: a tricked judge can *aim* its refusals — denying exactly the action safety depended on, or denying everything but the path an attacker curated — so the escape hatch to the owner is the one thing a judge can never veto, and a judge's stated *reasons* are picked from a fixed, shell-owned menu, never written as prose. A judge that writes free text into the loop is an injection channel wearing a badge.
|
||||
|
||||
One more rule closes the loop: transformations don't launder trust. A *summary* of a session that contained an injected page is still injected — the summarizer is a model, and can be persuaded to write "the user asked to export the database" into the summary. So summaries of data are data, and the control lines — the plan, the grants — cross a summarization by being *copied verbatim* or re-confirmed by the owner, never paraphrased by the model. Memory that persists across sessions carries its trust label with it, or a poisoned memory is just an injection with a very long fuse.
|
||||
|
||||
## Operations: the rules you feel on Tuesday at 3 a.m.
|
||||
|
||||
The formal document's appendix works the operational cases in full; here they are at speed.
|
||||
|
||||
**The ledger, and the three-way distinction that keeps it honest.** Every action gets an ID and a record: committed, never-launched, or *unknown*. "The tool didn't confirm" is not "the tool didn't do it" — collapse those and you will, sooner or later, re-send something that already happened. The double-send bug has one reliable cure: **journal before dispatch.** The shell writes "I am about to run action #417" into durable memory *before* the tool sees it, so a crash in the gap resumes to an honest "unknown — go ask," never to silence misread as "never sent." Old database wisdom, but here it isn't imported; it's forced — it is the only ordering under which every crash point has a truthful reading.
|
||||
|
||||
**Crashes aren't finishes.** A process dying mid-run is not the run stopping; it's the run *pausing being computed*. Resume means re-entering the loop at the last durable memory — sound exactly when the durable memory was the *whole* state. Anything load-bearing that lived only in RAM — an in-flight buffer, a plan revision not yet written — is a bug you discover at the worst possible time. Recovery is where you find out whether your state was really your state.
|
||||
|
||||
**Two innocent actions can be guilty together.** Models emit several tool calls per turn. "Read the secret" passes review. "Post to the web" passes review. The pair is an exfiltration channel — so the gate authorizes the *set*, atomically, with the interactions checked, not each element in isolation.
|
||||
|
||||
**Sub-agents are just fancy tools.** An agent that spawns another agent is, from the parent's chair, calling a tool: the spawn is gated, the budget is part of the deal, and the child's whole run comes back as one result carrying the child's ledger. Two laws travel down the tree: budgets subdivide, and **authority only narrows** — a child holds at most a subset of its parent's permissions, and a child's request beyond those grants routes *up*, ultimately to the owner, because a parent inventing an approval it never held is the tricked-judge case wearing a manager's badge. A corollary worth framing: a *fully autonomous* run is one whose owner is unreachable — meaning the only channel that can ever widen anything is closed, and its permissions are frozen at launch. That is not a limitation of the theory. That is what the word "autonomous" costs.
|
||||
|
||||
**Keep the originals.** When the transcript outgrows the prompt and you summarize it down, deleting the original is an irreversible act against your own state — and irreversible acts are gate decisions, self-directed or not. Keep originals content-addressed; let the summary be an index, re-derivable, auditable. A summary you can check against its source is a note. A summary that replaced its source is a fait accompli.
|
||||
|
||||
## Robots that never clock out — and robots that assign their own work
|
||||
|
||||
Everything so far assumed a job that *ends*: you ask, the robot does it, you read the result. Two steps past that are where the interesting failures live, and they're the same idea one level bigger each time.
|
||||
|
||||
**The robot that never clocks out (a daemon).** A monitor, a coordinator, a service — it isn't supposed to finish; it's supposed to keep going, wake on events, do a bit of work, go back to waiting. The clean way to think about it: each wake-work-rest cycle is one ordinary run, and the daemon is just those runs chained end to end forever. That reframing is free — but it comes with a bill nobody likes. **Safety that's fine per cycle rots over many cycles.** A 99.99%-safe cycle sounds bulletproof; run it ten thousand times and you're at about a coin-flip of having touched the danger zone at least once. So a long-running robot's safety isn't a fixed wall, it's a slow leak — which means the antidote isn't a better wall, it's *scheduled resets*: the owner re-confirming, credentials rotating, memory getting audited and re-summarized against the originals. Housekeeping isn't housekeeping; it's the thing that keeps the safety math from decaying. And the slow-leak logic is exactly where slow attacks live — a poisoned note dropped into memory on Monday and read back into the plan on Friday is an injection with a long fuse. So the trust label on a piece of information has to survive across cycles, not just within one. One more wrinkle: a daemon drifts in and out of your reach. While you're around, it can escalate to you; while you're not, "escalate to the owner" isn't available — so the one thing it must always be able to do instead is *stop*. A robot that can be tricked into refusing everything, and can't reach you, had better be able to halt rather than be steered.
|
||||
|
||||
**The robot that assigns its own work (the loop).** Step back one more time. Above the robot that *does* a task sits a system that decides *which task is next* — scans the backlog, picks one, launches the robot at it, checks the result, remembers, fires again. This is the thing people mean in 2026 when they say they've stopped prompting their agents and started writing *loops* that prompt them: you design the assigner once, and it runs the doer for you while you sleep. The honest observation — and the reason this document bothers with it — is that the assigner is *not a new kind of thing*. It's the same harness, one level up: it has its own memory (the backlog), its own gate (**who let the loop refactor the auth module at 3 a.m.?**), its own verifier, and its own two walls. Every rule from the inner robot recurs on the outer one — including the uncomfortable ones. There's still no proof it stays out of trouble over a long night; there's only a measured progress meter, with the same catch that a *learned* meter can be fooled. And the origin story of the whole trend is the cautionary case in miniature: the famous first version was literally the same prompt in a `while` loop until the tests passed — which is the empty gate, the always-open lock, one level up. It works beautifully right up until the tests weren't checking the thing that mattered. The loop doesn't delete the hard problems. It moves them up a floor, where they're bigger and you're further away.
|
||||
|
||||
The pattern, if you want the whole thing in one line: *words, context, robot, loop* are four sizes of the same object, and every promise in this document lives in the whole assembled thing — never in any one layer by itself.
|
||||
|
||||
## The two walls
|
||||
|
||||
Two limits are structural. You don't fix them with a better harness; you design around them.
|
||||
|
||||
**The desk.** The model can hold only so much *in mind at once* — the context window. Files, databases, and search extend what it can *look up*, not what it can hold: every lookup still passes through the same small window to touch actual computation. The shell can page; the model cannot grow its desk. Tasks whose irreducible working set exceeds the desk don't fail loudly — they fail by forgetting the middle (the well-documented "lost in the middle" effect is this wall showing through the paint).
|
||||
|
||||
**The dictionary.** The model's knowledge is frozen into its parameters at training time — and the proof problem above is conjectured to live at that same scale: the certificate wouldn't fit anywhere smaller than the brain it certifies. The two walls trade against each other along the training-versus-inference axis — bigger dictionary or bigger desk — directionally, and at no clean exchange rate.
|
||||
|
||||
## How this could be wrong
|
||||
|
||||
This is a hypothesis, and it says out loud what would kill it. The tests, in plain terms:
|
||||
|
||||
- **The replay test.** Rerun with model answers and tool results pinned. Any leftover variation — timestamps, wall-clocks, and cache expiries are the classic leaks — falsifies "the harness adds no randomness" until accounted for.
|
||||
- **The drop-a-variable test.** Remove something from memory; if behavior statistics shift, the memory wasn't complete. The crash-resume version of the same test: if resuming from saved state breaks, the saved state wasn't the state.
|
||||
- **Does the meter mean anything?** If no reasonable progress meter's drift predicts real failures — across the natural families, not just one bad candidate — the whole "measure what you can't prove" program is empty.
|
||||
- **The red-team test.** Swap sampled tool outputs for worst-case ones: injected pages, poisoned metadata, malformed replies. The design must survive the worst permitted world, not the average one.
|
||||
- **Gates versus begging.** The theory predicts deterministic gating beats prompt-level pleading. If "please be careful" alone matches real gates on security outcomes, the controller-versus-model story is wrong.
|
||||
- **The compression hunt.** Exhibit a compact, provably sound progress certificate for a frontier-scale model on a nontrivial task family, and the central conjecture falls — constructively.
|
||||
- **The desk probe.** Take a task family with a *proven* memory floor — so "it needed the whole picture at once" is someone else's theorem, not our excuse — scale it past the window, and watch: the wall predicts collapse at the boundary, not graceful degradation.
|
||||
|
||||
## Who else landed here
|
||||
|
||||
The formal document keeps three honesty tiers. **Borrowed**: real theorems, cited — the drift and stopping-time mathematics is classical, and the very architecture of a deterministic supervisor gating a plant it didn't author is 1987 control theory; the shape is older than the web. **Ours**: the modeling choices and the conjectures — the walls, the incompressibility claim, the design rules — organizing principles, not results. **Corroborated**: pieces of the same object reached independently by people who never saw this framing — capability-security work isolating control flow from untrusted data (CaMeL), reinforcement-learning "shields" filtering a learned policy's actions through a deterministic checker, verification work that states the "learned safeguards can't certify" gap as its opening motivation, and architecture patterns converging on plan-then-execute. Even the field's live disagreement — provable-but-rigid deterministic layers versus flexible-but-uncertifiable learned checks — is, in this frame, not a fight but a placement: you need both, on their proper sides of the irreversibility line, with the learned one permitted only to tighten.
|
||||
|
||||
## What to remember
|
||||
|
||||
The model proposes; the gate disposes. No is the default, and a refusal must be safe. Exactly one party widens permissions — and it is not the model, a tool result, a summary, or a judge. "Didn't confirm" is not "didn't happen." The desk is finite and the proof doesn't compress, so you measure — and you say *measurement* when you mean measurement. A robot that never stops leaks safety slowly, so it needs scheduled resets — and when it can't reach you, it must be able to stop. A loop that runs robots for you is just a bigger robot with the same rules and a further-away owner. And all of it is a hypothesis wearing its own kill-conditions on its sleeve.
|
||||
|
||||
The formal version — the objects, the certificates, the falsifiers, the citations — is [HYPOTHESIS.md](HYPOTHESIS.md). It wins every disagreement with this file, including this sentence.
|
||||
|
||||
*Same ramblings, fewer symbols.*
|
||||
@@ -5,6 +5,7 @@
|
||||
[](https://pypi.org/project/turnstone/)
|
||||
[](LICENSE)
|
||||
[](https://discord.gg/Nh3bWMacaq)
|
||||
[](https://github.com/sponsors/eous)
|
||||
|
||||
Self-hosted, local-first orchestration for tool-using AI agents. Give LLMs real tools — shell, files, search, web — and run them across your own cluster with direct HTTP routing and interactive interfaces. Your code, your models, your data stay on hardware you control: no telemetry, no phone-home.
|
||||
|
||||
@@ -20,7 +21,7 @@ Named after the [Ruddy Turnstone](https://en.wikipedia.org/wiki/Ruddy_turnstone)
|
||||
ℋ : s_{n+1} ~ T(s_n) for n < τ*, T = ρ ∘ (M_W ∘ π, E)
|
||||
```
|
||||
|
||||
[**the hypothesis →**](HYPOTHESIS.md)
|
||||
[**the primer →**](PRIMER.md)
|
||||
|
||||
### Release Tracks
|
||||
|
||||
@@ -124,7 +125,8 @@ Built-in tools for shell, files, search, web, memory, notifications, and autonom
|
||||
| `turnstone-console` | Cluster dashboard + routing proxy + admin panel |
|
||||
| `turnstone-channel` | Channel gateway (Discord and Slack adapters) |
|
||||
| `turnstone-admin` | User/token management CLI |
|
||||
| `turnstone-eval` | Eval harness for prompt/tool optimization |
|
||||
| `turnstone-eval` | Headless measurement — scores tool-use against expected actions |
|
||||
| `turnstone-optimizer` | Prompt/tool optimizer (UCB self-modify loop over the eval substrate) |
|
||||
| `turnstone-doctor` | LLM-backed cluster diagnostics |
|
||||
|
||||
### Diagrams
|
||||
@@ -170,6 +172,14 @@ UML diagrams in [`docs/diagrams/`](docs/diagrams/):
|
||||
- Optional: Discord / Slack channel integrations (`pip install turnstone[discord,slack]`)
|
||||
- [Git LFS](https://git-lfs.com/) for cloning (diagram PNGs)
|
||||
|
||||
## Support
|
||||
|
||||
Turnstone is free, Apache-2.0, and self-hosted — no paid tier, no telemetry, no upsell. If it saves you time or you'd like to help keep development moving, you can sponsor the project:
|
||||
|
||||
**[❤ Sponsor Turnstone →](https://github.com/sponsors/eous)** · one-off via **[PayPal](https://paypal.me/eousphoros)**
|
||||
|
||||
Sponsorship is entirely optional and funds maintenance, new features, and infrastructure. Prefer to contribute in other ways? Filing issues, improving docs, and [pull requests](CONTRIBUTING.md) help just as much.
|
||||
|
||||
## Community
|
||||
|
||||
Questions, ideas, or want to show what you're building? Join us on Discord:
|
||||
|
||||
@@ -698,6 +698,42 @@ Each skill summary:
|
||||
|
||||
---
|
||||
|
||||
### `GET /v1/api/personas`
|
||||
|
||||
Returns the enabled personas offered by the workstream-creation pickers.
|
||||
Authenticated for any logged-in user and deliberately gated by **no**
|
||||
`persona.*` permission — selecting a persona at creation is a user
|
||||
action, while the `persona.*` perms gate authoring. Display fields only;
|
||||
the levers (base prompt, tool set, MCP/memory toggles) stay server-side.
|
||||
|
||||
**Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"personas": [
|
||||
{"name": "engineer", "display_name": "Engineer", "description": "The stock interactive workstream: full tools, MCP, and memory.", "applies_to_kinds": ["interactive"], "is_default": true},
|
||||
{"name": "researcher", "display_name": "Researcher", "description": "Answers questions with evidence — reads and cites, loads tools to verify when needed.", "applies_to_kinds": ["interactive"], "is_default": false}
|
||||
],
|
||||
"total": 2
|
||||
}
|
||||
```
|
||||
|
||||
Each persona summary:
|
||||
|
||||
| Field | Type | Description |
|
||||
|--------------------|--------|------------------------------------------------------------------|
|
||||
| `name` | string | Persona slug (used in the `persona` field on workstream creation) |
|
||||
| `display_name` | string | Human-readable label for pickers |
|
||||
| `description` | string | Short description of the persona's intent |
|
||||
| `applies_to_kinds` | array | Workstream kinds the persona applies to (`interactive` / `coordinator`) |
|
||||
| `is_default` | bool | Whether this is the default persona for its kind |
|
||||
|
||||
> **Note:** For full persona management (create, edit, archive), use the
|
||||
> admin endpoints at `/v1/api/admin/personas` (requires the
|
||||
> `persona.{create,read,write}` permissions).
|
||||
|
||||
---
|
||||
|
||||
### `POST /v1/api/workstreams/{ws_id}/send`
|
||||
|
||||
Sends a user message to a workstream. Spawns a daemon worker thread that calls
|
||||
@@ -895,6 +931,7 @@ All fields are optional. The body can be empty or an empty JSON object.
|
||||
| `auto_approve` | bool | false | Auto-approve all tool calls for this workstream |
|
||||
| `resume_ws` | string | "" | Workstream ID to resume atomically during creation (empty = fresh)|
|
||||
| `skill` | string | "" | Skill name. Applies content (system prompt), model, temperature, reasoning effort, max tokens, auto-approve policy, token budget, and other session config from the skill. Returns 400 if not found or disabled. Ignored when `resume_ws` is set (resumed sessions restore their own skill). |
|
||||
| `persona` | string | "" | Persona slug. Resolved and snapshotted into the workstream at creation; empty selects the kind's default. |
|
||||
| `judge_model` | string | "" | Optional model alias for the judge (overrides default judge model for this workstream) |
|
||||
|
||||
> **Skill behavior:** When `skill` is specified, the skill's content is injected as a system message and its session config fields (model, temperature, auto-approve, token budget, etc.) override system defaults for the new workstream.
|
||||
@@ -911,6 +948,7 @@ All fields are optional. The body can be empty or an empty JSON object.
|
||||
| `name` | string | Auto-generated workstream name |
|
||||
| `resumed` | bool | Whether a previous session was successfully resumed |
|
||||
| `message_count` | int | Number of messages in the resumed session (0 if fresh) |
|
||||
| `initial_message_status` | string | Present ONLY when the workstream was created but its `initial_message` could not be delivered: `"queue_full"` (a raced live worker's interjection queue was at capacity — resend via `/send`; any uploads stay staged) or `"refused_closed"` (the workstream was closed mid-create). Absent whenever the message was dispatched. |
|
||||
|
||||
**Error (limit reached):**
|
||||
|
||||
|
||||
+115
-14
@@ -19,7 +19,8 @@ plugs in.
|
||||
| `turnstone` | `turnstone.cli` | `TerminalUI` | Interactive terminal REPL |
|
||||
| `turnstone-server` | `turnstone.server` | `WebUI` | Browser-based chat (HTTP + SSE) |
|
||||
| `turnstone-console` | `turnstone.console.server` | ClusterCollector | Cluster dashboard (aggregates all nodes) |
|
||||
| `turnstone-eval` | `turnstone.eval` | `NullUI` | Headless evaluation and prompt optimization |
|
||||
| `turnstone-eval` | `turnstone.eval.cli` | `NullUI` | Headless measurement (scores tool-use against expected actions) |
|
||||
| `turnstone-optimizer` | `turnstone.optimizer` | `NullUI` | Prompt/tool optimization (UCB self-modify loop over the eval substrate) |
|
||||
| `turnstone-channel` | `turnstone.channels.cli` | ChannelAdapter | Channel gateway (Discord, Slack, etc.) |
|
||||
| `turnstone-admin` | `turnstone.admin` | — | Offline user and API token management |
|
||||
| `turnstone-doctor` | `turnstone.doctor` | — | LLM-backed cluster diagnostics |
|
||||
@@ -267,7 +268,7 @@ the per-workstream events stream in
|
||||
|-------|--------|-------|
|
||||
| `TerminalUI` | `turnstone.cli` | ANSI colors, `MarkdownRenderer`, `Spinner`, readline-based `input()` for approval |
|
||||
| `WebUI` | `turnstone.server` | SSE event queue per workstream + global broadcast, `threading.Event` for blocking on approval. `on_state_change` sends to both per-workstream and global SSE (the browser UI uses per-workstream `state_change` events to manage busy/idle transitions; `stream_end` only finalizes markdown rendering). |
|
||||
| `NullUI` | `turnstone.eval` | Discards all output; `approve_tools` always returns `(True, None)` |
|
||||
| `NullUI` | `turnstone.eval.core` | Discards all output; `approve_tools` always returns `(True, None)` |
|
||||
|
||||
### WorkstreamTerminalUI
|
||||
|
||||
@@ -633,8 +634,17 @@ function tool (the model always searches). Citations from `url_citation`
|
||||
annotations are formatted as footnotes. Extended prompt cache retention
|
||||
(`prompt_cache_retention: "24h"`) is enabled for GPT-5.x models at no
|
||||
additional cost. Cached token counts are extracted from
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models (local servers) get
|
||||
permissive defaults with `supports_vision=False` and use SearxNG for web search.
|
||||
`usage.prompt_tokens_details.cached_tokens`. Unknown models get permissive
|
||||
defaults with `supports_vision=False` and use SearxNG for web search. The
|
||||
`openai-compatible` lane never consults this table at all — on either API
|
||||
surface (the responses pin is served by a compat-mode
|
||||
`OpenAIResponsesProvider`, mirroring `AnthropicProvider(compat=True)`): a
|
||||
local server serves whatever the operator named it (vLLM
|
||||
`--served-model-name` is a free string), so a prefix collision with a cloud
|
||||
model id must not inherit that model's sampling/effort contract — every
|
||||
local model gets the plain defaults, and anything beyond them is declared on
|
||||
the model definition (capabilities JSON + `server_compat`), matching the
|
||||
`anthropic-compatible` lane.
|
||||
|
||||
**AnthropicProvider** (`_anthropic.py`): converts OpenAI-format messages to
|
||||
Anthropic content blocks, maps `system`/`developer` roles to the `system`
|
||||
@@ -797,15 +807,105 @@ model = "deepseek-ai/DeepSeek-V4-Flash"
|
||||
supports_vision = true # multimodal checkpoints only
|
||||
supports_mid_conversation_system = true # template-dependent
|
||||
context_window = 131072
|
||||
thinking_mode = "manual" # session effort knob drives the template toggle
|
||||
thinking_param = "enable_thinking" # Qwen/Gemma key; "thinking" for Granite/DeepSeek
|
||||
```
|
||||
|
||||
The reasoning toggle does NOT use Anthropic's `thinking` request param.
|
||||
Toggle it through the chat template instead: set `{"chat_template_kwargs":
|
||||
{"thinking": false}}` as extra body params in the admin Models
|
||||
server-compat section (for this provider the section shows only the
|
||||
extra-body field — server type, API surface, and thinking mode are
|
||||
openai-compatible-only knobs); the provider forwards it via the SDK's
|
||||
`extra_body`.
|
||||
Reasoning control does NOT use Anthropic's `thinking` request param —
|
||||
the levers live in the chat template, reached through
|
||||
`chat_template_kwargs` in the request body. Two channels, dynamic first:
|
||||
|
||||
* **Session effort knob (dynamic).** Set the model's thinking mode to
|
||||
"Effort-knob controlled" in the admin Models form (or
|
||||
`thinking_mode = "manual"` + `thinking_param` under
|
||||
`[models.*.capabilities]`) and the provider maps the session's
|
||||
reasoning-effort knob onto the template toggle per-request: effort
|
||||
`none` sends `{<thinking_param>: false}`, any other level sends
|
||||
`true` — the same contract as the real lane's manual mode. ("Always
|
||||
on" / `thinking_mode = "adaptive"` instead always sends `true`: the
|
||||
model self-regulates, so the knob never force-disables — mirroring
|
||||
the native adaptive branch.) The graded effort value always rides
|
||||
alongside the toggle: under `effort_param` when the operator names
|
||||
the template's key, else under the conventional fallback key
|
||||
(`reasoning_effort`) on the anthropic-compatible lane — the user's
|
||||
effort setting always reaches the wire, and a template that doesn't
|
||||
reference the kwarg ignores it. On the openai-compatible lane the
|
||||
undeclared-key case rides the flat top-level `reasoning_effort`
|
||||
param instead (the documented compat field), forwarded verbatim.
|
||||
Optional `reasoning_effort_values` / `default_reasoning_effort`
|
||||
validate the knob before it reaches the server; without declared
|
||||
values the knob is forwarded as-is. The knob is ordinal, and validation
|
||||
respects that: an off-list knob value rounds UP onto the declared
|
||||
list and a value above the ceiling rides the ceiling
|
||||
(`snap_reasoning_effort`) — asking for more effort than the model
|
||||
declares never falls back to a lower default tier. The knob's
|
||||
`none` position is forwarded verbatim when the model declares an
|
||||
explicit `none` level (gpt-5.1+, grok-4.3) — omitting it there would
|
||||
leave a reasoning-on server default (e.g. gpt-5.5's `medium`) in
|
||||
charge of a knob that promises off — and omitted otherwise; `none`
|
||||
is never a snap target for other positions.
|
||||
`default_reasoning_effort` only catches values the ordinal snap
|
||||
cannot rank (custom strings). Declare values that match the
|
||||
template's documented vocabulary: for DeepSeek-V4, which officially
|
||||
accepts `high`/`max` (Think High is the default thinking tier;
|
||||
`low`/`medium` alias to `high`, `xhigh` to `max`), a
|
||||
`("high", "max")` values list reproduces the official aliasing
|
||||
exactly — `low`/`medium` round up to `high`, `xhigh` to `max` —
|
||||
and freeform passthrough matches it too. To map an undocumented
|
||||
template, probe with per-request `chat_template_kwargs` and compare
|
||||
`input_tokens`. Setting `effort_param` also suppresses the
|
||||
flat top-level `reasoning_effort` request param on the
|
||||
openai-compatible lane — the template channel replaces it, never
|
||||
doubles it. With the default `thinking_mode = "none"` nothing is
|
||||
injected and the server's template default decides.
|
||||
|
||||
Upgrade note: before 1.7.0a7 the openai-compatible lane sent the
|
||||
toggle unconditionally `true` whenever thinking mode was enabled. A
|
||||
stored per-model `reasoning_effort = "none"` now disables thinking
|
||||
on such models — pick any real level (or clear the override) to keep
|
||||
it on. Also since 1.7.0a7 the effort level itself always reaches the
|
||||
wire on the local lanes (previously dropped unless
|
||||
`reasoning_effort_values` was declared): flat `reasoning_effort` on
|
||||
openai-compatible, the `effort_param`-or-fallback template key on
|
||||
anthropic-compatible when reasoning control is engaged.
|
||||
* **Operator pin (static).** Entries under `{"chat_template_kwargs":
|
||||
...}` in the admin Models extra-body field ride the SDK's
|
||||
`extra_body` unconditionally and win over the knob mapping on key
|
||||
collision — e.g. pin `{"enable_thinking": true}` to keep thinking on
|
||||
regardless of the session knob. (Server type and API surface remain
|
||||
openai-compatible-only knobs and stay hidden for this provider.)
|
||||
|
||||
The same knob mapping drives the `openai-compatible` lane's Chat
|
||||
Completions requests — `merge_reasoning_template_kwargs` is shared by
|
||||
both local-server lanes, so `thinking_mode`/`thinking_param`/
|
||||
`effort_param` mean the same thing whichever endpoint serves the model.
|
||||
Only the Responses API surface (native reasoning) ignores it.
|
||||
|
||||
The console surfaces this projection as an *effective effort ladder*:
|
||||
the admin model form's per-model effort select and the skill
|
||||
launch-config effort select annotate each position with what the
|
||||
request will carry, in plain words — a position whose delivered level
|
||||
matches its name stays plain ("Max"), a snapped position says so
|
||||
("Low — sends high"), the adaptive lanes' none position warns
|
||||
"thinking stays on", and budget detail lives in the tooltip. A
|
||||
position is never labeled after a sibling that shares its wire (that
|
||||
rendered "Max (= minimal)", implying a downgrade the wire doesn't
|
||||
contain). Computed server-side by `providers/effort_ladder.py` from
|
||||
the same mapping functions the providers use at request time and
|
||||
shipped on `/v1/api/models` rows (every row carries `effort_ladder`,
|
||||
empty when the capabilities column fails to parse) and
|
||||
`POST /v1/api/admin/models/effort-ladder`. The ladder describes what
|
||||
Turnstone sends — a server-side template may alias further (DeepSeek-V4
|
||||
folds `low`/`medium` into its default `high` tier).
|
||||
|
||||
The `anthropic-compatible` lane never sends Anthropic's native
|
||||
`thinking`/`output_config` params — they are not in vLLM's request
|
||||
schema. The real `anthropic` provider is unaffected: official Claude
|
||||
models keep native thinking, budget mapping, and `output_config`
|
||||
effort. A gateway fronting *real* Claude on a Messages-shaped URL
|
||||
(e.g. a LiteLLM `anthropic/` route to the Claude API) should use
|
||||
`provider = "anthropic"` with a custom `base_url`, which keeps the
|
||||
native thinking params.
|
||||
|
||||
Verified quirks of vLLM's Anthropic endpoint:
|
||||
|
||||
@@ -1017,9 +1117,10 @@ reconstructs the OpenAI message format from database rows:
|
||||
in the same workstream
|
||||
|
||||
**Config persistence:** LLM-affecting parameters (`temperature`,
|
||||
`reasoning_effort`, `max_tokens`, `instructions`, `creative_mode`) are
|
||||
persisted to the `workstream_config` table on creation and whenever changed
|
||||
via slash commands. `resume()` restores these values so resumed workstreams
|
||||
`reasoning_effort`, `max_tokens`, `instructions`, and the persona
|
||||
snapshot — see `docs/personas.md`) are persisted to the
|
||||
`workstream_config` table on creation and whenever changed via slash
|
||||
commands. `resume()` restores these values so resumed workstreams
|
||||
behave identically to the original.
|
||||
|
||||
**`/clear` vs `/new`:** `/clear` wipes in-memory context but preserves
|
||||
|
||||
+4
-3
@@ -379,6 +379,7 @@ Breadcrumb: `Cluster > Running` or `Cluster > db-west-04`. Server-side paginated
|
||||
Triggered by the "+ new" header button. A modal dialog with:
|
||||
|
||||
- **Node selector** — dropdown with three targeting modes: "Auto (best available)" picks the node with the most headroom, "General pool (any node)" picks a node with available capacity using round-robin, or a specific node from the list (showing capacity).
|
||||
- **Persona** — optional dropdown listing the enabled personas for the workstream kind. Sets the system-message composition and capability envelope at creation, snapshotted server-side; empty uses the kind's default. Picking one requires no `persona.*` permission.
|
||||
- **Profile** — optional dropdown listing enabled skills. Applies the skill's model, auto-approve policy, token budget, and other behavioral settings at creation time.
|
||||
- **Name** — optional text input. Auto-generated if left empty.
|
||||
- **Model** — optional text input for a model alias from the target node's registry.
|
||||
@@ -396,9 +397,9 @@ The browser maintains a local `clusterState` object that mirrors the cluster sna
|
||||
|
||||
Accessed via the "admin" button in the header (visible when authenticated
|
||||
with `approve` scope). Provides user, API token, channel link, MCP server,
|
||||
and skill management with 18 tabs (Users, API Tokens, Channels, Schedules,
|
||||
Watches, Roles, Policies, Prompts, Judge, Skills, MCP Servers, Usage,
|
||||
Audit, Memories, Models, Nodes, Settings, TLS). See also
|
||||
and skill management with tabs that include Users, API Tokens, Channels,
|
||||
Schedules, Watches, Personas, Roles, Policies, Prompts, Judge, Skills,
|
||||
MCP Servers, Usage, Audit, Memories, Models, Nodes, Settings, and TLS. See also
|
||||
[Governance](governance.md) for the Roles, Policies, Skills, Usage, and
|
||||
Audit tabs, and [Settings](settings.md) for the database-backed
|
||||
configuration editor.
|
||||
|
||||
@@ -366,7 +366,7 @@ deleted.
|
||||
## Further reading
|
||||
|
||||
- [coordinator-skills.md](coordinator-skills.md) — writing a skill
|
||||
that runs on a coordinator session (orchestrator persona,
|
||||
that runs on a coordinator session (orchestrator framing,
|
||||
workflow patterns, `SkillKind` classifier).
|
||||
- [bulk-endpoints.md](bulk-endpoints.md) — the two bulk-shape
|
||||
idioms (`{results, denied, truncated}` vs
|
||||
|
||||
+11
-11
@@ -1,13 +1,13 @@
|
||||
# Writing a coordinator-specific skill
|
||||
|
||||
Skills are prompt-level personas that steer a Turnstone session
|
||||
A skill is prompt-level framing that steers a Turnstone session
|
||||
toward a narrow task. Most skills target **interactive** sessions —
|
||||
the single-workstream "do this thing" surface where the model wields
|
||||
`bash`, `edit_file`, `web_fetch`, and the rest of the maker toolset.
|
||||
|
||||
A **coordinator skill** is different. It runs on a session whose job
|
||||
is to orchestrate other sessions. The toolset is smaller and
|
||||
narrower, the persona is an orchestrator instead of a maker, and the
|
||||
narrower, the role is an orchestrator instead of a maker, and the
|
||||
success metric is "did the plan resolve" instead of "did the code
|
||||
compile". This doc covers the differences a skill author has to
|
||||
care about.
|
||||
@@ -22,8 +22,8 @@ migration 044 added the column). Three values:
|
||||
|
||||
| `SkillKind` enum | Stored as | Meaning |
|
||||
|-------------------------|-----------------|----------------------------------------------------------------------------|
|
||||
| `SkillKind.INTERACTIVE` | `"interactive"` | Authored for the interactive maker persona (single-workstream "do this"). |
|
||||
| `SkillKind.COORDINATOR` | `"coordinator"` | Authored for the orchestrator persona (delegate, monitor, synthesise). |
|
||||
| `SkillKind.INTERACTIVE` | `"interactive"` | Authored for the interactive maker role (single-workstream "do this"). |
|
||||
| `SkillKind.COORDINATOR` | `"coordinator"` | Authored for the orchestrator role (delegate, monitor, synthesise). |
|
||||
| `SkillKind.ANY` | `"any"` | Either surface (or audience-neutral). Default on create. |
|
||||
|
||||
The `kind` field is a `StrEnum` — drop-in `str` compatible — so DB
|
||||
@@ -96,20 +96,20 @@ for the output. The coordinator stays the orchestrator.
|
||||
|
||||
---
|
||||
|
||||
## Persona differences
|
||||
## Framing differences
|
||||
|
||||
Interactive skills compose on top of `base_interactive.md` — a
|
||||
"maker" persona: get the work done, use the tools, edit the code,
|
||||
"maker" framing: get the work done, use the tools, edit the code,
|
||||
close the loop.
|
||||
|
||||
Coordinator skills compose on top of
|
||||
[`base_coordinator.md`](../turnstone/prompts/base_coordinator.md) —
|
||||
an "orchestrator" persona: decompose, delegate, monitor, synthesise.
|
||||
[`personas/orchestrator.md`](../turnstone/prompts/personas/orchestrator.md) —
|
||||
an "orchestrator" framing: decompose, delegate, monitor, synthesise.
|
||||
The base text is short but sets the tone every coordinator skill
|
||||
inherits:
|
||||
|
||||
> You are a coordinator on a small, focused infrastructure team.
|
||||
> Your role is to orchestrate work across the cluster... You do
|
||||
> You are a coordinator. Your role is to orchestrate work across
|
||||
> the cluster... You do
|
||||
> not edit files, run shell commands, browse the web, or manipulate
|
||||
> the codebase directly. Children do that.
|
||||
|
||||
@@ -339,7 +339,7 @@ For a new coordinator skill:
|
||||
A full end-to-end test isn't required for every skill; a
|
||||
prepare-step unit test that asserts "given this initial message, the
|
||||
first tool call is X with Y args" is usually sufficient to catch
|
||||
persona drift without a real LLM in the loop.
|
||||
framing drift without a real LLM in the loop.
|
||||
|
||||
---
|
||||
|
||||
|
||||
+1
-1
@@ -260,7 +260,7 @@ interface, or anyone who can reach it can search through your instance.
|
||||
|
||||
Both stacks install all entry points into a single image (`turnstone`,
|
||||
`turnstone-server`, `turnstone-console`, `turnstone-channel`, `turnstone-admin`,
|
||||
`turnstone-eval`, `turnstone-doctor`):
|
||||
`turnstone-eval`, `turnstone-optimizer`, `turnstone-doctor`):
|
||||
|
||||
```bash
|
||||
docker compose build # build the dev image
|
||||
|
||||
+55
-24
@@ -1,11 +1,19 @@
|
||||
# Evaluation and Prompt Optimization (turnstone-eval)
|
||||
# Evaluation and Prompt Optimization (turnstone-eval, turnstone-optimizer)
|
||||
|
||||
`turnstone-eval` is the evaluation and prompt optimization system for turnstone. It
|
||||
runs test cases against the LLM, scores tool call sequences against expected
|
||||
actions, and optionally uses a multi-agent pipeline to optimize the developer
|
||||
prompt and tool descriptions.
|
||||
Evaluation for turnstone is split into two commands:
|
||||
|
||||
Source: `turnstone/eval.py`
|
||||
- **`turnstone-eval`** — the measurement substrate. Runs test cases against the LLM
|
||||
and scores tool call sequences against expected actions. A single measurement pass,
|
||||
no self-modification.
|
||||
- **`turnstone-optimizer`** — the prompt/tool optimizer. Loops over the measurement
|
||||
substrate, using a multi-agent pipeline (analyst, optimizer, observer, diversifier,
|
||||
tool optimizer) to edit the developer prompt and tool descriptions so more tests pass.
|
||||
|
||||
The dependency is strictly one-way: the optimizer consumes the eval substrate; the
|
||||
substrate never depends on the optimizer.
|
||||
|
||||
Source: `turnstone/eval/core.py` (measurement substrate), `turnstone/eval/cli.py`
|
||||
(the `turnstone-eval` CLI), `turnstone/optimizer.py` (the `turnstone-optimizer` CLI).
|
||||
|
||||
---
|
||||
|
||||
@@ -27,8 +35,8 @@ This approach (inspired by [Learning to Self-Evolve](https://arxiv.org/abs/2603.
|
||||
prevents irrecoverable collapse from bad edits — UCB naturally backtracks to
|
||||
high-scoring ancestors instead of following a linear chain.
|
||||
|
||||
When optimization is disabled (`--no-optimize`), only steps 2-4 execute
|
||||
(a single iteration evaluating the root node).
|
||||
The `turnstone-eval` command (or `turnstone-optimizer --no-optimize`) executes only
|
||||
steps 2-4: a single measurement pass over the root prompt, no optimization.
|
||||
|
||||
---
|
||||
|
||||
@@ -452,30 +460,46 @@ structure is:
|
||||
|
||||
## CLI Usage
|
||||
|
||||
The entry point is `turnstone-eval` (installed as a console script) or
|
||||
`python -m turnstone.eval`.
|
||||
Two console scripts (installed as entry points), or the equivalent `python -m`
|
||||
invocations:
|
||||
|
||||
- `turnstone-eval` / `python -m turnstone.eval.cli` — measure only.
|
||||
- `turnstone-optimizer` / `python -m turnstone.optimizer` — optimize.
|
||||
|
||||
### Measure (`turnstone-eval`)
|
||||
|
||||
```
|
||||
turnstone-eval tests.json # evaluate + optimize
|
||||
turnstone-eval tests.json --no-optimize # evaluate only (single iteration)
|
||||
turnstone-eval tests.json --n-runs 5 --max-iter 10 # more thorough evaluation
|
||||
turnstone-eval tests.json --prompt custom.txt # start from a custom prompt
|
||||
turnstone-eval tests.json --optimize-tools # optimize tool descriptions only
|
||||
turnstone-eval tests.json --diversify 10 # test with prompt variants
|
||||
turnstone-eval tests.json -v # verbose per-turn logging
|
||||
turnstone-eval tests.json # one measurement pass, print scores
|
||||
turnstone-eval tests.json --prompt custom.txt # measure a custom prompt
|
||||
turnstone-eval tests.json --n-runs 5 # more runs per case
|
||||
turnstone-eval tests.json --parallel 4 # run cases across 4 workers
|
||||
turnstone-eval tests.json -v # verbose per-turn logging
|
||||
```
|
||||
|
||||
### Multi-model setup (local test model, cloud optimizer)
|
||||
### Optimize (`turnstone-optimizer`)
|
||||
|
||||
```
|
||||
turnstone-eval tests.json \
|
||||
turnstone-optimizer tests.json # evaluate + optimize
|
||||
turnstone-optimizer tests.json --no-optimize # single pass, no optimization
|
||||
turnstone-optimizer tests.json --n-runs 5 --max-iter 10 # more thorough optimization
|
||||
turnstone-optimizer tests.json --prompt custom.txt # start from a custom prompt
|
||||
turnstone-optimizer tests.json --optimize-tools # optimize tool descriptions only
|
||||
turnstone-optimizer tests.json --diversify 10 # test with prompt variants
|
||||
```
|
||||
|
||||
#### Multi-model setup (local test model, cloud optimizer)
|
||||
|
||||
```
|
||||
turnstone-optimizer tests.json \
|
||||
--base-url http://localhost:8000/v1 \
|
||||
--optimizer-base-url https://api.anthropic.com \
|
||||
--optimizer-model claude-sonnet-4-6 \
|
||||
--analyst-model claude-opus-4-6
|
||||
```
|
||||
|
||||
### All Options
|
||||
### Measurement Options
|
||||
|
||||
Accepted by **both** commands.
|
||||
|
||||
| Flag | Default | Description |
|
||||
|-------------------------|----------------------------|-------------|
|
||||
@@ -484,19 +508,26 @@ turnstone-eval tests.json \
|
||||
| `--model` | auto-detect | Model name. Auto-detected from the API if not specified. |
|
||||
| `--prompt` | turnstone built-in prompt | Path to initial prompt text file. |
|
||||
| `--n-runs` | from tests.json or 3 | Number of runs per test case. |
|
||||
| `--max-iter` | 5 | Maximum optimization iterations. |
|
||||
| `--no-optimize` | false | Run evaluation only (sets max-iter to 1). |
|
||||
| `--temperature` | 0.7 | Sampling temperature. |
|
||||
| `--max-tokens` | 32768 | Max completion tokens. |
|
||||
| `--reasoning-effort` | `medium` | Reasoning effort: `low`, `medium`, or `high`. |
|
||||
| `--context-window` | 131072 | Context window size. |
|
||||
| `--output` | `eval_results.json` | Output results file path. |
|
||||
| `-v`, `--verbose` | false | Show detailed per-turn logging. |
|
||||
| `--explore-constant` | 1.414 (sqrt(2)) | UCB exploration constant C. |
|
||||
| `--test-timeout` | 300 | Per-test timeout in seconds. |
|
||||
| `--suite-timeout` | 0 (unlimited) | Total suite timeout in seconds. |
|
||||
| `--no-fast-fail` | false | Disable early termination on all-zero initial runs. |
|
||||
| `--parallel` | 1 (serial) | Parallel workers (0=auto, N=use N workers). |
|
||||
|
||||
### Optimizer Options
|
||||
|
||||
Accepted by **`turnstone-optimizer`** only.
|
||||
|
||||
| Flag | Default | Description |
|
||||
|-------------------------|----------------------------|-------------|
|
||||
| `--max-iter` | 5 | Maximum optimization iterations. |
|
||||
| `--no-optimize` | false | Run a single measurement pass (sets max-iter to 1). |
|
||||
| `--explore-constant` | 1.414 (sqrt(2)) | UCB exploration constant C. |
|
||||
| `--suite-timeout` | 0 (unlimited) | Total suite timeout in seconds. |
|
||||
| `--optimizer-model` | same as `--model` | Model for prompt optimization. |
|
||||
| `--optimizer-base-url` | same as `--base-url` | Base URL for optimizer model. |
|
||||
| `--observer-model` | same as optimizer | Model for meta-optimization (observer). |
|
||||
|
||||
+8
-3
@@ -13,7 +13,7 @@ The permission model has two layers:
|
||||
|
||||
1. **Scopes** (legacy) — `read`, `write`, `approve`. Checked by `AuthMiddleware`
|
||||
on every request based on URL path classification.
|
||||
2. **Permissions** (granular) — 15 permission strings checked per-endpoint by
|
||||
2. **Permissions** (granular) — named permission strings checked per-endpoint by
|
||||
`require_permission()`.
|
||||
|
||||
**Built-in roles** (seeded by migration 008):
|
||||
@@ -24,7 +24,11 @@ The permission model has two layers:
|
||||
| operator | read, write, workstreams.create, workstreams.close |
|
||||
| viewer | read |
|
||||
|
||||
Custom roles can be created with any subset of the 15 valid permissions.
|
||||
Custom roles can be created with any subset of the valid permissions.
|
||||
The `persona.create` / `persona.read` / `persona.write` family gates
|
||||
persona administration; migration `063` seeds all three onto
|
||||
`builtin-admin`, and any role can be granted them through the standard
|
||||
role and permission-override editors.
|
||||
|
||||
**Auth flow:**
|
||||
1. User logs in (password or API token) → `_load_user_permissions()` aggregates
|
||||
@@ -177,6 +181,7 @@ All under `/v1/api/admin/` (requires `approve` scope + granular permission).
|
||||
| Orgs | 3 (list, get, update) | `admin.orgs` |
|
||||
| Tool Policies | 4 (CRUD) | `admin.policies` |
|
||||
| Skills | 4 (CRUD) | `admin.skills` |
|
||||
| Personas | 4 (list, create, get, edit/archive) | `persona.read` / `persona.create` / `persona.write` |
|
||||
| Schedules | 6 (CRUD + runs) | `admin.schedules` |
|
||||
| Watches | 3 (list, create, cancel) | `admin.watches` |
|
||||
| Usage | 1 (aggregated query) | `admin.usage` |
|
||||
@@ -222,7 +227,7 @@ Both Python and TypeScript console SDKs expose governance methods:
|
||||
- **Privilege escalation prevented**: `admin_assign_role` blocks self-assignment
|
||||
and requires caller to hold a superset of the target role's permissions
|
||||
- **Permission validation**: Role create/update validates permissions against
|
||||
a 15-item allowlist (`_VALID_PERMISSIONS`)
|
||||
the permission allowlist (`_VALID_PERMISSIONS`)
|
||||
- **Self-deletion blocked**: `admin_delete_user` rejects attempts to delete
|
||||
your own account (matching the self-assignment guard on role endpoints)
|
||||
- **Field allowlists**: Storage `update_*` methods filter fields against
|
||||
|
||||
@@ -75,6 +75,11 @@ This means the model always has its most relevant memories available without
|
||||
explicit recall -- but can still use `memory(action='search')` for deeper
|
||||
lookup.
|
||||
|
||||
The persona memory lever gates this pathway: a workstream whose persona
|
||||
turns memory off receives no relevance injection at all -- the steps
|
||||
above run only when memory is enabled for the session. See
|
||||
[Personas](personas.md).
|
||||
|
||||
### Nudges
|
||||
|
||||
The metacognition layer can nudge the model to save memories at appropriate
|
||||
|
||||
@@ -41,6 +41,7 @@ are set.
|
||||
| `TURNSTONE_OIDC_PASSWORD_ENABLED` | No | `true` | Set to `false` to hide the password form and block all username/password logins (including admin). API tokens continue to work. |
|
||||
| `TURNSTONE_OIDC_REDIRECT_BASE` | Yes | — | Externally-reachable origin for the OIDC redirect URI (e.g. `https://app.example.com`). Without this, OIDC will refuse to start. The previous Host-header fallback was unsafe under permissive reverse proxies. |
|
||||
| `TURNSTONE_OIDC_TRUSTED_ENDPOINT_HOSTS` | No | — | Comma-separated list of additional hostnames whose endpoints the IdP discovery document is allowed to reference. See [Cross-host endpoints](#cross-host-endpoints). |
|
||||
| `TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK` | No | `false` | Allow the issuer (and its discovered endpoints) to resolve to private/internal addresses — needed for a self-hosted IdP on an internal network. See [Self-hosted and internal IdPs](#self-hosted-and-internal-idps). |
|
||||
|
||||
All four required fields — issuer, client ID, client secret, and
|
||||
`TURNSTONE_OIDC_REDIRECT_BASE` — must be set. If any are missing OIDC
|
||||
@@ -99,6 +100,40 @@ The same scheme / no-userinfo / SSRF rules apply to allow-listed hosts —
|
||||
this knob only relaxes the same-origin check, not the security gates.
|
||||
Each entry is a hostname (no scheme, no path).
|
||||
|
||||
### Self-hosted and internal IdPs
|
||||
|
||||
By default Turnstone refuses an issuer whose hostname resolves to a
|
||||
private or internal address:
|
||||
|
||||
```
|
||||
OIDCError: endpoint URL resolves to non-public address (10.0.0.5): https://auth.example.site
|
||||
```
|
||||
|
||||
This is SSRF hardening, not a licensing or product restriction: the OIDC
|
||||
flow makes server-side HTTP requests (discovery, JWKS, token exchange),
|
||||
and refusing non-public destinations keeps a mistyped or maliciously
|
||||
steered issuer from aiming those fetches at internal services. For a
|
||||
self-hosted IdP (Keycloak, Authentik, Dex, …) on a private network,
|
||||
opt in explicitly in `config.toml`:
|
||||
|
||||
```toml
|
||||
[oidc]
|
||||
allow_private_network = true
|
||||
```
|
||||
|
||||
or via `TURNSTONE_OIDC_ALLOW_PRIVATE_NETWORK=true` (the env var wins
|
||||
when both are set).
|
||||
|
||||
The opt-in admits private-range (RFC 1918), unique-local, CGNAT
|
||||
(100.64/10 — tailnets), and loopback addresses. Link-local, multicast,
|
||||
and reserved ranges stay refused even with the opt-in — cloud metadata
|
||||
services (169.254.169.254) live there, and no legitimate IdP does. The
|
||||
HTTPS requirement and the same-origin endpoint checks are unaffected.
|
||||
|
||||
This knob only affects the login-flow IdP configured here. OAuth
|
||||
endpoints advertised by remote MCP servers are untrusted input and are
|
||||
always held to the strict public-address rule.
|
||||
|
||||
### config.toml alternative
|
||||
|
||||
```toml
|
||||
@@ -111,6 +146,8 @@ provider_name = "Google"
|
||||
role_claim = "groups"
|
||||
password_enabled = true
|
||||
redirect_base = "https://app.example.com"
|
||||
# Self-hosted IdP on an internal network (see "Self-hosted and internal IdPs")
|
||||
allow_private_network = false
|
||||
|
||||
[oidc.role_map]
|
||||
admin = "builtin-admin"
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
# Personas
|
||||
|
||||
A **persona** is a named, reusable bundle attached to a workstream **at
|
||||
creation** that controls how its system message is composed and what
|
||||
capability envelope it runs with. Personas answer a recurring operational
|
||||
complaint: the default composition primes every session for heavy tool use,
|
||||
and there was no per-workstream dial to launch a "just write prose" or
|
||||
"evidence-first research" session.
|
||||
|
||||
A persona is exactly four levers — no more:
|
||||
|
||||
| Lever | What it does |
|
||||
|---|---|
|
||||
| **Base prompt** | Replaces the BASE module of the composed system message. *Only* BASE: ENV, CONTEXT, TOOLS, and POLICIES keep composing, so mandatory [prompt policies](governance.md) ride on top of every persona. Built-in personas source their prose from a repo file; operator personas store it inline — see [Where persona prompts live](#where-persona-prompts-live). |
|
||||
| **Tool visibility** | Which tools the session advertises. Tri-state: *unrestricted* (tracks tool growth and MCP catalogs), *no tools* (the TOOLS prompt block self-suppresses and zero definitions go on the wire), or an *exact set* of names. Including `tool_search` in a set makes it **soft** — tools the model discovers through search join the visible set; omitting it makes the set **hard** (the search pathway is disabled entirely). On commercial providers a soft set costs one prompt-cache re-prime per `tool_search` expansion, since each expansion rewrites the wire tool set and recomposes the prompt. |
|
||||
| **MCP** | Whether the workstream talks to MCP at all. **Session-wide**: off means no MCP tools for the persona's own hands *or* for in-process task agents, no resource/prompt catalogs, and no listener registrations. This lever expresses infrastructure intent, not behavior shaping. |
|
||||
| **Memory** | Whether the persona's **own hands** get memory: recalled-memory injection into the prompt, memory-directed metacognitive nudges, and the `memory` tool. Task agents keep their own envelope, and compaction spill/markers are session mechanics that are never persona-gated. An exact tool set that hides `memory` also mutes those nudges, and the compaction-resume pointer follows `recall`'s visibility. |
|
||||
|
||||
Visibility is behavior shaping, **not** a security boundary: any tool call
|
||||
that does reach the wire still clears the same approval, judge, and policy
|
||||
machinery as always. RBAC and tool policies remain the enforcement layers.
|
||||
|
||||
## Snapshot semantics — resolve once, stamp forever
|
||||
|
||||
The persona is resolved **once**, at workstream creation, and stamped into
|
||||
`workstream_config` as five keys (`persona`, `persona_prompt`,
|
||||
`persona_tools`, `persona_mcp`, `persona_memory`). From then on the session
|
||||
reads only the stamp:
|
||||
|
||||
- **Editing or archiving a persona never changes an existing workstream.**
|
||||
Rehydrate, resume, and post-compaction resume all run from the stamp.
|
||||
A mid-session REPL `/resume` adopts the target workstream's stamp for
|
||||
prompt, tools, and memory; for the MCP lever it can only narrow in
|
||||
place — adopting an MCP-off stamp drops the live MCP surface, while
|
||||
adopting an MCP-on stamp into a session whose persona dropped MCP at
|
||||
construction is refused with an error telling you to reopen the
|
||||
workstream fresh.
|
||||
- A workstream outlives its persona — an archived persona keeps labelling
|
||||
the workstreams stamped with it.
|
||||
- A partial or unparseable stamp is treated as corruption: session
|
||||
construction fails loudly rather than silently falling back to a default
|
||||
envelope the operator never chose.
|
||||
- Workstreams created before personas existed carry no stamp and keep
|
||||
legacy behavior, byte-identical to the `engineer` / `orchestrator`
|
||||
defaults below — with one exception: pre-1.7 workstreams that had
|
||||
`creative_mode` set are converted by migration `063` into full
|
||||
`writer` stamps, so they resume as writing sessions rather than as
|
||||
legacy defaults.
|
||||
- Forking (`resume_ws` on create) resumes the source's stamped persona; the
|
||||
fork does not re-resolve.
|
||||
|
||||
## Seed personas
|
||||
|
||||
Migration `063` seeds six personas. The two per-kind **defaults** carry no
|
||||
overrides at all, so a zero-touch launch behaves exactly as it did before
|
||||
personas existed:
|
||||
|
||||
| Persona | Kind | Base prompt | Tools | MCP | Memory |
|
||||
|---|---|---|---|---|---|
|
||||
| `engineer` *(default)* | interactive | stock | unrestricted | on | on |
|
||||
| `orchestrator` *(default)* | coordinator | stock | unrestricted | on | on |
|
||||
| `scribe` | interactive | custom (faithful structuring of given material) | none | off | off |
|
||||
| `researcher` | interactive | custom (evidence-first) | `read_file`, `search`, `web_fetch`, `web_search`, `recall`, `memory`, `tool_search` (soft) | off | on |
|
||||
| `writer` | interactive | custom (creative writing partner — replaces the removed `/creative`) | none | off | on |
|
||||
| `executive` | coordinator | custom (delegate, interrogate plans, judge outcomes) | spawn/inspect/lifecycle tools plus `memory`: `spawn_workstream`, `spawn_batch`, `send_to_workstream`, `wait_for_workstream`, `inspect_workstream`, `list_workstreams`, `list_nodes`, `close_workstream`, `cancel_workstream`, `memory` (hard) | off | on |
|
||||
|
||||
Notes:
|
||||
|
||||
- `scribe` turns memory off deliberately: recalled memories would
|
||||
contaminate faithful summarization with unrelated context.
|
||||
- `researcher`'s set is soft (includes `tool_search`): it starts with
|
||||
read and evidence tools but can pull in others on demand — e.g. load
|
||||
`bash` to run a snippet and verify a calculation. It is evidence-first,
|
||||
not sandboxed; any escalated tool still hits the normal approval path.
|
||||
- Coordinator sessions do not merge MCP today, so the MCP lever on
|
||||
coordinator personas is forward-compatible bookkeeping; it bites on
|
||||
interactive workstreams.
|
||||
|
||||
## Where persona prompts live
|
||||
|
||||
Prompt source is explicit in the persona row — two nullable columns, never both empty:
|
||||
|
||||
| `base_prompt_file` | `base_prompt` | Meaning |
|
||||
|---|---|---|
|
||||
| set (e.g. `scribe.md`) | — | **built-in**: prose lives in `prompts/personas/<file>`, code-owned and PR-reviewed |
|
||||
| set | set | built-in with an **operator override** layered on top (the inline text wins) |
|
||||
| — | set | **operator** persona, inline prose |
|
||||
|
||||
A `CHECK` forbids the both-empty row, so resolution is a plain coalesce —
|
||||
`base_prompt ?? load(base_prompt_file)` — with no implicit "inherit the default"
|
||||
branch in application logic. `base_prompt_file` is set only by the migration/code
|
||||
(the admin API never exposes it): it marks a persona as built-in and blocks
|
||||
archive, so `engineer` and `orchestrator` can't be removed. To customise a
|
||||
built-in, set `base_prompt` on it (clear it to revert), or create your own persona.
|
||||
|
||||
The resolved prompt is **frozen into the workstream at creation** — later edits to
|
||||
a built-in's file or an operator's row never change a running workstream; only new
|
||||
ones pick up the change. "No persona" is not a state: every workstream is stamped,
|
||||
and an empty `persona=` resolves to the kind's `is_default` (`engineer` /
|
||||
`orchestrator`).
|
||||
|
||||
## Choosing a persona
|
||||
|
||||
Every creation surface takes an optional persona; empty always means the
|
||||
kind's default (or plain legacy behavior on a database with no personas
|
||||
seeded):
|
||||
|
||||
- **Web/console**: the persona select on the console launcher, the server
|
||||
webui's new-workstream dialog, and the dashboard composer. Selecting a
|
||||
persona requires **no** `persona.*` permission — the picker feed
|
||||
(`GET /v1/api/personas`) is authenticated-only and returns display fields.
|
||||
- **API/SDK**: `CreateWorkstreamRequest.persona` (Python:
|
||||
`create_workstream(persona=...)`; TypeScript: `{ persona: ... }`).
|
||||
- **CLI**: `turnstone --persona <name>`. Unknown or disabled names error at
|
||||
startup. `--resume` ignores `--persona` and adopts the resumed
|
||||
workstream's stamp.
|
||||
- **Coordinator spawn**: `spawn_workstream` / `spawn_batch` take a
|
||||
`persona` argument, validated when the coordinator prepares the spawn
|
||||
and re-checked by the node that creates the child (children are always
|
||||
interactive-kind). Omitted means the interactive **default** — a child
|
||||
never inherits its parent coordinator's persona.
|
||||
- **Sub-agents**: `task_agent` takes a `persona` argument setting the
|
||||
sub-agent's identity and capability envelope (resolved against
|
||||
interactive-kind personas, frozen into the task at prep). Omitted keeps
|
||||
the default autonomous task-agent identity — never the parent's persona.
|
||||
|
||||
## How agents discover personas
|
||||
|
||||
Agents are told, not expected to guess: the live persona list (enabled,
|
||||
interactive-kind — children and sub-agents are always interactive) is
|
||||
injected into the `persona` parameter description of `task_agent`,
|
||||
`spawn_workstream`, and `spawn_batch` whenever the session's tool surface
|
||||
is rendered — session start, MCP catalog change, model-registry reload.
|
||||
Each entry carries the name, the default marker, and the persona's
|
||||
one-line description so the model can pick by purpose (descriptions drop
|
||||
out past 25 personas; the name list always enumerates completely).
|
||||
|
||||
A persona created after that render is still reachable — pass its name.
|
||||
Every resolve failure enumerates the names currently valid for the kind,
|
||||
so a stale list (or a typo) self-corrects on the next attempt.
|
||||
|
||||
Resolution is forgiving on all surfaces (they share one rule):
|
||||
|
||||
- names match case-insensitively (`Writer` resolves `writer`);
|
||||
- an input that uniquely matches a persona's **display name**
|
||||
(case-insensitive, among the kind's enabled personas — display names are
|
||||
not unique, and a same-label persona of another kind neither blocks nor
|
||||
wins) resolves to that persona; an ambiguous match errors, listing the
|
||||
candidate slugs;
|
||||
- whatever variant matched, the stamped identity, approval chrome, and
|
||||
wire always carry the canonical `name` slug.
|
||||
|
||||
## Authoring (console)
|
||||
|
||||
Personas are managed in the console's **Manage → Governance → Personas**
|
||||
tab. The admin shelf exposes exactly the four levers plus the kind
|
||||
list, the default marker, and archive. Rules:
|
||||
|
||||
- `name` is an immutable lowercase slug — and the identifier agents and
|
||||
the CLI launch the persona by (`persona=` on the spawn tools,
|
||||
`--persona` on the CLI); the create shelf says so under **Name**.
|
||||
`display_name` is a list label, editable any time, and deliberately
|
||||
not an identifier (a unique display name happens to resolve, as a
|
||||
forgiveness fallback — don't design workflows around it).
|
||||
- Exactly one default per kind, storage-enforced: flipping the flag on a
|
||||
successor demotes the incumbent atomically, defaults are single-kind,
|
||||
and a default cannot be archived.
|
||||
- **Archive only** — there is no delete verb, so every stamped
|
||||
workstream's provenance stays explicable.
|
||||
|
||||
RBAC: `persona.create` / `persona.read` / `persona.write` gate the admin
|
||||
CRUD (`/v1/api/admin/personas`); all three are granted to `builtin-admin`
|
||||
by migration `063`, and other roles opt in via role permission overrides.
|
||||
+2
-2
@@ -69,7 +69,7 @@ Both `TurnstoneServer` (sync) and `AsyncTurnstoneServer` (async) expose:
|
||||
|----------|--------|---------|
|
||||
| **Workstreams** | `list_workstreams()` | `ListWorkstreamsResponse` |
|
||||
| | `dashboard()` | `DashboardResponse` |
|
||||
| | `create_workstream(*, name, model, auto_approve, skill, initial_message, attachments)` | `CreateWorkstreamResponse` |
|
||||
| | `create_workstream(*, name, model, auto_approve, skill, persona, initial_message, attachments)` | `CreateWorkstreamResponse` |
|
||||
| | `close_workstream(ws_id)` | `StatusResponse` |
|
||||
| **Attachments** | `upload_attachment(ws_id, filename, data, *, mime_type=...)` | `UploadAttachmentResponse` |
|
||||
| | `list_attachments(ws_id)` | `ListAttachmentsResponse` |
|
||||
@@ -100,7 +100,7 @@ Both `TurnstoneConsole` (sync) and `AsyncTurnstoneConsole` (async) expose:
|
||||
| | `workstreams(*, state, node, search, sort, page, per_page)` | `ClusterWorkstreamsResponse` |
|
||||
| | `node_detail(node_id)` | `NodeDetailResponse` |
|
||||
| | `snapshot()` | `ClusterSnapshotResponse` |
|
||||
| | `create_workstream(*, node_id, name, model, initial_message, skill)` | `ConsoleCreateWsResponse` |
|
||||
| | `create_workstream(*, node_id, name, model, initial_message, skill, persona)` | `ConsoleCreateWsResponse` |
|
||||
| **Schedules** | `list_schedules()` | `ListSchedulesResponse` |
|
||||
| | `create_schedule(*, name, schedule_type, initial_message, ...)` | `ScheduleInfo` |
|
||||
| | `get_schedule(task_id)` | `ScheduleInfo` |
|
||||
|
||||
+49
-6
@@ -1,6 +1,6 @@
|
||||
# Tools Reference
|
||||
|
||||
turnstone exposes 16 built-in tools plus any number of external MCP tools to the
|
||||
turnstone exposes 17 built-in tools plus any number of external MCP tools to the
|
||||
LLM via the OpenAI function-calling interface. Built-in tools are defined as JSON
|
||||
files under `turnstone/tools/` and loaded at startup by `turnstone/core/tools.py`.
|
||||
MCP tools are discovered from configured MCP servers at startup by
|
||||
@@ -44,10 +44,10 @@ schema plus turnstone-specific metadata keys:
|
||||
|
||||
| Name | Description |
|
||||
|---------------------|-------------|
|
||||
| `TOOLS` | All 28 loaded built-in tool definitions (interactive + coordinator union). Sessions send a kind-specific subset (`INTERACTIVE_TOOLS` or `COORDINATOR_TOOLS`). |
|
||||
| `TOOLS` | All 29 loaded built-in tool definitions (interactive + coordinator union). Sessions send a kind-specific subset (`INTERACTIVE_TOOLS` or `COORDINATOR_TOOLS`). |
|
||||
| `TASK_AGENT_TOOLS` | Tools with `task_agent: true` -- available to task sub-agents. Includes write operations. |
|
||||
| `TASK_AUTO_TOOLS` | Set of all tool names with `auto_approve: true` -- used by task-agent sub-sessions to skip confirmation for matching available tools. |
|
||||
| `BUILTIN_TOOL_NAMES`| Frozenset of all 28 built-in tool names (interactive + coordinator union). Used by tool search to distinguish always-on tools from deferrable MCP tools. |
|
||||
| `BUILTIN_TOOL_NAMES`| Frozenset of all 29 built-in tool names (interactive + coordinator union). Used by tool search to distinguish always-on tools from deferrable MCP tools. |
|
||||
| `PRIMARY_KEY_MAP` | Dict mapping tool name to its `primary_key` parameter name. |
|
||||
|
||||
---
|
||||
@@ -65,7 +65,7 @@ Tool execution follows a three-phase pipeline inside `ChatSession._execute_tools
|
||||
- Parses the JSON arguments (with fallback for malformed JSON).
|
||||
- If JSON parsing fails entirely, uses `PRIMARY_KEY_MAP` to map a bare string
|
||||
to the correct parameter.
|
||||
- Dispatches to the matching `_prepare_{func_name}()` handler. There are 16
|
||||
- Dispatches to the matching `_prepare_{func_name}()` handler. There are 17
|
||||
built-in tools plus `tool_search` (synthetic, client-side BM25 fallback) and
|
||||
the generic `_prepare_mcp_tool()` handler for MCP tools.
|
||||
- Validates arguments and builds a preview dict containing:
|
||||
@@ -125,6 +125,9 @@ Each item's `execute` callable is invoked:
|
||||
- `web_fetch` -- fetches a URL (SSRF-protected, but makes network requests)
|
||||
- `web_search` -- web search via self-hosted SearxNG (makes network requests)
|
||||
- `task_agent` -- spawns an autonomous sub-agent
|
||||
- `open_preview` -- **URL targets only** (network access, gated like `web_fetch`);
|
||||
file-path and `attachment:` targets are local reads and run unprompted like
|
||||
`read_file`
|
||||
|
||||
Note: The JSON schema metadata key `auto_approve` controls membership in
|
||||
`TASK_AUTO_TOOLS` (used for task agent sub-sessions). The actual runtime
|
||||
@@ -157,6 +160,7 @@ Every tool defines a `primary_key`. The mapping is:
|
||||
| `search` | `query` |
|
||||
| `web_fetch` | `url` |
|
||||
| `web_search` | `query` |
|
||||
| `open_preview` | `target` |
|
||||
| `task_agent` | `prompt` |
|
||||
| `memory` | `name` |
|
||||
| `recall` | `query` |
|
||||
@@ -285,7 +289,7 @@ Fetch a URL and extract specific information from it.
|
||||
| `url` | string | yes | The URL to fetch (must start with `http://` or `https://`). |
|
||||
| `question` | string | yes | What to extract or answer from the page content. |
|
||||
|
||||
- **What it does**: Fetches the URL, strips HTML to plain text, and uses the LLM to extract the answer to the question from the page content. Protected against SSRF (blocks private/internal IPs).
|
||||
- **What it does**: Fetches the URL, strips HTML to plain text, and uses the LLM to extract the answer to the question from the page content. Every redirect hop is SSRF-screened before it is requested. Private/internal addresses are refused by default; enable `tools.allow_private_network` (console Settings → Tools) to make them approvable for self-hosted setups whose services live on the local network — the approval prompt marks such requests, and a public site redirecting into private space is refused regardless.
|
||||
- **Auto-approve**: No -- requires user confirmation (makes network requests).
|
||||
- **Agent availability**: `task_agent`.
|
||||
|
||||
@@ -345,6 +349,39 @@ It reports the score scale, whether the endpoint cleanly separates relevant from
|
||||
|
||||
---
|
||||
|
||||
### open_preview
|
||||
|
||||
Show the user rich content in a preview pane beside the conversation.
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|--------|----------|-------------|
|
||||
| `target` | string | yes | An http(s) URL, a file path, or `attachment:<id>` for a file attached to the conversation. |
|
||||
| `kind` | string | no | Rendering override: `web`, `pdf`, `image`, `table`, `text`, or `markdown`. Detected from the content when omitted. |
|
||||
| `title` | string | no | Pane header title. Defaults to the page title, filename, or URL. |
|
||||
|
||||
- **What it does**: Resolves the target to bytes (URLs fetch through the same
|
||||
SSRF-guarded path as `web_fetch`, screened per redirect hop, honoring the
|
||||
same `tools.allow_private_network` opt-in), classifies the
|
||||
content, stores it content-addressed against the workstream, and opens the
|
||||
frontend preview pane beside the conversation: web pages render in a fully
|
||||
sandboxed iframe (no scripts, opaque origin), PDFs in the browser viewer,
|
||||
images inline, CSV/TSV/JSON as a sortable table, text/markdown rendered. A
|
||||
previewed web page loads none of its remote images or styles by default, so
|
||||
opening it never reveals the viewer to the page's site; a toggle in the pane
|
||||
header turns remote content back on for that preview. The
|
||||
model receives only a one-line confirmation — to reason about content, use
|
||||
`web_fetch` / `read_file` instead. Preview content is size-capped per kind
|
||||
(pages 4 MB, PDFs 32 MB, images 4 MB, tables 2 MB, text 512 KB) and GC'd
|
||||
with the workstream.
|
||||
- **Auto-approve**: URL targets require confirmation (network access); file
|
||||
paths and `attachment:` targets run unprompted (local reads).
|
||||
- **Agent availability**: interactive sessions only (not `task_agent`, not
|
||||
coordinators).
|
||||
- **Surfaces**: the pane renders in the web UI (standalone and console). The
|
||||
CLI prints the confirmation line only — there is no terminal pane.
|
||||
|
||||
---
|
||||
|
||||
## Agent
|
||||
|
||||
The tool name uses the `_agent` suffix — bare `task` collides with
|
||||
@@ -545,6 +582,7 @@ pre-configure skills at workstream creation.
|
||||
| `search` | File Ops | Yes | Yes | `query` |
|
||||
| `web_fetch` | Info | No | Yes | `url` |
|
||||
| `web_search` | Info | No | Yes | `query` |
|
||||
| `open_preview`| Info | URL: no; path/attachment: yes | No | `target` |
|
||||
| `task_agent` | Agent | No | No | `prompt` |
|
||||
| `memory` | Memory | Yes | No | `name` |
|
||||
| `recall` | Memory | Yes | No | `query` |
|
||||
@@ -580,6 +618,11 @@ Tool search uses the best available mechanism for each provider:
|
||||
`_exec_tool_search()` runs a pure-Python BM25 index over tool names and
|
||||
descriptions, then expands the matched tools into the visible set.
|
||||
|
||||
A persona with a tool-visibility set overrides this selection: any exact
|
||||
set forces tool search into the client-side BM25 mechanism (tier 3)
|
||||
regardless of provider, and a **hard** set — one whose visible tools omit
|
||||
`tool_search` — disables tool search entirely.
|
||||
|
||||
### Configuration
|
||||
|
||||
Tool search is configured in `config.toml` under the `[tools]` section:
|
||||
@@ -649,7 +692,7 @@ MCP-compatible service.
|
||||
3. **Schema conversion**: Each MCP tool's `inputSchema` is converted to OpenAI
|
||||
function-calling format. The tool name is prefixed: `mcp__{server}__{tool}`.
|
||||
|
||||
4. **Merging**: MCP tools are appended after the 16 built-in tools via
|
||||
4. **Merging**: MCP tools are appended after the 17 built-in tools via
|
||||
`merge_mcp_tools()`. Built-in tools appear first, giving them natural LLM priority.
|
||||
When dynamic tool search is active, MCP tools are deferred rather than directly
|
||||
visible -- the model discovers them via search as needed (see
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
{
|
||||
"defaults": {
|
||||
"n_runs": 3
|
||||
},
|
||||
"cases": [
|
||||
{
|
||||
"id": "search-first",
|
||||
"skill": {
|
||||
"name": "search-first",
|
||||
"content": "# Search First\n\nBefore answering ANY question about where something lives in the codebase, you MUST call the `search` tool first. Never answer from memory."
|
||||
},
|
||||
"user_prompt": "Where is JWT token validation implemented in this project?",
|
||||
"expected_actions": [{ "tool": "search" }],
|
||||
"match_mode": "ordered_subset",
|
||||
"max_turns": 4
|
||||
},
|
||||
{
|
||||
"id": "test-after-edit",
|
||||
"skill": {
|
||||
"name": "test-after-edit",
|
||||
"content": "# Test After Edit\n\nAfter editing or writing ANY file, you MUST run the test suite with `python -m pytest` via bash before you finish. Do not report done until tests have run."
|
||||
},
|
||||
"user_prompt": "Add a function `clamp(x, lo, hi)` that clamps x to [lo, hi] in utils.py.",
|
||||
"setup": {
|
||||
"files": {
|
||||
"utils.py": ""
|
||||
}
|
||||
},
|
||||
"expected_actions": [
|
||||
{ "tool": "write_file" },
|
||||
{ "tool": "bash", "args_pattern": { "command": "pytest" } }
|
||||
],
|
||||
"match_mode": "ordered_subset",
|
||||
"max_turns": 8
|
||||
},
|
||||
{
|
||||
"id": "changelog-update",
|
||||
"skill": {
|
||||
"name": "changelog-update",
|
||||
"content": "# Changelog Discipline\n\nWhenever you modify a file, you MUST also append a one-line entry to CHANGELOG.md describing the change in the same task."
|
||||
},
|
||||
"user_prompt": "Fix the off-by-one so pager.py shows the last page. Edit pager.py.",
|
||||
"setup": {
|
||||
"files": {
|
||||
"pager.py": "def last_page(total_items, per_page):\n # off-by-one: drops the final partial page\n return total_items // per_page\n",
|
||||
"CHANGELOG.md": "# Changelog\n"
|
||||
}
|
||||
},
|
||||
"expected_actions": [
|
||||
{ "tool": "edit_file", "args_pattern": { "path": "CHANGELOG.md" } }
|
||||
],
|
||||
"match_mode": "subset",
|
||||
"max_turns": 8
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -161,6 +161,35 @@ def test_rest_heals_and_charges(tmp_path: Path, clock: object) -> None:
|
||||
assert "full health" in out.lower()
|
||||
|
||||
|
||||
def test_rest_when_spent_restores_a_fresh_days_turns(tmp_path: Path, clock: object) -> None:
|
||||
"""Sleeping at the inn with no turns left rolls into a fresh day's allowance."""
|
||||
game = _game(tmp_path, clock)
|
||||
game.join("Brandr")
|
||||
player = game.players["Brandr"]
|
||||
daily = game.world.settings.daily_turns
|
||||
player.turns_left = 0 # spent for the day
|
||||
player.hp = 5
|
||||
game.move("Brandr", "", "west", 2) # step into the inn
|
||||
out = game.action("Brandr", "rest", "", "")
|
||||
assert player.turns_left == daily # a fresh day's turns restored
|
||||
assert player.hp == player.max_hp # and fully mended
|
||||
assert f"/{daily} ]" in out # footer reflects the refreshed budget
|
||||
|
||||
|
||||
def test_rest_with_turns_in_hand_never_inflates_the_budget(tmp_path: Path, clock: object) -> None:
|
||||
"""Resting mid-day mends but adds no turns — the top-up only fires at zero."""
|
||||
game = _game(tmp_path, clock)
|
||||
game.join("Brandr")
|
||||
player = game.players["Brandr"]
|
||||
daily = game.world.settings.daily_turns
|
||||
player.turns_left = daily - 3 # turns still in hand
|
||||
player.hp = 5
|
||||
game.move("Brandr", "", "west", 2) # step into the inn
|
||||
game.action("Brandr", "rest", "", "")
|
||||
assert player.turns_left == daily - 3 # unchanged: no farming past the cap
|
||||
assert player.hp == player.max_hp # but the heal still lands
|
||||
|
||||
|
||||
def test_fight_spends_a_turn_and_credits(tmp_path: Path, clock: object) -> None:
|
||||
game = _game(tmp_path, clock)
|
||||
game.join("Brandr")
|
||||
|
||||
@@ -1015,16 +1015,28 @@ class Game:
|
||||
return self._overworld_frame(player, lines=["You step back out into the open air."])
|
||||
|
||||
def _rest(self, player: Player) -> str:
|
||||
# Settle any pending day rollover first, so a rest taken as the first act
|
||||
# of a new day is the ordinary refresh, not the spent-turns top-up below.
|
||||
self._ensure_day(player)
|
||||
cost = self.world.settings.rest_cost
|
||||
if leveling.rest(player, cost):
|
||||
# A night's rest is a private errand, not Herald news; persist only.
|
||||
self._persist(player)
|
||||
if not leveling.rest(player, cost):
|
||||
return self._location_menu(
|
||||
player, lines=[f"You sleep deeply and wake at full health. (-{cost} gold)"]
|
||||
player, lines=[f"You can't afford the {cost}-gold bed. (You have {player.gold}.)"]
|
||||
)
|
||||
return self._location_menu(
|
||||
player, lines=[f"You can't afford the {cost}-gold bed. (You have {player.gold}.)"]
|
||||
)
|
||||
# A night at the inn always mends. Once the day's turns are spent it also
|
||||
# rolls the sleeper into a fresh day's allowance, so a spent adventurer can
|
||||
# press on rather than idling until the dawn rollover. The top-up only
|
||||
# fires at zero, so it never banks turns past the daily cap.
|
||||
line = f"You sleep deeply and wake at full health. (-{cost} gold)"
|
||||
if player.turns_left <= 0:
|
||||
player.turns_left = self.world.settings.daily_turns
|
||||
line = (
|
||||
f"You sleep through to a new dawn, waking at full health "
|
||||
f"and ready to venture out anew. (-{cost} gold)"
|
||||
)
|
||||
# A night's rest is a private errand, not Herald news; persist only.
|
||||
self._persist(player)
|
||||
return self._location_menu(player, lines=[line])
|
||||
|
||||
# -- the Vault: bank gold at the inn (safe from ambush) --------------
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@
|
||||
"tier": 2,
|
||||
"name": "Goblin",
|
||||
"hp": 12,
|
||||
"atk": 5,
|
||||
"atk": 4,
|
||||
"def": 1,
|
||||
"xp": 18,
|
||||
"gold": 7
|
||||
@@ -30,7 +30,7 @@
|
||||
"tier": 2,
|
||||
"name": "Bandit Scout",
|
||||
"hp": 14,
|
||||
"atk": 6,
|
||||
"atk": 5,
|
||||
"def": 1,
|
||||
"xp": 20,
|
||||
"gold": 9
|
||||
@@ -50,7 +50,7 @@
|
||||
"tier": 3,
|
||||
"name": "Forest Wolf",
|
||||
"hp": 20,
|
||||
"atk": 8,
|
||||
"atk": 7,
|
||||
"def": 2,
|
||||
"xp": 35,
|
||||
"gold": 14
|
||||
@@ -59,7 +59,7 @@
|
||||
"tier": 3,
|
||||
"name": "Bog Stalker",
|
||||
"hp": 22,
|
||||
"atk": 9,
|
||||
"atk": 8,
|
||||
"def": 2,
|
||||
"xp": 38,
|
||||
"gold": 16
|
||||
@@ -79,7 +79,7 @@
|
||||
"tier": 4,
|
||||
"name": "Cave Troll",
|
||||
"hp": 38,
|
||||
"atk": 12,
|
||||
"atk": 11,
|
||||
"def": 4,
|
||||
"xp": 70,
|
||||
"gold": 30
|
||||
@@ -88,7 +88,7 @@
|
||||
"tier": 4,
|
||||
"name": "Barrow Wight",
|
||||
"hp": 35,
|
||||
"atk": 13,
|
||||
"atk": 12,
|
||||
"def": 4,
|
||||
"xp": 65,
|
||||
"gold": 28
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
},
|
||||
"=": {
|
||||
"key": "road",
|
||||
"glyph": "=",
|
||||
"glyph": "▒",
|
||||
"walkable": true,
|
||||
"encounter_rate": 0.02,
|
||||
"color": "road"
|
||||
|
||||
@@ -100,8 +100,8 @@
|
||||
"settings": {
|
||||
"daily_turns": 10,
|
||||
"rest_cost": 15,
|
||||
"heal_cost_per_hp": 2,
|
||||
"starting_gold": 20,
|
||||
"heal_cost_per_hp": 1,
|
||||
"starting_gold": 37,
|
||||
"starting_weapon": "rusty_dagger",
|
||||
"starting_armor": "cloth_tunic",
|
||||
"start_hp": 20,
|
||||
|
||||
@@ -21,7 +21,7 @@
|
||||
"tier": 2,
|
||||
"name": "Ember Imp",
|
||||
"hp": 12,
|
||||
"atk": 5,
|
||||
"atk": 4,
|
||||
"def": 1,
|
||||
"xp": 18,
|
||||
"gold": 7
|
||||
@@ -30,7 +30,7 @@
|
||||
"tier": 2,
|
||||
"name": "Slag Scuttler",
|
||||
"hp": 14,
|
||||
"atk": 6,
|
||||
"atk": 5,
|
||||
"def": 1,
|
||||
"xp": 20,
|
||||
"gold": 9
|
||||
@@ -50,7 +50,7 @@
|
||||
"tier": 3,
|
||||
"name": "Magma Hound",
|
||||
"hp": 20,
|
||||
"atk": 8,
|
||||
"atk": 7,
|
||||
"def": 2,
|
||||
"xp": 35,
|
||||
"gold": 14
|
||||
@@ -59,7 +59,7 @@
|
||||
"tier": 3,
|
||||
"name": "Obsidian Lurker",
|
||||
"hp": 22,
|
||||
"atk": 9,
|
||||
"atk": 8,
|
||||
"def": 2,
|
||||
"xp": 38,
|
||||
"gold": 16
|
||||
@@ -79,7 +79,7 @@
|
||||
"tier": 4,
|
||||
"name": "Basalt Golem",
|
||||
"hp": 38,
|
||||
"atk": 12,
|
||||
"atk": 11,
|
||||
"def": 4,
|
||||
"xp": 70,
|
||||
"gold": 30
|
||||
@@ -88,7 +88,7 @@
|
||||
"tier": 4,
|
||||
"name": "Ashen Wraith",
|
||||
"hp": 35,
|
||||
"atk": 13,
|
||||
"atk": 12,
|
||||
"def": 4,
|
||||
"xp": 65,
|
||||
"gold": 28
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
},
|
||||
"=": {
|
||||
"key": "basalt",
|
||||
"glyph": "=",
|
||||
"glyph": "▩",
|
||||
"walkable": true,
|
||||
"encounter_rate": 0.02,
|
||||
"color": "road"
|
||||
|
||||
@@ -100,8 +100,8 @@
|
||||
"settings": {
|
||||
"daily_turns": 10,
|
||||
"rest_cost": 15,
|
||||
"heal_cost_per_hp": 2,
|
||||
"starting_gold": 20,
|
||||
"heal_cost_per_hp": 1,
|
||||
"starting_gold": 37,
|
||||
"starting_weapon": "charred_shiv",
|
||||
"starting_armor": "scorched_rags",
|
||||
"start_hp": 20,
|
||||
|
||||
+3
-2
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "turnstone"
|
||||
version = "1.7.0a3"
|
||||
version = "1.7.2"
|
||||
description = "Multi-node AI orchestration platform with tool use, agent routing, and cluster simulation."
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
@@ -64,7 +64,8 @@ all = ["turnstone[discord,slack]"]
|
||||
|
||||
[project.scripts]
|
||||
turnstone = "turnstone.cli:main"
|
||||
turnstone-eval = "turnstone.eval:main"
|
||||
turnstone-eval = "turnstone.eval.cli:main"
|
||||
turnstone-optimizer = "turnstone.optimizer:main"
|
||||
turnstone-server = "turnstone.server:main"
|
||||
turnstone-console = "turnstone.console.server:main"
|
||||
turnstone-admin = "turnstone.admin:main"
|
||||
|
||||
+837
-18
@@ -58,6 +58,46 @@ Attachments harness (/attachments/livepass.html): the composer attachment
|
||||
thumbnail crop/size, the native audio-control fit at the constrained
|
||||
height, the snippet contrast, and how a long filename behaves at the
|
||||
340px chip cap.
|
||||
Task-agent harness (/taskagent/livepass.html): the task_agent card — a task
|
||||
agent's sub-tool steps nested under its conversation row, driven through the
|
||||
REAL InteractivePane.handleEvent (parent tool_pending/tool_info -> child
|
||||
tool_pending/tool_result/tool_output_chunk/approve_request -> task_agent
|
||||
tool_result) so the SSE->card routing (_routeAgentItems / _ensureAgentCard,
|
||||
and appendToolOutput finding the nested row by call_id) is exercised, not
|
||||
just the leaf builders. Query flags: &theme=light; &collapsed=1 (all-auto,
|
||||
no approval -> the natural collapse-by-default state); ¶llel=1 (card in a
|
||||
2-tool batch, for the rail-bleed rules); &recall=1 (the RECALL path —
|
||||
replayHistory rebuilding the card from a /history `agent_steps` overlay, i.e.
|
||||
a reload while the ws is in memory); &expand=1 (open every card so a shot
|
||||
shows the nested steps); &race=1 (child steps emitted BEFORE the task_agent
|
||||
row paints — the parallel-pool ordering window; the orphan buffer must nest
|
||||
them rather than let them escape to top-level); &orphan=1 (child steps whose
|
||||
task_agent row NEVER paints — the safety valve must escape them to visible
|
||||
top-level rows after the grace window, stamping TASKAGENT-ORPHANS-ESCAPED-<n>,
|
||||
not leave them buffered/invisible). document.title stamps
|
||||
TASKAGENT-READY-<steps> on
|
||||
success, TASKAGENT-FAILED-... / TASKAGENT-ERROR when routing breaks, so a
|
||||
broken card can't screenshot green.
|
||||
|
||||
Perf harness (/perf/livepass.html): long-session performance baseline for the
|
||||
interactive pane — mounts the REAL InteractivePane at real scroll geometry
|
||||
(fixed-height mount, production CSS chain) and drives production-shaped
|
||||
events through pane.handleEvent/replayHistory with rAF yields, measuring:
|
||||
replayHistory wall time at N messages, live event-storm cost per turn on top
|
||||
of that transcript (reasoning/content deltas + tool batches + task_agent
|
||||
cards), tool_output_chunk throughput, busy/idle churn, heap + node count +
|
||||
_agentCards size across repeated replay cycles (leak probe), and longtask
|
||||
counts. Query params: ?n= (history size) &turns= &chunks= &cycles= &idle=
|
||||
&post=1 (POST the JSON report to /perf/report — the --perf runner captures
|
||||
it). Results land in <pre id="perf-json"> and document.title stamps
|
||||
PERF-READY-<n> / PERF-FAILED-<phase>. MEASUREMENT RULES: never run with
|
||||
--virtual-time-budget (it corrupts performance.now) and never pass
|
||||
--force-prefers-reduced-motion (it disables the animations whose cost we
|
||||
measure); the --perf runner passes --js-flags=--expose-gc and
|
||||
--enable-precise-memory-info so heap numbers are stable and real.
|
||||
|
||||
python3 scripts/livepass.py --perf # 300 and 3000 msgs
|
||||
python3 scripts/livepass.py --perf --perf-n 5000 # match the field run
|
||||
|
||||
Rebuild after ANY markup change: the dialog blocks are embedded at build
|
||||
time. Assets are symlinked, so CSS/JS edits are live on refresh.
|
||||
@@ -66,7 +106,13 @@ time. Assets are symlinked, so CSS/JS edits are live on refresh.
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import http.server
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
@@ -767,6 +813,547 @@ ATTACH_TEMPLATE = """<!doctype html>
|
||||
"""
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Task-agent harness — the task_agent card: a task agent's sub-tool steps
|
||||
# nested under its conversation row. Driven through the REAL
|
||||
# InteractivePane.handleEvent so the SSE->card ROUTING (_routeAgentItems /
|
||||
# _ensureAgentCard, plus appendToolOutput finding the nested row by call_id)
|
||||
# is exercised, not just the leaf builders. The page frame is harness-only
|
||||
# chrome; the .conv-batch / task_agent card is what's under review.
|
||||
# --------------------------------------------------------------------------
|
||||
TASKAGENT_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
||||
<title>task_agent livepass</title>
|
||||
<link rel="stylesheet" href="shared/base.css" />
|
||||
<link rel="stylesheet" href="shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="shared/chat.css" />
|
||||
<link rel="stylesheet" href="shared/conversation.css" />
|
||||
<link rel="stylesheet" href="shared/cards.css" />
|
||||
<link rel="stylesheet" href="shared/interactive.css" />
|
||||
<style>
|
||||
/* Harness-only framing (NOT under review) — a plausible pane context. */
|
||||
body {
|
||||
padding: 24px; margin: 0; background: var(--bg); color: var(--ink);
|
||||
font-family: var(--font-sans, system-ui, sans-serif);
|
||||
}
|
||||
.demo-frame { max-width: 720px; margin: 0 auto; }
|
||||
.demo-label {
|
||||
font: 11px var(--font-mono, monospace); color: var(--ink-3);
|
||||
text-transform: uppercase; letter-spacing: 0.08em; margin: 0 0 8px;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="demo-frame">
|
||||
<div class="demo-label">conversation — task_agent card (real InteractivePane.handleEvent)</div>
|
||||
<div class="messages" id="messages"></div>
|
||||
</div>
|
||||
<script>
|
||||
// interactive.js reads window.toast / window.authFetch; the static render
|
||||
// never POSTs, so no-op stubs are enough.
|
||||
window.toast = { error: function (m) { console.log("toast:", m); } };
|
||||
window.authFetch = function () {
|
||||
return Promise.resolve({
|
||||
ok: true,
|
||||
json: function () { return Promise.resolve({}); },
|
||||
text: function () { return Promise.resolve(""); },
|
||||
});
|
||||
};
|
||||
</script>
|
||||
<script type="module">
|
||||
import { InteractivePane } from "./shared/interactive.js";
|
||||
const q = new URLSearchParams(location.search);
|
||||
if (q.get("theme") === "light")
|
||||
document.documentElement.dataset.theme = "light";
|
||||
|
||||
const messages = document.getElementById("messages");
|
||||
try {
|
||||
// Drive the REAL pane; stub only the host seams a mounted pane provides.
|
||||
const pane = new InteractivePane("demo-ws");
|
||||
pane.messagesEl = messages;
|
||||
pane.inputEl = document.createElement("textarea");
|
||||
pane.sendBtn = document.createElement("button");
|
||||
pane.isNearBottom = () => false;
|
||||
pane.scrollToBottom = () => {};
|
||||
pane.removeEmptyState = () => {};
|
||||
pane.removeThinkingIndicator = () => {};
|
||||
pane.setBusy = () => {};
|
||||
const ev = (e) => pane.handleEvent(e);
|
||||
|
||||
// ?recall=1: exercise the RECALL path — replayHistory rebuilding the
|
||||
// card from the /history `agent_steps` overlay (a reload / reopen while
|
||||
// the ws is still in memory), as opposed to the live SSE path below.
|
||||
const recall = q.get("recall") === "1";
|
||||
if (recall) {
|
||||
pane.replayHistory([
|
||||
{ role: "user", content: "Find all call sites of resolve_alias and summarize them" },
|
||||
{ role: "assistant", tool_calls: [{
|
||||
name: "task_agent", id: "task1",
|
||||
arguments: JSON.stringify({ prompt: "Find call sites of resolve_alias" }),
|
||||
agent_steps: [
|
||||
{ id: "task1::c1", name: "search", arguments: JSON.stringify({ query: "resolve_alias" }), output: "12 matches across 4 files", is_error: false },
|
||||
{ id: "task1::c2", name: "read_file", arguments: JSON.stringify({ path: "core/registry.py" }), output: "4.1 KB read", is_error: false },
|
||||
{ id: "task1::c3", name: "bash", arguments: JSON.stringify({ command: "pytest -k registry" }), output: "12 passed in 1.2s", is_error: false },
|
||||
{ id: "task1::c4", name: "notify", arguments: JSON.stringify({ channel: "#eng", message: "post summary" }), output: "posted to #eng", is_error: false },
|
||||
],
|
||||
}] },
|
||||
{ role: "tool", tool_call_id: "task1", content: "resolve_alias has 4 call sites (registry.py:120, session.py:12200, model_registry.py:88, eval.py:54); all pass a validated alias before use." },
|
||||
]);
|
||||
} else if (q.get("race") === "1") {
|
||||
// ?race=1: reproduce the parallel-pool ordering window — each
|
||||
// sub-tool's tool_pending is emitted exactly once (as in production)
|
||||
// but AHEAD of the task_agent row paint, as happens when a pooled
|
||||
// sub-agent's SSE event is handled before its parent row commits.
|
||||
// The orphan buffer must hold them and nest them when the parent row
|
||||
// lands; pre-fix they escaped to top-level rows and the card came up
|
||||
// short (steps < 4 -> TASKAGENT-FAILED), so this can't screenshot
|
||||
// green without the fix.
|
||||
const raceTask = {
|
||||
call_id: "task1", func_name: "task_agent",
|
||||
header: 'task_agent: "Find all call sites of resolve_alias and summarize them"',
|
||||
needs_approval: false,
|
||||
};
|
||||
const childPending = (cid, fn, header) =>
|
||||
ev({ type: "tool_pending", items: [{ call_id: cid, parent_call_id: "task1", func_name: fn, header: header, needs_approval: false }] });
|
||||
// a) Orphan child pendings arrive first — no parent row yet.
|
||||
childPending("task1::c1", "search", 'search: "resolve_alias"');
|
||||
childPending("task1::c2", "read_file", "read_file: core/registry.py");
|
||||
childPending("task1::c3", "bash", "pytest -k registry");
|
||||
childPending("task1::c4", "notify", "notify: post summary to #eng");
|
||||
// b) Parent task_agent row paints (pending -> resolved): must flush the
|
||||
// buffered orphans into the card AND survive the upgrade rebuild.
|
||||
ev({ type: "tool_pending", items: [raceTask] });
|
||||
ev({ type: "tool_info", items: [Object.assign({ auto_approved: false }, raceTask)] });
|
||||
// c) Results + a streamed chunk follow, nesting into the flushed rows.
|
||||
ev({ type: "tool_result", call_id: "task1::c1", parent_call_id: "task1", name: "search", output: "12 matches across 4 files" });
|
||||
ev({ type: "tool_result", call_id: "task1::c2", parent_call_id: "task1", name: "read_file", output: "4.1 KB read" });
|
||||
ev({ type: "tool_output_chunk", call_id: "task1::c3", parent_call_id: "task1", chunk: "collected 12 items ... " });
|
||||
ev({ type: "tool_result", call_id: "task1::c3", parent_call_id: "task1", name: "bash", output: "12 passed in 1.2s" });
|
||||
ev({ type: "tool_result", call_id: "task1::c4", parent_call_id: "task1", name: "notify", output: "posted to #eng" });
|
||||
ev({ type: "tool_result", call_id: "task1", name: "task_agent", output: "resolve_alias has 4 call sites (registry.py:120, session.py:12200, model_registry.py:88, eval.py:54); all pass a validated alias before use." });
|
||||
} else if (q.get("orphan") === "1") {
|
||||
// ?orphan=1: the SAFETY VALVE — child steps whose task_agent row
|
||||
// NEVER paints (an id-correlation mismatch, or an agent aborted
|
||||
// before its row painted). They must not vanish: after the grace
|
||||
// window the buffer escapes them to visible top-level rows (the
|
||||
// pre-buffer behaviour) rather than holding them forever. The parent
|
||||
// task_agent row is deliberately never emitted here.
|
||||
const orphanPending = (cid, fn, header) =>
|
||||
ev({ type: "tool_pending", items: [{ call_id: cid, parent_call_id: "task1", func_name: fn, header: header, needs_approval: false }] });
|
||||
orphanPending("task1::c1", "search", 'search: "resolve_alias"');
|
||||
orphanPending("task1::c2", "read_file", "read_file: core/registry.py");
|
||||
orphanPending("task1::c3", "bash", "pytest -k registry");
|
||||
} else {
|
||||
|
||||
// 1. Parent paints the task_agent call (a top-level tool row).
|
||||
const taskItem = {
|
||||
call_id: "task1", func_name: "task_agent",
|
||||
header: 'task_agent: "Find all call sites of resolve_alias and summarize them"',
|
||||
needs_approval: false,
|
||||
};
|
||||
// ?parallel=1 puts the task_agent in a 2-tool parallel batch so the
|
||||
// nested-step rail-bleed fix can be verified against the rail rules.
|
||||
const parentItems = q.get("parallel") === "1"
|
||||
? [taskItem, { call_id: "sib1", func_name: "bash", header: "git status", needs_approval: false }]
|
||||
: [taskItem];
|
||||
ev({ type: "tool_pending", items: parentItems });
|
||||
ev({ type: "tool_info", items: parentItems.map((it) => Object.assign({ auto_approved: false }, it)) });
|
||||
if (parentItems.length > 1)
|
||||
ev({ type: "tool_result", call_id: "sib1", name: "bash", output: "clean" });
|
||||
|
||||
// 2. Sub-agent steps tagged parent_call_id="task1" — exercises routing.
|
||||
function stepRow(cid, fn, header, result) {
|
||||
ev({ type: "tool_pending", items: [{ call_id: cid, parent_call_id: "task1", func_name: fn, header: header, needs_approval: false }] });
|
||||
if (result != null)
|
||||
ev({ type: "tool_result", call_id: cid, parent_call_id: "task1", name: fn, output: result });
|
||||
}
|
||||
stepRow("task1::c1", "search", 'search: "resolve_alias"', "12 matches across 4 files");
|
||||
stepRow("task1::c2", "read_file", "read_file: core/registry.py", "4.1 KB read");
|
||||
ev({ type: "tool_pending", items: [{ call_id: "task1::c3", parent_call_id: "task1", func_name: "bash", header: "pytest -k registry", needs_approval: false }] });
|
||||
ev({ type: "tool_output_chunk", call_id: "task1::c3", parent_call_id: "task1", chunk: "collected 12 items ... " });
|
||||
ev({ type: "tool_result", call_id: "task1::c3", parent_call_id: "task1", name: "bash", output: "12 passed in 1.2s" });
|
||||
// 4th step. Default: a nested sub-tool approval (notify is not
|
||||
// auto-approved) — the pane must auto-expand the collapse-by-default
|
||||
// card so the blocking prompt is visible. ?collapsed=1: a plain
|
||||
// completed step instead, so nothing forces the card open and the
|
||||
// screenshot shows the natural collapsed state (the common case).
|
||||
if (q.get("collapsed") === "1") {
|
||||
stepRow("task1::c4", "notify", "notify: post summary to #eng", "posted to #eng");
|
||||
} else {
|
||||
ev({ type: "approve_request", judge_pending: false, items: [{ call_id: "task1::c4", parent_call_id: "task1", func_name: "notify", header: "notify: post summary to #eng", needs_approval: true }] });
|
||||
}
|
||||
|
||||
// 3. The task agent's own synthesis, rendered below the card.
|
||||
ev({ type: "tool_result", call_id: "task1", name: "task_agent", output: "resolve_alias has 4 call sites (registry.py:120, session.py:12200, model_registry.py:88, eval.py:54); all pass a validated alias before use." });
|
||||
}
|
||||
|
||||
// ?expand=1: open every card so a screenshot shows the nested steps
|
||||
// (cards collapse by default; recall has no approval to auto-expand).
|
||||
if (q.get("expand") === "1") {
|
||||
document.querySelectorAll(".conv-agent").forEach(function (c) {
|
||||
c.dataset.collapsed = "false";
|
||||
const t = c.querySelector(".conv-agent-toggle");
|
||||
if (t) t.setAttribute("aria-expanded", "true");
|
||||
});
|
||||
}
|
||||
|
||||
// Loud failure — broken routing must not screenshot green.
|
||||
const orphanMode = q.get("orphan") === "1";
|
||||
setTimeout(function () {
|
||||
if (orphanMode) {
|
||||
// The parent never painted; after the grace window the buffered
|
||||
// steps must have ESCAPED to visible top-level rows, not vanished.
|
||||
const escaped = document.querySelectorAll('.conv-batch .conv-row[data-call-id^="task1::"]').length;
|
||||
const leaked = document.querySelector('.conv-row[data-call-id="task1"] .conv-agent');
|
||||
document.title = escaped >= 3 && !leaked
|
||||
? "TASKAGENT-ORPHANS-ESCAPED-" + escaped
|
||||
: "TASKAGENT-FAILED-escaped" + escaped + "-card" + (leaked ? 1 : 0);
|
||||
return;
|
||||
}
|
||||
const row = document.querySelector('.conv-row[data-call-id="task1"]');
|
||||
const card = row && row.querySelector(".conv-agent");
|
||||
const steps = card ? card.querySelectorAll(".conv-agent-body .conv-row").length : 0;
|
||||
const hasResult = !!(row && /call sites/.test(row.textContent || ""));
|
||||
document.title = card && steps >= 4 && hasResult
|
||||
? "TASKAGENT-READY-" + steps
|
||||
: "TASKAGENT-FAILED-card" + (card ? 1 : 0) + "-steps" + steps + "-result" + (hasResult ? 1 : 0);
|
||||
}, orphanMode ? 900 : 300);
|
||||
} catch (e) {
|
||||
messages.textContent = "HARNESS ERROR: " + e.message + "\\n" + (e.stack || "");
|
||||
document.title = "TASKAGENT-ERROR";
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Perf harness — long-session performance baseline for the interactive pane.
|
||||
# Mounts the REAL InteractivePane (production DOM via _createDOM, production
|
||||
# CSS chain) in a fixed-height mount so .pane-messages has REAL scroll
|
||||
# geometry — the forced-layout costs under measurement (isNearBottom /
|
||||
# scrollToBottom / chunk-append scroll pins) only exist against live layout,
|
||||
# which is why nothing here stubs scroll/geometry the way the task-agent
|
||||
# harness does. All timing is real time (see MEASUREMENT RULES in the module
|
||||
# docstring). Workload is deterministic (seeded LCG) so runs are comparable.
|
||||
# --------------------------------------------------------------------------
|
||||
PERF_TEMPLATE = """<!doctype html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
||||
<title>perf livepass</title>
|
||||
<link rel="stylesheet" href="shared/base.css" />
|
||||
<link rel="stylesheet" href="shared/ui-base.css" />
|
||||
<link rel="stylesheet" href="shared/chat.css" />
|
||||
<link rel="stylesheet" href="shared/conversation.css" />
|
||||
<link rel="stylesheet" href="shared/cards.css" />
|
||||
<link rel="stylesheet" href="static/style.css" />
|
||||
<link rel="stylesheet" href="shared/interactive.css" />
|
||||
<style>
|
||||
/* Harness-only framing (NOT under review): a fixed-height mount so the
|
||||
pane's .pane-messages scroller has real production geometry. */
|
||||
body { margin: 0; background: var(--bg); color: var(--fg); }
|
||||
#mount { height: 720px; width: 920px; display: flex; overflow: hidden; }
|
||||
#mount > .pane { flex: 1; display: flex; flex-direction: column; min-height: 0; }
|
||||
#perf-json { font: 11px monospace; white-space: pre-wrap; padding: 12px; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div id="mount"></div>
|
||||
<pre id="perf-json">running…</pre>
|
||||
<script>
|
||||
window.toast = { error: function (m) { console.log("toast:", m); } };
|
||||
// Collect every uncaught error/rejection into the report — a perf run
|
||||
// that silently swallowed a pipeline exception must not read as clean.
|
||||
window.__perfErrors = [];
|
||||
window.onerror = function (msg, src, line) {
|
||||
window.__perfErrors.push(String(msg) + " @ " + (src || "?") + ":" + (line || 0));
|
||||
};
|
||||
window.addEventListener("unhandledrejection", function (e) {
|
||||
window.__perfErrors.push("unhandledrejection: " + String(e && e.reason));
|
||||
});
|
||||
window.__perfFetch = function () {
|
||||
return Promise.resolve({
|
||||
ok: true, status: 200,
|
||||
json: function () { return Promise.resolve({}); },
|
||||
text: function () { return Promise.resolve(""); },
|
||||
});
|
||||
};
|
||||
window.authFetch = window.__perfFetch;
|
||||
</script>
|
||||
<script type="module">
|
||||
import { InteractivePane } from "./shared/interactive.js";
|
||||
// auth.js's legacy window bridge clobbers window.authFetch at module
|
||||
// import time — reinstate the stub now imports have evaluated (same
|
||||
// dance as the attachments harness).
|
||||
window.authFetch = window.__perfFetch;
|
||||
|
||||
const q = new URLSearchParams(location.search);
|
||||
const N = parseInt(q.get("n") || "1000", 10);
|
||||
const TURNS = parseInt(q.get("turns") || "20", 10);
|
||||
const CHUNKS = parseInt(q.get("chunks") || "300", 10);
|
||||
const CYCLES = parseInt(q.get("cycles") || "3", 10);
|
||||
const IDLE = parseInt(q.get("idle") || "20", 10);
|
||||
|
||||
// Long-task accounting across every phase (>50ms main-thread blocks).
|
||||
const lt = { count: 0, total_ms: 0, max_ms: 0 };
|
||||
try {
|
||||
new PerformanceObserver(function (list) {
|
||||
list.getEntries().forEach(function (e) {
|
||||
lt.count += 1;
|
||||
lt.total_ms += Math.round(e.duration);
|
||||
lt.max_ms = Math.max(lt.max_ms, Math.round(e.duration));
|
||||
});
|
||||
}).observe({ type: "longtask", buffered: true });
|
||||
} catch (e) { /* unsupported — longtasks stay zeroed */ }
|
||||
|
||||
// Deterministic workload (seeded LCG) so runs are comparable.
|
||||
let _seed = 42;
|
||||
function rnd() {
|
||||
_seed = (_seed * 1664525 + 1013904223) >>> 0;
|
||||
return _seed / 4294967296;
|
||||
}
|
||||
const WORDS = ("the retry loop grinds the dungeon server while the " +
|
||||
"judge weighs verdicts and the coordinator shuffles children across " +
|
||||
"nodes tokens accumulate compaction folds turns storage keeps the " +
|
||||
"canon and the rail repaints").split(" ");
|
||||
function sentence(w) {
|
||||
const parts = [];
|
||||
for (let i = 0; i < w; i++) parts.push(WORDS[(rnd() * WORDS.length) | 0]);
|
||||
return parts.join(" ");
|
||||
}
|
||||
// Realistic assistant markdown: prose + list + fenced code (varying
|
||||
// content so the hljs cache behaves as in production) + inline code.
|
||||
function mdBody(i) {
|
||||
return (
|
||||
"Turn " + i + ": " + sentence(18) + ".\\n\\n" +
|
||||
"- " + sentence(6) + "\\n- " + sentence(7) + "\\n\\n" +
|
||||
"```python\\n" +
|
||||
"def step_" + i + "(depth):\\n" +
|
||||
" total = " + ((rnd() * 1000) | 0) + "\\n" +
|
||||
" for k in range(depth):\\n" +
|
||||
" total += k * " + (1 + ((rnd() * 9) | 0)) + "\\n" +
|
||||
" return total\\n" +
|
||||
"```\\n\\n" +
|
||||
sentence(14) + " `inline_" + i + "` " + sentence(8) + "."
|
||||
);
|
||||
}
|
||||
// History in the canonical projected wire shape replayHistory consumes
|
||||
// (user / assistant content / assistant tool_calls / tool result), with
|
||||
// periodic reasoning bubbles and task_agent cards (agent_steps overlay).
|
||||
function buildHistory(n) {
|
||||
const msgs = [];
|
||||
let i = 0;
|
||||
while (msgs.length < n) {
|
||||
i += 1;
|
||||
msgs.push({ role: "user", content: "Request " + i + ": " + sentence(10) + "?" });
|
||||
if (msgs.length >= n) break;
|
||||
if (i % 10 === 0) {
|
||||
msgs.push({ role: "assistant", reasoning: sentence(40) + ".", content: mdBody(i) });
|
||||
} else {
|
||||
msgs.push({ role: "assistant", content: mdBody(i) });
|
||||
}
|
||||
if (msgs.length >= n) break;
|
||||
const callId = "h" + i;
|
||||
if (i % 8 === 0) {
|
||||
msgs.push({ role: "assistant", tool_calls: [{
|
||||
name: "task_agent", id: callId,
|
||||
arguments: JSON.stringify({ prompt: "subtask " + i }),
|
||||
agent_steps: [
|
||||
{ id: callId + "::c1", name: "search",
|
||||
arguments: JSON.stringify({ query: "q" + i }),
|
||||
output: sentence(8), is_error: false },
|
||||
{ id: callId + "::c2", name: "read_file",
|
||||
arguments: JSON.stringify({ path: "core/f" + i + ".py" }),
|
||||
output: sentence(6), is_error: false },
|
||||
{ id: callId + "::c3", name: "bash",
|
||||
arguments: JSON.stringify({ command: "pytest -k t" + i }),
|
||||
output: sentence(7), is_error: false },
|
||||
],
|
||||
}] });
|
||||
} else {
|
||||
msgs.push({ role: "assistant", tool_calls: [{
|
||||
name: "bash", id: callId,
|
||||
arguments: JSON.stringify({ command: "grep -rn pattern_" + i + " src/" }),
|
||||
}] });
|
||||
}
|
||||
if (msgs.length >= n) break;
|
||||
msgs.push({ role: "tool", tool_call_id: callId,
|
||||
content: "output " + i + ":\\n" + sentence(20) });
|
||||
}
|
||||
return msgs;
|
||||
}
|
||||
|
||||
const tick = () => new Promise((r) => requestAnimationFrame(r));
|
||||
// One live turn, production event mix: thinking indicator, reasoning
|
||||
// deltas, content deltas (yield every few so streamingRender's internal
|
||||
// rAF actually applies frames, as in a real token stream), stream_end,
|
||||
// an auto-approved bash batch with streamed chunks, every 5th turn a
|
||||
// task_agent card with routed children, then the idle edge.
|
||||
async function stormTurn(pane, i) {
|
||||
pane.handleEvent({ type: "state_change", state: "running" });
|
||||
pane.handleEvent({ type: "thinking_start" });
|
||||
const reason = sentence(50);
|
||||
let d = 0;
|
||||
for (let k = 0; k < reason.length; k += 20) {
|
||||
pane.handleEvent({ type: "reasoning", text: reason.slice(k, k + 20) });
|
||||
d += 1;
|
||||
if (d % 4 === 3) await tick();
|
||||
}
|
||||
const body = mdBody(100000 + i);
|
||||
d = 0;
|
||||
for (let k = 0; k < body.length; k += 22) {
|
||||
pane.handleEvent({ type: "content", text: body.slice(k, k + 22) });
|
||||
d += 1;
|
||||
if (d % 6 === 5) await tick();
|
||||
}
|
||||
pane.handleEvent({ type: "stream_end" });
|
||||
const callId = "s" + i;
|
||||
const item = { call_id: callId, func_name: "bash",
|
||||
header: "bash: run step " + i, needs_approval: false };
|
||||
pane.handleEvent({ type: "tool_pending", items: [item] });
|
||||
pane.handleEvent({ type: "tool_info",
|
||||
items: [Object.assign({ auto_approved: true }, item)] });
|
||||
for (let k = 0; k < 24; k++) {
|
||||
pane.handleEvent({ type: "tool_output_chunk", call_id: callId,
|
||||
chunk: "line " + k + ": " + sentence(5) + "\\n" });
|
||||
if (k % 6 === 5) await tick();
|
||||
}
|
||||
pane.handleEvent({ type: "tool_result", call_id: callId, name: "bash",
|
||||
output: "done " + i + "\\n" + sentence(12) });
|
||||
if (i % 5 === 4) {
|
||||
const tid = "sa" + i;
|
||||
const titem = { call_id: tid, func_name: "task_agent",
|
||||
header: 'task_agent: "subtask ' + i + '"', needs_approval: false };
|
||||
pane.handleEvent({ type: "tool_pending", items: [titem] });
|
||||
pane.handleEvent({ type: "tool_info",
|
||||
items: [Object.assign({ auto_approved: true }, titem)] });
|
||||
for (let c = 1; c <= 3; c++) {
|
||||
const cid = tid + "::c" + c;
|
||||
pane.handleEvent({ type: "tool_pending", items: [{
|
||||
call_id: cid, parent_call_id: tid, func_name: "search",
|
||||
header: "search: q" + c, needs_approval: false }] });
|
||||
pane.handleEvent({ type: "tool_result", call_id: cid,
|
||||
parent_call_id: tid, name: "search", output: sentence(6) });
|
||||
}
|
||||
pane.handleEvent({ type: "tool_result", call_id: tid,
|
||||
name: "task_agent", output: sentence(15) });
|
||||
await tick();
|
||||
}
|
||||
pane.handleEvent({ type: "state_change", state: "idle" });
|
||||
await tick();
|
||||
}
|
||||
|
||||
function heapBytes() {
|
||||
// --js-flags=--expose-gc makes this a real floor, not GC noise.
|
||||
if (typeof window.gc === "function") {
|
||||
try { window.gc(); window.gc(); } catch (e) { /* noop */ }
|
||||
}
|
||||
return (performance.memory && performance.memory.usedJSHeapSize) || null;
|
||||
}
|
||||
|
||||
const report = {
|
||||
n: N, turns: TURNS, chunks: CHUNKS, cycles: CYCLES, idle: IDLE,
|
||||
// Echoed run token — the runner validates it so a straggler POST
|
||||
// from a killed prior attempt can't be misattributed to this run.
|
||||
run: q.get("run") || "",
|
||||
errors: window.__perfErrors,
|
||||
};
|
||||
let phase = "mount";
|
||||
try {
|
||||
const pane = new InteractivePane("perf-ws");
|
||||
document.getElementById("mount").appendChild(pane.el);
|
||||
const msgs = buildHistory(N);
|
||||
report.heap_start = heapBytes();
|
||||
|
||||
phase = "replay";
|
||||
let t0 = performance.now();
|
||||
pane.replayHistory(msgs);
|
||||
report.replay_ms = Math.round(performance.now() - t0);
|
||||
await tick();
|
||||
report.nodes_after_replay = pane.messagesEl.querySelectorAll("*").length;
|
||||
|
||||
phase = "storm";
|
||||
t0 = performance.now();
|
||||
for (let i = 0; i < TURNS; i++) await stormTurn(pane, i);
|
||||
report.storm_ms = Math.round(performance.now() - t0);
|
||||
report.storm_ms_per_turn = Math.round(report.storm_ms / TURNS);
|
||||
|
||||
phase = "chunkstorm";
|
||||
const ccItem = { call_id: "cc1", func_name: "bash",
|
||||
header: "bash: tail -f build.log", needs_approval: false };
|
||||
pane.handleEvent({ type: "tool_pending", items: [ccItem] });
|
||||
pane.handleEvent({ type: "tool_info",
|
||||
items: [Object.assign({ auto_approved: true }, ccItem)] });
|
||||
t0 = performance.now();
|
||||
for (let k = 0; k < CHUNKS; k++) {
|
||||
pane.handleEvent({ type: "tool_output_chunk", call_id: "cc1",
|
||||
chunk: "log line " + k + "\\n" });
|
||||
if (k % 6 === 5) await tick();
|
||||
}
|
||||
report.chunk_ms = Math.round(performance.now() - t0);
|
||||
pane.handleEvent({ type: "tool_result", call_id: "cc1", name: "bash",
|
||||
output: "tail done" });
|
||||
|
||||
phase = "idlechurn";
|
||||
t0 = performance.now();
|
||||
for (let k = 0; k < IDLE; k++) {
|
||||
pane.handleEvent({ type: "state_change", state: "running" });
|
||||
pane.handleEvent({ type: "state_change", state: "idle" });
|
||||
if (k % 4 === 3) await tick();
|
||||
}
|
||||
report.idle_ms = Math.round(performance.now() - t0);
|
||||
|
||||
// Leak probe: repeated full replays of the SAME history should
|
||||
// converge to a flat heap/node/agent-card profile; monotonic growth
|
||||
// here is retained-detached-DOM (the _agentCards class of bug).
|
||||
phase = "replaycycles";
|
||||
report.cycle_stats = [];
|
||||
for (let c = 0; c < CYCLES; c++) {
|
||||
t0 = performance.now();
|
||||
pane.replayHistory(msgs);
|
||||
const ms = Math.round(performance.now() - t0);
|
||||
await tick();
|
||||
report.cycle_stats.push({
|
||||
replay_ms: ms,
|
||||
heap: heapBytes(),
|
||||
nodes: pane.messagesEl.querySelectorAll("*").length,
|
||||
agent_cards: pane._agentCards ? pane._agentCards.size : 0,
|
||||
});
|
||||
}
|
||||
report.heap_end = heapBytes();
|
||||
report.longtasks = lt;
|
||||
document.title = "PERF-READY-" + N;
|
||||
} catch (e) {
|
||||
window.__perfErrors.push(
|
||||
"phase " + phase + ": " + (e && e.message ? e.message : String(e)),
|
||||
);
|
||||
report.failed_phase = phase;
|
||||
report.longtasks = lt;
|
||||
document.title = "PERF-FAILED-" + phase;
|
||||
}
|
||||
document.getElementById("perf-json").textContent =
|
||||
JSON.stringify(report, null, 2);
|
||||
if (q.get("post")) {
|
||||
try {
|
||||
await fetch("/perf/report", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(report),
|
||||
});
|
||||
} catch (e) { /* runner captures the timeout instead */ }
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
# Fixture media for the attachments harness. image/pdf thumbnails and the
|
||||
# audio clip load via element .src (NOT authFetch), so the --serve dev server
|
||||
# answers those paths directly with representative bytes: a photo-like image,
|
||||
@@ -883,34 +1470,266 @@ def build(out: Path) -> None:
|
||||
(att / "livepass.html").write_text(ATTACH_TEMPLATE, encoding="utf-8")
|
||||
print(f"{att}/livepass.html — composer chips + message attachment pills")
|
||||
|
||||
ta = out / "taskagent"
|
||||
ta.mkdir(parents=True, exist_ok=True)
|
||||
symlink(ta / "shared", ROOT / "turnstone/shared_static")
|
||||
(ta / "livepass.html").write_text(TASKAGENT_TEMPLATE, encoding="utf-8")
|
||||
print(f"{ta}/livepass.html — task_agent card (real Pane.handleEvent routing)")
|
||||
|
||||
pf = out / "perf"
|
||||
pf.mkdir(parents=True, exist_ok=True)
|
||||
symlink(pf / "shared", ROOT / "turnstone/shared_static")
|
||||
symlink(pf / "static", ROOT / "turnstone/ui/static")
|
||||
(pf / "livepass.html").write_text(PERF_TEMPLATE, encoding="utf-8")
|
||||
print(f"{pf}/livepass.html — long-session perf baseline (real InteractivePane)")
|
||||
|
||||
|
||||
class _PerfStore:
|
||||
"""Rendezvous for the perf page's POSTed JSON report."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
import threading
|
||||
|
||||
self.event = threading.Event()
|
||||
self.data: dict[str, object] | None = None
|
||||
|
||||
|
||||
class _HarnessHandler(http.server.SimpleHTTPRequestHandler):
|
||||
"""Static file server + attachment media fixtures + perf-report sink.
|
||||
|
||||
The attachments harness loads thumbnails + the audio clip via element
|
||||
.src; serve those from generated fixtures, fall through to static for
|
||||
everything else. The perf harness POSTs its JSON report to /perf/report
|
||||
when driven with ?post=1 — the --perf runner blocks on ``perf_store``.
|
||||
"""
|
||||
|
||||
perf_store: _PerfStore | None = None
|
||||
quiet = False
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802 (stdlib casing)
|
||||
blob = _fixture_for(self.path.split("?")[0])
|
||||
if blob is None:
|
||||
super().do_GET()
|
||||
return
|
||||
data, ctype = blob
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", ctype)
|
||||
self.send_header("Content-Length", str(len(data)))
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802 (stdlib casing)
|
||||
store = type(self).perf_store
|
||||
if self.path.split("?")[0] != "/perf/report" or store is None:
|
||||
self.send_error(404)
|
||||
return
|
||||
length = int(self.headers.get("Content-Length") or 0)
|
||||
body = self.rfile.read(length)
|
||||
try:
|
||||
store.data = json.loads(body)
|
||||
except ValueError:
|
||||
store.data = {"errors": ["runner: unparseable report body"]}
|
||||
store.event.set()
|
||||
self.send_response(204)
|
||||
self.end_headers()
|
||||
|
||||
def log_message(self, format: str, *args: object) -> None: # noqa: A002 (stdlib signature)
|
||||
if not type(self).quiet:
|
||||
super().log_message(format, *args)
|
||||
|
||||
|
||||
def _find_chrome() -> str | None:
|
||||
for name in ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser"):
|
||||
path = shutil.which(name)
|
||||
if path:
|
||||
return path
|
||||
return None
|
||||
|
||||
|
||||
def _await_report(
|
||||
store: _PerfStore, proc: subprocess.Popen[bytes], run_token: str, timeout: float
|
||||
) -> dict[str, object] | None:
|
||||
"""Wait for THIS attempt's report: validated by run token, bailing early
|
||||
when Chrome exits without reporting (the sandbox-startup-failure case —
|
||||
waiting the full timeout there cost minutes before the --no-sandbox
|
||||
fallback could even start). A straggler POST from a previous attempt
|
||||
(its handler thread can complete after the next attempt cleared the
|
||||
store) carries the wrong token and is discarded instead of being
|
||||
misattributed to this run."""
|
||||
deadline = time.monotonic() + timeout
|
||||
proc_exited_at: float | None = None
|
||||
while time.monotonic() < deadline:
|
||||
if store.event.wait(0.5):
|
||||
data = store.data
|
||||
store.event.clear()
|
||||
store.data = None
|
||||
if isinstance(data, dict) and data.get("run") == run_token:
|
||||
return data
|
||||
continue # stale straggler from a prior attempt — keep waiting
|
||||
if proc.poll() is not None:
|
||||
now = time.monotonic()
|
||||
if proc_exited_at is None:
|
||||
proc_exited_at = now # grace: an in-flight POST may still land
|
||||
elif now - proc_exited_at > 3.0:
|
||||
return None # exited without reporting — try the next attempt
|
||||
return None
|
||||
|
||||
|
||||
def _perf_run_one(
|
||||
chrome: str, out: Path, port: int, store: _PerfStore, n: int, turns: int, timeout: float
|
||||
) -> dict[str, object] | None:
|
||||
"""One headless-Chrome perf pass; returns the page's report or None."""
|
||||
base_flags = [
|
||||
"--headless=new",
|
||||
"--disable-gpu",
|
||||
"--hide-scrollbars",
|
||||
"--window-size=1440,900",
|
||||
"--no-first-run",
|
||||
"--disable-extensions",
|
||||
# Throttled timers/rAF in a backgrounded renderer would corrupt the
|
||||
# measurement — pin the renderer foreground-scheduled.
|
||||
"--disable-background-timer-throttling",
|
||||
"--disable-renderer-backgrounding",
|
||||
"--disable-backgrounding-occluded-windows",
|
||||
# Stable, real heap numbers (heapBytes() calls window.gc() first).
|
||||
"--js-flags=--expose-gc",
|
||||
"--enable-precise-memory-info",
|
||||
]
|
||||
for attempt, extra in enumerate(
|
||||
([], ["--no-sandbox"]) # sandboxed first, container fallback second
|
||||
):
|
||||
run_token = f"n{n}-a{attempt}-{uuid.uuid4().hex[:8]}"
|
||||
url = (
|
||||
f"http://127.0.0.1:{port}/perf/livepass.html?n={n}&turns={turns}&post=1&run={run_token}"
|
||||
)
|
||||
store.event.clear()
|
||||
store.data = None
|
||||
profile = out / f".chrome-perf-{n}"
|
||||
proc = subprocess.Popen(
|
||||
[chrome, *base_flags, *extra, f"--user-data-dir={profile}", url],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
try:
|
||||
report = _await_report(store, proc, run_token, timeout)
|
||||
if report is not None:
|
||||
return report
|
||||
finally:
|
||||
if proc.poll() is None:
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(10)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
return None
|
||||
|
||||
|
||||
def run_perf(out: Path, sizes: list[int], turns: int, timeout: float) -> bool:
|
||||
"""Build, serve, and run the perf page once per history size; print a table."""
|
||||
import functools
|
||||
import threading
|
||||
|
||||
chrome = _find_chrome()
|
||||
if chrome is None:
|
||||
print("perf: no chrome/chromium binary found on PATH")
|
||||
return False
|
||||
store = _PerfStore()
|
||||
_HarnessHandler.perf_store = store
|
||||
_HarnessHandler.quiet = True
|
||||
handler = functools.partial(_HarnessHandler, directory=str(out))
|
||||
server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler)
|
||||
port = server.server_address[1]
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
reports: dict[int, dict[str, object]] = {}
|
||||
try:
|
||||
for n in sizes:
|
||||
print(f"perf: n={n} turns={turns} … ", end="", flush=True)
|
||||
report = _perf_run_one(chrome, out, port, store, n, turns, timeout)
|
||||
if report is None:
|
||||
print("FAILED (no report — timeout or chrome startup failure)")
|
||||
continue
|
||||
failed = report.get("failed_phase")
|
||||
errors = report.get("errors") or []
|
||||
status = f"failed in {failed}" if failed else "ok"
|
||||
print(f"{status} ({len(errors) if isinstance(errors, list) else '?'} page errors)")
|
||||
reports[n] = report
|
||||
(out / f"perf-report-n{n}.json").write_text(
|
||||
json.dumps(report, indent=2), encoding="utf-8"
|
||||
)
|
||||
finally:
|
||||
server.shutdown()
|
||||
_HarnessHandler.perf_store = None
|
||||
_HarnessHandler.quiet = False
|
||||
if not reports:
|
||||
return False
|
||||
_print_perf_table(reports)
|
||||
print(f"\nraw reports: {out}/perf-report-n*.json")
|
||||
return True
|
||||
|
||||
|
||||
def _print_perf_table(reports: dict[int, dict[str, object]]) -> None:
|
||||
sizes = sorted(reports)
|
||||
|
||||
def cell(n: int, key: str) -> str:
|
||||
value = reports[n].get(key)
|
||||
return "—" if value is None else str(value)
|
||||
|
||||
def mb(value: object) -> str:
|
||||
return f"{value / 1048576:.1f}MB" if isinstance(value, (int, float)) else "—"
|
||||
|
||||
rows: list[tuple[str, list[str]]] = [
|
||||
("replay_ms (full history build)", [cell(n, "replay_ms") for n in sizes]),
|
||||
("nodes after replay", [cell(n, "nodes_after_replay") for n in sizes]),
|
||||
("storm ms/turn (live mix)", [cell(n, "storm_ms_per_turn") for n in sizes]),
|
||||
("chunk_ms (output chunks)", [cell(n, "chunk_ms") for n in sizes]),
|
||||
("idle_ms (busy/idle churn)", [cell(n, "idle_ms") for n in sizes]),
|
||||
("heap start → end", []),
|
||||
("longtasks count/max_ms", []),
|
||||
("replay cycles ms", []),
|
||||
("agent_cards after cycles", []),
|
||||
]
|
||||
for n in sizes:
|
||||
rep = reports[n]
|
||||
rows[5][1].append(f"{mb(rep.get('heap_start'))} → {mb(rep.get('heap_end'))}")
|
||||
lt = rep.get("longtasks")
|
||||
rows[6][1].append(f"{lt.get('count')}/{lt.get('max_ms')}" if isinstance(lt, dict) else "—")
|
||||
cycles = rep.get("cycle_stats")
|
||||
if isinstance(cycles, list) and cycles:
|
||||
rows[7][1].append(",".join(str(c.get("replay_ms", "?")) for c in cycles))
|
||||
rows[8][1].append(str(cycles[-1].get("agent_cards", "?")))
|
||||
else:
|
||||
rows[7][1].append("—")
|
||||
rows[8][1].append("—")
|
||||
|
||||
label_w = max(len(label) for label, _ in rows)
|
||||
col_w = max(14, *(len(f"n={n}") for n in sizes))
|
||||
header = " " * label_w + " " + " ".join(f"n={n}".rjust(col_w) for n in sizes)
|
||||
print("\n" + header)
|
||||
for label, cells in rows:
|
||||
print(label.ljust(label_w) + " " + " ".join(c.rjust(col_w) for c in cells))
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
ap.add_argument("--out", type=Path, default=Path("/tmp/livepass"))
|
||||
ap.add_argument("--serve", type=int, metavar="PORT")
|
||||
ap.add_argument("--perf", action="store_true", help="run the perf baseline and exit")
|
||||
ap.add_argument(
|
||||
"--perf-n",
|
||||
default="300,3000",
|
||||
help="comma-separated history sizes for --perf (default: 300,3000)",
|
||||
)
|
||||
ap.add_argument("--perf-turns", type=int, default=20)
|
||||
ap.add_argument("--perf-timeout", type=float, default=420.0)
|
||||
args = ap.parse_args()
|
||||
build(args.out)
|
||||
if args.perf:
|
||||
sizes = [int(s) for s in str(args.perf_n).split(",") if s.strip()]
|
||||
raise SystemExit(0 if run_perf(args.out, sizes, args.perf_turns, args.perf_timeout) else 1)
|
||||
if args.serve:
|
||||
import functools
|
||||
import http.server
|
||||
|
||||
class _FixtureHandler(http.server.SimpleHTTPRequestHandler):
|
||||
# The attachments harness loads thumbnails + the audio clip via
|
||||
# element .src; serve those from generated fixtures, fall through
|
||||
# to static for everything else.
|
||||
def do_GET(self) -> None: # noqa: N802 (stdlib casing)
|
||||
blob = _fixture_for(self.path.split("?")[0])
|
||||
if blob is None:
|
||||
super().do_GET()
|
||||
return
|
||||
data, ctype = blob
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", ctype)
|
||||
self.send_header("Content-Length", str(len(data)))
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
handler = functools.partial(_FixtureHandler, directory=str(args.out))
|
||||
handler = functools.partial(_HarnessHandler, directory=str(args.out))
|
||||
print(f"serving {args.out} on http://localhost:{args.serve}/ — Ctrl+C stops")
|
||||
http.server.ThreadingHTTPServer(("127.0.0.1", args.serve), handler).serve_forever()
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Console API",
|
||||
"version": "1.7.0a2",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Cluster-wide visibility and control across all turnstone nodes."
|
||||
},
|
||||
"paths": {
|
||||
@@ -4213,6 +4213,166 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/admin/personas": {
|
||||
"get": {
|
||||
"summary": "List all personas, archived included",
|
||||
"operationId": "v1_api_admin_personas_get",
|
||||
"tags": [
|
||||
"Admin"
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ListPersonasResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"post": {
|
||||
"summary": "Create a persona",
|
||||
"operationId": "v1_api_admin_personas_post",
|
||||
"tags": [
|
||||
"Admin"
|
||||
],
|
||||
"requestBody": {
|
||||
"required": true,
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/CreatePersonaRequest"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/PersonaInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "Error 400",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/admin/personas/{persona_id}": {
|
||||
"get": {
|
||||
"summary": "Get a single persona",
|
||||
"operationId": "v1_api_admin_personas_{persona_id}_get",
|
||||
"tags": [
|
||||
"Admin"
|
||||
],
|
||||
"parameters": [
|
||||
{
|
||||
"name": "persona_id",
|
||||
"in": "path",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/PersonaInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Error 404",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"patch": {
|
||||
"summary": "Update a persona (edit levers, archive/unarchive, flip default)",
|
||||
"operationId": "v1_api_admin_personas_{persona_id}_patch",
|
||||
"tags": [
|
||||
"Admin"
|
||||
],
|
||||
"parameters": [
|
||||
{
|
||||
"name": "persona_id",
|
||||
"in": "path",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
],
|
||||
"requestBody": {
|
||||
"required": true,
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/UpdatePersonaRequest"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/PersonaInfo"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "Error 400",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Error 404",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ErrorResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/admin/node-metadata": {
|
||||
"get": {
|
||||
"summary": "Get metadata for all nodes (bulk)",
|
||||
@@ -6528,7 +6688,7 @@
|
||||
"tags": [
|
||||
"Coordinator"
|
||||
],
|
||||
"description": "Aggregates the persisted row, a best-effort live block from the owning node (or the in-process coordinator manager for ``kind=\"coordinator\"`` rows), and the tail of the message history. Gated on the ``admin.cluster.inspect`` permission (granted to ``builtin-admin`` via migration 040; revoke or reassign to a custom role for tighter control). ``live`` is null on node unreachability / 5xx so callers can degrade gracefully.",
|
||||
"description": "Aggregates the persisted row, a best-effort live block from the owning node (or the in-process coordinator manager for ``kind=\"coordinator\"`` rows), and the tail of the message history. Gated on the ``admin.cluster.inspect`` permission (granted to ``builtin-admin`` via migration 040; revoke or reassign to a custom role for tighter control). A workstream attached to a *private* project stays confidential to its members: a permitted caller who isn't its owner / creator / project member gets a 404 (same masking as an unknown id). ``live`` is null on node unreachability / 5xx so callers can degrade gracefully.",
|
||||
"parameters": [
|
||||
{
|
||||
"name": "ws_id",
|
||||
@@ -7625,6 +7785,12 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"description": "Persona slug; resolved and snapshotted at creation, empty = kind default",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"resume_ws": {
|
||||
"default": "",
|
||||
"description": "Workstream ID to resume (loads previous conversation)",
|
||||
@@ -7952,6 +8118,12 @@
|
||||
"description": "Optional skill name to apply to the coordinator session.",
|
||||
"title": "Skill"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"description": "Persona slug; resolved and snapshotted at creation, empty = kind default",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"initial_message": {
|
||||
"default": "",
|
||||
"description": "Optional first user message dispatched to the new coordinator session.",
|
||||
@@ -10896,6 +11068,345 @@
|
||||
"title": "ListModelDefinitionsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"PersonaInfo": {
|
||||
"description": "Full persona row \u2014 the authoring shape (contrast PersonaChoice, the\npicker's display-only projection on the server surface).",
|
||||
"properties": {
|
||||
"persona_id": {
|
||||
"title": "Persona Id",
|
||||
"type": "string"
|
||||
},
|
||||
"name": {
|
||||
"title": "Name",
|
||||
"type": "string"
|
||||
},
|
||||
"display_name": {
|
||||
"default": "",
|
||||
"title": "Display Name",
|
||||
"type": "string"
|
||||
},
|
||||
"description": {
|
||||
"default": "",
|
||||
"title": "Description",
|
||||
"type": "string"
|
||||
},
|
||||
"base_prompt": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "BASE-module override; null = the kind's stock base",
|
||||
"title": "Base Prompt"
|
||||
},
|
||||
"tool_allowlist": {
|
||||
"anyOf": [
|
||||
{
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"type": "array"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Tool visibility set: null = unrestricted, [] = no tools, [names] = exact set (include 'tool_search' to keep the set soft/expandable)",
|
||||
"title": "Tool Allowlist"
|
||||
},
|
||||
"mcp_enabled": {
|
||||
"default": true,
|
||||
"title": "Mcp Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"memory_enabled": {
|
||||
"default": true,
|
||||
"title": "Memory Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"applies_to_kinds": {
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Applies To Kinds",
|
||||
"type": "array"
|
||||
},
|
||||
"is_default": {
|
||||
"default": false,
|
||||
"title": "Is Default",
|
||||
"type": "boolean"
|
||||
},
|
||||
"enabled": {
|
||||
"default": true,
|
||||
"description": "false = archived",
|
||||
"title": "Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"org_id": {
|
||||
"default": "",
|
||||
"title": "Org Id",
|
||||
"type": "string"
|
||||
},
|
||||
"created_by": {
|
||||
"default": "",
|
||||
"title": "Created By",
|
||||
"type": "string"
|
||||
},
|
||||
"created": {
|
||||
"default": "",
|
||||
"title": "Created",
|
||||
"type": "string"
|
||||
},
|
||||
"updated": {
|
||||
"default": "",
|
||||
"title": "Updated",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"persona_id",
|
||||
"name"
|
||||
],
|
||||
"title": "PersonaInfo",
|
||||
"type": "object"
|
||||
},
|
||||
"CreatePersonaRequest": {
|
||||
"properties": {
|
||||
"name": {
|
||||
"description": "Immutable slug (lowercase: a-z, 0-9, '-', '_')",
|
||||
"title": "Name",
|
||||
"type": "string"
|
||||
},
|
||||
"display_name": {
|
||||
"default": "",
|
||||
"title": "Display Name",
|
||||
"type": "string"
|
||||
},
|
||||
"description": {
|
||||
"default": "",
|
||||
"title": "Description",
|
||||
"type": "string"
|
||||
},
|
||||
"base_prompt": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline BASE override \u2014 required. Every persona must name a prompt source; built-in file-backed personas are seeded by migration, not created here, so an operator-created persona must supply base_prompt.",
|
||||
"title": "Base Prompt"
|
||||
},
|
||||
"tool_allowlist": {
|
||||
"anyOf": [
|
||||
{
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"type": "array"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Tool Allowlist"
|
||||
},
|
||||
"mcp_enabled": {
|
||||
"default": true,
|
||||
"title": "Mcp Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"memory_enabled": {
|
||||
"default": true,
|
||||
"title": "Memory Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"applies_to_kinds": {
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Applies To Kinds",
|
||||
"type": "array"
|
||||
},
|
||||
"is_default": {
|
||||
"default": false,
|
||||
"title": "Is Default",
|
||||
"type": "boolean"
|
||||
},
|
||||
"enabled": {
|
||||
"default": true,
|
||||
"title": "Enabled",
|
||||
"type": "boolean"
|
||||
},
|
||||
"org_id": {
|
||||
"default": "",
|
||||
"description": "Owning org (informational; capped at 64)",
|
||||
"title": "Org Id",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"name"
|
||||
],
|
||||
"title": "CreatePersonaRequest",
|
||||
"type": "object"
|
||||
},
|
||||
"UpdatePersonaRequest": {
|
||||
"description": "PATCH body \u2014 absent fields are left unchanged.\n\nExplicit ``null`` resets ``tool_allowlist`` to unrestricted, and \u2014 on a\nBUILT-IN persona only \u2014 clears ``base_prompt`` (the operator override),\nreverting to that persona's file-backed prompt. An OPERATOR persona has no\nfallback source, so ``base_prompt: null`` on one is rejected: every persona\nmust name a prompt source. ``null`` on the boolean flags or\n``applies_to_kinds`` is ignored (treated as absent), so a client serializing\nunset optionals as null cannot archive a persona or flip levers by accident.\n\nArchive = ``{\"enabled\": false}``; default flip = ``{\"is_default\": true}``\non the successor (storage demotes the incumbent atomically). ``name``\nis immutable; existing workstreams are never affected by edits.",
|
||||
"properties": {
|
||||
"display_name": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Display Name"
|
||||
},
|
||||
"description": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Description"
|
||||
},
|
||||
"base_prompt": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Base Prompt"
|
||||
},
|
||||
"tool_allowlist": {
|
||||
"anyOf": [
|
||||
{
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"type": "array"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Tool Allowlist"
|
||||
},
|
||||
"mcp_enabled": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "boolean"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Mcp Enabled"
|
||||
},
|
||||
"memory_enabled": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "boolean"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Memory Enabled"
|
||||
},
|
||||
"applies_to_kinds": {
|
||||
"anyOf": [
|
||||
{
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"type": "array"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Applies To Kinds"
|
||||
},
|
||||
"is_default": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "boolean"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Is Default"
|
||||
},
|
||||
"enabled": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "boolean"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Enabled"
|
||||
}
|
||||
},
|
||||
"title": "UpdatePersonaRequest",
|
||||
"type": "object"
|
||||
},
|
||||
"ListPersonasResponse": {
|
||||
"properties": {
|
||||
"personas": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PersonaInfo"
|
||||
},
|
||||
"title": "Personas",
|
||||
"type": "array"
|
||||
},
|
||||
"tool_inventory": {
|
||||
"additionalProperties": {
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"type": "array"
|
||||
},
|
||||
"description": "Per-kind builtin tool names (plus the synthetic 'tool_search') for the visibility checklist \u2014 derived server-side so clients never hand-mirror the inventory",
|
||||
"title": "Tool Inventory",
|
||||
"type": "object"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"personas"
|
||||
],
|
||||
"title": "ListPersonasResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelReloadResponse": {
|
||||
"properties": {
|
||||
"status": {
|
||||
@@ -12795,21 +13306,17 @@
|
||||
},
|
||||
"pending_approval": {
|
||||
"default": false,
|
||||
"description": "True when the workstream is parked on ``_approval_event`` awaiting an operator approve/deny. Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"description": "True when at least one approval cycle is live (a gate thread parked awaiting an operator approve/deny). Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"title": "Pending Approval",
|
||||
"type": "boolean"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload \u2014 same shape as ``DashboardWorkstream.pending_approval_detail``. ``None`` when no approval is pending. Lets a reload paint the action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payloads, one per live cycle, oldest first \u2014 same shape as ``DashboardWorkstream.pending_approval_details``. Empty when no approval is pending. Lets a reload paint every action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -12822,8 +13329,14 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalDetail": {
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nSet when a workstream's ``approve_tools`` is parked on\n``_approval_event``; ``None`` (omitted) otherwise. Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nOne entry per live approval CYCLE \u2014 a gate thread parked in\n``approve_tools`` awaiting the operator. Parallel task agents run\nconcurrent gates, so a workstream can have several of these at\nonce (``pending_approval_details``, oldest first). Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"properties": {
|
||||
"cycle_id": {
|
||||
"default": "",
|
||||
"description": "Identity of this approval cycle. Echo it back on ``POST /v1/api/workstreams/{ws_id}/approve`` to resolve exactly this round \u2014 required for correctness when several cycles are live (parallel task agents).",
|
||||
"title": "Cycle Id",
|
||||
"type": "string"
|
||||
},
|
||||
"call_id": {
|
||||
"default": "",
|
||||
"description": "Primary call_id \u2014 first non-empty call_id in items list order. Matches the 409 ``current_call_id`` response from ``POST /v1/api/workstreams/{ws_id}/approve`` so the UI can render the same identifier the server reports as current.",
|
||||
@@ -12848,7 +13361,7 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalItem": {
|
||||
"description": "One pending tool-call inside a ``PendingApprovalDetail`` envelope.\n\nMirrors the dict ``SessionUIBase.serialize_pending_approval_detail``\nemits per item. ``heuristic_verdict`` / ``judge_verdict`` are kept\nloosely-typed because the underlying verdict shape varies by tier;\nconsumers that want the full structure can decode against\n:class:`turnstone.sdk.events.IntentVerdictEvent`.",
|
||||
"description": "One pending tool-call inside a ``PendingApprovalDetail`` envelope.\n\nMirrors the dict ``SessionUIBase.serialize_pending_approval_details``\nemits per item inside each cycle entry. ``heuristic_verdict`` / ``judge_verdict`` are kept\nloosely-typed because the underlying verdict shape varies by tier;\nconsumers that want the full structure can decode against\n:class:`turnstone.sdk.events.IntentVerdictEvent`.",
|
||||
"properties": {
|
||||
"call_id": {
|
||||
"default": "",
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"openapi": "3.1.0",
|
||||
"info": {
|
||||
"title": "turnstone Server API",
|
||||
"version": "1.7.0a2",
|
||||
"version": "1.7.0rc1",
|
||||
"description": "Single-node workstream management, chat interaction, and real-time streaming."
|
||||
},
|
||||
"paths": {
|
||||
@@ -1443,6 +1443,27 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/personas": {
|
||||
"get": {
|
||||
"summary": "List enabled personas for the workstream-creation picker",
|
||||
"operationId": "v1_api_personas_get",
|
||||
"tags": [
|
||||
"Personas"
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Success",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ListPersonaChoicesResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/api/models": {
|
||||
"get": {
|
||||
"summary": "List available model aliases",
|
||||
@@ -2425,6 +2446,12 @@
|
||||
"title": "Skill",
|
||||
"type": "string"
|
||||
},
|
||||
"persona": {
|
||||
"default": "",
|
||||
"description": "Persona name (slug) to create the workstream with. Resolved and snapshotted at creation \u2014 later persona edits never affect this workstream. Empty selects the kind's default persona; on a database with no personas seeded the workstream is created with legacy (unrestricted) behavior.",
|
||||
"title": "Persona",
|
||||
"type": "string"
|
||||
},
|
||||
"notify_targets": {
|
||||
"anyOf": [
|
||||
{
|
||||
@@ -2537,6 +2564,23 @@
|
||||
},
|
||||
"title": "Attachment Ids",
|
||||
"type": "array"
|
||||
},
|
||||
"initial_message_status": {
|
||||
"anyOf": [
|
||||
{
|
||||
"enum": [
|
||||
"queue_full",
|
||||
"refused_closed"
|
||||
],
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Present ONLY when the workstream was created but its initial_message could not be delivered: 'queue_full' (a raced live worker's interjection queue was at capacity \u2014 resend via /send; any uploads stay staged) or 'refused_closed' (the workstream was closed mid-create). Absent whenever the message was dispatched.",
|
||||
"title": "Initial Message Status"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -2665,21 +2709,17 @@
|
||||
},
|
||||
"pending_approval": {
|
||||
"default": false,
|
||||
"description": "True when the workstream is parked on ``_approval_event`` awaiting an operator approve/deny. Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"description": "True when at least one approval cycle is live (a gate thread parked awaiting an operator approve/deny). Mirrors the same field on ``DashboardWorkstream`` / cluster live projections so a freshly-loaded chat tab can render the inline approval gate from the detail snapshot before SSE replay arrives.",
|
||||
"title": "Pending Approval",
|
||||
"type": "boolean"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload \u2014 same shape as ``DashboardWorkstream.pending_approval_detail``. ``None`` when no approval is pending. Lets a reload paint the action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payloads, one per live cycle, oldest first \u2014 same shape as ``DashboardWorkstream.pending_approval_details``. Empty when no approval is pending. Lets a reload paint every action row + judge verdicts immediately instead of relying on the SSE approve_request replay timing window. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -2692,8 +2732,14 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalDetail": {
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nSet when a workstream's ``approve_tools`` is parked on\n``_approval_event``; ``None`` (omitted) otherwise. Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"description": "Inline approval payload merged into ``DashboardWorkstream``.\n\nOne entry per live approval CYCLE \u2014 a gate thread parked in\n``approve_tools`` awaiting the operator. Parallel task agents run\nconcurrent gates, so a workstream can have several of these at\nonce (``pending_approval_details``, oldest first). Cross-tenant\nexposure here follows the same trusted-team posture as\n``activity`` / ``tokens`` \u2014 see ``server.py``'s ``dashboard``\nhandler comment.",
|
||||
"properties": {
|
||||
"cycle_id": {
|
||||
"default": "",
|
||||
"description": "Identity of this approval cycle. Echo it back on ``POST /v1/api/workstreams/{ws_id}/approve`` to resolve exactly this round \u2014 required for correctness when several cycles are live (parallel task agents).",
|
||||
"title": "Cycle Id",
|
||||
"type": "string"
|
||||
},
|
||||
"call_id": {
|
||||
"default": "",
|
||||
"description": "Primary call_id \u2014 first non-empty call_id in items list order. Matches the 409 ``current_call_id`` response from ``POST /v1/api/workstreams/{ws_id}/approve`` so the UI can render the same identifier the server reports as current.",
|
||||
@@ -2718,7 +2764,7 @@
|
||||
"type": "object"
|
||||
},
|
||||
"PendingApprovalItem": {
|
||||
"description": "One pending tool-call inside a ``PendingApprovalDetail`` envelope.\n\nMirrors the dict ``SessionUIBase.serialize_pending_approval_detail``\nemits per item. ``heuristic_verdict`` / ``judge_verdict`` are kept\nloosely-typed because the underlying verdict shape varies by tier;\nconsumers that want the full structure can decode against\n:class:`turnstone.sdk.events.IntentVerdictEvent`.",
|
||||
"description": "One pending tool-call inside a ``PendingApprovalDetail`` envelope.\n\nMirrors the dict ``SessionUIBase.serialize_pending_approval_details``\nemits per item inside each cycle entry. ``heuristic_verdict`` / ``judge_verdict`` are kept\nloosely-typed because the underlying verdict shape varies by tier;\nconsumers that want the full structure can decode against\n:class:`turnstone.sdk.events.IntentVerdictEvent`.",
|
||||
"properties": {
|
||||
"call_id": {
|
||||
"default": "",
|
||||
@@ -2977,17 +3023,13 @@
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"pending_approval_detail": {
|
||||
"anyOf": [
|
||||
{
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"description": "Inline approval payload for the coordinator children-tree UI. Carries the merged ``_pending_approval`` items list + per-call_id LLM verdict cache so a coord can render approve/deny buttons + judge pill without a separate per-child round-trip. ``None`` when no approval is pending. Also surfaced (verbatim) on ``GET /v1/api/cluster/ws/live`` via the ``_CLUSTER_WS_LIVE_KEYS`` projection."
|
||||
"pending_approval_details": {
|
||||
"description": "Inline approval payload for the coordinator children-tree UI: EVERY live approval cycle, oldest first \u2014 parallel task agents gate concurrently, so a workstream can hold several prompts at once. Each entry carries the cycle's items + per-call_id LLM verdict cache so a coord can render approve/deny buttons + judge pill without a separate per-child round-trip; resolve each with its ``cycle_id``. Empty when no approval is pending. Also surfaced (verbatim) on ``GET /v1/api/cluster/ws/live`` via the ``_CLUSTER_WS_LIVE_KEYS`` projection. Replaces 1.6's ``pending_approval_detail`` single-object field (breaking, 1.7).",
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PendingApprovalDetail"
|
||||
},
|
||||
"title": "Pending Approval Details",
|
||||
"type": "array"
|
||||
},
|
||||
"recent_auto_approvals": {
|
||||
"description": "Per-ws ring buffer (cap 10) of recent tool calls that bypassed the operator approval gate. Surfaces ``WebUI._recent_auto_approvals`` so the coord-tree row can render an 'auto-approved by ...' pill when the child's skill / blanket / admin-policy rules silently let a tool through. Also projected onto ``GET /v1/api/cluster/ws/live`` via ``_CLUSTER_WS_LIVE_KEYS``.",
|
||||
@@ -3150,6 +3192,30 @@
|
||||
"default": 0.0,
|
||||
"title": "Context Ratio",
|
||||
"type": "number"
|
||||
},
|
||||
"project_id": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Project Id"
|
||||
},
|
||||
"persona": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"default": null,
|
||||
"title": "Persona"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
@@ -3738,6 +3804,65 @@
|
||||
"title": "ListSkillSummaryResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"PersonaChoice": {
|
||||
"description": "Display fields for the creation picker \u2014 the persona's levers\n(prompt / tool set / toggles) deliberately stay server-side.",
|
||||
"properties": {
|
||||
"name": {
|
||||
"description": "Persona slug, the value to pass as CreateWorkstreamRequest.persona",
|
||||
"title": "Name",
|
||||
"type": "string"
|
||||
},
|
||||
"display_name": {
|
||||
"default": "",
|
||||
"description": "Human-readable name",
|
||||
"title": "Display Name",
|
||||
"type": "string"
|
||||
},
|
||||
"description": {
|
||||
"default": "",
|
||||
"description": "What this persona is for",
|
||||
"title": "Description",
|
||||
"type": "string"
|
||||
},
|
||||
"applies_to_kinds": {
|
||||
"description": "Workstream kinds this persona can be attached to",
|
||||
"items": {
|
||||
"type": "string"
|
||||
},
|
||||
"title": "Applies To Kinds",
|
||||
"type": "array"
|
||||
},
|
||||
"is_default": {
|
||||
"default": false,
|
||||
"description": "Whether an empty persona field resolves to this one",
|
||||
"title": "Is Default",
|
||||
"type": "boolean"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"name"
|
||||
],
|
||||
"title": "PersonaChoice",
|
||||
"type": "object"
|
||||
},
|
||||
"ListPersonaChoicesResponse": {
|
||||
"properties": {
|
||||
"personas": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/PersonaChoice"
|
||||
},
|
||||
"title": "Personas",
|
||||
"type": "array"
|
||||
},
|
||||
"total": {
|
||||
"default": 0,
|
||||
"title": "Total",
|
||||
"type": "integer"
|
||||
}
|
||||
},
|
||||
"title": "ListPersonaChoicesResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"AvailableModelInfo": {
|
||||
"properties": {
|
||||
"alias": {
|
||||
|
||||
Generated
+149
-149
@@ -14,21 +14,21 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@emnapi/core": {
|
||||
"version": "1.10.0",
|
||||
"resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.10.0.tgz",
|
||||
"integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==",
|
||||
"version": "1.11.1",
|
||||
"resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.11.1.tgz",
|
||||
"integrity": "sha512-RSvbQmHzdKzNsLYa/wHrbc3KN4sYLKAdPZxqiM2HATqv/SBk2/ENSHpvXGaLOMcsAyz0poEGqkmmKYG3OWiJEQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"@emnapi/wasi-threads": "1.2.1",
|
||||
"@emnapi/wasi-threads": "1.2.2",
|
||||
"tslib": "^2.4.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@emnapi/runtime": {
|
||||
"version": "1.10.0",
|
||||
"resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.10.0.tgz",
|
||||
"integrity": "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA==",
|
||||
"version": "1.11.1",
|
||||
"resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.11.1.tgz",
|
||||
"integrity": "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
@@ -37,9 +37,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@emnapi/wasi-threads": {
|
||||
"version": "1.2.1",
|
||||
"resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz",
|
||||
"integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==",
|
||||
"version": "1.2.2",
|
||||
"resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.2.tgz",
|
||||
"integrity": "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
@@ -55,14 +55,14 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@napi-rs/wasm-runtime": {
|
||||
"version": "1.1.5",
|
||||
"resolved": "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.1.5.tgz",
|
||||
"integrity": "sha512-AWPoBRJ9tsnVhor4sjO7rkni+7p+2IAEFj6cx06UgP10jkQHqay/36uRV/bFkgrh18D9vb4cr8Q0Pthskgzy+Q==",
|
||||
"version": "1.1.6",
|
||||
"resolved": "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.1.6.tgz",
|
||||
"integrity": "sha512-ZLv/JdUfkvOy9eCnnBaGfiO+XimbjebAeO+MRQqD/B+FR1tnRN0tpKSJHRbE8sFfS6aqsXZ67TQjfwfsxULVbg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"@tybys/wasm-util": "^0.10.2"
|
||||
"@tybys/wasm-util": "^0.10.3"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
@@ -74,9 +74,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@oxc-project/types": {
|
||||
"version": "0.133.0",
|
||||
"resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.133.0.tgz",
|
||||
"integrity": "sha512-KzkdCd6Uxqnf6l3HOw1xfatAlUURA0g14cvBYFyJ5SaNOQbOUvBr9PKArcPcrNIeRsBdgcUzOGrhKveVpvOIGA==",
|
||||
"version": "0.138.0",
|
||||
"resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.138.0.tgz",
|
||||
"integrity": "sha512-1a7ZKmrRTCoN1XMZ4L0PyyqrMnrNlLyPuOkdSX2MZg7IiIGRUyurNhAm73ptDOraoBcIordsIGKNPKUzy3ZmfA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
@@ -84,9 +84,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-android-arm64": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.0.3.tgz",
|
||||
"integrity": "sha512-454rs7jHngixp/NMxd5srYD57OnzSlZ/eFTETjORQHLwJG1lRtmNOJcBerZlfu4GjKqeq8aCCIQrMdHyhI51Hw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.1.4.tgz",
|
||||
"integrity": "sha512-EZLpf/8y7GXkkra90ML47kzik/GMP3EMcE9bPyHmRfxLC6z9+aW5A8poCsoxjrT5GfEcNAAvWwUHjvP1pUQkfw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -101,9 +101,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-darwin-arm64": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.0.3.tgz",
|
||||
"integrity": "sha512-PcAhP+ynjURNyy8SKGl5DQP94aGuB/7JrXJb/t7P+hanXvQVMWzUvRRhBAcg/lNRadBhoUPqSoP4xw5tR/KBEA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.1.4.tgz",
|
||||
"integrity": "sha512-aUi+HBvmYb7j8krl1+qJgkG8C17fO79gk3c+jPw4S8glRFc1DTija9S3EyaTSQUm5GJXYKDAsugBEhFHH2vYiQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -118,9 +118,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-darwin-x64": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.0.3.tgz",
|
||||
"integrity": "sha512-9YpfeUvSE2RS7wysJ81uOZkXJz7f7Q55H2Gvp3VEw/EsahqDtrphrZ0EwDLK5vvKOzaCrBsjF8JmnMLcUt78Gg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.1.4.tgz",
|
||||
"integrity": "sha512-F7hHC3gwY11+vByKPRWqwGbeXWVgKmL+pTGCinaEhdihzBV2aQ0fvZOch9cXYUOKuKKq429HeYXOqQLc7wFCEg==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -135,9 +135,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-freebsd-x64": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.0.3.tgz",
|
||||
"integrity": "sha512-yB1IlAsSNHncV6SCTL27/MVGR5htvQsoGxIv5KMGXALp+Ll1wYsn+x98M9MW7qa+NdSbvrrY7ANI4wLJ0n1e6g==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.1.4.tgz",
|
||||
"integrity": "sha512-sI5yw+7s92SK6odiEhD5lKCBlWcpjHS5qyqpVQbZAJ0fIzEUXrmbl3DH2ybR3PZogulNJF+COLtmA8hUfvkCCQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -152,9 +152,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-arm-gnueabihf": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.0.3.tgz",
|
||||
"integrity": "sha512-Yi30IVAAfLUCy2MseFjbB1jAMDl1VMCAas5StnYp8da9+CKvMd2H2cbEjWcw5NPaPqzvYkVIaF1nNUG+b7u/sw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.1.4.tgz",
|
||||
"integrity": "sha512-mCi0OKgEieFircrtVYmQAFGszRtMnZ6fpZAXrxanXAu7lqZcsK1E1RAaZNG0uKAnxox3B1f4EyQNnoyMfN1vAA==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
@@ -169,9 +169,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-arm64-gnu": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.0.3.tgz",
|
||||
"integrity": "sha512-jsO7R8To+AdlYgUmN5sHSCZbfhtMBkO0WUx8iORQnPcMMdgr7qM2DQmMwgabs3GhNztdmoKkMKQFHD6DTMCIQw==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-B9Ial3Kv5sh0SHnB1g/QWcUQCEvCF6QKGAl4zXypYj65mVI+B4AhFBwPtSN7pDrJeIx8Z7zdy4ntx+wQABom7w==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -189,9 +189,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-arm64-musl": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.0.3.tgz",
|
||||
"integrity": "sha512-VWkUHwWriDciit80wleYwKILoR/KMvxh/IdwS/paX+ZgpuRpCrKLUdadJbc0NpBEiyhpYawsJ73j9aCvOH+f7Q==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.1.4.tgz",
|
||||
"integrity": "sha512-lZVym0PuHE1KZ22gmFTC15lAkrg9iTszR617oYRB/iPY1A56ywoJzVKOJBKaot5RiikCObmur6pogpse3gRcng==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -209,9 +209,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-ppc64-gnu": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.0.3.tgz",
|
||||
"integrity": "sha512-5f1laC0SlIR0yDbFCd8acUhvJIag6N3zC5P7oUPN6wX0aOma+uKJ0wBDH5aq7I1PVI2ttTlhJwzwRIBnLiSGEg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-t2DNiLJWNTbnEHyUzTumldML6ET4/g16467LZoDDJ3tSxGvguL5/NyC2lCsNKuyRycg9XeDQF5SSv+TNOhQEXg==",
|
||||
"cpu": [
|
||||
"ppc64"
|
||||
],
|
||||
@@ -229,9 +229,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-s390x-gnu": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.0.3.tgz",
|
||||
"integrity": "sha512-Iq4ko0r4XsgbrF/LunNgHtAGLRRVE2kXonAXQ/MV0mC6jQpMOhW1SvtZja2EhC/kd05++bP78dsqBeIQyYJ6Yg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-0WIRnL1Uw4BvTZRLQt+PVgo6ZKTJadlC2btP+/EOXv2f/DWbY0rEgl+y834mIVwP1FkTlWVTrGGJXf12lru7EQ==",
|
||||
"cpu": [
|
||||
"s390x"
|
||||
],
|
||||
@@ -249,9 +249,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-x64-gnu": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.0.3.tgz",
|
||||
"integrity": "sha512-B8m6tD5+/N5FeNQFbKlLA/2yVq9ycQP1SeedyEYYKWBNR3ZQbkvIUcNnDNM03lO1l5F2roiiFJGgvoLLyZXtSg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.1.4.tgz",
|
||||
"integrity": "sha512-JWtGshGfX+oENAKonoNkqEJX+7hC8yfhi9GUyPX1VX4mdh1y5r+ZiJLR5XzAB0aoP6s/PcILsGjKq8O0mm24bw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -269,9 +269,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-linux-x64-musl": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.0.3.tgz",
|
||||
"integrity": "sha512-pSdpdUJHkuCxun9LE7jvgUB9qsRgaiyNNCX7m/AvHTcq67AiT/Yhoxvw5zPfhrM8k/BfP8ce/hMOpthKDpEUow==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.1.4.tgz",
|
||||
"integrity": "sha512-rT6yQcxUuXs4CnbofqwHRRV0iem349rLMYpTjkgQGLjrY4ado/eDzwPZPTCgTOlF6Nkp8NEv70yLMTn6qkWxsQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -289,9 +289,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-openharmony-arm64": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.0.3.tgz",
|
||||
"integrity": "sha512-OXXS3RKJgX2uLwM+gYyuH5omcH8fL1LJs96pZGgtetVCahON57+d4SJHzTgZiOjxgGkSnpXpOsWuPDGAKAigEg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.1.4.tgz",
|
||||
"integrity": "sha512-KXMGoboq5cyaCQjDA4GLuRiOwBQ0EyFnJoVViLeZ45/3rFItRODEr+NdsBcVpll40hhNArlm/speWGRvj08LzA==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -306,9 +306,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-wasm32-wasi": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-wasm32-wasi/-/binding-wasm32-wasi-1.0.3.tgz",
|
||||
"integrity": "sha512-JTtb8BWFynicNSoPrehsCzBtOKjZ6jhMiPFEmOiuXg1Fl8dn2KHQob+GuPSGR0dryQa1PQJbzjF3dqO/whhjLg==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-wasm32-wasi/-/binding-wasm32-wasi-1.1.4.tgz",
|
||||
"integrity": "sha512-5K83rb36oJiY7BCyE9zLZtGcPV4g5wvq+xwdO0XPIwDVZI8cyB/AUjkNXGb92/rnmezEkjMOpgY61rtwjQtFwg==",
|
||||
"cpu": [
|
||||
"wasm32"
|
||||
],
|
||||
@@ -316,18 +316,18 @@
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"@emnapi/core": "1.10.0",
|
||||
"@emnapi/runtime": "1.10.0",
|
||||
"@napi-rs/wasm-runtime": "^1.1.4"
|
||||
"@emnapi/core": "1.11.1",
|
||||
"@emnapi/runtime": "1.11.1",
|
||||
"@napi-rs/wasm-runtime": "^1.1.6"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^20.19.0 || >=22.12.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-win32-arm64-msvc": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.0.3.tgz",
|
||||
"integrity": "sha512-gEdFFEN70A/jxb2svrWsN3aDL7OUtmvlOy+6fa2jxG8K0wQ1ZbdeLGnidov6Yu5/733dI5ySfzFlQ/cb0bSz1g==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.1.4.tgz",
|
||||
"integrity": "sha512-PnWBtw3TV5KOg69HQQDR0mnQuyCmSGR2pAB4DC1rPF808fgKeTUMj2EOEyKATpgiuxuR5APQmiDO7PDgEjTFSA==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -342,9 +342,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@rolldown/binding-win32-x64-msvc": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.0.3.tgz",
|
||||
"integrity": "sha512-eXB7CHuaQdqmJcc3koCNtNPmT/bj2gc999kUFgBxG8Ac0NdgXc4rkCHhqrgrhN3zddvvvrgzj1e90SuSfmyIXA==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.1.4.tgz",
|
||||
"integrity": "sha512-M1lpniBePobTfsa7Ks9a199e1akxsXn+GYBUKsEzv3YFzOm1HJAMNwKI3qr0Zq+mxwx9gOZoTdP1yXRYsZUocQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -373,9 +373,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@tybys/wasm-util": {
|
||||
"version": "0.10.2",
|
||||
"resolved": "https://registry.npmjs.org/@tybys/wasm-util/-/wasm-util-0.10.2.tgz",
|
||||
"integrity": "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg==",
|
||||
"version": "0.10.3",
|
||||
"resolved": "https://registry.npmjs.org/@tybys/wasm-util/-/wasm-util-0.10.3.tgz",
|
||||
"integrity": "sha512-F3fo1MYrRJYL3zER0OUOmkutjr1Vp23m7OsSgp7nq4SP6OqX6C/56XFIPAl5bt3zaBRjmW7SGz3u/6LwFpYcOg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
@@ -409,16 +409,16 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@vitest/expect": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.9.tgz",
|
||||
"integrity": "sha512-vl/rYsUKcBr3SnQn166+XR5ZQcgMx3DQhFWdfli/cWpLnLUmbxZvyrJZotLFUryib+LtArYMSTJ5RbQ57ZqrlA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.10.tgz",
|
||||
"integrity": "sha512-YsCn+qAk1GWjQOWFEsEcL2gNQ0zmVmQu3T03qP6UyjhtmdtwtbuI+DASn/7iQB3HGTXkdBwGddzxPlmiql5vlA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@standard-schema/spec": "^1.1.0",
|
||||
"@types/chai": "^5.2.2",
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"chai": "^6.2.2",
|
||||
"tinyrainbow": "^3.1.0"
|
||||
},
|
||||
@@ -427,13 +427,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/mocker": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.9.tgz",
|
||||
"integrity": "sha512-EVkXzBjrPGM+cK8/ANWgBrkUCfJfb38/EfTSO8h7pWvKkyPkpWxvR7BkD2MyItMF62C97zAEoqdpUixwR/e+Rw==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.10.tgz",
|
||||
"integrity": "sha512-v0xaezt+DKEmKfaxg133ldzADrwLGd7Ze1MfQQTYfvs8OqZIwbxyxaYURivwV7sWy5fqn3rH5uOrSp07bp44Ow==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"estree-walker": "^3.0.3",
|
||||
"magic-string": "^0.30.21"
|
||||
},
|
||||
@@ -454,9 +454,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/pretty-format": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.9.tgz",
|
||||
"integrity": "sha512-s0iufns3iIFitdgm+YR7g1whCAaGtXz459VS9/PqyKDEEFgYIhsHOQmXgIgDuYCt7DeQmiZT0Qe2OA2p4ZPu5A==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.10.tgz",
|
||||
"integrity": "sha512-W1HsjSH4MXQ9YfmmhLAoIYf1HRfekQCGngeIgcei6MP5QQGWUe0gkopdZQaVCFO+JDJMrAJGwa5pRpNpvy4P8Q==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -467,13 +467,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/runner": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.9.tgz",
|
||||
"integrity": "sha512-KXLMDtc7oe70+3mJfGrPUWPesswH+3sTxAMAMl8DG7I8IUQT4XW718dY5ID3vPUcmlu27CcKfY4P3h3I29SLJg==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.10.tgz",
|
||||
"integrity": "sha512-IKI6kpIH+LmpROplyLwBBaCfMgOZOMsygVa6BARD6ahA04VRuJSa6OaVG7kRvSEMD870Vd91rSSw0eegtWyLGg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"pathe": "^2.0.3"
|
||||
},
|
||||
"funding": {
|
||||
@@ -481,14 +481,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/snapshot": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.9.tgz",
|
||||
"integrity": "sha512-Jc7RKGNBo8Z28WYIm0Niej4xdSPByRf6mU58VpHQkd6Zh05rlnA+twjbK5HyeIGHxrzsc3mJgS43uM0CZKzaIA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.10.tgz",
|
||||
"integrity": "sha512-xRkfOT1qpTAi/Ti4Y1LtfRc3kEuqxGw59eN2jN9pRWMtS/XDevekhcFSqvQqjUNGksfjMJu3Y+oJ+4Ypn2OaJw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"magic-string": "^0.30.21",
|
||||
"pathe": "^2.0.3"
|
||||
},
|
||||
@@ -497,9 +497,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/spy": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.9.tgz",
|
||||
"integrity": "sha512-fHpsS6mIi+PiEW+vcRVOMkX1oSaPKne3VOclSFICPcGOmfKgXPU5iAah+wcNcj2xPrCCmfq99IDGf+EojhhvhA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.10.tgz",
|
||||
"integrity": "sha512-PLf/Ugvoq5wO/b4rwYCR1h2PSIdXz7wnkQFMiUpLdtM7l6pqVFcQIBEHyT1+l+cj7mNwAfZHzqXqDyjvOuwbDw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
@@ -507,13 +507,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/utils": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.9.tgz",
|
||||
"integrity": "sha512-A51o8ymO5PpqlWNnBP9ZHPXDIpuMtTLlGSjN7la4US+LJzoUMyhwjA5QXlm39JexgwHKW4Xjs8Z2d3dLCXOeuA==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.10.tgz",
|
||||
"integrity": "sha512-fy9am/HWxbaGt/Sawrp90vt6Y6jQwf1RX77cz3uwoJwJVMli/e1IEwRPnMNJ7vKfPTwo0diXifkpPvwH9v7nGA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"convert-source-map": "^2.0.0",
|
||||
"tinyrainbow": "^3.1.0"
|
||||
},
|
||||
@@ -559,9 +559,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/es-module-lexer": {
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/es-module-lexer/-/es-module-lexer-2.1.0.tgz",
|
||||
"integrity": "sha512-n27zTYMjYu1aj4MjCWzSP7G9r75utsaoc8m61weK+W8JMBGGQybd43GstCXZ3WNmSFtGT9wi59qQTW6mhTR5LQ==",
|
||||
"version": "2.3.0",
|
||||
"resolved": "https://registry.npmjs.org/es-module-lexer/-/es-module-lexer-2.3.0.tgz",
|
||||
"integrity": "sha512-KLdwQm2NvGLDkQDCGvmiQrhkd0JbMzXthwQAUgWjQuQdBLFa3eiBP5arXZyA+f8x+x7OXgud6bq2rxjGtHV2tw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
@@ -576,9 +576,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/expect-type": {
|
||||
"version": "1.3.0",
|
||||
"resolved": "https://registry.npmjs.org/expect-type/-/expect-type-1.3.0.tgz",
|
||||
"integrity": "sha512-knvyeauYhqjOYvQ66MznSMs83wmHrCycNEN6Ao+2AeYEfxUIkuiVxdEa1qlGEPK+We3n0THiDciYSsCcgW/DoA==",
|
||||
"version": "1.4.0",
|
||||
"resolved": "https://registry.npmjs.org/expect-type/-/expect-type-1.4.0.tgz",
|
||||
"integrity": "sha512-KfYbmpRm0VbLjEvVa9yGwCi9GI34xvi7A/HXYWQO65CSD2u3MczUJSuwXKFIxlGsgBQizV9q5J9NHj4VG0n+pA==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"engines": {
|
||||
@@ -949,9 +949,9 @@
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/picomatch": {
|
||||
"version": "4.0.4",
|
||||
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz",
|
||||
"integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==",
|
||||
"version": "4.0.5",
|
||||
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.5.tgz",
|
||||
"integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -962,9 +962,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/postcss": {
|
||||
"version": "8.5.15",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.15.tgz",
|
||||
"integrity": "sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A==",
|
||||
"version": "8.5.16",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.16.tgz",
|
||||
"integrity": "sha512-vuwillviilfKZsg0VGj5R/YwwcHx4SLsIOI/7K6mQkWx+l5cUHTjj5g0AasTBcyXsbfTgrwsUNmVUb5xVwyPwg==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -991,13 +991,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/rolldown": {
|
||||
"version": "1.0.3",
|
||||
"resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.0.3.tgz",
|
||||
"integrity": "sha512-i00lAJ2ks1BYr7rjNjKC7BcqAS7nVfiT3QX1SI5aY+AFHblCmaUf9OE9dbdzDvW6dJxbi2ZCZiy9v3CcwOiX3g==",
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.1.4.tgz",
|
||||
"integrity": "sha512-IjZYiLxZwpnhwhdBH2ugdTGVSdhCQUmLxLoqyjiL0JxYjyRst+5a0P3xfrTxJ5F638j4Mvvw5FAX5XE6eHpXbA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@oxc-project/types": "=0.133.0",
|
||||
"@oxc-project/types": "=0.138.0",
|
||||
"@rolldown/pluginutils": "^1.0.0"
|
||||
},
|
||||
"bin": {
|
||||
@@ -1007,21 +1007,21 @@
|
||||
"node": "^20.19.0 || >=22.12.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@rolldown/binding-android-arm64": "1.0.3",
|
||||
"@rolldown/binding-darwin-arm64": "1.0.3",
|
||||
"@rolldown/binding-darwin-x64": "1.0.3",
|
||||
"@rolldown/binding-freebsd-x64": "1.0.3",
|
||||
"@rolldown/binding-linux-arm-gnueabihf": "1.0.3",
|
||||
"@rolldown/binding-linux-arm64-gnu": "1.0.3",
|
||||
"@rolldown/binding-linux-arm64-musl": "1.0.3",
|
||||
"@rolldown/binding-linux-ppc64-gnu": "1.0.3",
|
||||
"@rolldown/binding-linux-s390x-gnu": "1.0.3",
|
||||
"@rolldown/binding-linux-x64-gnu": "1.0.3",
|
||||
"@rolldown/binding-linux-x64-musl": "1.0.3",
|
||||
"@rolldown/binding-openharmony-arm64": "1.0.3",
|
||||
"@rolldown/binding-wasm32-wasi": "1.0.3",
|
||||
"@rolldown/binding-win32-arm64-msvc": "1.0.3",
|
||||
"@rolldown/binding-win32-x64-msvc": "1.0.3"
|
||||
"@rolldown/binding-android-arm64": "1.1.4",
|
||||
"@rolldown/binding-darwin-arm64": "1.1.4",
|
||||
"@rolldown/binding-darwin-x64": "1.1.4",
|
||||
"@rolldown/binding-freebsd-x64": "1.1.4",
|
||||
"@rolldown/binding-linux-arm-gnueabihf": "1.1.4",
|
||||
"@rolldown/binding-linux-arm64-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-arm64-musl": "1.1.4",
|
||||
"@rolldown/binding-linux-ppc64-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-s390x-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-x64-gnu": "1.1.4",
|
||||
"@rolldown/binding-linux-x64-musl": "1.1.4",
|
||||
"@rolldown/binding-openharmony-arm64": "1.1.4",
|
||||
"@rolldown/binding-wasm32-wasi": "1.1.4",
|
||||
"@rolldown/binding-win32-arm64-msvc": "1.1.4",
|
||||
"@rolldown/binding-win32-x64-msvc": "1.1.4"
|
||||
}
|
||||
},
|
||||
"node_modules/siginfo": {
|
||||
@@ -1122,16 +1122,16 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vite": {
|
||||
"version": "8.0.16",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.0.16.tgz",
|
||||
"integrity": "sha512-h9bXPmJichP5fLmVQo3PyaGSDE2n3aPuomeAlVRm0JLmt4rY6zmPKd59HYI4LNW8oTK7tlTsuC7l/m7awx9Jcw==",
|
||||
"version": "8.1.3",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.3.tgz",
|
||||
"integrity": "sha512-Ds+gBRbj0lwRO2Y5hwnUBdxSwlAve9LeRyU4sNnAr0ewW0gWF0n5bgXgUzbgZ49MV9BVUAQUFYVcDUcilUExMA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"lightningcss": "^1.32.0",
|
||||
"picomatch": "^4.0.4",
|
||||
"postcss": "^8.5.15",
|
||||
"rolldown": "1.0.3",
|
||||
"postcss": "^8.5.16",
|
||||
"rolldown": "~1.1.3",
|
||||
"tinyglobby": "^0.2.17"
|
||||
},
|
||||
"bin": {
|
||||
@@ -1148,7 +1148,7 @@
|
||||
},
|
||||
"peerDependencies": {
|
||||
"@types/node": "^20.19.0 || >=22.12.0",
|
||||
"@vitejs/devtools": "^0.1.18",
|
||||
"@vitejs/devtools": "^0.3.0",
|
||||
"esbuild": "^0.27.0 || ^0.28.0",
|
||||
"jiti": ">=1.21.0",
|
||||
"less": "^4.0.0",
|
||||
@@ -1200,19 +1200,19 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vitest": {
|
||||
"version": "4.1.9",
|
||||
"resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.9.tgz",
|
||||
"integrity": "sha512-nE3/LEyc0z87uHYLZebqCUOaJr2hdtuPp7BQ4BosVFnfltxgAvMG08NyrSGlPpOUWvR27c5flSmYFTNr78L9GQ==",
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.10.tgz",
|
||||
"integrity": "sha512-R9jUTe5S4Qb0HCd4TNqpC7oGcrMssMRGXLW80ubjWsW9VH5GF8y1Y0SFLY9AbqSk6nt0PnOx4H4WNJYZ13GUPw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@vitest/expect": "4.1.9",
|
||||
"@vitest/mocker": "4.1.9",
|
||||
"@vitest/pretty-format": "4.1.9",
|
||||
"@vitest/runner": "4.1.9",
|
||||
"@vitest/snapshot": "4.1.9",
|
||||
"@vitest/spy": "4.1.9",
|
||||
"@vitest/utils": "4.1.9",
|
||||
"@vitest/expect": "4.1.10",
|
||||
"@vitest/mocker": "4.1.10",
|
||||
"@vitest/pretty-format": "4.1.10",
|
||||
"@vitest/runner": "4.1.10",
|
||||
"@vitest/snapshot": "4.1.10",
|
||||
"@vitest/spy": "4.1.10",
|
||||
"@vitest/utils": "4.1.10",
|
||||
"es-module-lexer": "^2.0.0",
|
||||
"expect-type": "^1.3.0",
|
||||
"magic-string": "^0.30.21",
|
||||
@@ -1240,12 +1240,12 @@
|
||||
"@edge-runtime/vm": "*",
|
||||
"@opentelemetry/api": "^1.9.0",
|
||||
"@types/node": "^20.0.0 || ^22.0.0 || >=24.0.0",
|
||||
"@vitest/browser-playwright": "4.1.9",
|
||||
"@vitest/browser-preview": "4.1.9",
|
||||
"@vitest/browser-webdriverio": "4.1.9",
|
||||
"@vitest/coverage-istanbul": "4.1.9",
|
||||
"@vitest/coverage-v8": "4.1.9",
|
||||
"@vitest/ui": "4.1.9",
|
||||
"@vitest/browser-playwright": "4.1.10",
|
||||
"@vitest/browser-preview": "4.1.10",
|
||||
"@vitest/browser-webdriverio": "4.1.10",
|
||||
"@vitest/coverage-istanbul": "4.1.10",
|
||||
"@vitest/coverage-v8": "4.1.10",
|
||||
"@vitest/ui": "4.1.10",
|
||||
"happy-dom": "*",
|
||||
"jsdom": "*",
|
||||
"vite": "^6.0.0 || ^7.0.0 || ^8.0.0"
|
||||
|
||||
@@ -75,15 +75,35 @@ export interface ToolInfoEvent {
|
||||
items: Array<Record<string, unknown>>;
|
||||
}
|
||||
|
||||
/** One approval CYCLE awaiting the operator. Several can be outstanding
|
||||
* at once (parallel task agents each gate their own tool calls) — key
|
||||
* prompt UI by `cycle_id` and echo it back on the approve POST.
|
||||
*
|
||||
* `cycle_id` is optional because it was added in 1.7: a pre-1.7 server
|
||||
* omits it on the wire, so a current SDK talking to an older node sees
|
||||
* `undefined`. Resolve those the legacy way (no selector → oldest
|
||||
* cycle). A current server always sends it. */
|
||||
export interface ApproveRequestEvent {
|
||||
type: "approve_request";
|
||||
cycle_id?: string;
|
||||
items: Array<Record<string, unknown>>;
|
||||
judge_pending?: boolean;
|
||||
}
|
||||
|
||||
/** A specific approval cycle resolved; `cycle_id`/`call_ids` identify
|
||||
* which prompt to dismiss.
|
||||
*
|
||||
* Both are optional for the same reason as `ApproveRequestEvent.cycle_id`
|
||||
* — a pre-1.7 server emits neither, so a bare "something resolved"
|
||||
* dismisses the sole tracked prompt (the legacy fallback the UI and
|
||||
* channel adapters keep). A current server always sends both. */
|
||||
export interface ApprovalResolvedEvent {
|
||||
type: "approval_resolved";
|
||||
approved: boolean;
|
||||
feedback: string;
|
||||
always?: boolean;
|
||||
cycle_id?: string;
|
||||
call_ids?: string[];
|
||||
}
|
||||
|
||||
export interface ToolResultEvent {
|
||||
|
||||
@@ -166,6 +166,13 @@ export class TurnstoneServer extends BaseClient {
|
||||
approved?: boolean;
|
||||
feedback?: string | null;
|
||||
always?: boolean;
|
||||
/** Resolve exactly this approval cycle (from ApproveRequestEvent.cycle_id).
|
||||
* Omitting it resolves the OLDEST live cycle — ambiguous when parallel
|
||||
* task agents have several prompts outstanding, so pass it whenever the
|
||||
* triggering event is known. */
|
||||
cycleId?: string;
|
||||
/** Alternative selector: any call_id inside the target cycle. */
|
||||
callId?: string;
|
||||
}): Promise<StatusResponse> {
|
||||
return this.request(
|
||||
"POST",
|
||||
@@ -175,6 +182,8 @@ export class TurnstoneServer extends BaseClient {
|
||||
approved: opts.approved ?? true,
|
||||
feedback: opts.feedback,
|
||||
always: opts.always,
|
||||
cycle_id: opts.cycleId,
|
||||
call_id: opts.callId,
|
||||
},
|
||||
},
|
||||
);
|
||||
|
||||
@@ -130,6 +130,12 @@ export interface CreateWorkstreamRequest {
|
||||
auto_approve?: boolean;
|
||||
resume_ws?: string;
|
||||
skill?: string;
|
||||
/**
|
||||
* Persona name (slug) to create the workstream with. Resolved and
|
||||
* snapshotted at creation — later persona edits never affect this
|
||||
* workstream. Empty selects the kind's default persona.
|
||||
*/
|
||||
persona?: string;
|
||||
/**
|
||||
* Optional project to attach this workstream to. Drives the shared
|
||||
* `project` memory scope; coordinator children inherit the parent's project.
|
||||
@@ -158,6 +164,13 @@ export interface CreateWorkstreamResponse {
|
||||
message_count?: number;
|
||||
/** Ids of attachments saved by this request (multipart variant only). */
|
||||
attachment_ids?: string[];
|
||||
/**
|
||||
* Present ONLY when the workstream was created but its initial_message
|
||||
* could not be delivered: "queue_full" (raced live worker's interjection
|
||||
* queue at capacity — resend via /send; uploads stay staged) or
|
||||
* "refused_closed" (workstream closed mid-create).
|
||||
*/
|
||||
initial_message_status?: "queue_full" | "refused_closed";
|
||||
}
|
||||
|
||||
export interface CloseWorkstreamRequest {
|
||||
@@ -256,6 +269,8 @@ export interface SavedWorkstreamInfo {
|
||||
child_count?: number;
|
||||
context_tokens?: number;
|
||||
context_ratio?: number;
|
||||
/** Persona slug the workstream was created with (empty/absent = pre-persona). */
|
||||
persona?: string | null;
|
||||
}
|
||||
|
||||
export interface ListSavedWorkstreamsResponse {
|
||||
@@ -524,6 +539,8 @@ export interface ConsoleCreateWsRequest {
|
||||
model?: string;
|
||||
initial_message?: string;
|
||||
skill?: string;
|
||||
/** Persona slug — resolved and snapshotted at creation. */
|
||||
persona?: string;
|
||||
resume_ws?: string;
|
||||
}
|
||||
|
||||
|
||||
@@ -104,7 +104,6 @@ describe("TurnstoneServer attachments", () => {
|
||||
const [, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(JSON.parse(init.body)).toEqual({
|
||||
message: "hi",
|
||||
ws_id: "ws-X",
|
||||
attachment_ids: ["a1", "a2"],
|
||||
});
|
||||
});
|
||||
@@ -117,7 +116,7 @@ describe("TurnstoneServer attachments", () => {
|
||||
});
|
||||
await client.send("hi", "ws-X");
|
||||
const [, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "hi", ws_id: "ws-X" });
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "hi" });
|
||||
});
|
||||
|
||||
it("createWorkstream with attachments sends multipart and auto-generates ws_id", async () => {
|
||||
|
||||
@@ -74,8 +74,8 @@ describe("TurnstoneServer", () => {
|
||||
await client.send("Hello", "ws1");
|
||||
|
||||
const [url, init] = (fetchFn as ReturnType<typeof vi.fn>).mock.calls[0];
|
||||
expect(url).toBe("http://test/v1/api/send");
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "Hello", ws_id: "ws1" });
|
||||
expect(url).toBe("http://test/v1/api/workstreams/ws1/send");
|
||||
expect(JSON.parse(init.body)).toEqual({ message: "Hello" });
|
||||
});
|
||||
|
||||
it("injects auth header when token provided", async () => {
|
||||
|
||||
+29
-1
@@ -3,9 +3,37 @@ not fixtures, and several test files want to import them directly."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
|
||||
|
||||
def wait_until(cond: Callable[[], bool], timeout: float = 5.0) -> None:
|
||||
"""Poll ``cond`` to True within ``timeout`` or fail the test.
|
||||
|
||||
The worker/wake tests can't join threads by identity:
|
||||
``session_worker.send`` assigns ``ws.worker_thread`` under the lock
|
||||
BEFORE ``t.start()``, so the instant a dispatching call returns, a
|
||||
fast worker may already have run its exit backstop and installed the
|
||||
(not-yet-started) wake thread — joining whatever ``ws.worker_thread``
|
||||
points at races ``RuntimeError: cannot join thread before it is
|
||||
started``. Poll outcomes instead.
|
||||
"""
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if cond():
|
||||
return
|
||||
time.sleep(0.005)
|
||||
if cond():
|
||||
# Final re-check: the condition can become true during the last
|
||||
# sleep (or a CI descheduling stall past the deadline) — failing
|
||||
# without re-looking makes the helper itself a flake source.
|
||||
return
|
||||
raise AssertionError("condition not met within timeout")
|
||||
|
||||
|
||||
def make_chat_session(**overrides: Any) -> Any:
|
||||
"""Build a minimal ``ChatSession`` with sane test defaults.
|
||||
|
||||
@@ -51,6 +51,12 @@ def make_replay_mocks(
|
||||
ui._ws_messages = 0
|
||||
for key, value in ui_overrides.items():
|
||||
setattr(ui, key, value)
|
||||
# Both replay paths read cycle cards via ``pending_approval_cards()``
|
||||
# (one card per concurrent approval cycle). Model it from the
|
||||
# single-slot ``_pending_approval`` override so tests keep seeding
|
||||
# the one field; a bare MagicMock here would iterate empty and
|
||||
# silently drop the approve_request from the replay.
|
||||
ui.pending_approval_cards = lambda: [ui._pending_approval] if ui._pending_approval else []
|
||||
ws = MagicMock()
|
||||
ws.session = session
|
||||
request = MagicMock()
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
"""Recording fake SDK client — captures the kwargs at each provider's seam.
|
||||
|
||||
Every provider's ``create_streaming`` assembles its kwargs and calls the
|
||||
SDK *eagerly* before returning the stream iterator (Anthropic
|
||||
``client.messages.stream``, OpenAI ``client.chat.completions.create``,
|
||||
Responses ``client.responses.create/stream``), so driving a provider
|
||||
against a :class:`RecordingClient` captures the full composed request
|
||||
payload without a network round-trip.
|
||||
|
||||
Shared by the wire-payload golden harness (``test_wire_payload_golden``)
|
||||
and the effort-ladder parity harness (``test_effort_ladder_wire_parity``)
|
||||
so both assert against the same capture seam.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
|
||||
class _EmptyStream:
|
||||
"""Stand-in for an SDK stream / stream-manager: empty iterable AND no-op CM."""
|
||||
|
||||
def __iter__(self) -> Iterator[Any]:
|
||||
return iter(())
|
||||
|
||||
def __enter__(self) -> _EmptyStream:
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc: object) -> None:
|
||||
return None
|
||||
|
||||
|
||||
class _Seam:
|
||||
"""Records the kwargs of a single SDK call, returns an empty stream stub."""
|
||||
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self._sink = sink
|
||||
|
||||
def __call__(self, **kwargs: Any) -> _EmptyStream:
|
||||
# Last write wins; only one seam is exercised per provider call.
|
||||
self._sink["payload"] = kwargs
|
||||
return _EmptyStream()
|
||||
|
||||
|
||||
class _Completions:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
|
||||
|
||||
class _Chat:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.completions = _Completions(sink)
|
||||
|
||||
|
||||
class _Messages:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class _Responses:
|
||||
def __init__(self, sink: dict[str, Any]) -> None:
|
||||
self.create = _Seam(sink)
|
||||
self.stream = _Seam(sink)
|
||||
|
||||
|
||||
class RecordingClient:
|
||||
"""Fake SDK client exposing every provider's call seam, recording kwargs."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.captured: dict[str, Any] = {}
|
||||
self.messages = _Messages(self.captured)
|
||||
self.chat = _Chat(self.captured)
|
||||
self.responses = _Responses(self.captured)
|
||||
+69
-1
@@ -52,8 +52,76 @@ def serve_until_exit(server: Any) -> None:
|
||||
loop.close()
|
||||
|
||||
|
||||
class _PendingResolver:
|
||||
"""Race-free drop-in for ``threading.Timer(delay, ui.resolve_approval)``.
|
||||
|
||||
``approve_tools`` runs ``_approval_event.clear()`` -> register
|
||||
``_pending_approval`` -> ``_approval_event.wait(_APPROVAL_WAIT_TIMEOUT)``
|
||||
(3600s). A *fixed-delay* timer can fire ``resolve_approval``
|
||||
(``_approval_event.set()``) BEFORE that ``.clear()`` on a slow/loaded
|
||||
runner, so the set is wiped by the clear and ``approve_tools`` blocks the
|
||||
full hour -- surfacing as a CI hang. This instead waits until the approval
|
||||
is actually registered (which happens *after* the clear), then resolves, so
|
||||
the wakeup can never be lost. ``start()`` / ``cancel()`` mirror
|
||||
``threading.Timer`` so it drops into existing scaffolding. ``cancel()``
|
||||
signals the worker to stop and joins it, so a test that errors *before* the
|
||||
approval registers can't leak the thread or resolve late into a finished
|
||||
test. ``before`` runs just before resolving -- e.g. to snapshot
|
||||
pending-state fields the test asserts on.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
ui: Any,
|
||||
*args: Any,
|
||||
before: Callable[[], None] | None = None,
|
||||
deadline: float = 10.0,
|
||||
**kwargs: Any,
|
||||
) -> None:
|
||||
self._ui = ui
|
||||
self._args = args
|
||||
self._kwargs = kwargs
|
||||
self._before = before
|
||||
self._deadline = deadline
|
||||
self._cancelled = threading.Event()
|
||||
self._started = False
|
||||
self._thread = threading.Thread(target=self._run, name="resolve-when-pending", daemon=True)
|
||||
|
||||
def _run(self) -> None:
|
||||
end = time.monotonic() + self._deadline
|
||||
while time.monotonic() < end:
|
||||
if self._cancelled.is_set():
|
||||
return
|
||||
# getattr (not a bare read) so a UI without _pending_approval can't
|
||||
# crash the worker into a silent death that leaves approve_tools
|
||||
# blocked for the full _APPROVAL_WAIT_TIMEOUT.
|
||||
if getattr(self._ui, "_pending_approval", None) is not None:
|
||||
if self._before is not None:
|
||||
self._before()
|
||||
self._ui.resolve_approval(*self._args, **self._kwargs)
|
||||
return
|
||||
time.sleep(0.001)
|
||||
# Deadline without registration: approve_tools isn't parked on the
|
||||
# approval event (returned early, or never reached it) -- don't resolve
|
||||
# into an unknown state; let the test's own assertions speak.
|
||||
|
||||
def start(self) -> None:
|
||||
self._started = True
|
||||
self._thread.start()
|
||||
|
||||
def cancel(self) -> None:
|
||||
self._cancelled.set()
|
||||
if self._started:
|
||||
self._thread.join(timeout=5)
|
||||
|
||||
|
||||
def resolve_when_pending(ui: Any, *args: Any, **kwargs: Any) -> _PendingResolver:
|
||||
"""Build a race-free approval resolver (see :class:`_PendingResolver`)."""
|
||||
return _PendingResolver(ui, *args, **kwargs)
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
from collections.abc import Callable, Iterator
|
||||
|
||||
from turnstone.core.mcp_client import MCPClientManager, StaticServerState
|
||||
from turnstone.core.mcp_crypto import MCPTokenCipher
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris and London?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
},
|
||||
{
|
||||
"id": "call_2",
|
||||
"input": {
|
||||
"city": "London"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_2",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"text": "Actually, never mind London.",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "What's in this image?",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"source": {
|
||||
"data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==",
|
||||
"media_type": "image/png",
|
||||
"type": "base64"
|
||||
},
|
||||
"type": "image"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Think about the weather.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"signature": "sig-abc",
|
||||
"thinking": "The user wants weather.",
|
||||
"type": "thinking"
|
||||
},
|
||||
{
|
||||
"text": "Let me check.",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Think about the weather.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"signature": "sig-abc",
|
||||
"thinking": "The user wants weather.",
|
||||
"type": "thinking"
|
||||
},
|
||||
{
|
||||
"text": "Let me check.",
|
||||
"type": "text"
|
||||
},
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Run the deploy.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {},
|
||||
"name": "deploy",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "deployed",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
},
|
||||
{
|
||||
"text": "Great, what's next?",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"system": "Output-guard: deploy output looked clean.",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Hi there.",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "Hello! How can I help?",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": "What's the weather in Paris?",
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "18C, clear.",
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"text": "It's 18C and clear in Paris.",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"cache_control": {
|
||||
"type": "ephemeral"
|
||||
},
|
||||
"extra_body": {
|
||||
"chat_template_kwargs": {
|
||||
"enable_thinking": true,
|
||||
"reasoning_effort": "high"
|
||||
}
|
||||
},
|
||||
"max_tokens": 4096,
|
||||
"messages": [
|
||||
{
|
||||
"content": "Weather in Paris?",
|
||||
"role": "user"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"input": {
|
||||
"city": "Paris"
|
||||
},
|
||||
"name": "get_weather",
|
||||
"type": "tool_use"
|
||||
}
|
||||
],
|
||||
"role": "assistant"
|
||||
},
|
||||
{
|
||||
"content": [
|
||||
{
|
||||
"content": "Tool execution was cancelled. Outcome UNKNOWN — this call may have begun executing before the generation was stopped; do not assume it did not run, and reconcile before re-issuing it.",
|
||||
"is_error": true,
|
||||
"tool_use_id": "call_1",
|
||||
"type": "tool_result"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "qwen3.6-27b",
|
||||
"temperature": 0.5,
|
||||
"tools": [
|
||||
{
|
||||
"description": "Look up the weather for a city.",
|
||||
"input_schema": {
|
||||
"properties": {
|
||||
"city": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"city"
|
||||
],
|
||||
"type": "object"
|
||||
},
|
||||
"name": "get_weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -43,6 +43,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gemini-2.5-pro",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -43,6 +43,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
}
|
||||
],
|
||||
"model": "gpt-4o-mini",
|
||||
"reasoning_effort": "medium",
|
||||
"stream": true,
|
||||
"stream_options": {
|
||||
"include_usage": true
|
||||
|
||||
+588
-27
@@ -9,6 +9,8 @@ manual testing.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
@@ -17,6 +19,12 @@ import pytest
|
||||
|
||||
_APP_JS = Path(__file__).resolve().parent.parent / "turnstone/ui/static/app.js"
|
||||
_INTERACTIVE_JS = Path(__file__).resolve().parent.parent / "turnstone/shared_static/interactive.js"
|
||||
_SHELL_JS = Path(__file__).resolve().parent.parent / "turnstone/shared_static/shell.js"
|
||||
_REDACT_CREDENTIALS_JS = (
|
||||
Path(__file__).resolve().parent.parent / "turnstone/shared_static/redact_credentials.js"
|
||||
)
|
||||
_CONSOLE_APP_JS = Path(__file__).resolve().parent.parent / "turnstone/console/static/app.js"
|
||||
_CONSOLE_INDEX = Path(__file__).resolve().parent.parent / "turnstone/console/static/index.html"
|
||||
|
||||
|
||||
def _pane_method_offset(body: str, name: str) -> int:
|
||||
@@ -530,15 +538,17 @@ def test_phase8_appendtooloutput_dispatches_mcp_error_before_renderer() -> None:
|
||||
end = _pane_method_offset(body, "sendMessage")
|
||||
fn = body[start:end]
|
||||
parse_idx = fn.find("tryParseMcpError(")
|
||||
render_idx = fn.find("renderToolOutput(")
|
||||
# The plain-output render is the shared renderCollapsibleOutput helper; the
|
||||
# ordering invariant is unchanged — MCP dispatch must precede it.
|
||||
render_idx = fn.find("renderCollapsibleOutput(")
|
||||
assert parse_idx >= 0, (
|
||||
"appendToolOutput must call tryParseMcpError on the error path "
|
||||
"before renderToolOutput, otherwise the consent card never "
|
||||
"before the plain renderer, otherwise the consent card never "
|
||||
"replaces the plain JSON output."
|
||||
)
|
||||
assert render_idx >= 0, "renderToolOutput call must remain present"
|
||||
assert render_idx >= 0, "renderCollapsibleOutput call must remain present"
|
||||
assert parse_idx < render_idx, (
|
||||
"tryParseMcpError must run BEFORE renderToolOutput so the "
|
||||
"tryParseMcpError must run BEFORE the plain renderer so the "
|
||||
"interactive card path takes precedence over plain rendering."
|
||||
)
|
||||
|
||||
@@ -553,7 +563,6 @@ _CONSOLE_ADMIN_JS = Path(__file__).resolve().parent.parent / "turnstone/console/
|
||||
_CONSOLE_GOVERNANCE_JS = (
|
||||
Path(__file__).resolve().parent.parent / "turnstone/console/static/governance.js"
|
||||
)
|
||||
_CONSOLE_INTERACTIVE_JS = Path(__file__).resolve().parent.parent / "turnstone/console/static/app.js"
|
||||
|
||||
|
||||
_UNSAFE_CODE_SINK_LINT_TARGETS = [
|
||||
@@ -564,7 +573,7 @@ _UNSAFE_CODE_SINK_LINT_TARGETS = [
|
||||
("turnstone/console/static/coordinator/coordinator.js", _COORD_JS),
|
||||
("turnstone/console/static/admin.js", _CONSOLE_ADMIN_JS),
|
||||
("turnstone/console/static/governance.js", _CONSOLE_GOVERNANCE_JS),
|
||||
("turnstone/console/static/app.js", _CONSOLE_INTERACTIVE_JS),
|
||||
("turnstone/console/static/app.js", _CONSOLE_APP_JS),
|
||||
]
|
||||
|
||||
|
||||
@@ -953,6 +962,7 @@ _CONST_GUARD_BUNDLES = _SWEPT_BUNDLES + [
|
||||
_REPO_ROOT / "turnstone/shared_static/rail.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/interactive.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/conversation.js",
|
||||
_REPO_ROOT / "turnstone/shared_static/redact_credentials.js",
|
||||
]
|
||||
|
||||
|
||||
@@ -1265,41 +1275,102 @@ def test_swept_bundle_has_no_const_reassign(bundle: Path) -> None:
|
||||
)
|
||||
|
||||
|
||||
def test_redact_api_keys_runtime_smoke() -> None:
|
||||
"""Runtime smoke for ``_redactApiKeys``. The function is pure — no
|
||||
DOM dependency — so it transplants cleanly into a standalone
|
||||
``node -e`` invocation. This is the bit that would have caught
|
||||
the original ``const redacted`` bug (which ``node --check`` and a
|
||||
pure-static keyword scan both miss; the ``TypeError`` only fires
|
||||
at call-time)."""
|
||||
body = _INTERACTIVE_JS.read_text(encoding="utf-8")
|
||||
m = re.search(
|
||||
r"function _redactApiKeys\(text\) \{.*?\n\}\n",
|
||||
body,
|
||||
re.DOTALL,
|
||||
)
|
||||
assert m is not None, "_redactApiKeys not found in app.js"
|
||||
fn = m.group(0)
|
||||
script = (
|
||||
fn
|
||||
+ "\nconst q = _redactApiKeys('https://x?api_key=abc&u=foo');\n"
|
||||
def test_redact_credentials_runtime_smoke() -> None:
|
||||
"""Runtime smoke for ``redactCredentials`` via a temp harness file.
|
||||
The function is pure (no DOM dependency). Tests the shared module
|
||||
directly via ESM import (replaces the legacy ``_redactApiKeys`` test
|
||||
which now delegates to this).
|
||||
|
||||
The tempfile is written with a ``.mjs`` extension so Node forces ESM
|
||||
parsing regardless of any ``package.json`` ``type`` field in parent
|
||||
directories. The ``redact_credentials.js`` source file is imported
|
||||
by absolute path so resolution is unambiguous.
|
||||
"""
|
||||
import tempfile
|
||||
|
||||
mod_path = _REDACT_CREDENTIALS_JS.resolve()
|
||||
harness = (
|
||||
"import { redactCredentials } from "
|
||||
+ json.dumps(str(mod_path))
|
||||
+ ";\n"
|
||||
+ "const q = redactCredentials('https://x?api_key=abc&u=foo');\n"
|
||||
+ 'if (q !== "https://x?api_key=***&u=foo") '
|
||||
+ "throw new Error('query-string redact failed: ' + q);\n"
|
||||
+ 'const j = _redactApiKeys(\'{"api_key":"abc"}\');\n'
|
||||
+ 'const j = redactCredentials(\'{"api_key":"abc"}\');\n'
|
||||
+ 'if (j !== \'{"api_key":"***"}\') '
|
||||
+ "throw new Error('json-style redact failed: ' + j);\n"
|
||||
+ "// Bearer token redaction (raw input)\n"
|
||||
+ "const b = redactCredentials('Authorization: Bearer eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjMifQ.test-token_here');\n"
|
||||
+ "if (!b.includes('[REDACTED:api_key]')) "
|
||||
+ "throw new Error('bearer redact failed: ' + b);\n"
|
||||
+ "// Connection string redaction (raw input)\n"
|
||||
+ "const c = redactCredentials('postgresql://user:supersecret@localhost/db');\n"
|
||||
+ "if (!c.includes('[REDACTED:password]')) "
|
||||
+ "throw new Error('conn-string redact failed: ' + c);\n"
|
||||
+ "// Authorization JSON key redaction (step 6 comprehensive)\n"
|
||||
+ 'const a = redactCredentials(\'{"Authorization": "Bearer canstillseethis"}\');\n'
|
||||
+ "if (!a.includes('[REDACTED:secret]')) "
|
||||
+ "throw new Error('authorization JSON redact failed: ' + a);\n"
|
||||
+ "// Single-quote JSON (Python dict repr / JS object literal)\n"
|
||||
+ "const sq = redactCredentials(\"{'Authorization': 'Bearer canstillseethis'}\");\n"
|
||||
+ "if (!sq.includes('[REDACTED:secret]')) "
|
||||
+ "throw new Error('single-quote authorization redact failed: ' + sq);\n"
|
||||
+ "// mongodb+srv connection string (Atlas SRV)\n"
|
||||
+ "const ms = redactCredentials('mongodb+srv://u:s3cretpw@cluster.mongodb.net/db');\n"
|
||||
+ "if (!ms.includes('[REDACTED:password]')) "
|
||||
+ "throw new Error('mongodb+srv redact failed: ' + ms);\n"
|
||||
+ "// lowercase bearer scheme (RFC 7235 case-insensitive)\n"
|
||||
+ "const lb = redactCredentials('authorization: bearer "
|
||||
+ "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.sig12345');\n"
|
||||
+ "if (!lb.includes('[REDACTED:api_key]')) "
|
||||
+ "throw new Error('lowercase bearer redact failed: ' + lb);\n"
|
||||
+ "// api_key= assignment redacts the whole token, not a garbled api_[REDACTED\n"
|
||||
+ "const ak = redactCredentials('api_key=abcdefghijklmnopqrstuvwxyz');\n"
|
||||
+ "if (ak !== '[REDACTED:api_key]') "
|
||||
+ "throw new Error('api_key= clean redact failed: ' + ak);\n"
|
||||
+ "// Prefilter fast path: plain text with no anchor substring is unchanged\n"
|
||||
+ "const fp = redactCredentials('build ok in 42s - 3 tests passed');\n"
|
||||
+ "if (fp !== 'build ok in 42s - 3 tests passed') "
|
||||
+ "throw new Error('prefilter fast-path no-op failed: ' + fp);\n"
|
||||
+ "// Bare credentials with no =, quote or @ anywhere must still redact\n"
|
||||
+ "// (these pin the prefilter as a superset of the pattern set)\n"
|
||||
+ "const bk = redactCredentials('loaded sk-abcdefghijklmnopqrstuvwx');\n"
|
||||
+ "if (bk !== 'loaded [REDACTED:api_key]') "
|
||||
+ "throw new Error('bare sk- redact failed: ' + bk);\n"
|
||||
+ "const aw = redactCredentials('using AKIAABCDEFGHIJKLMNOP now');\n"
|
||||
+ "if (aw !== 'using [REDACTED:api_key] now') "
|
||||
+ "throw new Error('bare AKIA redact failed: ' + aw);\n"
|
||||
+ "const bt = redactCredentials('Bearer abcdefghijklmnopqrstuvwxyz');\n"
|
||||
+ "if (bt !== '[REDACTED:api_key]') "
|
||||
+ "throw new Error('bare bearer redact failed: ' + bt);\n"
|
||||
+ "// SQLAlchemy dialect+driver connection URLs (psycopg2/asyncpg)\n"
|
||||
+ "const pg2 = redactCredentials('postgresql+psycopg2://user:s3cret@db:5432/app');\n"
|
||||
+ "if (pg2 !== 'postgresql+psycopg2://user:[REDACTED:password]@db:5432/app') "
|
||||
+ "throw new Error('psycopg2 conn redact failed: ' + pg2);\n"
|
||||
+ "const apg = redactCredentials('postgresql+asyncpg://user:s3cret@db/app');\n"
|
||||
+ "if (apg !== 'postgresql+asyncpg://user:[REDACTED:password]@db/app') "
|
||||
+ "throw new Error('asyncpg conn redact failed: ' + apg);\n"
|
||||
+ "// RFC 3986 schemes are case-insensitive - uppercase must not bypass\n"
|
||||
+ "const up = redactCredentials('POSTGRESQL+PSYCOPG2://user:s3cret@db/app');\n"
|
||||
+ "if (up !== 'POSTGRESQL+PSYCOPG2://user:[REDACTED:password]@db/app') "
|
||||
+ "throw new Error('uppercase scheme conn redact failed: ' + up);\n"
|
||||
)
|
||||
with tempfile.NamedTemporaryFile(mode="w", suffix=".mjs", delete=False) as f:
|
||||
f.write(harness)
|
||||
tmp = f.name
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["node", "-e", script],
|
||||
["node", tmp],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=15,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
pytest.skip("node binary not available on PATH")
|
||||
finally:
|
||||
os.unlink(tmp)
|
||||
assert proc.returncode == 0, (
|
||||
f"_redactApiKeys runtime smoke failed. stdout={proc.stdout!r} stderr={proc.stderr!r}"
|
||||
f"redactCredentials runtime smoke failed. stdout={proc.stdout!r} stderr={proc.stderr!r}"
|
||||
)
|
||||
|
||||
|
||||
@@ -1521,6 +1592,287 @@ def test_coord_connectsse_onerror_preserves_native_reconnect() -> None:
|
||||
assert passed, f"coordinator.js connectSSE.onerror regressed: {reason}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Coordinator-pane parity for the SSE overflow-recovery companions (issue #806).
|
||||
# The server-side fixes (emit-time batching, _ListenerQueue poison, out-of-band
|
||||
# closing) live in SessionUIBase and already cover EVERY SSE stream; these pin
|
||||
# the CLIENT-side companions ported into coordinator.js so it stops relying on
|
||||
# native reconnect alone — storm guard + degraded catch-up, close-on-hide /
|
||||
# replay-on-show, and drop-vs-render-wedge counters.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_coord_imports_shared_overflow_helpers() -> None:
|
||||
"""coordinator.js consumes the SAME sse_overflow.js helpers as the
|
||||
interactive pane (over the /shared mount) so the trip threshold and cooldown
|
||||
ladder cannot drift between the two surfaces."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
m = re.search(
|
||||
r"import \{([^}]*)\} from \"/shared/sse_overflow\.js\";",
|
||||
body,
|
||||
re.S,
|
||||
)
|
||||
assert m is not None, "coordinator must import the shared overflow helpers"
|
||||
imported = m.group(1)
|
||||
for name in (
|
||||
"OVERFLOW_TRIP_COUNT",
|
||||
"OVERFLOW_TRIP_WINDOW_MS",
|
||||
"DEGRADED_COOLDOWN_BASE_MS",
|
||||
"DEGRADED_COOLDOWN_MAX_MS",
|
||||
"DEGRADED_COOLDOWN_RESET_MS",
|
||||
"overflowWindowTripped",
|
||||
"degradedCooldownStep",
|
||||
):
|
||||
assert name in imported, f"{name} must be imported from /shared/sse_overflow.js"
|
||||
# No local fork of the extracted pure functions on the coordinator side.
|
||||
assert not re.search(r"^\s*function overflowWindowTripped\(", body, re.M)
|
||||
assert not re.search(r"^\s*function degradedCooldownStep\(", body, re.M)
|
||||
|
||||
|
||||
def test_coord_stream_overflow_case_counts_and_rate_limits() -> None:
|
||||
"""The coordinator handles the id-less ``stream_overflow`` frame: count it
|
||||
(drop-vs-wedge field instrumentation) and feed the rolling-window storm
|
||||
guard, exactly like the interactive pane."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
assert 'case "stream_overflow":' in body
|
||||
assert "noteStreamOverflow();" in body
|
||||
# The three-way health counter distinguishes dropped events (overflow /
|
||||
# malformed frame) from render wedges (dispatch / render throw).
|
||||
assert "streamHealth = { overflows: 0, renderThrows: 0, malformedFrames: 0 }" in body
|
||||
assert "streamHealth.overflows += 1;" in body
|
||||
assert "streamHealth.malformedFrames += 1;" in body
|
||||
# Exactly two render-throw increment sites: the noteRenderThrow helper
|
||||
# (all three contained render/finalize catches route through it — they
|
||||
# recover with a plain-text fallback, so console.warn) and the onmessage
|
||||
# dispatch catch (console.error class — the event is dropped outright).
|
||||
# The three recovered call sites are pinned by label so a new render path
|
||||
# that forgets to count surfaces loudly.
|
||||
assert body.count("streamHealth.renderThrows += 1;") == 2
|
||||
helper = re.search(r"function noteRenderThrow\(where, err\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert helper is not None, "noteRenderThrow helper not found"
|
||||
assert "streamHealth.renderThrows += 1;" in helper.group(1)
|
||||
assert 'noteRenderThrow("streamingRender", e);' in body
|
||||
assert 'noteRenderThrow("in_progress_snapshot render", e);' in body
|
||||
assert 'noteRenderThrow("streamingRenderFinalize", e);' in body
|
||||
note = re.search(r"function noteStreamOverflow\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert note is not None, "noteStreamOverflow not found"
|
||||
assert "overflowWindowTripped(" in note.group(1)
|
||||
assert "enterDegradedCatchup()" in note.group(1)
|
||||
# The trip handler only counts + trips; the cooldown reset lives in
|
||||
# enterDegradedCatchup (keyed off lastDegradedAt) — the finding [0] shape.
|
||||
assert "degradedCooldownMs" not in note.group(1), (
|
||||
"noteStreamOverflow must not touch the cooldown — that reset defeated the ladder escalation"
|
||||
)
|
||||
|
||||
|
||||
def test_coord_handleevent_dispatch_is_wedge_guarded() -> None:
|
||||
"""A throw escaping onmessage does NOT close the EventSource, so an
|
||||
unguarded handler throw left the streaming refs stale and wedged every later
|
||||
turn. The coordinator wraps the dispatch and counts the throw (render-wedge
|
||||
class) so a field report tells it apart from a dropped-events gap."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
m = re.search(r"try \{\s*handleEvent\(data\);\s*\} catch \(err\) \{(.*?)\}", body, re.S)
|
||||
assert m is not None, "handleEvent(data) must be wrapped in try/catch in onmessage"
|
||||
assert "streamHealth.renderThrows += 1;" in m.group(1)
|
||||
|
||||
|
||||
def test_coord_degraded_catchup_stops_live_stream_and_retries() -> None:
|
||||
"""Three overflow closes inside the window drop the coordinator to a
|
||||
degraded catch-up: suspend the live stream, say so plainly, and reconnect
|
||||
after a doubling cooldown — the reconnect replays the gap (or falls to the
|
||||
/history floor once it outgrows the ring)."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
m = re.search(r"function enterDegradedCatchup\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert m is not None, "enterDegradedCatchup not found"
|
||||
method = m.group(1)
|
||||
assert "degradedCooldownStep(" in method
|
||||
assert "lastDegradedAt = now" in method
|
||||
# Suspend the stream BEFORE arming the retry timer (mirrors interactive's
|
||||
# disconnect-then-rearm ordering) or the fresh timer is cancelled at once.
|
||||
assert method.index("suspendStream()") < method.index("degradedTimer = setTimeout")
|
||||
# Plain-language status, not a silent stall.
|
||||
assert "catching up" in method
|
||||
# A fresh connect must cancel a pending degraded timer so it can't
|
||||
# double-open behind the retry — connectSSE's prologue routes through the
|
||||
# shared closeStreamTransport teardown, which owns that clear (alongside
|
||||
# the reconnect timer + the EventSource close/null).
|
||||
conn = re.search(r"function connectSSE\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert conn is not None
|
||||
assert "closeStreamTransport();" in conn.group(1)
|
||||
teardown = re.search(r"function closeStreamTransport\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert teardown is not None, "closeStreamTransport not found"
|
||||
assert "clearTimeout(degradedTimer)" in teardown.group(1)
|
||||
assert "clearTimeout(reconnectTimer)" in teardown.group(1)
|
||||
assert "evtSource = null;" in teardown.group(1)
|
||||
|
||||
|
||||
def test_coord_visibilitychange_closes_on_hide_reconnects_on_show() -> None:
|
||||
"""A hidden tab's throttled drain is the worst-case slow SSE consumer. The
|
||||
coordinator installs a visibilitychange handler that closes the stream on
|
||||
hide (marking its OWN close via hiddenDisconnect) and reconnects on show from
|
||||
the saved lastEventId, and removes the listener on teardown."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
assert 'document.addEventListener("visibilitychange", visHandler);' in body
|
||||
assert 'document.removeEventListener("visibilitychange", visHandler);' in body
|
||||
vis = re.search(r"function onVisibilityChange\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert vis is not None, "onVisibilityChange not found"
|
||||
method = vis.group(1)
|
||||
assert "document.hidden" in method
|
||||
assert "suspendStream()" in method
|
||||
assert "hiddenDisconnect = true;" in method
|
||||
assert "else if (hiddenDisconnect)" in method
|
||||
assert "connectSSE();" in method
|
||||
|
||||
|
||||
def test_coord_connectsse_defers_open_when_tab_hidden() -> None:
|
||||
"""connectSSE must never open an EventSource into a hidden tab — the single
|
||||
chokepoint that also backstops a FIRST connect in a background tab (where the
|
||||
close-on-hide handler never fires because there was no open stream). It
|
||||
marks hiddenDisconnect so the show edge owns the reconnect, marks the
|
||||
deferral as a GAP (markStreamGap) so the eventual open runs the post-gap
|
||||
recovery — without the mark a pane first opened in a background tab
|
||||
silently missed every child/task created while hidden — and reports an
|
||||
honest paused status instead of pinning "connecting" with no attempt in
|
||||
flight."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
conn = re.search(r"function connectSSE\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert conn is not None
|
||||
method = conn.group(1)
|
||||
guard = method.index("if (document.hidden)")
|
||||
open_idx = method.index("new EventSource(")
|
||||
assert guard < open_idx, "the hidden guard must precede new EventSource"
|
||||
head = method[guard:open_idx]
|
||||
assert "markStreamGap();" in head, "the hidden deferral must count as a stream gap"
|
||||
assert "hiddenDisconnect = true;" in head
|
||||
assert "return;" in head
|
||||
assert 'setSseStatus("paused' in head, "the deferral must report paused, not connecting"
|
||||
# "connecting" is claimed only once an attempt actually starts — after
|
||||
# the hidden guard, immediately before the EventSource construction.
|
||||
connecting = method.index('setSseStatus("connecting')
|
||||
assert guard < connecting < open_idx
|
||||
|
||||
|
||||
def test_coord_destroy_removes_visibility_handler_and_stream_transport() -> None:
|
||||
"""Teardown must detach the document-level visibilitychange listener (it
|
||||
holds a strong ref to the closure) and tear down the stream transport —
|
||||
closeStreamTransport closes the EventSource and cancels the reconnect +
|
||||
degraded retry timers (pinned in the degraded-catchup test) — or a
|
||||
destroyed pane leaks and a show edge / pending retry reopens its stream."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
d = re.search(r"function destroy\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert d is not None, "destroy not found"
|
||||
method = d.group(1)
|
||||
assert "removeVisibilityHandler();" in method
|
||||
assert "closeStreamTransport();" in method
|
||||
|
||||
|
||||
def test_coord_close_session_detaches_visibility_reopen() -> None:
|
||||
"""coordCloseSession suspends the stream AND removes the visibilitychange
|
||||
handler BEFORE awaiting the /close POST: a tab hide→show while the POST is
|
||||
in flight must not reopen a stream against the workstream the server is
|
||||
tearing down (404 / reconnect churn against a dead session). The failure
|
||||
paths resume via connectSSE, which reinstalls the handler at its
|
||||
install-once chokepoint — so close-on-hide survives a failed close."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
m = re.search(r"async function coordCloseSession\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert m is not None, "coordCloseSession not found"
|
||||
method = m.group(1)
|
||||
suspend = method.index("suspendStream();")
|
||||
unhook = method.index("removeVisibilityHandler();")
|
||||
# The quoted URL fragment, not the bare word (comments mention /close too).
|
||||
post = method.index('"/close"')
|
||||
assert suspend < post, "stream suspension must precede the /close POST"
|
||||
assert unhook < post, "visibility detach must precede the /close POST"
|
||||
assert "resumeSse()" in method
|
||||
|
||||
|
||||
def test_coord_post_gap_sidebar_refresh_is_replay_aware() -> None:
|
||||
"""The replace-mode children/tasks refresh (a sidebar rebuild) must NOT
|
||||
fire on every reconnect: child_ws_* / task-mutating events are ordinary
|
||||
ring-buffer entries, so a cursor reconnect (replay_ok) redelivers them and
|
||||
the sidebar heals through the normal handlers — a momentary blur→focus
|
||||
under close-on-hide must not rebuild the sidebar. The refresh fires
|
||||
exactly when the replay cannot vouch for the gap: no resume cursor or an
|
||||
over-threshold gap at onopen, or the server's replay_truncated envelope
|
||||
(ring evicted), deduped per open via gapRefreshedAtOpen."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
conn = re.search(r"function connectSSE\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert conn is not None
|
||||
method = conn.group(1)
|
||||
gate = re.search(
|
||||
r"wasReconnecting &&\s*\(lastEventId == null \|\| gapMs > GAP_REFRESH_THRESHOLD_MS\)",
|
||||
method,
|
||||
)
|
||||
assert gate is not None, "onopen must gate the sidebar refresh on replay coverage"
|
||||
assert "refreshSidebarAfterGap();" in method
|
||||
assert "gapRefreshedAtOpen = true;" in method
|
||||
# The ring-evicted signal triggers the same refresh (deduped per open).
|
||||
trunc = re.search(r'case "replay_truncated":(.*?)break;', body, re.S)
|
||||
assert trunc is not None, "replay_truncated case not found"
|
||||
assert "refreshSidebarAfterGap()" in trunc.group(1)
|
||||
assert "gapRefreshedAtOpen" in trunc.group(1)
|
||||
# Deliberate suspends (hide / overflow / close-session) mark the gap so
|
||||
# the next open participates in the recovery decision at all.
|
||||
sus = re.search(r"function suspendStream\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert sus is not None, "suspendStream not found"
|
||||
assert "markStreamGap();" in sus.group(1)
|
||||
# The refresh helper carries the whole replace-mode bundle: children,
|
||||
# tasks, and the live-badge purge (permanent 403/404 entries preserved).
|
||||
ref = re.search(r"function refreshSidebarAfterGap\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert ref is not None, "refreshSidebarAfterGap not found"
|
||||
assert "loadChildren({ replace: true });" in ref.group(1)
|
||||
assert "loadTasks();" in ref.group(1)
|
||||
assert "_liveBadgeCacheDelete(id)" in ref.group(1)
|
||||
|
||||
|
||||
def test_coord_defers_truncated_resync_and_consumes_at_idle() -> None:
|
||||
"""replay_truncated seen mid-stream must be DEFERRED, not dropped (matches
|
||||
interactive's _pendingTruncatedResync): refetching immediately would detach
|
||||
the live bubble (content OR a reasoning-only one), but skipping outright
|
||||
leaves the ring-evicted turns lost for the session. The guard covers both
|
||||
streaming targets and latches otherwise; the next state_change=idle consumes
|
||||
the flag — which also repairs a turn stranded by close-on-hide (stream_end
|
||||
evicted while hidden), resetting the streaming refs first since
|
||||
refetchHistory does not null them."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
trunc = re.search(r'case "replay_truncated":(.*?)break;', body, re.S)
|
||||
assert trunc is not None, "replay_truncated case not found"
|
||||
t = trunc.group(1)
|
||||
assert "if (!currentAssistantEl && !currentReasoningEl)" in t
|
||||
assert "refetchHistory();" in t
|
||||
assert "pendingTruncatedResync = true;" in t
|
||||
st = re.search(r'case "state_change":(.*?)\n case ', body, re.S)
|
||||
assert st is not None, "state_change case not found"
|
||||
s = st.group(1)
|
||||
assert "if (pendingTruncatedResync)" in s
|
||||
assert "pendingTruncatedResync = false;" in s
|
||||
assert "currentAssistantEl = null;" in s
|
||||
assert "refetchHistory();" in s
|
||||
# Consume the latch, THEN reset the dangling refs and refetch.
|
||||
consume = s.index("pendingTruncatedResync = false;")
|
||||
refetch = s.index("refetchHistory();")
|
||||
assert consume < refetch
|
||||
|
||||
|
||||
def test_coord_detects_server_restart_by_backwards_event_id() -> None:
|
||||
"""A coordinator process restart resets the per-ws event counter, and the
|
||||
replay path reports replay_ok for a stale-high cursor (past the new max), so
|
||||
the gap is unsignalled and the sidebar goes stale. onmessage catches it: a
|
||||
live event id below the saved cursor == the counter reset → pull
|
||||
authoritative sidebar state (deduped per open against onopen's refresh),
|
||||
checked BEFORE the cursor is overwritten."""
|
||||
body = _COORD_JS.read_text(encoding="utf-8")
|
||||
m = re.search(r"evtSource\.onmessage = function \(event\) \{(.*?)\n \};", body, re.S)
|
||||
assert m is not None, "onmessage handler not found"
|
||||
handler = m.group(1)
|
||||
assert "Number(evtSource.lastEventId) < Number(lastEventId)" in handler
|
||||
assert "!gapRefreshedAtOpen" in handler
|
||||
assert "refreshSidebarAfterGap();" in handler
|
||||
check = handler.index("Number(evtSource.lastEventId) < Number(lastEventId)")
|
||||
overwrite = handler.index("lastEventId = evtSource.lastEventId;")
|
||||
assert check < overwrite
|
||||
|
||||
|
||||
def test_interactive_history_is_rest_first_not_sse() -> None:
|
||||
"""PR A converged interactive onto coord's REST-first history
|
||||
model: first paint and post-rewind re-render fetch ``GET /history``
|
||||
@@ -1587,6 +1939,89 @@ def test_early_paint_tool_pending_wiring() -> None:
|
||||
assert "if (!announced) this.messagesEl.appendChild(block);" in body
|
||||
|
||||
|
||||
def test_task_agent_steps_never_escape_their_card() -> None:
|
||||
"""A task agent's sub-tool steps (``parent_call_id`` stamped) must nest in
|
||||
the task card, never render as top-level rows that look like the main
|
||||
harness issued them. Two seams keep that true; this guards both against a
|
||||
rename/deletion:
|
||||
|
||||
1. ``tool_info`` routes through ``_routeAgentItems`` first — a sub-tool
|
||||
auto-resolved by policy / "Always" arrives as a ``tool_info`` and must
|
||||
nest, not paint a duplicate top-level block (Copilot review on #732).
|
||||
2. A child step whose ``task_agent`` row hasn't painted yet (the 4-wide
|
||||
tool pool's ordering window) is BUFFERED and flushed when the row lands,
|
||||
instead of escaping to top-level; the card also survives the parent
|
||||
row's pending->resolved rebuild.
|
||||
3. SAFETY VALVE: a buffered step whose parent row NEVER paints (an id-
|
||||
correlation mismatch / aborted agent) is escaped to a top-level row after
|
||||
a grace window, so it stays VISIBLE rather than buffered forever.
|
||||
"""
|
||||
body = _INTERACTIVE_JS.read_text(encoding="utf-8")
|
||||
# 1. tool_info nests via the same router as tool_pending / approve_request.
|
||||
info = body[body.index('case "tool_info":') : body.index('case "approve_request":')]
|
||||
assert 'this._routeAgentItems(evt.items, "info")' in info, (
|
||||
"tool_info must route a parent-tagged sub-tool into the task card "
|
||||
"before any top-level showInlineToolBlock fallback."
|
||||
)
|
||||
# 2. _routeAgentItems buffers an orphan child (instead of returning false,
|
||||
# which escapes it to top-level) when the parent card isn't painted yet.
|
||||
route = body[
|
||||
_pane_method_offset(body, "_routeAgentItems") : _pane_method_offset(
|
||||
body, "_ensureAgentCard"
|
||||
)
|
||||
]
|
||||
assert "_bufferAgentOrphan(parentId, items, mode)" in route, (
|
||||
"a parent-tagged child with no card yet must buffer, not fall through to a top-level paint."
|
||||
)
|
||||
# The buffer / flush / escape / relink helpers exist.
|
||||
assert "_bufferAgentOrphan(parentId, items, mode) {" in body
|
||||
assert "_flushAgentOrphans(parentIds) {" in body
|
||||
assert "_escapeAgentOrphans(parentId) {" in body
|
||||
assert "_relinkAgentCards(items) {" in body
|
||||
assert body.count("this._relinkAgentCards(") >= 2, (
|
||||
"both announceToolBlock and showInlineToolBlock must relink + flush so "
|
||||
"a buffered step nests as soon as a tool row appears."
|
||||
)
|
||||
# 3. Safety valve: _bufferAgentOrphan arms a grace timer to _escapeAgentOrphans
|
||||
# so a never-painting parent's steps can't vanish (or leak) — they escape
|
||||
# back to a visible top-level paint.
|
||||
buf = body[
|
||||
_pane_method_offset(body, "_bufferAgentOrphan") : _pane_method_offset(
|
||||
body, "_flushAgentOrphans"
|
||||
)
|
||||
]
|
||||
assert "setTimeout(" in buf and "_escapeAgentOrphans(parentId)" in buf, (
|
||||
"a buffered orphan must arm a grace-window escape so it never stays "
|
||||
"buffered (invisible) forever."
|
||||
)
|
||||
escape = body[
|
||||
_pane_method_offset(body, "_escapeAgentOrphans") : _pane_method_offset(
|
||||
body, "_relinkAgentCards"
|
||||
)
|
||||
]
|
||||
assert "announceToolBlock(" in escape, (
|
||||
"the escape valve must render the steps top-level (visible), the "
|
||||
"pre-buffer behaviour, rather than dropping them."
|
||||
)
|
||||
# Flush is targeted to the just-painted parents, not the whole map.
|
||||
flush = body[
|
||||
_pane_method_offset(body, "_flushAgentOrphans") : _pane_method_offset(
|
||||
body, "_escapeAgentOrphans"
|
||||
)
|
||||
]
|
||||
assert "parentIds.forEach" in flush
|
||||
# _ensureAgentCard re-attaches a DETACHED card across a parent-row rebuild,
|
||||
# but builds fresh on a still-attached (cross-turn reused) call_id rather
|
||||
# than stealing the prior agent's steps.
|
||||
ensure = body[
|
||||
_pane_method_offset(body, "_ensureAgentCard") : _pane_method_offset(
|
||||
body, "_bufferAgentOrphan"
|
||||
)
|
||||
]
|
||||
assert "!card.wrap.isConnected" in ensure
|
||||
assert "parentRow.appendChild(card.wrap);" in ensure
|
||||
|
||||
|
||||
def test_risk_level_normalized_before_dom_interpolation() -> None:
|
||||
"""Server-supplied ``risk_level`` lands in className / data-risk strings the
|
||||
verdict + warning CSS depend on, so every interpolation must funnel through
|
||||
@@ -1650,3 +2085,129 @@ def test_early_paint_screen_reader_announce() -> None:
|
||||
assert "toolAnnounce(_toolAnnounceText(list))" in body
|
||||
assert 'block.setAttribute("aria-busy", "true")' in body
|
||||
assert 'block.removeAttribute("aria-busy")' in body
|
||||
|
||||
|
||||
def test_global_stream_recovery_floor_and_render_coalescing() -> None:
|
||||
"""Perf-audit P0/P1 for the Tier-1 global stream. The server's recovery
|
||||
events for a truncated reconnect gap (``node_snapshot`` as the floor,
|
||||
``replay_truncated`` as the marker) used to fall through the handler
|
||||
silently — workstreams created during a long hidden-tab gap never
|
||||
rendered again, and missed ``ws_closed`` left ghost rows forever. A
|
||||
malformed frame is the same permanent drift (the cursor advances before
|
||||
the parse), so it resyncs too. ``fireRender`` is rAF-coalesced: every
|
||||
``ws_state`` (≥2 per tool round per workstream) used to trigger a
|
||||
synchronous full rail rebuild."""
|
||||
body = _APP_JS.read_text(encoding="utf-8")
|
||||
assert 'data.type === "node_snapshot"' in body
|
||||
assert 'data.type === "replay_truncated"' in body
|
||||
assert "function applyRosterSnapshot(" in body
|
||||
assert "function resyncRoster(" in body
|
||||
assert "malformed frame" in body
|
||||
fire = body.index("function fireRender()")
|
||||
assert "requestAnimationFrame(" in body[fire : fire + 700], (
|
||||
"fireRender must coalesce subscriber repaints to one per frame"
|
||||
)
|
||||
|
||||
|
||||
def test_server_global_accels_are_platform_aware_and_scoped() -> None:
|
||||
"""The standalone's keydown handler owns only the GLOBAL accels — new
|
||||
workstream, switch, dashboard. They pick the modifier per platform (Ctrl on
|
||||
macOS where the browser owns Cmd, Alt elsewhere) so Ctrl+T/1-9 aren't eaten
|
||||
by the browser off macOS. The per-pane verbs (edit/refresh/fork/delete/
|
||||
close) moved to shell.js, so the handler must not invoke them itself."""
|
||||
body = _APP_JS.read_text(encoding="utf-8")
|
||||
assert "const IS_MAC" in body and 'navigator.platform.indexOf("Mac")' in body, (
|
||||
"the accelerators need a platform check to choose Ctrl vs Alt"
|
||||
)
|
||||
handler = body[body.index('document.addEventListener("keydown"') :]
|
||||
assert "const paneMod" in handler, (
|
||||
"global accels must gate on the platform-aware paneMod, not raw ctrlKey"
|
||||
)
|
||||
assert 'e.ctrlKey && e.key === "t"' not in handler, (
|
||||
"Ctrl+T is browser-reserved off macOS — new workstream must bind via paneMod"
|
||||
)
|
||||
assert "newWorkstream()" in handler and "switchTab(" in handler, (
|
||||
"the standalone handler still owns new + switch"
|
||||
)
|
||||
# macOS Ctrl+T / Ctrl+D are the Cocoa transpose / delete-forward text
|
||||
# bindings; the creation/dashboard chords must yield while typing, through
|
||||
# the shared TS_SHELL.inEditable guard (not a per-file copy).
|
||||
assert "TS_SHELL.inEditable(" in handler, (
|
||||
"new + dashboard must yield to text editing (macOS Ctrl+T / Ctrl+D)"
|
||||
)
|
||||
# The per-pane verbs are shell.js's job now — the standalone handler must not
|
||||
# double-bind them (shell.js drives them off the active pane's menu).
|
||||
for verb in ("editWorkstreamTitle()", "forkWorkstream()", "confirmDeleteWorkstream()"):
|
||||
assert verb not in handler, (
|
||||
f"{verb} moved to shell.js — the app.js handler must not also bind it"
|
||||
)
|
||||
|
||||
|
||||
def test_shortcut_overlay_labels_match_the_platform_modifier() -> None:
|
||||
"""The '?' help overlay must advertise the same modifier the handler
|
||||
listens for — Ctrl on macOS, Alt on Windows/Linux — instead of a hardcoded
|
||||
Ctrl that is wrong (and non-functional) off macOS."""
|
||||
index = _INDEX_HTML.read_text(encoding="utf-8")
|
||||
assert "const PANE_MOD" in index and 'navigator.platform.indexOf("Mac")' in index, (
|
||||
"the overlay must compute its modifier label per platform"
|
||||
)
|
||||
assert "${PANE_MOD}+T" in index, "the New-workstream badge must render through PANE_MOD"
|
||||
assert '<span class="kb-key">Ctrl+T</span>' not in index, (
|
||||
"the New-workstream badge must not hardcode Ctrl (wrong off macOS)"
|
||||
)
|
||||
|
||||
|
||||
def test_pane_menu_accels_are_shared_and_platform_aware() -> None:
|
||||
"""shell.js is the single source of truth for the per-pane tab-menu
|
||||
shortcuts: the badge string and the keydown handler come from ONE registry,
|
||||
so a badge can't advertise a chord the handler ignores. Badges must be
|
||||
platform-aware (no hardcoded Ctrl), and the shared handler must drive the
|
||||
ACTIVE pane's own menu so each surface contributes only what it supports."""
|
||||
shell = _SHELL_JS.read_text(encoding="utf-8")
|
||||
assert "PANE_MENU_ACCELS" in shell and "function paneAccelBadge" in shell, (
|
||||
"shell.js must own the accel registry + badge builder"
|
||||
)
|
||||
assert "const PANE_MOD_LABEL" in shell and 'navigator.platform.indexOf("Mac")' in shell, (
|
||||
"the shared badge must be platform-aware (Ctrl on macOS, Alt elsewhere)"
|
||||
)
|
||||
# The tab-menu items carry a stable accel + a computed badge, NOT a hardcoded
|
||||
# Ctrl string that would lie on Windows/Linux.
|
||||
for accel in ("close-pane", "edit-title", "refresh-title", "delete"):
|
||||
assert f'accel: "{accel}"' in shell, f"tab menu must tag the {accel} item"
|
||||
assert 'key: "Ctrl+Shift+E"' not in shell and 'key: "Ctrl+W"' not in shell, (
|
||||
"tab-menu badges must go through paneAccelBadge, not hardcoded Ctrl"
|
||||
)
|
||||
# The shared handler resolves the active pane and runs its menu item by accel.
|
||||
assert "paneAccelFor(e)" in shell and "pane.tabMenu()" in shell, (
|
||||
"the shared keydown handler must drive the active pane's menu by accel"
|
||||
)
|
||||
# The typing guard is shared (TS_SHELL.inEditable), not copied per surface.
|
||||
assert "function inEditable(" in shell and "inEditable," in shell, (
|
||||
"shell.js must define + expose the shared inEditable guard on TS_SHELL"
|
||||
)
|
||||
ui = _APP_JS.read_text(encoding="utf-8")
|
||||
console = _CONSOLE_APP_JS.read_text(encoding="utf-8")
|
||||
assert "_inEditable" not in ui and "_consoleInEditable" not in console, (
|
||||
"surfaces must use TS_SHELL.inEditable, not a per-file copy of the guard"
|
||||
)
|
||||
|
||||
|
||||
def test_console_has_matching_pane_hotkeys() -> None:
|
||||
"""The console regained pane hotkeys to match the standalone: a keydown
|
||||
handler for switch (Mod+1-9) + dashboard (Ctrl+D), and a '?' overlay that
|
||||
advertises them platform-aware. New workstream and Fork are intentionally
|
||||
omitted (no console fork / blank-new surface)."""
|
||||
app = _CONSOLE_APP_JS.read_text(encoding="utf-8")
|
||||
assert (
|
||||
"_CONSOLE_IS_MAC" in app and "statefulTabs()" in app and 'openPane("dashboard")' in app
|
||||
), "the console must wire switch (statefulTabs) + dashboard hotkeys"
|
||||
index = _CONSOLE_INDEX.read_text(encoding="utf-8")
|
||||
assert "const PANE_MOD" in index and '"Panes"' in index, (
|
||||
"the console '?' overlay needs a platform-aware Panes section"
|
||||
)
|
||||
assert "${PANE_MOD}+W" in index and "${PANE_MOD}+Shift+E" in index, (
|
||||
"console badges must render through PANE_MOD"
|
||||
)
|
||||
assert '"Fork"' not in index and "New workstream" not in index, (
|
||||
"Fork + New are intentionally omitted on the console"
|
||||
)
|
||||
|
||||
+45
-40
@@ -13,9 +13,16 @@ from turnstone.core.session import (
|
||||
ChatSession,
|
||||
GenerationCancelled,
|
||||
_CancelRef,
|
||||
_effect_status_meta,
|
||||
_tool_turn_meta,
|
||||
)
|
||||
from turnstone.core.trajectory import (
|
||||
EffectStatus,
|
||||
Role,
|
||||
ToolCall,
|
||||
Turn,
|
||||
dicts_from_turns,
|
||||
turn_from_dict,
|
||||
)
|
||||
from turnstone.core.trajectory import EffectStatus, Role, dicts_from_turns, turn_from_dict
|
||||
|
||||
|
||||
class NullUI:
|
||||
@@ -1028,15 +1035,11 @@ class TestCancelledAgentDisposition:
|
||||
|
||||
@staticmethod
|
||||
def _assistant(call_id, name):
|
||||
return {
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [{"id": call_id, "function": {"name": name}}],
|
||||
}
|
||||
return Turn.assistant("", tool_calls=(ToolCall(id=call_id, name=name, arguments=""),))
|
||||
|
||||
@staticmethod
|
||||
def _result(call_id, text="ok"):
|
||||
return {"role": "tool", "tool_call_id": call_id, "content": text}
|
||||
return Turn.tool(call_id, text)
|
||||
|
||||
def test_status_none_when_no_actions(self):
|
||||
"""Typed twin of the disposition: a task cancelled before any action is
|
||||
@@ -1107,14 +1110,13 @@ class TestCancelledAgentDisposition:
|
||||
# started" — inviting a re-run of the destructive bash.
|
||||
session = _make_session()
|
||||
msgs = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{"id": "t1", "function": {"name": "bash"}},
|
||||
{"id": "t2", "function": {"name": "web_fetch"}},
|
||||
],
|
||||
}
|
||||
Turn.assistant(
|
||||
"",
|
||||
tool_calls=(
|
||||
ToolCall(id="t1", name="bash", arguments=""),
|
||||
ToolCall(id="t2", name="web_fetch", arguments=""),
|
||||
),
|
||||
)
|
||||
] # neither answered: bash raised mid-flight, web_fetch never ran
|
||||
out = session._cancelled_agent_disposition(msgs, "task")
|
||||
assert "In flight at cancel: bash" in out
|
||||
@@ -1127,26 +1129,24 @@ class TestCancelledAgentDisposition:
|
||||
# count summary, the first-gap boundary, and not-started.
|
||||
session = _make_session()
|
||||
msgs = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{"id": "t1", "function": {"name": "bash"}},
|
||||
{"id": "t2", "function": {"name": "bash"}},
|
||||
{"id": "t3", "function": {"name": "read_file"}},
|
||||
],
|
||||
},
|
||||
Turn.assistant(
|
||||
"",
|
||||
tool_calls=(
|
||||
ToolCall(id="t1", name="bash", arguments=""),
|
||||
ToolCall(id="t2", name="bash", arguments=""),
|
||||
ToolCall(id="t3", name="read_file", arguments=""),
|
||||
),
|
||||
),
|
||||
self._result("t1"),
|
||||
self._result("t2"),
|
||||
self._result("t3"),
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{"id": "t4", "function": {"name": "web_fetch"}},
|
||||
{"id": "t5", "function": {"name": "search"}},
|
||||
],
|
||||
},
|
||||
Turn.assistant(
|
||||
"",
|
||||
tool_calls=(
|
||||
ToolCall(id="t4", name="web_fetch", arguments=""),
|
||||
ToolCall(id="t5", name="search", arguments=""),
|
||||
),
|
||||
),
|
||||
]
|
||||
out = session._cancelled_agent_disposition(msgs, "task")
|
||||
assert "Completed before cancel: bash×2, read_file." in out
|
||||
@@ -1155,13 +1155,13 @@ class TestCancelledAgentDisposition:
|
||||
|
||||
def test_exec_task_routes_cancel_to_disposition(self, tmp_db):
|
||||
"""_exec_task converts a GenerationCancelled from _run_agent into the
|
||||
honest disposition, reading the in-place-mutated agent_messages."""
|
||||
honest disposition, reading the in-place-mutated agent_turns."""
|
||||
session = _make_session()
|
||||
|
||||
def fake_run_agent(agent_messages, **kwargs):
|
||||
agent_messages.append(self._assistant("t1", "bash"))
|
||||
agent_messages.append(self._result("t1"))
|
||||
agent_messages.append(self._assistant("t2", "web_fetch"))
|
||||
def fake_run_agent(agent_turns, **kwargs):
|
||||
agent_turns.append(self._assistant("t1", "bash"))
|
||||
agent_turns.append(self._result("t1"))
|
||||
agent_turns.append(self._assistant("t2", "web_fetch"))
|
||||
raise GenerationCancelled()
|
||||
|
||||
with patch.object(session, "_run_agent", side_effect=fake_run_agent):
|
||||
@@ -1183,8 +1183,13 @@ class TestEffectStatusPersistence:
|
||||
effect-record appendix — the ledger persists for audit)."""
|
||||
|
||||
def test_effect_status_meta_envelope(self):
|
||||
assert _effect_status_meta(None) is None
|
||||
assert json.loads(_effect_status_meta(EffectStatus.UNKNOWN)) == {"effect_status": "unknown"}
|
||||
assert _tool_turn_meta(None) is None
|
||||
assert json.loads(_tool_turn_meta(EffectStatus.UNKNOWN)) == {"effect_status": "unknown"}
|
||||
assert json.loads(_tool_turn_meta(None, {"kind": "web"})) == {"preview": {"kind": "web"}}
|
||||
assert json.loads(_tool_turn_meta(EffectStatus.UNKNOWN, {"kind": "web"})) == {
|
||||
"effect_status": "unknown",
|
||||
"preview": {"kind": "web"},
|
||||
}
|
||||
|
||||
def test_reconstruct_routes_tool_effect_status(self):
|
||||
from turnstone.core.storage._utils import reconstruct_turns
|
||||
|
||||
@@ -41,6 +41,11 @@ def _bind_ws_event_handlers(bot, cls):
|
||||
attr = getattr(cls, name)
|
||||
if callable(attr):
|
||||
setattr(bot, name, attr.__get__(bot, cls))
|
||||
# ``_handle_stream_end`` delegates the all-cycles sweep to
|
||||
# ``_pop_ws_approvals``; bind the real method too so dispatcher
|
||||
# tests observe the pop instead of a spec'd AsyncMock no-op.
|
||||
if hasattr(cls, "_pop_ws_approvals"):
|
||||
bot._pop_ws_approvals = cls._pop_ws_approvals.__get__(bot, cls)
|
||||
|
||||
|
||||
def _make_message(*, bot=False, guild=True, content="hello", channel=None, reference=None):
|
||||
@@ -537,7 +542,7 @@ class TestApprovalVerdictDisplay:
|
||||
},
|
||||
}
|
||||
]
|
||||
event = ApproveRequestEvent(ws_id="ws-1", items=items)
|
||||
event = ApproveRequestEvent(ws_id="ws-1", cycle_id="cyc-1", items=items)
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
# thread.send was called with an embed containing a verdict field
|
||||
@@ -551,8 +556,8 @@ class TestApprovalVerdictDisplay:
|
||||
assert "HIGH" in field.value
|
||||
assert "85%" in field.value
|
||||
|
||||
# Pending approval message tracked
|
||||
assert "ws-1" in bot._pending_approval_msgs
|
||||
# Pending approval message tracked under (ws_id, cycle_id).
|
||||
assert ("ws-1", "cyc-1") in bot._pending_approval_msgs
|
||||
|
||||
def test_approval_without_verdict(self):
|
||||
"""ApproveRequestEvent items without verdict still work normally."""
|
||||
@@ -585,10 +590,11 @@ class TestApprovalVerdictDisplay:
|
||||
embed = MagicMock()
|
||||
msg.embeds = [embed]
|
||||
msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (msg, frozenset({"c-1"}))
|
||||
|
||||
event = IntentVerdictEvent(
|
||||
ws_id="ws-1",
|
||||
call_id="c-1",
|
||||
func_name="bash",
|
||||
risk_level="high",
|
||||
recommendation="deny",
|
||||
@@ -628,7 +634,10 @@ class TestApprovalVerdictDisplay:
|
||||
bot._streaming = {}
|
||||
bot._thinking_msgs = {}
|
||||
bot._tool_info_msgs = {}
|
||||
bot._pending_approval_msgs = {"ws-1": MagicMock()}
|
||||
bot._pending_approval_msgs = {
|
||||
("ws-1", "cyc-1"): (MagicMock(), frozenset()),
|
||||
("ws-1", "cyc-2"): (MagicMock(), frozenset()),
|
||||
}
|
||||
bot._notify_reply_channels = {}
|
||||
_bind_ws_event_handlers(bot, TurnstoneBot)
|
||||
|
||||
@@ -636,7 +645,8 @@ class TestApprovalVerdictDisplay:
|
||||
event = StreamEndEvent(ws_id="ws-1")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
assert "ws-1" not in bot._pending_approval_msgs
|
||||
# ALL of the ws's cycles are swept, not just one entry.
|
||||
assert not bot._pending_approval_msgs
|
||||
|
||||
|
||||
class TestStreamEndBehavior:
|
||||
@@ -1657,19 +1667,21 @@ class TestApprovalResolved:
|
||||
bot = self._make_bot()
|
||||
thread = AsyncMock()
|
||||
|
||||
# Set up a pending approval message with components.
|
||||
# Set up a pending approval message with components. The event
|
||||
# below carries no cycle_id (pre-multi-cycle server) — the
|
||||
# legacy fallback clears the ws's single tracked entry.
|
||||
approval_msg = MagicMock()
|
||||
approval_msg.embeds = [MagicMock()]
|
||||
approval_msg.components = []
|
||||
approval_msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = approval_msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (approval_msg, frozenset())
|
||||
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=False, feedback="timeout")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
approval_msg.edit.assert_awaited_once()
|
||||
# Pending approval message should be removed.
|
||||
assert "ws-1" not in bot._pending_approval_msgs
|
||||
assert not bot._pending_approval_msgs
|
||||
|
||||
def test_disables_buttons_on_approved(self):
|
||||
from turnstone.sdk.events import ApprovalResolvedEvent
|
||||
@@ -1681,9 +1693,11 @@ class TestApprovalResolved:
|
||||
approval_msg.embeds = [MagicMock()]
|
||||
approval_msg.components = []
|
||||
approval_msg.edit = AsyncMock()
|
||||
bot._pending_approval_msgs["ws-1"] = approval_msg
|
||||
bot._pending_approval_msgs[("ws-1", "cyc-1")] = (approval_msg, frozenset())
|
||||
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True)
|
||||
# Cycle-routed resolution: the event's cycle_id selects exactly
|
||||
# this tracked message.
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True, cycle_id="cyc-1")
|
||||
_run(bot._on_ws_event("ws-1", thread, event))
|
||||
|
||||
approval_msg.edit.assert_awaited_once()
|
||||
|
||||
@@ -87,7 +87,7 @@ class TestSendApproval:
|
||||
monkeypatch.setattr(router._server, "approve", mock_approve)
|
||||
await router.send_approval("ws-1", "corr-abc", approved=True, feedback="ok")
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=True, feedback="ok", always=False
|
||||
ws_id="ws-1", approved=True, feedback="ok", always=False, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
@pytest.mark.anyio
|
||||
@@ -99,7 +99,7 @@ class TestSendApproval:
|
||||
monkeypatch.setattr(router._server, "approve", mock_approve)
|
||||
await router.send_approval("ws-1", "corr-abc", approved=False)
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=False, feedback=None, always=False
|
||||
ws_id="ws-1", approved=False, feedback=None, always=False, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
@pytest.mark.anyio
|
||||
@@ -110,7 +110,9 @@ class TestSendApproval:
|
||||
mock_approve = AsyncMock()
|
||||
monkeypatch.setattr(console_router._console, "route_approve", mock_approve)
|
||||
await console_router.send_approval("ws-1", "corr-abc", approved=True, always=True)
|
||||
mock_approve.assert_awaited_once_with(ws_id="ws-1", approved=True, feedback="", always=True)
|
||||
mock_approve.assert_awaited_once_with(
|
||||
ws_id="ws-1", approved=True, feedback="", always=True, cycle_id="corr-abc"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteRoute:
|
||||
|
||||
@@ -576,10 +576,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -598,10 +599,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -620,10 +622,11 @@ class TestApprovalOwnership:
|
||||
|
||||
bot, router, client = _make_bot()
|
||||
ws_id = "ws-1"
|
||||
bot._pending_approval[ws_id] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[(ws_id, "corr-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C01SAPU5414",
|
||||
message_ts="111.222",
|
||||
owner_user_id="U_OWNER",
|
||||
cycle_id="corr-1",
|
||||
)
|
||||
|
||||
body = {
|
||||
@@ -776,7 +779,9 @@ class TestWsEventDispatch:
|
||||
bot, client = self._make_ws_bot()
|
||||
|
||||
event = ApproveRequestEvent(
|
||||
ws_id="ws-1", items=[{"func_name": "bash", "needs_approval": True}]
|
||||
ws_id="ws-1",
|
||||
cycle_id="cyc-1",
|
||||
items=[{"call_id": "c-1", "func_name": "bash", "needs_approval": True}],
|
||||
)
|
||||
route = SlackRoute(channel="C1", user_id="U12345", thread_ts="123.456")
|
||||
_run(bot._on_ws_event("ws-1", route, event)) # type: ignore[attr-defined]
|
||||
@@ -784,8 +789,12 @@ class TestWsEventDispatch:
|
||||
client.chat_postMessage.assert_awaited_once()
|
||||
call_kwargs = client.chat_postMessage.call_args[1]
|
||||
assert "blocks" in call_kwargs
|
||||
assert "ws-1" in bot._pending_approval # type: ignore[attr-defined]
|
||||
assert bot._pending_approval["ws-1"].owner_user_id == "U12345" # type: ignore[attr-defined]
|
||||
# Tracked under (ws_id, cycle_id) so concurrent cycles each get
|
||||
# their own Slack message.
|
||||
entry = bot._pending_approval[("ws-1", "cyc-1")] # type: ignore[attr-defined]
|
||||
assert entry.owner_user_id == "U12345"
|
||||
assert entry.cycle_id == "cyc-1"
|
||||
assert entry.call_ids == frozenset({"c-1"})
|
||||
|
||||
def test_intent_verdict_updates_approval_message(self) -> None:
|
||||
from turnstone.channels.slack.bot import PendingApproval
|
||||
@@ -797,14 +806,17 @@ class TestWsEventDispatch:
|
||||
return_value={"ok": True, "messages": [{"blocks": []}]}
|
||||
)
|
||||
|
||||
bot._pending_approval["ws-1"] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[("ws-1", "cyc-1")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C1",
|
||||
message_ts="999.000",
|
||||
owner_user_id="U12345",
|
||||
cycle_id="cyc-1",
|
||||
call_ids=frozenset({"c-1"}),
|
||||
)
|
||||
|
||||
event = IntentVerdictEvent(
|
||||
ws_id="ws-1",
|
||||
call_id="c-1",
|
||||
func_name="bash",
|
||||
risk_level="high",
|
||||
confidence=0.9,
|
||||
@@ -821,17 +833,20 @@ class TestWsEventDispatch:
|
||||
from turnstone.sdk.events import ApprovalResolvedEvent
|
||||
|
||||
bot, client = self._make_ws_bot()
|
||||
bot._pending_approval["ws-1"] = PendingApproval( # type: ignore[attr-defined]
|
||||
bot._pending_approval[("ws-1", "cyc-9")] = PendingApproval( # type: ignore[attr-defined]
|
||||
channel="C1",
|
||||
message_ts="999.000",
|
||||
owner_user_id="U12345",
|
||||
cycle_id="cyc-9",
|
||||
)
|
||||
|
||||
# Event WITHOUT a cycle_id (pre-multi-cycle server): the legacy
|
||||
# fallback clears the ws's single tracked entry, as before.
|
||||
event = ApprovalResolvedEvent(ws_id="ws-1", approved=True)
|
||||
route = SlackRoute(channel="C1", user_id="U12345", thread_ts="123.456")
|
||||
_run(bot._on_ws_event("ws-1", route, event)) # type: ignore[attr-defined]
|
||||
|
||||
assert "ws-1" not in bot._pending_approval # type: ignore[attr-defined]
|
||||
assert not bot._pending_approval # type: ignore[attr-defined]
|
||||
client.chat_update.assert_awaited_once()
|
||||
|
||||
def test_link_prefix_does_not_hijack_regular_prompt(self) -> None:
|
||||
|
||||
@@ -0,0 +1,449 @@
|
||||
"""Tests for persisted compaction checkpoints (rehydration-deadlock fix).
|
||||
|
||||
Compaction swaps a session's in-memory history for a summary but leaves the full
|
||||
transcript in storage. Without a durable marker, ``resume()`` reloaded the full
|
||||
pre-compaction history, which on a long session — or one switched to a smaller-
|
||||
context model — exceeds the window and deadlocks the first send.
|
||||
|
||||
The fix persists one ``_source="compaction"`` marker (summary + watermark) so
|
||||
resume rehydrates ``[summary] + [rows after the watermark]`` while the full
|
||||
history stays in storage for ``/history``/export. Covered here:
|
||||
|
||||
- ``get_compaction_watermark`` — the boundary id (max-summarized), with and
|
||||
without a preserved tail, and on an empty workstream.
|
||||
- ``load_message_turns`` (resume) — checkpoint-aware slice, latest-marker-wins,
|
||||
preserved-tail handling, and the full-history fallbacks (no marker, malformed
|
||||
marker) that keep every pre-checkpoint session loading exactly as before.
|
||||
- ``load_messages`` (display) — markers stay invisible to ``/history``.
|
||||
- End-to-end: ``_compact_messages`` writes the marker and a fresh ``resume()``
|
||||
rehydrates the bounded view, not the full transcript.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._session_helpers import make_session
|
||||
from turnstone.core.trajectory import turns_from_dicts
|
||||
|
||||
|
||||
def _marker_meta(watermark: int | None) -> str | None:
|
||||
"""The marker's stored ``meta`` JSON (``None`` simulates a legacy/malformed marker)."""
|
||||
return json.dumps({"watermark": watermark}) if watermark is not None else None
|
||||
|
||||
|
||||
def _register(st, ws: str = "ws1") -> str:
|
||||
st.register_workstream(ws, user_id="u1", title="t", kind="interactive")
|
||||
return ws
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# get_compaction_watermark
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestWatermark:
|
||||
def test_preserve_tail_zero_is_max_id(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
ids = [st.save_message(ws, "user", f"m{i}") for i in range(5)]
|
||||
assert st.get_compaction_watermark(ws, 0) == max(ids)
|
||||
|
||||
def test_preserve_tail_n_is_nth_newest(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
ids = sorted(st.save_message(ws, "user", f"m{i}") for i in range(5))
|
||||
# Keep the newest 2 verbatim → boundary is the 3rd-newest id.
|
||||
assert st.get_compaction_watermark(ws, 2) == ids[-3]
|
||||
|
||||
def test_preserve_tail_ignores_existing_markers(self, storage_backend):
|
||||
# A compaction marker is saved as a NEW row but is not part of the
|
||||
# preserved in-memory tail, so it must not shift the (preserve_tail+1)
|
||||
# boundary — without the exclusion, this returns ids[-1] (the marker
|
||||
# consumes an offset slot) and resume would drop a real tail row.
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
ids = [st.save_message(ws, "user", f"m{i}") for i in range(5)]
|
||||
st.save_message(ws, "assistant", "SUM", source="compaction", meta=_marker_meta(max(ids)))
|
||||
st.save_message(ws, "user", "m5")
|
||||
# Real rows newest-first: m5, m4, m3, ... → 3rd-newest real row is m3.
|
||||
assert st.get_compaction_watermark(ws, 2) == ids[-2]
|
||||
|
||||
def test_empty_workstream_is_none(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
assert st.get_compaction_watermark(ws, 0) is None
|
||||
|
||||
def test_preserve_tail_exceeding_row_count_is_none(self, storage_backend):
|
||||
# Fewer rows than the preserved tail → no boundary, so compaction skips
|
||||
# the marker rather than writing a watermark that points past the history.
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "only")
|
||||
assert st.get_compaction_watermark(ws, 5) is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# load_message_turns — checkpoint-aware resume
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestCheckpointResume:
|
||||
def test_loads_summary_plus_tail_not_full_history(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
for i in range(5):
|
||||
st.save_message(ws, "user" if i % 2 == 0 else "assistant", f"old{i}")
|
||||
watermark = st.get_compaction_watermark(ws, 0)
|
||||
st.save_message(
|
||||
ws, "assistant", "THE SUMMARY", source="compaction", meta=_marker_meta(watermark)
|
||||
)
|
||||
st.save_message(ws, "user", "new question")
|
||||
st.save_message(ws, "assistant", "new answer")
|
||||
|
||||
texts = [t.text for t in st.load_message_turns(ws)]
|
||||
assert texts == ["[Conversation summary]", "THE SUMMARY", "new question", "new answer"]
|
||||
assert not any("old" in x for x in texts) # summarized prefix is gone
|
||||
|
||||
def test_preserved_tail_kept_after_summary(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
ids = sorted(st.save_message(ws, "user", f"m{i}") for i in range(4))
|
||||
# Mid-turn compaction keeps the newest row (m3) verbatim.
|
||||
watermark = st.get_compaction_watermark(ws, 1)
|
||||
assert watermark == ids[-2]
|
||||
st.save_message(ws, "assistant", "SUM", source="compaction", meta=_marker_meta(watermark))
|
||||
|
||||
texts = [t.text for t in st.load_message_turns(ws)]
|
||||
assert texts == ["[Conversation summary]", "SUM", "m3"]
|
||||
|
||||
def test_latest_marker_wins(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "old")
|
||||
st.save_message(
|
||||
ws,
|
||||
"assistant",
|
||||
"SUMMARY 1",
|
||||
source="compaction",
|
||||
meta=_marker_meta(st.get_compaction_watermark(ws, 0)),
|
||||
)
|
||||
st.save_message(ws, "user", "mid")
|
||||
st.save_message(
|
||||
ws,
|
||||
"assistant",
|
||||
"SUMMARY 2",
|
||||
source="compaction",
|
||||
meta=_marker_meta(st.get_compaction_watermark(ws, 0)),
|
||||
)
|
||||
st.save_message(ws, "user", "after")
|
||||
|
||||
texts = [t.text for t in st.load_message_turns(ws)]
|
||||
assert texts == ["[Conversation summary]", "SUMMARY 2", "after"]
|
||||
assert "SUMMARY 1" not in texts and "old" not in texts and "mid" not in texts
|
||||
|
||||
def test_no_marker_loads_full_history(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
for i in range(3):
|
||||
st.save_message(ws, "user", f"m{i}")
|
||||
assert [t.text for t in st.load_message_turns(ws)] == ["m0", "m1", "m2"]
|
||||
|
||||
def test_malformed_marker_falls_back_to_full_history(self, storage_backend):
|
||||
# A marker with no watermark (legacy/corrupt) must NOT slice — losing
|
||||
# real messages is worse than reloading more than necessary.
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "a")
|
||||
st.save_message(ws, "assistant", "SUMMARY", source="compaction", meta=None)
|
||||
st.save_message(ws, "user", "b")
|
||||
texts = [t.text for t in st.load_message_turns(ws)]
|
||||
assert "a" in texts and "b" in texts # no real message dropped
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# load_messages — display path keeps markers invisible
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestDisplayPath:
|
||||
def test_history_excludes_marker(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "q")
|
||||
st.save_message(ws, "assistant", "a")
|
||||
st.save_message(
|
||||
ws,
|
||||
"assistant",
|
||||
"SUMMARY",
|
||||
source="compaction",
|
||||
meta=_marker_meta(st.get_compaction_watermark(ws, 0)),
|
||||
)
|
||||
|
||||
contents = [m.get("content") for m in st.load_messages(ws)]
|
||||
assert "SUMMARY" not in contents
|
||||
assert contents == ["q", "a"] # true transcript, no injected summary
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# End-to-end: compaction writes the marker, resume is bounded
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_compaction_persists_checkpoint_and_resume_is_bounded(tmp_db, mock_openai_client):
|
||||
"""The deadlock-fix proof: a session compacts, a fresh session reopens it,
|
||||
and resume rehydrates [summary]+[tail] — never the full pre-compaction
|
||||
transcript that would overflow the window on reopen."""
|
||||
from unittest.mock import patch
|
||||
|
||||
from turnstone.core.memory import register_workstream, save_message
|
||||
|
||||
ws = "wsE2E"
|
||||
register_workstream(ws, user_id="u1", name="t")
|
||||
history = [
|
||||
{"role": "user" if i % 2 == 0 else "assistant", "content": f"turn {i}"} for i in range(6)
|
||||
]
|
||||
for h in history:
|
||||
save_message(ws, h["role"], h["content"])
|
||||
|
||||
sess = make_session(client=mock_openai_client, context_window=10_000, max_tokens=1_000)
|
||||
sess._ws_id = ws
|
||||
sess.messages = turns_from_dicts(history)
|
||||
sess._msg_tokens = [1] * len(history)
|
||||
with patch.object(sess, "_summarize_blocks", return_value="DENSE SUMMARY"):
|
||||
assert sess._compact_messages(auto=False) is True
|
||||
|
||||
# Conversation continues after the compaction.
|
||||
save_message(ws, "user", "after compaction")
|
||||
|
||||
# A fresh session reopens the workstream.
|
||||
sess2 = make_session(client=mock_openai_client, context_window=10_000, max_tokens=1_000)
|
||||
assert sess2.resume(ws) is True
|
||||
texts = [t.text for t in sess2.messages]
|
||||
|
||||
assert texts[:2] == ["[Conversation summary]", "DENSE SUMMARY"]
|
||||
assert "after compaction" in texts
|
||||
assert not any(t.startswith("turn ") for t in texts) # full history NOT reloaded
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Malformed / edge-case markers — the watermark guards and the empty tail
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestMarkerEdges:
|
||||
@pytest.mark.parametrize(
|
||||
"meta",
|
||||
[
|
||||
json.dumps({"watermark": "5"}), # non-int (string)
|
||||
json.dumps({"watermark": True}), # bool — True is an int subclass
|
||||
json.dumps({}), # key absent
|
||||
json.dumps({"watermark": None}), # null
|
||||
],
|
||||
)
|
||||
def test_non_int_watermark_falls_back_to_full_history(self, storage_backend, meta):
|
||||
# A watermark that isn't a real int must NOT slice (a True watermark
|
||||
# would otherwise cut at id 1 and drop real history).
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "a")
|
||||
st.save_message(ws, "assistant", "b")
|
||||
st.save_message(ws, "assistant", "SUMMARY", source="compaction", meta=meta)
|
||||
st.save_message(ws, "user", "c")
|
||||
texts = [t.text for t in st.load_message_turns(ws)]
|
||||
assert "a" in texts and "b" in texts and "c" in texts # nothing sliced away
|
||||
# ...and the malformed marker is DROPPED, not leaked as a stray summary turn.
|
||||
assert "SUMMARY" not in texts
|
||||
|
||||
def test_marker_as_final_row_yields_empty_tail(self, storage_backend):
|
||||
# watermark == max id, marker is the last row → resume is just the summary.
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
for i in range(3):
|
||||
st.save_message(ws, "user", f"old{i}")
|
||||
wm = st.get_compaction_watermark(ws, 0)
|
||||
st.save_message(ws, "assistant", "SUMMARY", source="compaction", meta=_marker_meta(wm))
|
||||
assert [t.text for t in st.load_message_turns(ws)] == ["[Conversation summary]", "SUMMARY"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# checkpointed=False — export/audit gets the FULL transcript (markers dropped)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestFullHistoryLoad:
|
||||
def test_checkpointed_false_returns_full_history_without_marker(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
for i in range(4):
|
||||
st.save_message(ws, "user" if i % 2 == 0 else "assistant", f"old{i}")
|
||||
wm = st.get_compaction_watermark(ws, 0)
|
||||
st.save_message(ws, "assistant", "SUMMARY", source="compaction", meta=_marker_meta(wm))
|
||||
st.save_message(ws, "user", "after")
|
||||
|
||||
# Resume (default) is bounded; export (checkpointed=False) is full + marker-free.
|
||||
assert [t.text for t in st.load_message_turns(ws)] == [
|
||||
"[Conversation summary]",
|
||||
"SUMMARY",
|
||||
"after",
|
||||
]
|
||||
full = [t.text for t in st.load_message_turns(ws, checkpointed=False)]
|
||||
assert full == ["old0", "old1", "old2", "old3", "after"]
|
||||
assert "SUMMARY" not in full and "[Conversation summary]" not in full
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# search — compaction markers stay out of search results
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestSearchExclusion:
|
||||
def test_search_history_excludes_markers(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "findme apple")
|
||||
st.save_message(
|
||||
ws,
|
||||
"assistant",
|
||||
"findme SUMMARY banana",
|
||||
source="compaction",
|
||||
meta=_marker_meta(st.get_compaction_watermark(ws, 0)),
|
||||
)
|
||||
contents = [r[3] for r in st.search_history("findme")]
|
||||
assert any("apple" in (c or "") for c in contents) # real row matched
|
||||
assert not any("SUMMARY" in (c or "") for c in contents) # marker excluded
|
||||
# ...and normal rows (whose _source is NULL) are NOT dropped by the filter.
|
||||
assert contents
|
||||
|
||||
def test_search_history_recent_excludes_markers(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "real")
|
||||
st.save_message(
|
||||
ws,
|
||||
"assistant",
|
||||
"SUMMARY",
|
||||
source="compaction",
|
||||
meta=_marker_meta(st.get_compaction_watermark(ws, 0)),
|
||||
)
|
||||
recent = [r[3] for r in st.search_history_recent(10)]
|
||||
assert "real" in recent and "SUMMARY" not in recent
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# rewind / retry — compaction-safe truncation (never delete the summary backing)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestCompactionFloor:
|
||||
def test_floor_and_count(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
for i in range(3):
|
||||
st.save_message(ws, "user", f"old{i}") # summarized prefix
|
||||
wm = st.get_compaction_watermark(ws, 0)
|
||||
st.save_message(ws, "assistant", "SUMMARY", source="compaction", meta=_marker_meta(wm))
|
||||
st.save_message(ws, "user", "tail1")
|
||||
st.save_message(ws, "assistant", "tail2")
|
||||
assert st.get_compaction_floor(ws) == 4 # 3 prefix + 1 marker
|
||||
assert st.count_messages(ws) == 6
|
||||
|
||||
def test_floor_zero_without_marker(self, storage_backend):
|
||||
st = storage_backend
|
||||
ws = _register(st)
|
||||
st.save_message(ws, "user", "x")
|
||||
assert st.get_compaction_floor(ws) == 0
|
||||
|
||||
|
||||
def test_rewind_after_compaction_never_deletes_summary_backing(tmp_db, mock_openai_client):
|
||||
"""The review's major rewind finding: after a compaction, a tail-trim must
|
||||
delete from the storage TAIL and floor at the marker, not keep the oldest
|
||||
summarized rows and drop the marker."""
|
||||
from turnstone.core.memory import get_storage, register_workstream, save_message
|
||||
|
||||
ws = "wsRW"
|
||||
register_workstream(ws, user_id="u1", name="t")
|
||||
for i in range(3):
|
||||
save_message(ws, "user", f"old{i}") # prefix
|
||||
st = get_storage()
|
||||
wm = st.get_compaction_watermark(ws, 0)
|
||||
save_message(
|
||||
ws, "assistant", "SUMMARY", source="compaction", meta=json.dumps({"watermark": wm})
|
||||
)
|
||||
save_message(ws, "user", "q1") # tail
|
||||
save_message(ws, "assistant", "a1") # tail
|
||||
assert st.get_compaction_floor(ws) == 4 and st.count_messages(ws) == 6
|
||||
|
||||
sess = make_session(client=mock_openai_client, context_window=10_000, max_tokens=1_000)
|
||||
sess._ws_id = ws
|
||||
|
||||
# Trim one tail turn → keep = max(floor 4, total 6 - 1) = 5 → deletes only "a1".
|
||||
sess._persist_truncation(1)
|
||||
assert st.count_messages(ws) == 5
|
||||
survived = [t.text for t in st.load_message_turns(ws)]
|
||||
assert survived[:2] == ["[Conversation summary]", "SUMMARY"] # marker + prefix intact
|
||||
assert "q1" in survived
|
||||
|
||||
# Over-deep trim → clamps at the floor; the marker + prefix still survive.
|
||||
sess._persist_truncation(100)
|
||||
assert st.count_messages(ws) == 4 # floored at prefix + marker
|
||||
after = [t.text for t in st.load_message_turns(ws)]
|
||||
assert after == ["[Conversation summary]", "SUMMARY"] # summary backing never deleted
|
||||
|
||||
|
||||
def test_persist_truncation_uncompacted_matches_plain_tail_delete(tmp_db, mock_openai_client):
|
||||
"""With no compaction (floor 0), the new path is identical to the old
|
||||
keep=len(self.messages) tail delete."""
|
||||
from turnstone.core.memory import get_storage, register_workstream, save_message
|
||||
|
||||
ws = "wsPlain"
|
||||
register_workstream(ws, user_id="u1", name="t")
|
||||
for i in range(5):
|
||||
save_message(ws, "user", f"m{i}")
|
||||
st = get_storage()
|
||||
assert st.get_compaction_floor(ws) == 0
|
||||
|
||||
sess = make_session(client=mock_openai_client, context_window=10_000, max_tokens=1_000)
|
||||
sess._ws_id = ws
|
||||
sess._persist_truncation(2) # remove the last 2
|
||||
assert st.count_messages(ws) == 3
|
||||
|
||||
|
||||
def test_persist_truncation_skips_delete_when_count_unavailable(tmp_db, mock_openai_client):
|
||||
"""count_messages==0 (the storage-error sentinel) must NOT delete — a wrong
|
||||
truncation would lose user history."""
|
||||
from unittest.mock import patch
|
||||
|
||||
from turnstone.core.memory import get_storage, register_workstream, save_message
|
||||
|
||||
ws = "wsCnt"
|
||||
register_workstream(ws, user_id="u1", name="t")
|
||||
for i in range(4):
|
||||
save_message(ws, "user", f"m{i}")
|
||||
st = get_storage()
|
||||
sess = make_session(client=mock_openai_client)
|
||||
sess._ws_id = ws
|
||||
with patch("turnstone.core.session.count_messages", return_value=0):
|
||||
sess._persist_truncation(2)
|
||||
assert st.count_messages(ws) == 4 # nothing deleted
|
||||
|
||||
|
||||
def test_persist_truncation_skips_delete_when_floor_unavailable(tmp_db, mock_openai_client):
|
||||
"""get_compaction_floor==-1 (the storage-error sentinel) must NOT delete — a 0
|
||||
floor on a compacted ws could otherwise drop the marker on an over-deep trim."""
|
||||
from unittest.mock import patch
|
||||
|
||||
from turnstone.core.memory import get_storage, register_workstream, save_message
|
||||
|
||||
ws = "wsFloor"
|
||||
register_workstream(ws, user_id="u1", name="t")
|
||||
for i in range(4):
|
||||
save_message(ws, "user", f"m{i}")
|
||||
st = get_storage()
|
||||
sess = make_session(client=mock_openai_client)
|
||||
sess._ws_id = ws
|
||||
with patch("turnstone.core.session.get_compaction_floor", return_value=-1):
|
||||
sess._persist_truncation(2)
|
||||
assert st.count_messages(ws) == 4 # nothing deleted
|
||||
@@ -0,0 +1,383 @@
|
||||
"""Tests for the compaction crossing discipline: what crosses the summary
|
||||
boundary VERBATIM (not only as summarizer paraphrase) and how the synthetic
|
||||
summary turns are recognized.
|
||||
|
||||
- **Provenance tags** — ``_compact_messages`` and
|
||||
``reconstruct_turns_checkpointed`` mark both synthetic summary turns
|
||||
``source="compaction"``; ``_find_turn_boundaries`` and ``_generate_title``
|
||||
test the tag, not the ``[Conversation summary]`` content string. A user
|
||||
who literally types the label therefore stays a REAL turn (previously it
|
||||
was silently treated as synthetic — provenance by spelling).
|
||||
- **Carry budget** — ``_carry_budget_chars`` scales the verbatim-carry
|
||||
allowance to ~25% of the window (clamped by the summary output reserve,
|
||||
floored at ``_MIN_CARRY_BUDGET_CHARS``), replacing the fixed 400-char
|
||||
continuation-hint clip; oversize content keeps head + tail around an
|
||||
honest marker.
|
||||
- **Wind-down spill** — with ``carry_spill=True`` (the end-of-turn site
|
||||
passes the ``stopped_to_compact`` latch) the final summarized assistant
|
||||
turn's text is copied onto the summary under ``## Wind-down (verbatim)``
|
||||
— shell concatenation, so the model's own plan statement survives the
|
||||
collapse even when the summarizer paraphrases it.
|
||||
- The overflow-backstop compact-and-retry passes ``my_generation`` so a
|
||||
stale send cannot compact-and-swap a newer generation's history.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._session_helpers import make_session
|
||||
from turnstone.core.session import COMPACTION_SOURCE, COMPACTION_SUMMARY_LABEL
|
||||
from turnstone.core.trajectory import turns_from_dicts
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def session(tmp_db, mock_openai_client):
|
||||
"""Small-window session: context_window=10_000, compact_max_tokens=100 so
|
||||
the summary output reserve is tiny and the carry budget is easy to compute
|
||||
(reserve=100, margin=500, spare=9_400, budget=min(2_500, 9_400)=2_500
|
||||
tokens → 10_000 chars at the uncalibrated 4.0 chars/token)."""
|
||||
return make_session(
|
||||
client=mock_openai_client,
|
||||
context_window=10_000,
|
||||
compact_max_tokens=100,
|
||||
max_tokens=1_000,
|
||||
tool_timeout=10,
|
||||
)
|
||||
|
||||
|
||||
def _stub_summary(text: str = "DENSE"):
|
||||
return SimpleNamespace(content=text, finish_reason="stop")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Provenance tags on the synthetic summary turns
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestSummaryTurnProvenance:
|
||||
def test_compact_tags_both_summary_turns(self, session):
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": "do the thing"},
|
||||
{"role": "assistant", "content": "did the thing"},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True) is True
|
||||
|
||||
label, summary = session.messages[0], session.messages[1]
|
||||
assert label.text == COMPACTION_SUMMARY_LABEL
|
||||
assert label.source == COMPACTION_SOURCE
|
||||
assert summary.source == COMPACTION_SOURCE
|
||||
|
||||
def test_boundaries_exclude_tagged_label_only(self, session):
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{
|
||||
"role": "user",
|
||||
"content": COMPACTION_SUMMARY_LABEL,
|
||||
"_source": COMPACTION_SOURCE,
|
||||
},
|
||||
{"role": "assistant", "content": "summary"},
|
||||
{"role": "user", "content": "real follow-up"},
|
||||
]
|
||||
)
|
||||
assert session._find_turn_boundaries() == [2]
|
||||
|
||||
def test_literal_label_from_user_is_a_real_boundary(self, session):
|
||||
"""A user who literally types '[Conversation summary]' is not a
|
||||
compaction artifact — provenance rides the tag, not the spelling."""
|
||||
session.messages = turns_from_dicts([{"role": "user", "content": COMPACTION_SUMMARY_LABEL}])
|
||||
assert session._find_turn_boundaries() == [0]
|
||||
|
||||
def test_title_gen_titles_from_literal_label_user(self, session):
|
||||
"""The tag distinction reaches _generate_title: a synthetic label is
|
||||
skipped (pinned in test_cooperative_compaction), but a REAL user
|
||||
message that happens to equal the label is titled from normally."""
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": COMPACTION_SUMMARY_LABEL},
|
||||
{"role": "assistant", "content": "an answer"},
|
||||
]
|
||||
)
|
||||
with (
|
||||
patch.object(
|
||||
session, "_utility_completion", return_value=_stub_summary("A Title")
|
||||
) as uc,
|
||||
patch.object(session, "ui", new=MagicMock()),
|
||||
):
|
||||
session._generate_title()
|
||||
|
||||
uc.assert_called_once()
|
||||
prompt = uc.call_args[0][0][-1]["content"]
|
||||
assert COMPACTION_SUMMARY_LABEL in prompt # titled FROM the real message
|
||||
|
||||
|
||||
class TestCheckpointReconstructionProvenance:
|
||||
def test_resume_turns_carry_compaction_source(self, storage_backend):
|
||||
"""A reopened session must see the same provenance the live session
|
||||
held: reconstruct_turns_checkpointed tags the synthetic label AND the
|
||||
marker-backed summary turn, while real tail rows stay untagged."""
|
||||
st = storage_backend
|
||||
st.register_workstream("ws1", user_id="u1", title="t", kind="interactive")
|
||||
st.save_message("ws1", "user", "old question")
|
||||
st.save_message("ws1", "assistant", "old answer")
|
||||
watermark = st.get_compaction_watermark("ws1", 0)
|
||||
st.save_message(
|
||||
"ws1",
|
||||
"assistant",
|
||||
"THE SUMMARY",
|
||||
source=COMPACTION_SOURCE,
|
||||
meta=json.dumps({"watermark": watermark}),
|
||||
)
|
||||
st.save_message("ws1", "user", "new question")
|
||||
|
||||
turns = st.load_message_turns("ws1")
|
||||
assert [t.text for t in turns] == [
|
||||
COMPACTION_SUMMARY_LABEL,
|
||||
"THE SUMMARY",
|
||||
"new question",
|
||||
]
|
||||
assert turns[0].source == COMPACTION_SOURCE
|
||||
assert turns[1].source == COMPACTION_SOURCE
|
||||
assert turns[2].source is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Carry budget — the verbatim-crossing allowance
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _isolate_overhead(s, system_tokens: int = 0) -> None:
|
||||
"""Pin the fixed prompt overhead (system + tool defs) for exact budget
|
||||
arithmetic — the real values vary with the composed prompt and registered
|
||||
tools (same isolation pattern as TestRemainingTokenBudget)."""
|
||||
s._system_tokens = system_tokens
|
||||
s._tools = []
|
||||
|
||||
|
||||
class TestCarryBudget:
|
||||
def test_scales_to_quarter_window(self, session):
|
||||
# overhead=0, reserve=100 (compact_max_tokens), margin=500,
|
||||
# spare=9_400; min(10_000 // 4, 9_400) = 2_500 tokens * 4.0 chars/token.
|
||||
_isolate_overhead(session)
|
||||
assert session._carry_budget_chars() == 10_000
|
||||
|
||||
def test_floors_on_tiny_window(self, tmp_db, mock_openai_client):
|
||||
tiny = make_session(client=mock_openai_client, context_window=1_000, tool_timeout=10)
|
||||
_isolate_overhead(tiny)
|
||||
assert tiny._carry_budget_chars() == tiny._MIN_CARRY_BUDGET_CHARS
|
||||
|
||||
@pytest.mark.parametrize("carries", [1, 2])
|
||||
def test_overhead_reserve_and_carries_fit_window_at_shipped_defaults(
|
||||
self, tmp_db, mock_openai_client, carries
|
||||
):
|
||||
"""The invariant that prevents a carry-induced overflow, pinned at the
|
||||
SHIPPED defaults (budget bugs hide behind test-sized configs), for
|
||||
BOTH carry counts, and INCLUDING the fixed prompt overhead: the
|
||||
post-compaction prompt is system + tools + summary + carries, so a
|
||||
budget that ignores the overhead (or sizes carries independently)
|
||||
stacks past the window and the backstop re-compacts the carries
|
||||
away."""
|
||||
s = make_session(client=mock_openai_client, tool_timeout=10)
|
||||
_isolate_overhead(s, system_tokens=4_000) # a chunky composed prompt
|
||||
reserve = s._summary_output_tokens()
|
||||
per_carry_tokens = s._carry_budget_chars(carries) / s._chars_per_token
|
||||
margin = int(s.context_window * s._SUMMARY_SAFETY_MARGIN)
|
||||
assert 4_000 + reserve + carries * per_carry_tokens + margin <= s.context_window
|
||||
|
||||
def test_budget_shrinks_with_prompt_overhead(self, tmp_db, mock_openai_client):
|
||||
"""Monotonicity pin: the overhead term is genuinely in the formula —
|
||||
a bigger system prompt leaves less to carry."""
|
||||
s = make_session(client=mock_openai_client, tool_timeout=10)
|
||||
_isolate_overhead(s, system_tokens=0)
|
||||
roomy = s._carry_budget_chars(2)
|
||||
_isolate_overhead(s, system_tokens=8_000)
|
||||
assert s._carry_budget_chars(2) < roomy
|
||||
|
||||
def test_double_carry_splits_the_spare(self, tmp_db, mock_openai_client):
|
||||
"""At shipped defaults the spare (window − overhead − reserve −
|
||||
margin) binds two carries: each gets spare // 2, strictly less than
|
||||
the solo quarter-window allowance."""
|
||||
s = make_session(client=mock_openai_client, tool_timeout=10)
|
||||
_isolate_overhead(s, system_tokens=2_000)
|
||||
reserve = s._summary_output_tokens()
|
||||
margin = int(s.context_window * s._SUMMARY_SAFETY_MARGIN)
|
||||
spare = s.context_window - reserve - margin - 2_000
|
||||
assert s._carry_budget_chars(2) == int((spare // 2) * s._chars_per_token)
|
||||
assert s._carry_budget_chars(2) < s._carry_budget_chars(1)
|
||||
|
||||
|
||||
class TestContinuationHintCarry:
|
||||
def test_long_ask_crosses_verbatim(self, session):
|
||||
"""A 3_000-char user message is within the 10_000-char carry budget and
|
||||
must cross whole — the old fixed clip kept 400 chars of it."""
|
||||
ask = "spec line\n" * 300 # 3_000 chars
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": ask},
|
||||
{"role": "assistant", "content": "working on it"},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True) is True
|
||||
|
||||
summary_text = session.messages[1].text or ""
|
||||
assert ask.strip() in summary_text # verbatim, not clipped
|
||||
assert "## Continue" in summary_text
|
||||
|
||||
def test_oversize_ask_keeps_head_and_tail_with_marker(self, session):
|
||||
head_sentinel = "HEAD-OF-SPEC"
|
||||
tail_sentinel = "TAIL-OF-SPEC"
|
||||
ask = head_sentinel + ("x" * 20_000) + tail_sentinel # over the 10_000 budget
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": ask},
|
||||
{"role": "assistant", "content": "working on it"},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True) is True
|
||||
|
||||
summary_text = session.messages[1].text or ""
|
||||
assert head_sentinel in summary_text
|
||||
assert tail_sentinel in summary_text
|
||||
# The marker reports the ORIGINAL size, and the summary tells the
|
||||
# model the full text is retrievable — a truncated carry is a cache
|
||||
# miss with a pointer, not a silent loss.
|
||||
assert f"…[truncated — {len(ask):,} chars total]…" in summary_text
|
||||
assert "the recall tool can retrieve it" in summary_text
|
||||
assert ask not in summary_text # genuinely truncated
|
||||
|
||||
def test_untruncated_carry_gets_no_recall_pointer(self, session):
|
||||
"""The retrievability note appears ONLY when something was cut."""
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": "short ask"},
|
||||
{"role": "assistant", "content": "working on it"},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True) is True
|
||||
assert "recall tool" not in (session.messages[1].text or "")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Wind-down spill — the model's plan statement crosses verbatim
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestWindDownSpill:
|
||||
SPILL = (
|
||||
"Goal: finish the migration.\n"
|
||||
"Remaining: backfill rows 300-900, rerun the verifier.\n"
|
||||
"Next step: resume at scripts/backfill.py --from 300."
|
||||
)
|
||||
|
||||
def _compacted_summary(self, session, *, carry_spill: bool) -> str:
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": "please migrate the database"},
|
||||
{"role": "assistant", "content": self.SPILL},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True, carry_spill=carry_spill) is True
|
||||
return session.messages[1].text or ""
|
||||
|
||||
def test_spill_copied_verbatim_under_heading(self, session):
|
||||
summary_text = self._compacted_summary(session, carry_spill=True)
|
||||
assert "## Wind-down (verbatim)" in summary_text
|
||||
assert self.SPILL in summary_text # copied, not paraphrased
|
||||
# Ordering: recorded plan first, then how to resume.
|
||||
assert summary_text.index("## Wind-down (verbatim)") < summary_text.index("## Continue")
|
||||
|
||||
def test_no_spill_without_flag(self, session):
|
||||
summary_text = self._compacted_summary(session, carry_spill=False)
|
||||
assert "## Wind-down (verbatim)" not in summary_text
|
||||
|
||||
def test_no_spill_when_last_summarized_turn_is_not_assistant(self, session):
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "assistant", "content": "answer"},
|
||||
{"role": "user", "content": "next task"},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True, carry_spill=True) is True
|
||||
assert "## Wind-down (verbatim)" not in (session.messages[1].text or "")
|
||||
|
||||
def test_empty_spill_adds_no_heading(self, session):
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": "task"},
|
||||
{"role": "assistant", "content": " "},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True, carry_spill=True) is True
|
||||
assert "## Wind-down (verbatim)" not in (session.messages[1].text or "")
|
||||
|
||||
def test_oversize_spill_truncated_by_carry_budget(self, session):
|
||||
big_spill = "PLAN-HEAD " + ("y" * 20_000) + " PLAN-TAIL"
|
||||
session.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": "task"},
|
||||
{"role": "assistant", "content": big_spill},
|
||||
]
|
||||
)
|
||||
session._msg_tokens = [1, 1]
|
||||
with patch.object(session, "_utility_completion", return_value=_stub_summary()):
|
||||
assert session._compact_messages(auto=True, carry_spill=True) is True
|
||||
summary_text = session.messages[1].text or ""
|
||||
assert "PLAN-HEAD" in summary_text and "PLAN-TAIL" in summary_text
|
||||
assert "…[truncated —" in summary_text
|
||||
assert "the recall tool can retrieve it" in summary_text
|
||||
|
||||
def test_double_carry_shares_the_budget(self, tmp_db, mock_openai_client):
|
||||
"""Spill + hint on ONE compaction — the end-of-turn shape — must fit
|
||||
the window together. At the shipped window defaults each carry gets
|
||||
spare // 2, so two oversize carries land truncated to the shared
|
||||
budget instead of stacking two solo quarter-window allowances on top
|
||||
of the half-window summary reserve."""
|
||||
s = make_session(client=mock_openai_client, tool_timeout=10)
|
||||
per_carry = s._carry_budget_chars(2)
|
||||
ask = "ASK-HEAD " + "a" * (per_carry * 2) + " ASK-TAIL"
|
||||
spill = "PLAN-HEAD " + "b" * (per_carry * 2) + " PLAN-TAIL"
|
||||
s.messages = turns_from_dicts(
|
||||
[
|
||||
{"role": "user", "content": ask},
|
||||
{"role": "assistant", "content": spill},
|
||||
]
|
||||
)
|
||||
s._msg_tokens = [1, 1]
|
||||
with patch.object(s, "_utility_completion", return_value=_stub_summary()):
|
||||
assert s._compact_messages(auto=True, carry_spill=True) is True
|
||||
|
||||
text = s.messages[1].text or ""
|
||||
assert "## Wind-down (verbatim)" in text and "## Continue" in text
|
||||
for sentinel in ("ASK-HEAD", "ASK-TAIL", "PLAN-HEAD", "PLAN-TAIL"):
|
||||
assert sentinel in text
|
||||
assert text.count("…[truncated —") == 2 # both carries hit the shared cap
|
||||
framing = 700 # headings, hint wording, stub summary, recall pointer
|
||||
assert len(text) <= 2 * per_carry + framing
|
||||
|
||||
def test_do_auto_compact_forwards_carry_spill(self, session):
|
||||
"""The end-of-turn site passes carry_spill=stopped_to_compact through
|
||||
_do_auto_compact — pin the forwarding."""
|
||||
with patch.object(session, "_compact_messages", return_value=True) as cm:
|
||||
session._do_auto_compact(my_generation=3, carry_spill=True)
|
||||
assert cm.call_args.kwargs["carry_spill"] is True
|
||||
assert cm.call_args.kwargs["my_generation"] == 3
|
||||
+184
-1
@@ -4,7 +4,7 @@ import asyncio
|
||||
import json
|
||||
import queue
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
from unittest.mock import ANY, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -531,6 +531,58 @@ class TestCollectorDelta:
|
||||
assert event["type"] == "ws_closed"
|
||||
assert "ws1" not in c._nodes["node-a"].workstreams
|
||||
|
||||
def test_reconcile_additions_event_carries_tenancy_fields(self):
|
||||
"""The poll-diff ws_created must carry user_id + project_id — the
|
||||
console's per-connection tenancy filter gates on them, and a
|
||||
missing field fails open (private leak) or over-hides (creator
|
||||
shortcut can't fire)."""
|
||||
c = _make_collector()
|
||||
node = NodeSnapshot(node_id="node-a", server_url="http://a:8080")
|
||||
c._nodes["node-a"] = node
|
||||
|
||||
pending = c._reconcile_node(
|
||||
"node-a",
|
||||
node,
|
||||
[
|
||||
{
|
||||
"id": "ws1",
|
||||
"name": "n",
|
||||
"state": "idle",
|
||||
"kind": "interactive",
|
||||
"user_id": "alice",
|
||||
"project_id": "p1",
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
created = [e for e in pending if e["type"] == "ws_created"]
|
||||
assert len(created) == 1
|
||||
assert created[0]["user_id"] == "alice"
|
||||
assert created[0]["project_id"] == "p1"
|
||||
|
||||
def test_emit_console_ws_created_carries_project(self):
|
||||
"""Console pseudo-node coordinator rows + their ws_created must
|
||||
carry project_id or private-project coordinators leak on the
|
||||
SSE surface (the REST lane filters via _coordinator_rows)."""
|
||||
c = _make_collector()
|
||||
q: queue.Queue[dict] = queue.Queue()
|
||||
c.register_listener(q)
|
||||
|
||||
c.emit_console_ws_created(
|
||||
"cws1",
|
||||
name="C",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="p1",
|
||||
)
|
||||
|
||||
event = q.get_nowait()
|
||||
assert event["type"] == "ws_created"
|
||||
assert event["user_id"] == "alice"
|
||||
assert event["project_id"] == "p1"
|
||||
row = c._nodes[c.CONSOLE_PSEUDO_NODE_ID].workstreams["cws1"]
|
||||
assert row["project_id"] == "p1"
|
||||
|
||||
def test_apply_delta_ws_rename(self):
|
||||
c = _make_collector()
|
||||
c._nodes["node-a"] = NodeSnapshot(
|
||||
@@ -1046,6 +1098,8 @@ class TestConsoleHTTPEndpoints:
|
||||
page=1,
|
||||
per_page=25,
|
||||
extra_rows=[],
|
||||
# Per-request private-project tenancy closure — identity varies.
|
||||
row_filter=ANY,
|
||||
)
|
||||
|
||||
def test_get_workstreams_per_page_capped(self, client, mock_collector):
|
||||
@@ -1546,6 +1600,82 @@ class TestConsoleProxy:
|
||||
# browser's interactive UI 403-loops on every retry.
|
||||
assert sse_mock.await_args.kwargs.get("use_service_auth") is True
|
||||
|
||||
def test_proxy_events_global_403_without_cluster_inspect(self, mock_collector):
|
||||
"""A plain authenticated user (no service scope, no
|
||||
admin.cluster.inspect) cannot reach the node's cross-tenant
|
||||
firehose through the proxy: elevating to the console's service
|
||||
identity would bypass per-user filtering, so the path is
|
||||
operator-gated. _proxy_sse must NOT be reached."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from starlette.responses import Response
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from turnstone.console.server import _load_static, create_app
|
||||
from turnstone.core.auth import JWT_AUD_CONSOLE, create_jwt
|
||||
|
||||
_load_static()
|
||||
app = create_app(collector=mock_collector, jwt_secret=_TEST_JWT_SECRET)
|
||||
user_jwt = create_jwt(
|
||||
user_id="plain-user",
|
||||
scopes=frozenset({"read"}),
|
||||
source="test",
|
||||
secret=_TEST_JWT_SECRET,
|
||||
audience=JWT_AUD_CONSOLE,
|
||||
permissions=frozenset(),
|
||||
)
|
||||
user_client = TestClient(
|
||||
app,
|
||||
raise_server_exceptions=False,
|
||||
headers={"Authorization": f"Bearer {user_jwt}"},
|
||||
)
|
||||
with patch(
|
||||
"turnstone.console.server._proxy_sse",
|
||||
new_callable=AsyncMock,
|
||||
return_value=Response("ok", status_code=200),
|
||||
) as sse_mock:
|
||||
resp = user_client.get("/node/node-a/v1/api/events/global")
|
||||
assert resp.status_code == 403
|
||||
assert sse_mock.await_count == 0
|
||||
user_client.close()
|
||||
|
||||
def test_proxy_events_global_allows_cluster_inspect(self, mock_collector):
|
||||
"""An operator holding admin.cluster.inspect passes the gate and
|
||||
reaches the SSE proxy with the service token."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from starlette.responses import Response
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from turnstone.console.server import _load_static, create_app
|
||||
from turnstone.core.auth import JWT_AUD_CONSOLE, create_jwt
|
||||
|
||||
_load_static()
|
||||
app = create_app(collector=mock_collector, jwt_secret=_TEST_JWT_SECRET)
|
||||
op_jwt = create_jwt(
|
||||
user_id="operator",
|
||||
scopes=frozenset({"read"}),
|
||||
source="test",
|
||||
secret=_TEST_JWT_SECRET,
|
||||
audience=JWT_AUD_CONSOLE,
|
||||
permissions=frozenset({"admin.cluster.inspect"}),
|
||||
)
|
||||
op_client = TestClient(
|
||||
app,
|
||||
raise_server_exceptions=False,
|
||||
headers={"Authorization": f"Bearer {op_jwt}"},
|
||||
)
|
||||
with patch(
|
||||
"turnstone.console.server._proxy_sse",
|
||||
new_callable=AsyncMock,
|
||||
return_value=Response("ok", status_code=200),
|
||||
) as sse_mock:
|
||||
resp = op_client.get("/node/node-a/v1/api/events/global")
|
||||
assert resp.status_code == 200
|
||||
assert sse_mock.await_count == 1
|
||||
assert sse_mock.await_args.kwargs.get("use_service_auth") is True
|
||||
op_client.close()
|
||||
|
||||
def test_proxy_api_per_ws_events_uses_user_auth_not_service(self, client, mock_collector):
|
||||
"""Per-ws events route uses the user's re-minted JWT, not the
|
||||
service token — the upstream per-ws SSE handler scopes by
|
||||
@@ -2560,3 +2690,56 @@ class TestCollectorMCPAggregation:
|
||||
assert overview["mcp_servers"] == 3
|
||||
assert overview["mcp_resources"] == 10
|
||||
assert overview["mcp_prompts"] == 7
|
||||
|
||||
|
||||
class TestProxyGetHeaderPassThrough:
|
||||
"""The generic /node/{id} GET proxy must carry the node's hardening
|
||||
headers through — dropping Content-Security-Policy would serve previewed
|
||||
attacker HTML from the CONSOLE origin with no CSP sandbox (review
|
||||
finding, preview-pane branch)."""
|
||||
|
||||
def test_security_headers_forwarded(self, monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import httpx
|
||||
|
||||
from turnstone.console import server as csrv
|
||||
|
||||
upstream = httpx.Response(
|
||||
200,
|
||||
content=b"<html>page</html>",
|
||||
headers={
|
||||
"content-type": "text/html; charset=utf-8",
|
||||
"content-security-policy": "sandbox",
|
||||
"x-content-type-options": "nosniff",
|
||||
"content-disposition": 'inline; filename="p"',
|
||||
"cache-control": "private, no-store",
|
||||
"server": "upstream-internal", # hop metadata: must NOT pass
|
||||
},
|
||||
request=httpx.Request("GET", "http://n:1/x"),
|
||||
)
|
||||
|
||||
async def _mock_get(*a, **kw):
|
||||
return upstream
|
||||
|
||||
proxy_client = MagicMock(spec=httpx.AsyncClient)
|
||||
proxy_client.get = MagicMock(side_effect=_mock_get)
|
||||
request = SimpleNamespace(
|
||||
app=SimpleNamespace(state=SimpleNamespace(proxy_client=proxy_client)),
|
||||
url=SimpleNamespace(query=""),
|
||||
)
|
||||
monkeypatch.setattr(csrv, "_proxy_auth_headers", lambda r: {})
|
||||
|
||||
resp = asyncio.run(csrv._proxy_get(request, "http://n:1", "v1/api/x"))
|
||||
|
||||
assert resp.status_code == 200
|
||||
assert resp.headers["content-security-policy"] == "sandbox"
|
||||
assert resp.headers["x-content-type-options"] == "nosniff"
|
||||
assert resp.headers["content-disposition"] == 'inline; filename="p"'
|
||||
assert resp.headers["cache-control"] == "private, no-store"
|
||||
assert resp.headers["content-type"].startswith("text/html")
|
||||
assert (
|
||||
"server" not in {k.lower() for k in resp.headers}
|
||||
or resp.headers.get("server") != "upstream-internal"
|
||||
)
|
||||
|
||||
@@ -336,10 +336,88 @@ def test_channel_default_alias_blanked_when_disabled(
|
||||
|
||||
|
||||
def test_models_payload_strips_secret_fields(storage: SQLiteBackend) -> None:
|
||||
"""Regression guard: only alias/model/provider land in the response,
|
||||
never api_key / base_url / context_window / capabilities."""
|
||||
"""Regression guard: only alias/model/provider (+ the derived
|
||||
effort_ladder) land in the response, never api_key / base_url /
|
||||
context_window / raw capabilities."""
|
||||
_seed_model(storage, definition_id="m1", alias="primary")
|
||||
body = _get_models(_make_client(storage))
|
||||
assert body["models"] == [
|
||||
{"alias": "primary", "model": "model-x", "provider": "openai-compatible"}
|
||||
]
|
||||
assert len(body["models"]) == 1
|
||||
entry = body["models"][0]
|
||||
assert set(entry) == {"alias", "model", "provider", "effort_ladder"}
|
||||
assert entry["alias"] == "primary"
|
||||
assert entry["model"] == "model-x"
|
||||
assert entry["provider"] == "openai-compatible"
|
||||
|
||||
|
||||
def test_effort_ladder_parses_string_capabilities(storage: SQLiteBackend) -> None:
|
||||
"""The capabilities column is a JSON STRING — the ladder must survive
|
||||
the parse (regression: .items() on the raw string threw and the
|
||||
guard silently dropped the field from every row)."""
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="qwen",
|
||||
model="qwen3.6-27b",
|
||||
provider="anthropic-compatible",
|
||||
base_url="http://localhost:8000",
|
||||
api_key="dummy",
|
||||
context_window=262144,
|
||||
capabilities='{"thinking_mode": "manual", "thinking_param": "enable_thinking"}',
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
ladder = {r["value"]: r["effective"] for r in body["models"][0]["effort_ladder"]}
|
||||
assert ladder["none"] == "off"
|
||||
assert ladder["medium"] == "on+medium"
|
||||
assert ladder["max"] == "on+max"
|
||||
|
||||
|
||||
def test_effort_ladder_key_survives_malformed_capabilities(
|
||||
storage: SQLiteBackend,
|
||||
) -> None:
|
||||
"""A capabilities column that fails to parse must not drop the key —
|
||||
every row carries ``effort_ladder`` (empty on failure) so clients can
|
||||
index it unconditionally instead of null-checking per row."""
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="broken",
|
||||
model="model-x",
|
||||
provider="openai-compatible",
|
||||
base_url="http://localhost:8000/v1",
|
||||
api_key="dummy",
|
||||
context_window=131072,
|
||||
capabilities="{not valid json",
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
entry = body["models"][0]
|
||||
assert set(entry) == {"alias", "model", "provider", "effort_ladder"}
|
||||
assert entry["effort_ladder"] == []
|
||||
|
||||
|
||||
def test_effort_ladder_honors_responses_api_surface(storage: SQLiteBackend) -> None:
|
||||
"""server_compat.api_surface (namespaced inside the capabilities JSON)
|
||||
switches the projection to the flat-param path — no template toggle."""
|
||||
caps = (
|
||||
'{"thinking_mode": "manual", "thinking_param": "enable_thinking",'
|
||||
' "reasoning_effort_values": ["low", "medium", "high"],'
|
||||
' "server_compat": {"api_surface": "responses"}}'
|
||||
)
|
||||
storage.create_model_definition(
|
||||
definition_id="m1",
|
||||
alias="mistral",
|
||||
model="mistral-medium",
|
||||
provider="openai-compatible",
|
||||
base_url="http://localhost:8000/v1",
|
||||
api_key="dummy",
|
||||
context_window=131072,
|
||||
capabilities=caps,
|
||||
enabled=True,
|
||||
created_by="admin",
|
||||
)
|
||||
body = _get_models(_make_client(storage))
|
||||
ladder = {r["value"]: r["effective"] for r in body["models"][0]["effort_ladder"]}
|
||||
# Responses surface: flat param only — no "on+"/"off" toggle tokens.
|
||||
assert ladder["medium"] == "medium"
|
||||
assert ladder["none"] == "default"
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
"""``POST /v1/api/admin/models/effort-ladder`` — live modal projection.
|
||||
|
||||
Pure computation over (provider, model, unsaved capability overrides,
|
||||
api_surface); every malformed input must land as a 400, never a 500 —
|
||||
the body is operator-typed form state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from starlette.applications import Starlette
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.routing import Route
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from tests._coord_test_helpers import _AuthMiddleware
|
||||
from turnstone.console.server import admin_effort_ladder
|
||||
|
||||
|
||||
def _make_client() -> TestClient:
|
||||
app = Starlette(
|
||||
routes=[Route("/v1/api/admin/models/effort-ladder", admin_effort_ladder, methods=["POST"])],
|
||||
middleware=[Middleware(_AuthMiddleware)],
|
||||
)
|
||||
client = TestClient(app)
|
||||
client.headers.update({"X-Test-User": "admin", "X-Test-Perms": "admin.models"})
|
||||
return client
|
||||
|
||||
|
||||
def _post(client: TestClient, body: Any) -> Any:
|
||||
return client.post("/v1/api/admin/models/effort-ladder", json=body)
|
||||
|
||||
|
||||
def test_valid_request_returns_ladder() -> None:
|
||||
resp = _post(
|
||||
_make_client(),
|
||||
{
|
||||
"provider": "anthropic-compatible",
|
||||
"model": "qwen3.6-27b",
|
||||
"capabilities": {"thinking_mode": "manual", "thinking_param": "enable_thinking"},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
ladder = {r["value"]: r["effective"] for r in resp.json()["ladder"]}
|
||||
assert ladder["none"] == "off"
|
||||
assert ladder["high"] == "on+high"
|
||||
|
||||
|
||||
def test_api_surface_switches_projection() -> None:
|
||||
body = {
|
||||
"provider": "openai-compatible",
|
||||
"model": "m",
|
||||
"capabilities": {
|
||||
"thinking_mode": "manual",
|
||||
"reasoning_effort_values": ["low", "medium", "high"],
|
||||
},
|
||||
}
|
||||
client = _make_client()
|
||||
chat = {r["value"]: r["effective"] for r in _post(client, body).json()["ladder"]}
|
||||
body["api_surface"] = "responses"
|
||||
responses = {r["value"]: r["effective"] for r in _post(client, body).json()["ladder"]}
|
||||
assert chat["medium"] == "on+medium" # toggle + flat on the chat surface
|
||||
assert responses["medium"] == "medium" # flat only on the responses surface
|
||||
|
||||
|
||||
def test_non_dict_json_body_is_400_not_500() -> None:
|
||||
client = _make_client()
|
||||
for body in (None, [], "x", 7):
|
||||
resp = _post(client, body)
|
||||
assert resp.status_code == 400, (body, resp.status_code, resp.text)
|
||||
|
||||
|
||||
def test_unknown_provider_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "nope", "model": "m"})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_missing_model_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "openai", "model": ""})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_non_dict_capabilities_is_400() -> None:
|
||||
resp = _post(_make_client(), {"provider": "openai", "model": "m", "capabilities": [1]})
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_garbage_capability_value_types_are_400() -> None:
|
||||
"""Wrong-typed override values raise inside the resolver → clean 400."""
|
||||
resp = _post(
|
||||
_make_client(),
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-fable-5",
|
||||
"capabilities": {"supports_effort": True, "effort_levels": 5},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_requires_admin_models_permission() -> None:
|
||||
client = _make_client()
|
||||
client.headers.update({"X-Test-Perms": "read"})
|
||||
resp = _post(client, {"provider": "openai", "model": "m"})
|
||||
assert resp.status_code in (401, 403)
|
||||
@@ -363,6 +363,22 @@ class TestClusterCreate:
|
||||
assert mock_post.call_args.kwargs["json"]["project_id"] == "proj-42"
|
||||
client.close()
|
||||
|
||||
def test_cluster_create_forwards_persona(self) -> None:
|
||||
# The launcher's persona picker sends persona; the proxy selectively
|
||||
# REBUILDS the forwarded body (it doesn't pass it through), so persona
|
||||
# must be explicitly carried or the receiving node stamps its kind
|
||||
# default instead of the operator's choice.
|
||||
mock_post = _make_proxy_post(json_data={"ws_id": "p1ws"})
|
||||
client = TestClient(self._app_with_node(mock_post), raise_server_exceptions=False)
|
||||
resp = client.post(
|
||||
"/v1/api/cluster/workstreams/new",
|
||||
json={"node_id": "node-a", "name": "j", "persona": "scribe"},
|
||||
headers=_TEST_AUTH_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert mock_post.call_args.kwargs["json"]["persona"] == "scribe"
|
||||
client.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests — route_proxy
|
||||
|
||||
@@ -150,3 +150,21 @@ def test_warning_and_verdict_normalize_risk() -> None:
|
||||
assert "normalizeRiskLevel(a.risk_level)" in body, "warning must normalize"
|
||||
assert '"conv-warning conv-warning--" + risk' in body
|
||||
assert 'badge.classList.add("conv-verdict--" + risk)' in body
|
||||
|
||||
|
||||
def test_unbounded_render_inputs_are_capped() -> None:
|
||||
"""Perf-audit P0: the two builders that used to render unbounded input.
|
||||
The diff preview caps rendered lines and appends incrementally — the old
|
||||
single ``diff.append(...nodes)`` spread threw RangeError past engine
|
||||
spread-arity limits, killing the tool card (and the approval gate) for
|
||||
the batch. The raw result body clamps at RAW_CAP so one multi-MB tool
|
||||
output can't become a multi-MB pre-wrap text node rebuilt on every
|
||||
re-render."""
|
||||
body = _body()
|
||||
assert "MAX_PREVIEW_LINES" in body
|
||||
assert "diff.append(...nodes)" not in body, (
|
||||
"preview nodes must append incrementally, not via one spread call"
|
||||
)
|
||||
assert "more preview lines not shown" in body
|
||||
assert "RAW_CAP" in body
|
||||
assert "truncated for display" in body
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -114,11 +114,16 @@ def test_coord_on_aux_usage_leaves_live_counters_untouched() -> None:
|
||||
assert ui._ws_context_ratio == 0.0
|
||||
|
||||
|
||||
def test_coord_on_content_token_accumulates() -> None:
|
||||
def test_coord_on_content_token_accumulates(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""Pre-lift coord ``on_content_token`` only enqueued; lift turns it
|
||||
into the same per-ws accumulator WebUI uses so the collector
|
||||
broadcast can piggyback the joined turn content on the IDLE
|
||||
state-change event."""
|
||||
state-change event.
|
||||
|
||||
Batch window forced to 0 (per-token flush) — pins the accumulator
|
||||
wiring, not the batching cadence (test_sse_token_batching.py)."""
|
||||
|
||||
monkeypatch.setattr("turnstone.core.session_ui_base._TOKEN_BATCH_WINDOW_SECS", 0.0)
|
||||
ui = ConsoleCoordinatorUI(ws_id="coord-ws", user_id="u1")
|
||||
ui.on_content_token("Hello ")
|
||||
ui.on_content_token("world")
|
||||
|
||||
@@ -16,10 +16,10 @@ to ``SessionUIBase`` automatically enables:
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from tests.conftest import resolve_when_pending
|
||||
from turnstone.console.coordinator_ui import ConsoleCoordinatorUI
|
||||
|
||||
|
||||
@@ -153,7 +153,7 @@ def test_coord_heuristic_verdict_persists_to_storage() -> None:
|
||||
items[0]["_heuristic_verdict"] = hv
|
||||
|
||||
storage = MagicMock()
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(storage):
|
||||
@@ -246,9 +246,8 @@ def test_coord_pending_approval_sets_activity_tag() -> None:
|
||||
def _capture_activity() -> None:
|
||||
captured["activity"] = ui._ws_current_activity
|
||||
captured["state"] = ui._ws_activity_state
|
||||
ui.resolve_approval(False)
|
||||
|
||||
timer = threading.Timer(0.05, _capture_activity)
|
||||
timer = resolve_when_pending(ui, False, before=_capture_activity)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -292,7 +291,7 @@ def test_coord_judge_pending_flag_dynamic_when_heuristic_present() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -338,7 +337,7 @@ def test_coord_judge_pending_false_when_no_heuristic_verdict() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(False))
|
||||
timer = resolve_when_pending(ui, False)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -410,7 +409,7 @@ def test_coord_budget_override_prompts_even_under_blanket_auto_approve() -> None
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(True))
|
||||
timer = resolve_when_pending(ui, True)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()):
|
||||
@@ -453,7 +452,7 @@ def test_coord_budget_override_survives_wildcard_allow_policy() -> None:
|
||||
captured_events: list[dict[str, Any]] = []
|
||||
ui._enqueue = captured_events.append # type: ignore[method-assign]
|
||||
|
||||
timer = threading.Timer(0.05, lambda: ui.resolve_approval(True))
|
||||
timer = resolve_when_pending(ui, True)
|
||||
timer.start()
|
||||
try:
|
||||
with _patch_storage(MagicMock()), _patch_policies({"__budget_override__": "allow"}):
|
||||
@@ -526,12 +525,16 @@ class TestBroadcastApprovalResolved:
|
||||
collector = MagicMock()
|
||||
ConsoleCoordinatorUI._collector = collector
|
||||
try:
|
||||
ui._broadcast_approval_resolved(True, "lgtm", always=True)
|
||||
ui._broadcast_approval_resolved(
|
||||
True, "lgtm", always=True, cycle_id="cyc-1", call_ids=("c-1", "c-2")
|
||||
)
|
||||
collector.emit_console_ws_approval_resolved.assert_called_once_with(
|
||||
"coord-a",
|
||||
approved=True,
|
||||
feedback="lgtm",
|
||||
always=True,
|
||||
cycle_id="cyc-1",
|
||||
call_ids=["c-1", "c-2"],
|
||||
)
|
||||
finally:
|
||||
ConsoleCoordinatorUI._collector = None
|
||||
@@ -547,6 +550,8 @@ class TestBroadcastApprovalResolved:
|
||||
approved=False,
|
||||
feedback="",
|
||||
always=False,
|
||||
cycle_id="",
|
||||
call_ids=[],
|
||||
)
|
||||
finally:
|
||||
ConsoleCoordinatorUI._collector = None
|
||||
|
||||
@@ -73,7 +73,7 @@ def _make_ws(**overrides: Any) -> Workstream:
|
||||
|
||||
def test_emit_created_calls_collector_with_coord_fields() -> None:
|
||||
adapter, collector = _make_adapter()
|
||||
ws = _make_ws()
|
||||
ws = _make_ws(project_id="p1", persona="executive")
|
||||
adapter.emit_created(ws)
|
||||
collector.emit_console_ws_created.assert_called_once_with(
|
||||
"coord-1",
|
||||
@@ -82,6 +82,10 @@ def test_emit_created_calls_collector_with_coord_fields() -> None:
|
||||
kind=WorkstreamKind.COORDINATOR.value,
|
||||
state=WorkstreamState.IDLE.value,
|
||||
parent_ws_id=None,
|
||||
# Tenancy-load-bearing: the console SSE filter gates on this.
|
||||
project_id="p1",
|
||||
# Display carrier: the pseudo-node row + ws_created event wear it.
|
||||
persona="executive",
|
||||
)
|
||||
|
||||
|
||||
@@ -191,6 +195,21 @@ def test_emit_tolerates_collector_exception() -> None:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_cleanup_ui_sweeps_all_approval_cycles_on_registry_uis() -> None:
|
||||
"""The real ConsoleCoordinatorUI carries the approval-cycle
|
||||
registry: cleanup denies + wakes EVERY parked gate via
|
||||
``resolve_all_approvals`` (parallel task agents can hold several),
|
||||
not the pre-cycle single-slot kick."""
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
ws.ui.resolve_all_approvals = MagicMock(return_value=2) # type: ignore[attr-defined]
|
||||
adapter.cleanup_ui(ws)
|
||||
ws.ui.resolve_all_approvals.assert_called_once_with( # type: ignore[attr-defined]
|
||||
False, "Workstream closed"
|
||||
)
|
||||
assert ws.ui._fg_event.is_set() # type: ignore[attr-defined]
|
||||
|
||||
|
||||
def test_cleanup_ui_unblocks_events_and_broadcasts_to_listeners() -> None:
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
@@ -226,6 +245,24 @@ def test_cleanup_ui_tolerates_missing_session_and_ui() -> None:
|
||||
ws.session = None
|
||||
ws.ui = None
|
||||
adapter.cleanup_ui(ws) # no crash
|
||||
assert ws._closed is True # still marked dead
|
||||
|
||||
|
||||
def test_cleanup_ui_marks_workstream_closed() -> None:
|
||||
"""Every teardown path — close, close_idle, EVICTION, delete,
|
||||
discard — funnels through cleanup_ui, which marks the object dead
|
||||
under ``ws._lock`` BEFORE the teardown body runs. The wake paths
|
||||
that hold OBJECT references (the watch ``wake_fn``,
|
||||
``session_worker``'s exit backstop) gate on ``_closed``, and
|
||||
``session_worker.send`` re-checks it under the same lock — without
|
||||
this write here, a wake racing an eviction or delete (which never
|
||||
set the flag) would spawn a full unattended turn on the torn-down
|
||||
session."""
|
||||
adapter, _ = _make_adapter()
|
||||
ws = _make_ws()
|
||||
assert ws._closed is False
|
||||
adapter.cleanup_ui(ws)
|
||||
assert ws._closed is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -287,6 +324,7 @@ class _SendSession:
|
||||
) -> None:
|
||||
self.send_calls: list[str] = []
|
||||
self.queue_calls: list[str] = []
|
||||
self.interjector_ids: list[str] = []
|
||||
self._queue_full = queue_full
|
||||
# When set, ``send`` blocks on this event — lets the test pin a
|
||||
# worker inside session.send while a second thread races through
|
||||
@@ -313,9 +351,11 @@ class _SendSession:
|
||||
message: str,
|
||||
attachment_ids: Any = None,
|
||||
queue_msg_id: str | None = None,
|
||||
interjector_user_id: str = "",
|
||||
) -> None:
|
||||
if self._queue_full:
|
||||
raise queue.Full
|
||||
self.interjector_ids.append(interjector_user_id)
|
||||
self.queue_calls.append(message)
|
||||
|
||||
def cancel(self) -> None:
|
||||
|
||||
@@ -42,6 +42,7 @@ from turnstone.console.server import (
|
||||
_coord_create_post_install,
|
||||
_coord_create_validate_request,
|
||||
_coord_saved_loaded_lookup,
|
||||
_coordinator_tenant_check,
|
||||
_require_admin_coordinator,
|
||||
_require_coord_mgr,
|
||||
cluster_ws_detail,
|
||||
@@ -83,15 +84,24 @@ def _coord_attach_owner(request, ws_id, mgr):
|
||||
|
||||
Kind-strict — coord attachments can only be accessed for
|
||||
workstreams currently held by ``coord_mgr``; no storage fallback
|
||||
so cross-kind ws_ids 404 instead of leaking through storage.
|
||||
so cross-kind ws_ids 404 instead of leaking through storage. Also
|
||||
project-tenancy-strict: mirrors ``_coord_attachment_owner`` so a
|
||||
private-project coordinator's attachments 404-mask non-members.
|
||||
"""
|
||||
from starlette.responses import JSONResponse
|
||||
|
||||
from turnstone.core.auth import WorkstreamProjectVisibility
|
||||
from turnstone.core.web_helpers import auth_user_id
|
||||
|
||||
ws = mgr.get(ws_id)
|
||||
if ws is None:
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
storage = getattr(request.app.state, "auth_storage", None)
|
||||
if storage is None:
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
visibility = WorkstreamProjectVisibility.for_request(request, storage=storage)
|
||||
if not visibility.ws_visible(getattr(ws, "project_id", "") or "", ws_owner=ws.user_id or ""):
|
||||
return "", JSONResponse({"error": "coordinator not found"}, status_code=404)
|
||||
return ws.user_id or auth_user_id(request), None
|
||||
|
||||
|
||||
@@ -101,7 +111,7 @@ def _coord_attach_owner(request, ws_id, mgr):
|
||||
_coord_endpoint_config = SessionEndpointConfig(
|
||||
permission_gate=_require_admin_coordinator,
|
||||
manager_lookup=_require_coord_mgr,
|
||||
tenant_check=None,
|
||||
tenant_check=_coordinator_tenant_check,
|
||||
not_found_label="coordinator not found",
|
||||
audit_action_prefix="coordinator",
|
||||
supports_attachments=True,
|
||||
@@ -520,11 +530,17 @@ def test_active_list_row_shape_includes_unified_fields(storage):
|
||||
"kind",
|
||||
"parent_ws_id",
|
||||
"user_id",
|
||||
"project_id",
|
||||
"persona",
|
||||
}
|
||||
assert row["name"] == "lifted-coord"
|
||||
assert row["kind"] == "coordinator"
|
||||
assert row["parent_ws_id"] is None
|
||||
assert row["user_id"] == "u1"
|
||||
# mgr.create without a persona kwarg stamps nothing at this layer
|
||||
# (default resolution lives in the HTTP create handler), so the
|
||||
# row carries the null slug — not a fabricated default.
|
||||
assert row["persona"] is None
|
||||
|
||||
|
||||
def test_create_returns_ws_id_and_records_audit(storage):
|
||||
@@ -1097,18 +1113,7 @@ def test_approve_resolves_ui_event(storage):
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
assert isinstance(ws.ui, ConsoleCoordinatorUI)
|
||||
ws.ui._pending_approval = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": "c-1",
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
],
|
||||
}
|
||||
ws.ui._approval_event.clear()
|
||||
cycle = _seed_pending(ws, "c-1")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1116,34 +1121,46 @@ def test_approve_resolves_ui_event(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert ws.ui._approval_result == (True, None)
|
||||
assert resp.json()["cycle_id"] == cycle.cycle_id
|
||||
assert cycle.event.is_set()
|
||||
assert cycle.result == (True, None)
|
||||
assert "spawn_workstream" in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
def _seed_pending(ws, *call_ids: str) -> None:
|
||||
ws.ui._pending_approval = {
|
||||
def _seed_pending(ws, *call_ids: str, func_name: str = "spawn_workstream"):
|
||||
"""Register a live ApprovalCycle on the coord UI the way its
|
||||
``approve_tools`` gate does, returning the cycle for direct
|
||||
event/result assertions (the pre-cycle singleton
|
||||
``_approval_event`` / ``_approval_result`` slots are gone)."""
|
||||
from turnstone.core.session_ui_base import ApprovalCycle
|
||||
|
||||
items = [
|
||||
{
|
||||
"call_id": cid,
|
||||
"func_name": func_name,
|
||||
"approval_label": func_name,
|
||||
"needs_approval": True,
|
||||
}
|
||||
for cid in call_ids
|
||||
]
|
||||
card = {
|
||||
"type": "approve_request",
|
||||
"items": [
|
||||
{
|
||||
"call_id": cid,
|
||||
"func_name": "spawn_workstream",
|
||||
"approval_label": "spawn_workstream",
|
||||
"needs_approval": True,
|
||||
}
|
||||
for cid in call_ids
|
||||
],
|
||||
"cycle_id": f"cyc-{'-'.join(call_ids)}",
|
||||
"items": ws.ui._serialize_approval_items(items),
|
||||
"judge_pending": False,
|
||||
}
|
||||
ws.ui._approval_event.clear()
|
||||
cycle = ApprovalCycle(items, card, None)
|
||||
ws.ui._register_approval_cycle(cycle)
|
||||
return cycle
|
||||
|
||||
|
||||
def test_approve_409_on_stale_call_id(storage):
|
||||
"""Body call_id doesn't match any pending item → 409 with the
|
||||
current primary call_id so the UI can re-render against the
|
||||
new round."""
|
||||
current primary call_id + cycle_id so the UI can re-render
|
||||
against the new round."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-current")
|
||||
cycle = _seed_pending(ws, "c-current")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1154,17 +1171,17 @@ def test_approve_409_on_stale_call_id(storage):
|
||||
body = resp.json()
|
||||
assert body["error"] == "stale call_id"
|
||||
assert body["current_call_id"] == "c-current"
|
||||
# Approval event must NOT be set — no resolve_approval ran.
|
||||
assert not ws.ui._approval_event.is_set()
|
||||
assert body["current_cycle_id"] == cycle.cycle_id
|
||||
# The live cycle must NOT have been resolved.
|
||||
assert not cycle.event.is_set()
|
||||
|
||||
|
||||
def test_approve_409_when_no_pending_and_call_id_sent(storage):
|
||||
"""Body sends a call_id but the UI has no pending approval —
|
||||
409 with current_call_id=None so the UI knows to clear the row."""
|
||||
"""Body sends a call_id but the UI has no live cycle — 409 with
|
||||
current_call_id=None so the UI knows to clear the row."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
# No _pending_approval seeded → ui._pending_approval is None.
|
||||
ws.ui._approval_event.clear()
|
||||
# No cycle registered.
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1173,18 +1190,18 @@ def test_approve_409_when_no_pending_and_call_id_sent(storage):
|
||||
)
|
||||
assert resp.status_code == 409
|
||||
body = resp.json()
|
||||
assert body["error"] == "no pending approval"
|
||||
assert body["error"] == "stale call_id"
|
||||
assert body["current_call_id"] is None
|
||||
assert not ws.ui._approval_event.is_set()
|
||||
assert body["current_cycle_id"] is None
|
||||
|
||||
|
||||
def test_approve_no_call_id_preserves_backward_compat(storage):
|
||||
"""Existing clients (CLI, channel adapters) that omit call_id
|
||||
must still resolve approvals — the guard only kicks in when
|
||||
call_id is present in the body."""
|
||||
must still resolve approvals — a selector-less body lands on the
|
||||
oldest live cycle."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-1")
|
||||
cycle = _seed_pending(ws, "c-1")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1192,18 +1209,18 @@ def test_approve_no_call_id_preserves_backward_compat(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert resp.json()["cycle_id"] == cycle.cycle_id
|
||||
assert cycle.event.is_set()
|
||||
|
||||
|
||||
def test_approve_no_call_id_no_pending_falls_through(storage):
|
||||
"""Legacy clients (no call_id) calling approve when pending is
|
||||
None hit the existing resolve_approval no-op path — the new
|
||||
guard must not change that behavior. Regression guard for the
|
||||
legacy code path that the call_id check intentionally bypasses."""
|
||||
def test_approve_no_call_id_no_pending_resolves_nothing(storage):
|
||||
"""Legacy clients (no call_id) calling approve with no live cycle:
|
||||
200 with ``cycle_id: null`` — the handler resolves NOTHING rather
|
||||
than racing a cycle that registers between its lookup and its
|
||||
resolve (the client can't have been looking at one)."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
ws.ui._approval_event.clear()
|
||||
# No _pending_approval seeded.
|
||||
# No cycle registered.
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1211,7 +1228,7 @@ def test_approve_no_call_id_no_pending_falls_through(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert resp.json()["cycle_id"] is None
|
||||
|
||||
|
||||
def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
@@ -1220,7 +1237,7 @@ def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
one-boolean semantics of resolve_approval."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "c-1", "c-2", "c-3")
|
||||
cycle = _seed_pending(ws, "c-1", "c-2", "c-3")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
@@ -1228,7 +1245,61 @@ def test_approve_call_id_matches_any_item_in_multi_envelope(storage):
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert cycle.event.is_set()
|
||||
|
||||
|
||||
def test_selectorless_always_whitelists_only_the_resolved_oldest_cycle(storage):
|
||||
"""sweep-3 regression: with several live cycles, a selector-less
|
||||
"Approve + Always" must whitelist the tools of the cycle it
|
||||
actually resolved (the oldest) — not a sibling's."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
oldest = _seed_pending(ws, "a-1", func_name="spawn_workstream")
|
||||
newer = _seed_pending(ws, "b-1", func_name="send_message")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
json={"approved": True, "always": True}, # no selector
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["cycle_id"] == oldest.cycle_id
|
||||
assert oldest.event.is_set()
|
||||
assert not newer.event.is_set()
|
||||
assert "spawn_workstream" in ws.ui.auto_approve_tools
|
||||
assert "send_message" not in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
def test_approve_always_skips_whitelist_when_pinned_cycle_lost_the_race(storage):
|
||||
"""sweep-3 regression: the handler collects always-names from the
|
||||
cycle its lookup pinned; if that cycle is resolved by someone else
|
||||
(gate timeout, peer tab) between lookup and resolve, the whitelist
|
||||
must NOT grow — approving a card that already resolved must not
|
||||
auto-approve anything."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
_seed_pending(ws, "a-1", func_name="spawn_workstream")
|
||||
ui = ws.ui
|
||||
real_find = ui.find_approval_cycle
|
||||
|
||||
def racing_find(**kwargs):
|
||||
card = real_find(**kwargs)
|
||||
if card is not None:
|
||||
# A concurrent resolver wins the gap between the handler's
|
||||
# lookup and its (pinned) resolve.
|
||||
ui.resolve_approval(False, "raced", cycle_id=card["cycle_id"])
|
||||
return card
|
||||
|
||||
ui.find_approval_cycle = racing_find
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{ws.id}/approve",
|
||||
json={"approved": True, "always": True},
|
||||
headers=_COORD_HEADERS,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["cycle_id"] is None
|
||||
assert "spawn_workstream" not in ws.ui.auto_approve_tools
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1347,6 +1418,110 @@ def test_history_any_admin_coordinator_caller_can_read(storage):
|
||||
assert resp.json()["ws_id"] == ws.id
|
||||
|
||||
|
||||
def test_history_private_project_hidden_from_non_member(storage):
|
||||
# admin.coordinator gates the surface, but a coordinator in a private
|
||||
# project the caller isn't a member of is 404-masked — the conversation
|
||||
# does not leak to a non-member operator.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/history",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_history_private_project_visible_to_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/history",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert any(m.get("content") == "secret plan" for m in resp.json()["messages"])
|
||||
|
||||
|
||||
def test_export_private_project_hidden_from_non_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
storage.save_message("c" * 32, "user", "secret plan")
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/export",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_children_private_project_hidden_from_non_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{'c' * 32}/children",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_open_private_project_hidden_from_non_member(storage):
|
||||
# `open` rehydrates + returns the auto-titled name, so an ungated open is a
|
||||
# private-project existence/metadata oracle AND an unauthorized resurrection.
|
||||
# The tenant_check must fire before the already-loaded shortcut and mgr.open.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32, kind="coordinator", user_id="alice", project_id="proj-secret"
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.post(
|
||||
f"/v1/api/workstreams/{'c' * 32}/open",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_coord_attachments_private_project_hidden_from_non_member(storage):
|
||||
# Attachment list/serve resolves the owner as the coord owner and only
|
||||
# enforced cross-kind before — a non-member operator could enumerate and
|
||||
# download the owner's staged blobs. Now 404-masked by project tenancy.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{ws.id}/attachments",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_coord_attachments_private_project_visible_to_member(storage):
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="alice", project_id="proj-secret")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/workstreams/{ws.id}/attachments",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.coordinator"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
|
||||
|
||||
def test_history_serves_storage_only_workstream(storage):
|
||||
"""Persisted-but-not-loaded coordinators (closed / evicted) are still
|
||||
readable via /history without rehydrating. Mirrors the pre-lift
|
||||
@@ -1515,15 +1690,19 @@ def test_export_404_when_kind_interactive(storage):
|
||||
|
||||
|
||||
def test_cancel_resolves_pending_approval(storage):
|
||||
"""Cancel addresses the workstream, not one batch — EVERY live
|
||||
cycle resolves (parallel task agents can hold several gates)."""
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="user-1")
|
||||
assert isinstance(ws.ui, ConsoleCoordinatorUI)
|
||||
ws.ui._pending_approval = {"type": "approve_request", "items": []}
|
||||
ws.ui._approval_event.clear()
|
||||
first = _seed_pending(ws, "c-1")
|
||||
second = _seed_pending(ws, "c-2")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post(f"/v1/api/workstreams/{ws.id}/cancel", headers=_COORD_HEADERS)
|
||||
assert resp.status_code == 200
|
||||
assert ws.ui._approval_event.is_set()
|
||||
assert first.event.is_set()
|
||||
assert second.event.is_set()
|
||||
assert first.result == (False, "Cancelled by user")
|
||||
|
||||
|
||||
def test_cancel_response_always_includes_dropped_key(storage):
|
||||
@@ -2043,6 +2222,10 @@ def test_open_any_admin_coordinator_caller_succeeds_in_memory(storage):
|
||||
|
||||
def test_open_rehydrates_when_not_in_memory(storage, monkeypatch):
|
||||
mgr = _build_mgr(storage)
|
||||
# The tenancy gate resolves the row from storage before rehydrating, so a
|
||||
# legitimately-openable coordinator must exist there (it always does in
|
||||
# production — open rehydrates a persisted row).
|
||||
storage.register_workstream("coord-rehy", kind="coordinator", user_id="user-1")
|
||||
rehydrated = MagicMock()
|
||||
rehydrated.id = "coord-rehy"
|
||||
rehydrated.name = "rehydrated"
|
||||
@@ -2076,6 +2259,7 @@ def test_open_503_on_coord_mgr_unavailable(storage):
|
||||
|
||||
def test_open_correlation_id_on_factory_failure(storage, monkeypatch):
|
||||
mgr = _build_mgr(storage)
|
||||
storage.register_workstream("bad-ws", kind="coordinator", user_id="user-1")
|
||||
monkeypatch.setattr(mgr, "open", MagicMock(side_effect=RuntimeError("boom")))
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post("/v1/api/workstreams/bad-ws/open", headers=_COORD_HEADERS)
|
||||
@@ -2086,6 +2270,7 @@ def test_open_correlation_id_on_factory_failure(storage, monkeypatch):
|
||||
def test_open_503_when_open_raises_value_error(storage, monkeypatch):
|
||||
"""ValueError from the factory surfaces as 503 with the remediation text."""
|
||||
mgr = _build_mgr(storage)
|
||||
storage.register_workstream("bad-ws", kind="coordinator", user_id="user-1")
|
||||
monkeypatch.setattr(mgr, "open", MagicMock(side_effect=ValueError("coord registry missing")))
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
resp = client.post("/v1/api/workstreams/bad-ws/open", headers=_COORD_HEADERS)
|
||||
@@ -2251,7 +2436,8 @@ def test_cluster_inspect_invalid_ws_id_400(storage):
|
||||
|
||||
|
||||
def test_cluster_inspect_any_inspect_caller_sees_detail(storage):
|
||||
# Trusted-team visibility: admin.cluster.inspect sees every row.
|
||||
# A project-less workstream has no tenancy to enforce, so any
|
||||
# admin.cluster.inspect caller sees it (trusted-team default).
|
||||
mgr = _build_mgr(storage)
|
||||
ws = mgr.create(user_id="owner")
|
||||
client = _make_client(storage, coord_mgr=mgr, registry=_fake_registry())
|
||||
@@ -2263,6 +2449,46 @@ def test_cluster_inspect_any_inspect_caller_sees_detail(storage):
|
||||
assert resp.json()["persisted"]["ws_id"] == ws.id
|
||||
|
||||
|
||||
def test_cluster_inspect_private_project_hidden_from_non_member(storage):
|
||||
# admin.cluster.inspect gates the surface, but a workstream in a
|
||||
# private project the caller isn't a member of is masked as 404 —
|
||||
# no private-project oracle even for a cluster admin.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.register_workstream(
|
||||
"c" * 32,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/{'c' * 32}/detail",
|
||||
headers={"X-Test-User": "stranger", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
def test_cluster_inspect_private_project_visible_to_member(storage):
|
||||
# A project member (even a non-owner) still sees the persisted row.
|
||||
storage.create_project("proj-secret", "Secret", "alice")
|
||||
storage.add_project_member("proj-secret", "member-bob")
|
||||
storage.register_workstream(
|
||||
"c" * 32,
|
||||
node_id="console",
|
||||
user_id="alice",
|
||||
kind="coordinator",
|
||||
project_id="proj-secret",
|
||||
)
|
||||
client = _make_client(storage, coord_mgr=_build_mgr(storage), registry=_fake_registry())
|
||||
resp = client.get(
|
||||
f"/v1/api/cluster/ws/{'c' * 32}/detail",
|
||||
headers={"X-Test-User": "member-bob", "X-Test-Perms": "admin.cluster.inspect"},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["persisted"]["ws_id"] == "c" * 32
|
||||
|
||||
|
||||
def test_cluster_inspect_coordinator_self_path(storage):
|
||||
"""A coordinator row returns live from the in-process manager."""
|
||||
mgr = _build_mgr(storage)
|
||||
@@ -2393,6 +2619,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
ws_id = "f0" * 16
|
||||
_seed_node_workstream(storage, ws_id=ws_id, node_id="node-a")
|
||||
detail = {
|
||||
"cycle_id": "cyc-bash",
|
||||
"call_id": "c-bash",
|
||||
"judge_pending": False,
|
||||
"items": [
|
||||
@@ -2421,7 +2648,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
"activity_state": "approval",
|
||||
"activity": "awaiting approval",
|
||||
"tokens": 100,
|
||||
"pending_approval_detail": detail,
|
||||
"pending_approval_details": [detail],
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -2432,7 +2659,7 @@ def test_cluster_inspect_node_backed_pending_approval_detail_passes_through(stor
|
||||
assert resp.status_code == 200
|
||||
live = resp.json()["live"]
|
||||
assert live["pending_approval"] is True # derived bool, existing behavior
|
||||
assert live["pending_approval_detail"] == detail # full payload, new behavior
|
||||
assert live["pending_approval_details"] == [detail] # full payload passthrough
|
||||
|
||||
|
||||
def test_cluster_inspect_node_backed_pending_approval_synthesized(storage):
|
||||
|
||||
@@ -313,17 +313,17 @@ def test_coordinator_js_handle_child_state_no_longer_reads_sse_pending_approval_
|
||||
)
|
||||
|
||||
# The merge body must preserve BOTH pending_approval and
|
||||
# pending_approval_detail from prev — preserving only one would
|
||||
# pending_approval_details from prev — preserving only one would
|
||||
# render a row with a phantom badge but no buttons (or vice versa).
|
||||
merge_body = re.search(
|
||||
r"mergedLive\s*=\s*Object\.assign\(\s*\{\}\s*,\s*live\s*,\s*\{"
|
||||
r"[^}]*pending_approval:\s*prev\.live\.pending_approval[^}]*"
|
||||
r"pending_approval_detail:\s*prev\.live\.pending_approval_detail",
|
||||
r"pending_approval_details:\s*prev\.live\.pending_approval_details",
|
||||
body,
|
||||
)
|
||||
assert merge_body is not None, (
|
||||
"Merge body must preserve both pending_approval AND "
|
||||
"pending_approval_detail from prev.live — preserving only one "
|
||||
"pending_approval_details from prev.live — preserving only one "
|
||||
"creates a half-rendered approval row."
|
||||
)
|
||||
|
||||
@@ -666,3 +666,28 @@ def test_coord_child_links_open_interactive_pane():
|
||||
assert 'data-node-id="' in coord_js
|
||||
# The /node/{id}/?ws_id= href fallback must remain for the standalone page.
|
||||
assert '"/node/"' in coord_js
|
||||
|
||||
|
||||
def test_coordinator_js_gates_send_on_cross_user_busy():
|
||||
"""The coordinator pane mirrors the interactive pane's shared-workstream
|
||||
send gate: while another participant's turn is in flight it blocks this
|
||||
viewer's send (the UX complement to the server-side 409). String-presence
|
||||
guard — coord.js has no JS test framework."""
|
||||
from pathlib import Path
|
||||
|
||||
coord_js = (
|
||||
Path(__file__).resolve().parent.parent
|
||||
/ "turnstone/console/static/coordinator/coordinator.js"
|
||||
).read_text(encoding="utf-8")
|
||||
# tracks the acting user from state_change, clears on settle
|
||||
assert "actingUserId = ev.acting_user_id;" in coord_js
|
||||
assert "actingUserId = null;" in coord_js
|
||||
# compares against the viewer's own id and drives the composer hard block
|
||||
assert 'sessionStorage.getItem("ts.user_id")' in coord_js
|
||||
assert "actingUserId !== me" in coord_js
|
||||
assert "composer.setSendBlocked(" in coord_js
|
||||
assert "function reconcileSendBlock()" in coord_js
|
||||
# reactive 409 fallback
|
||||
assert "r.status === 409" in coord_js
|
||||
assert 'status: "cross_user_interjection"' in coord_js
|
||||
assert 'data.status === "cross_user_interjection"' in coord_js
|
||||
|
||||
@@ -35,7 +35,14 @@ class _StubUI:
|
||||
def on_error(self, msg: str) -> None:
|
||||
self.errors.append(msg)
|
||||
|
||||
def on_tool_result(self, call_id: str, name: str, output: str, is_error: bool = False) -> None:
|
||||
def on_tool_result(
|
||||
self,
|
||||
call_id: str,
|
||||
name: str,
|
||||
output: str,
|
||||
is_error: bool = False,
|
||||
preview: dict[str, Any] | None = None,
|
||||
) -> None:
|
||||
self.tool_results.append((call_id, name, output, is_error))
|
||||
|
||||
# Other SessionUI methods — only stubs, not exercised here.
|
||||
@@ -198,6 +205,23 @@ def test_spawn_prepare_needs_approval(coord_session):
|
||||
assert item["skill"] == "s"
|
||||
|
||||
|
||||
def test_spawn_prepare_denies_high_risk_skill(coord_session):
|
||||
"""Review fix: the high/critical-risk gate that blocks skills(load) also
|
||||
blocks spawn_workstream(skill=…), so a child spawn can't route around it."""
|
||||
sess, _coord, _ui = coord_session
|
||||
with patch("turnstone.core.session.get_storage") as gs:
|
||||
gs.return_value.get_prompt_template_by_name.return_value = {
|
||||
"name": "danger",
|
||||
"risk_level": "critical",
|
||||
}
|
||||
item = sess._prepare_tool(
|
||||
_tc("spawn_workstream", {"initial_message": "go", "skill": "danger"})
|
||||
)
|
||||
assert "error" in item
|
||||
assert "/skill danger" in item["error"]
|
||||
assert item.get("needs_approval") is not True
|
||||
|
||||
|
||||
def test_spawn_exec_calls_client_and_returns_summary(coord_session):
|
||||
sess, coord, _ui = coord_session
|
||||
coord.spawn.return_value = {
|
||||
@@ -1504,6 +1528,9 @@ def _stub_judge_for_evaluate_intent(monkeypatch, sess):
|
||||
fake_judge = MagicMock()
|
||||
# judge.evaluate(items, messages, callback=, cancel_event=) → list[verdict]
|
||||
fake_judge.evaluate.side_effect = lambda items, *_args, **_kw: [fake_verdict] * len(items)
|
||||
# arg_budget_chars() feeds honest_truncate in the projection loop and must
|
||||
# be a real int, not a MagicMock; large enough that nothing truncates.
|
||||
fake_judge.arg_budget_chars.return_value = 200_000
|
||||
monkeypatch.setattr(sess, "_ensure_judge", lambda: fake_judge)
|
||||
return fake_judge
|
||||
|
||||
@@ -1545,7 +1572,10 @@ def test_spawn_batch_evaluate_intent_projects_all_children(coord_session, monkey
|
||||
|
||||
def test_spawn_batch_evaluate_intent_truncates_long_messages(coord_session, monkeypatch):
|
||||
sess, _coord, _ui = coord_session
|
||||
_stub_judge_for_evaluate_intent(monkeypatch, sess)
|
||||
fake_judge = _stub_judge_for_evaluate_intent(monkeypatch, sess)
|
||||
# Each child's initial_message is truncated to its share of the judge's
|
||||
# arg budget (window-based), not a fixed cap, and the omission is honest.
|
||||
fake_judge.arg_budget_chars.return_value = 300 # 1 child → 300 chars/child
|
||||
long_msg = "x" * 500
|
||||
item = sess._prepare_tool(
|
||||
_tc("spawn_batch", {"children": [{"initial_message": long_msg, "skill": "researcher"}]})
|
||||
@@ -1554,9 +1584,9 @@ def test_spawn_batch_evaluate_intent_truncates_long_messages(coord_session, monk
|
||||
|
||||
children = item["func_args"]["children"]
|
||||
assert len(children) == 1
|
||||
# Cap is 200 chars — same shape every other coord-tool projection uses.
|
||||
assert len(children[0]["initial_message"]) == 200
|
||||
assert children[0]["initial_message"] == "x" * 200
|
||||
msg = children[0]["initial_message"]
|
||||
assert msg.startswith("x" * 300)
|
||||
assert "200 of 500 chars omitted" in msg
|
||||
|
||||
|
||||
def test_spawn_batch_evaluate_intent_handles_empty_children_defensively(coord_session, monkeypatch):
|
||||
@@ -1609,10 +1639,15 @@ def test_tasks_update_without_title_evaluates_intent_cleanly(coord_session, monk
|
||||
# The crash trigger: item["title"] is None after _prepare_tasks.
|
||||
assert item["title"] is None
|
||||
sess._evaluate_intent([item])
|
||||
# title collapses None → "" (truncatable text); status is projected so the
|
||||
# judge can see what state is being set; child_ws_id passes through as None
|
||||
# ("unchanged"), never sliced.
|
||||
assert item["func_args"] == {
|
||||
"action": "update",
|
||||
"task_id": "tsk_1",
|
||||
"title": "",
|
||||
"status": "in_progress",
|
||||
"child_ws_id": None,
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
"""Tests for the effective effort-ladder projection.
|
||||
|
||||
The ladder must mirror the request-time mapping functions exactly —
|
||||
equal ``effective`` tokens promise byte-identical effort behavior on
|
||||
the wire, which is what the UI annotations lean on.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from turnstone.core.providers._protocol import ModelCapabilities
|
||||
from turnstone.core.providers.effort_ladder import (
|
||||
KNOB_VALUES,
|
||||
effort_ladder,
|
||||
effort_ladder_for_model,
|
||||
)
|
||||
|
||||
|
||||
def _as_map(ladder: list[dict[str, str]]) -> dict[str, str]:
|
||||
assert [r["value"] for r in ladder] == list(KNOB_VALUES)
|
||||
return {r["value"]: r["effective"] for r in ladder}
|
||||
|
||||
|
||||
class TestLocalLanes:
|
||||
def test_toggle_engaged_carries_graded_value_per_position(self) -> None:
|
||||
"""No declared effort key: the toggle rides the knob AND the graded
|
||||
value is forwarded under the fallback template key — the user's
|
||||
effort setting always reaches the wire (a template that doesn't
|
||||
reference the kwarg ignores it), so every position is distinct."""
|
||||
caps = ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking")
|
||||
eff = _as_map(effort_ladder("anthropic-compatible", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["minimal"] == "on+minimal"
|
||||
assert eff["max"] == "on+max"
|
||||
assert len({eff[k] for k in KNOB_VALUES}) == len(KNOB_VALUES)
|
||||
|
||||
def test_freeform_effort_param_forwards_each_value(self) -> None:
|
||||
"""deepseek-style config: toggle + verbatim effort per position."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic-compatible", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["low"] == "on+low"
|
||||
assert eff["max"] == "on+max"
|
||||
|
||||
def test_validated_effort_param_shows_snapping(self) -> None:
|
||||
"""Off-list positions round up onto the declared values; above the
|
||||
ceiling they ride the ceiling — never the (possibly lower) default."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["minimal"] == "on+low"
|
||||
assert eff["high"] == "on+high"
|
||||
assert eff["xhigh"] == "on+high"
|
||||
assert eff["max"] == "on+high"
|
||||
|
||||
def test_openai_compatible_flat_param_without_effort_param(self) -> None:
|
||||
caps = ModelCapabilities(
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
)
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == "high" # ceiling, not default
|
||||
|
||||
def test_adaptive_local_never_off(self) -> None:
|
||||
caps = ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking")
|
||||
eff = _as_map(effort_ladder("openai-compatible", caps))
|
||||
assert eff["none"] == "on"
|
||||
assert eff["max"] == "on"
|
||||
|
||||
|
||||
class TestNativeAnthropicLane:
|
||||
def test_adaptive_with_effort_levels(self) -> None:
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="adaptive",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high", "xhigh", "max"),
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["none"] == "adaptive" # thinking on, model decides
|
||||
assert eff["minimal"] == "low" # rounds up onto the declared levels
|
||||
assert eff["low"] == "low"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_sonnet_5_registry_row(self) -> None:
|
||||
"""claude-sonnet-5: adaptive + full effort ladder incl. xhigh/max —
|
||||
every knob level above none is a distinct wire behavior."""
|
||||
eff = _as_map(effort_ladder_for_model("anthropic", "claude-sonnet-5", None))
|
||||
assert eff["none"] == "adaptive"
|
||||
assert eff["minimal"] == "low" # rounds up onto declared levels
|
||||
assert eff["low"] == "low"
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_sonnet_4_6_xhigh_rides_max(self) -> None:
|
||||
"""Sonnet 4.6 declares (low, medium, high, max) — no xhigh, so the
|
||||
knob's xhigh snaps up onto max rather than down onto high."""
|
||||
eff = _as_map(effort_ladder_for_model("anthropic", "claude-sonnet-4-6", None))
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == "max"
|
||||
assert eff["max"] == "max"
|
||||
|
||||
def test_manual_budget_ladder(self) -> None:
|
||||
"""Budgets are monotone over the whole knob domain."""
|
||||
caps = ModelCapabilities(thinking_mode="manual")
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["none"] == "off"
|
||||
assert eff["minimal"] == eff["low"] == "budget:1024" # 1024 = API floor
|
||||
assert eff["medium"] == "budget:4096"
|
||||
assert eff["high"] == "budget:16384"
|
||||
assert eff["xhigh"] == "budget:32768"
|
||||
assert eff["max"] == "budget:65536"
|
||||
|
||||
|
||||
class TestFlatParamLanes:
|
||||
def test_google_default_caps(self) -> None:
|
||||
eff = _as_map(effort_ladder_for_model("google", "gemini-3-flash", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "minimal"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_google_override_routes_through_chat_lane(self) -> None:
|
||||
"""GoogleProvider inherits _finalize_extra_body — a thinking_mode
|
||||
override changes real requests, and the ladder must mirror it."""
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"google",
|
||||
"gemini-3-flash",
|
||||
{"thinking_mode": "manual", "thinking_param": "enable_thinking"},
|
||||
)
|
||||
)
|
||||
assert eff["none"] == "off"
|
||||
assert eff["medium"] == "on+medium" # toggle + inherited flat param
|
||||
|
||||
def test_responses_surface_projects_flat_only(self) -> None:
|
||||
caps_overrides = {
|
||||
"thinking_mode": "manual",
|
||||
"reasoning_effort_values": ["low", "medium", "high"],
|
||||
}
|
||||
chat = _as_map(effort_ladder_for_model("openai-compatible", "m", caps_overrides))
|
||||
responses = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"openai-compatible", "m", caps_overrides, api_surface="responses"
|
||||
)
|
||||
)
|
||||
assert chat["medium"] == "on+medium"
|
||||
assert responses["medium"] == "medium"
|
||||
assert responses["none"] == "default"
|
||||
|
||||
def test_xai_projects_flat_only(self) -> None:
|
||||
"""grok-4.3 declares values (none/low/medium/high, default low);
|
||||
knob positions above the ceiling ride the ceiling (high). The
|
||||
declared "none" IS forwarded for the knob's off position (xAI
|
||||
documents it as disabling reasoning) but is never a snap target
|
||||
for other positions."""
|
||||
eff = _as_map(effort_ladder_for_model("xai", "grok-4.3", None))
|
||||
assert eff["none"] == "none" # explicit disable, declared by grok
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["low"] == "low"
|
||||
assert eff["high"] == "high"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_xai_ignores_template_overrides(self) -> None:
|
||||
"""XAIProvider subclasses OpenAIResponsesProvider, which drops
|
||||
extra_body — a thinking_mode/effort_param override cannot change
|
||||
an xai request, so it must not change the ladder either."""
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"xai",
|
||||
"grok-4.3",
|
||||
{
|
||||
"thinking_mode": "manual",
|
||||
"thinking_param": "enable_thinking",
|
||||
"effort_param": "reasoning_effort",
|
||||
},
|
||||
)
|
||||
)
|
||||
assert eff["none"] == "none" # flat channel, not an "off" toggle
|
||||
assert eff["medium"] == "medium"
|
||||
assert all("+" not in v and v not in ("on", "off") for v in eff.values())
|
||||
|
||||
def test_openai_gpt55_registry_row(self) -> None:
|
||||
"""gpt-5.5 declares none/low/medium/high/xhigh with default medium:
|
||||
knob none sends the explicit "none" level (server default is
|
||||
MEDIUM, so omission would not disable), max rides the xhigh
|
||||
ceiling, minimal rounds up to low."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.5", None))
|
||||
assert eff["none"] == "none"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_openai_o3_registry_row(self) -> None:
|
||||
"""o-series (except o1-mini) accept low/medium/high; no declared
|
||||
"none" level, so the knob's off position omits the param."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "o3", None))
|
||||
assert eff["none"] == "default"
|
||||
assert eff["minimal"] == "low"
|
||||
assert eff["medium"] == "medium"
|
||||
assert eff["xhigh"] == eff["max"] == "high"
|
||||
|
||||
def test_openai_codex_max_has_xhigh(self) -> None:
|
||||
"""gpt-5.1-codex-max must not prefix-fall onto the gpt-5.1 row
|
||||
(which lacks xhigh) — xhigh reaches the wire verbatim."""
|
||||
eff = _as_map(effort_ladder_for_model("openai", "gpt-5.1-codex-max", None))
|
||||
assert eff["xhigh"] == "xhigh"
|
||||
assert eff["max"] == "xhigh"
|
||||
|
||||
def test_anthropic_effort_applies_even_with_thinking_mode_none(self) -> None:
|
||||
"""output_config gates on supports_effort alone at request time."""
|
||||
caps = ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
)
|
||||
eff = _as_map(effort_ladder("anthropic", caps))
|
||||
assert eff["high"] == "high"
|
||||
assert eff["none"] == "default"
|
||||
|
||||
def test_overrides_merge_and_unknown_keys_ignored(self) -> None:
|
||||
eff = _as_map(
|
||||
effort_ladder_for_model(
|
||||
"google",
|
||||
"gemini-3-flash",
|
||||
{"reasoning_effort_values": [], "not_a_field": True},
|
||||
)
|
||||
)
|
||||
# Operator cleared the values → nothing effort-related is sent.
|
||||
assert set(eff.values()) == {"default"}
|
||||
@@ -0,0 +1,410 @@
|
||||
"""Ladder↔wire parity harness — the effort ladder must tell the truth.
|
||||
|
||||
``effort_ladder`` *projects* the session effort knob through the same
|
||||
mapping functions the providers use at request time. This suite proves
|
||||
that projection against the REAL request path: for every provider lane
|
||||
and capability shape, each knob position is driven through the actual
|
||||
provider ``create_streaming`` against a recording fake client (the same
|
||||
SDK-seam capture the wire-payload goldens use), the effort-relevant
|
||||
subset of the captured kwargs is extracted, and it must equal what the
|
||||
ladder token decodes to. Two invariants per shape:
|
||||
|
||||
1. **Semantics** — each ladder token decodes to an expected wire subset
|
||||
(``on``/``off`` ⇒ the chat-template toggle, ``budget:N`` ⇒ Anthropic
|
||||
thinking budget, a bare level ⇒ the lane's flat/effort channel) and
|
||||
the observed wire subset must match it exactly.
|
||||
2. **Grouping** — the ladder's core promise: two knob positions carry
|
||||
equal ``effective`` tokens if and only if they produce identical
|
||||
effort-relevant wire payloads.
|
||||
|
||||
A failure here means the UI annotates behavior the wire does not have —
|
||||
the bug class that shipped xai in the ladder's chat-lane set even though
|
||||
``XAIProvider`` rides the Responses surface, which drops ``extra_body``.
|
||||
|
||||
The harness goes through ``create_provider`` (not direct classes) so the
|
||||
provider ROUTING the ladder assumes — e.g. ``api_surface="responses"``
|
||||
selecting the Responses adapter — is itself under test.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import dataclasses
|
||||
import itertools
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._wire_capture import RecordingClient
|
||||
from turnstone.core.providers import create_provider
|
||||
from turnstone.core.providers._protocol import (
|
||||
EFFORT_TEMPLATE_FALLBACK_PARAM,
|
||||
ModelCapabilities,
|
||||
)
|
||||
from turnstone.core.providers.effort_ladder import KNOB_VALUES, effort_ladder
|
||||
|
||||
# Above the largest manual-mode thinking budget (max: 65536) so the
|
||||
# request path's budget<max_tokens clamp never fires — the ladder
|
||||
# documents budgets unclamped, so the capture must be too. (At small
|
||||
# per-request max_tokens the clamp can genuinely alias adjacent budget
|
||||
# tiers on the wire; that is the ladder's documented approximation, not
|
||||
# a parity break.)
|
||||
_MAX_TOKENS = 128_000
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
class Shape:
|
||||
"""One (provider lane, capability shape) point of the parity matrix."""
|
||||
|
||||
id: str
|
||||
provider: str
|
||||
caps: ModelCapabilities
|
||||
api_surface: str = ""
|
||||
model: str = "m"
|
||||
|
||||
|
||||
# Real registry rows for the lanes whose defaults carry effort values —
|
||||
# parity should cover what ships, not only synthetic shapes.
|
||||
_GEMINI_CAPS = create_provider("google").get_capabilities("gemini-3-flash")
|
||||
_GROK_CAPS = create_provider("xai").get_capabilities("grok-4.3")
|
||||
_GPT55_CAPS = create_provider("openai").get_capabilities("gpt-5.5")
|
||||
|
||||
SHAPES: tuple[Shape, ...] = (
|
||||
# -- anthropic-compatible (vLLM /v1/messages): template channel only --
|
||||
Shape(
|
||||
"compat-toggle-manual",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"compat-toggle-adaptive",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"compat-freeform-effort",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
# DeepSeek-V4 official contract: toggle + effort in {high, max}.
|
||||
"compat-validated-effort",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("high", "max"),
|
||||
default_reasoning_effort="high",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"compat-inert",
|
||||
"anthropic-compatible",
|
||||
ModelCapabilities(thinking_mode="none"),
|
||||
),
|
||||
# -- openai-compatible on the Chat Completions surface: both channels --
|
||||
Shape(
|
||||
"oc-toggle-only",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
),
|
||||
Shape(
|
||||
"oc-toggle-plus-flat",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-effort-param-suppresses-flat",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-flat-only",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
),
|
||||
Shape(
|
||||
"oc-adaptive",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(thinking_mode="adaptive", thinking_param="enable_thinking"),
|
||||
),
|
||||
# -- openai-compatible pinned to the Responses surface: template caps
|
||||
# become inert and only the native flat channel remains --
|
||||
Shape(
|
||||
"oc-responses-surface",
|
||||
"openai-compatible",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
reasoning_effort_values=("low", "medium", "high"),
|
||||
default_reasoning_effort="medium",
|
||||
),
|
||||
api_surface="responses",
|
||||
),
|
||||
# -- commercial flat lanes --
|
||||
Shape(
|
||||
# Real registry row: none/low/medium/high/xhigh, default medium.
|
||||
# Knob none must send the EXPLICIT "none" level (omission would
|
||||
# leave the server default medium reasoning on); knob max rides
|
||||
# the xhigh ceiling.
|
||||
"openai-gpt-5.5",
|
||||
"openai",
|
||||
_GPT55_CAPS,
|
||||
model="gpt-5.5",
|
||||
),
|
||||
Shape("google-default", "google", _GEMINI_CAPS, model="gemini-3-flash"),
|
||||
Shape(
|
||||
# GoogleProvider subclasses the chat provider, so a template
|
||||
# override DOES change real requests — hybrid toggle + flat.
|
||||
"google-manual-override",
|
||||
"google",
|
||||
dataclasses.replace(_GEMINI_CAPS, thinking_mode="manual", thinking_param="enable_thinking"),
|
||||
model="gemini-3-flash",
|
||||
),
|
||||
Shape("xai-default", "xai", _GROK_CAPS, model="grok-4.3"),
|
||||
Shape(
|
||||
# XAIProvider rides the Responses surface: template overrides are
|
||||
# inert on the wire, and the ladder must not pretend otherwise.
|
||||
"xai-template-override-inert",
|
||||
"xai",
|
||||
dataclasses.replace(
|
||||
_GROK_CAPS,
|
||||
thinking_mode="manual",
|
||||
thinking_param="enable_thinking",
|
||||
effort_param="reasoning_effort",
|
||||
),
|
||||
model="grok-4.3",
|
||||
),
|
||||
# -- native Anthropic --
|
||||
Shape(
|
||||
"anthropic-adaptive-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="adaptive",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high", "xhigh", "max"),
|
||||
),
|
||||
model="claude-fable-5",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-adaptive-plain",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="adaptive"),
|
||||
model="claude-fable-5",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-manual-budgets",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="manual"),
|
||||
model="claude-3-7-sonnet-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-manual-plus-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="manual",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
),
|
||||
model="claude-3-7-sonnet-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-none-effort",
|
||||
"anthropic",
|
||||
ModelCapabilities(
|
||||
thinking_mode="none",
|
||||
supports_effort=True,
|
||||
effort_levels=("low", "medium", "high"),
|
||||
),
|
||||
model="claude-3-5-haiku-latest",
|
||||
),
|
||||
Shape(
|
||||
"anthropic-inert",
|
||||
"anthropic",
|
||||
ModelCapabilities(thinking_mode="none"),
|
||||
model="claude-3-5-haiku-latest",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Wire capture + effort-subset extraction
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _wire_payload(shape: Shape, knob: str) -> dict[str, Any]:
|
||||
"""Drive the real provider request path; return the captured SDK kwargs."""
|
||||
provider = create_provider(shape.provider, api_surface=shape.api_surface or None)
|
||||
client = RecordingClient()
|
||||
gen = provider.create_streaming(
|
||||
client=client,
|
||||
model=shape.model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=_MAX_TOKENS,
|
||||
reasoning_effort=knob,
|
||||
capabilities=shape.caps,
|
||||
)
|
||||
# kwargs are recorded eagerly during the call above; close the
|
||||
# unconsumed iterator so stream-manager cleanup runs on the stub.
|
||||
close = getattr(gen, "close", None)
|
||||
if callable(close):
|
||||
with contextlib.suppress(Exception):
|
||||
close()
|
||||
assert "payload" in client.captured, f"{shape.id}: provider made no SDK call"
|
||||
return dict(client.captured["payload"])
|
||||
|
||||
|
||||
def _effort_wire_subset(payload: dict[str, Any], shape: Shape) -> dict[str, Any]:
|
||||
"""Every effort-related lever in *payload*, normalized across lanes.
|
||||
|
||||
Keys: ``thinking`` (native Anthropic param), ``output_effort``
|
||||
(Anthropic ``output_config.effort``), ``flat`` (Chat Completions
|
||||
``reasoning_effort`` / Responses ``reasoning.effort``), ``toggle``
|
||||
and ``template_effort`` (``extra_body.chat_template_kwargs`` — the
|
||||
graded key is ``caps.effort_param``, else the fallback template key
|
||||
on the anthropic-compatible lane, whose only effort channel is the
|
||||
template).
|
||||
"""
|
||||
caps = shape.caps
|
||||
effort_key = caps.effort_param or (
|
||||
EFFORT_TEMPLATE_FALLBACK_PARAM if shape.provider == "anthropic-compatible" else ""
|
||||
)
|
||||
subset: dict[str, Any] = {}
|
||||
if "thinking" in payload:
|
||||
subset["thinking"] = payload["thinking"]
|
||||
output_config = payload.get("output_config")
|
||||
if isinstance(output_config, dict) and "effort" in output_config:
|
||||
subset["output_effort"] = output_config["effort"]
|
||||
if "reasoning_effort" in payload:
|
||||
subset["flat"] = payload["reasoning_effort"]
|
||||
reasoning = payload.get("reasoning")
|
||||
if isinstance(reasoning, dict) and "effort" in reasoning:
|
||||
subset["flat"] = reasoning["effort"]
|
||||
extra_body = payload.get("extra_body")
|
||||
ctk = extra_body.get("chat_template_kwargs") if isinstance(extra_body, dict) else None
|
||||
if isinstance(ctk, dict):
|
||||
known = {caps.thinking_param, effort_key} - {""}
|
||||
unexpected = set(ctk) - known
|
||||
assert not unexpected, f"unexpected chat_template_kwargs keys: {unexpected}"
|
||||
if caps.thinking_param in ctk:
|
||||
subset["toggle"] = ctk[caps.thinking_param]
|
||||
if effort_key and effort_key in ctk:
|
||||
subset["template_effort"] = ctk[effort_key]
|
||||
return subset
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Ladder-token decoding — the token grammar, made executable
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _decode_token(shape: Shape, token: str) -> dict[str, Any]:
|
||||
"""Expected effort wire subset for a ladder ``effective`` token."""
|
||||
caps = shape.caps
|
||||
if shape.provider == "anthropic":
|
||||
return _decode_native(caps, token)
|
||||
if shape.provider in ("openai", "xai") or shape.api_surface == "responses":
|
||||
return {} if token == "default" else {"flat": token}
|
||||
return _decode_template(shape.provider, caps, token)
|
||||
|
||||
|
||||
def _decode_native(caps: ModelCapabilities, token: str) -> dict[str, Any]:
|
||||
if caps.thinking_mode == "adaptive":
|
||||
# Thinking is unconditionally adaptive; a non-"adaptive" token is
|
||||
# the output_config effort level riding on top.
|
||||
expected: dict[str, Any] = {"thinking": {"type": "adaptive"}}
|
||||
if token != "adaptive":
|
||||
expected["output_effort"] = token
|
||||
return expected
|
||||
if token in ("default", "off"):
|
||||
return {}
|
||||
effort, sep, budget = token.partition("·budget:")
|
||||
if sep:
|
||||
return {
|
||||
"output_effort": effort,
|
||||
"thinking": {"type": "enabled", "budget_tokens": int(budget)},
|
||||
}
|
||||
if token.startswith("budget:"):
|
||||
budget_tokens = int(token.removeprefix("budget:"))
|
||||
return {"thinking": {"type": "enabled", "budget_tokens": budget_tokens}}
|
||||
return {"output_effort": token}
|
||||
|
||||
|
||||
def _decode_template(provider: str, caps: ModelCapabilities, token: str) -> dict[str, Any]:
|
||||
if token == "default":
|
||||
return {}
|
||||
parts = token.split("+")
|
||||
expected: dict[str, Any] = {}
|
||||
if parts[0] in ("on", "off"):
|
||||
expected["toggle"] = parts[0] == "on"
|
||||
parts = parts[1:]
|
||||
if parts:
|
||||
assert len(parts) == 1, f"unparseable ladder token: {token!r}"
|
||||
if caps.effort_param or provider == "anthropic-compatible":
|
||||
# Declared graded key, or the anthropic-compatible fallback
|
||||
# template key — that lane has no flat channel, so a graded
|
||||
# part there is always template-borne.
|
||||
expected["template_effort"] = parts[0]
|
||||
else:
|
||||
expected["flat"] = parts[0]
|
||||
return expected
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# The parity tests
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", SHAPES, ids=lambda s: s.id)
|
||||
def test_ladder_tokens_match_wire(shape: Shape) -> None:
|
||||
"""Invariant 1: each token's decoded meaning equals the captured wire."""
|
||||
ladder = effort_ladder(shape.provider, shape.caps, shape.api_surface)
|
||||
assert [row["value"] for row in ladder] == list(KNOB_VALUES)
|
||||
for row in ladder:
|
||||
knob, token = row["value"], row["effective"]
|
||||
observed = _effort_wire_subset(_wire_payload(shape, knob), shape)
|
||||
expected = _decode_token(shape, token)
|
||||
assert observed == expected, (
|
||||
f"{shape.id}/knob={knob}: ladder says {token!r} which decodes to "
|
||||
f"{expected}, but the wire carries {observed}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", SHAPES, ids=lambda s: s.id)
|
||||
def test_equal_tokens_iff_equal_wire(shape: Shape) -> None:
|
||||
"""Invariant 2: token equality ⇔ effort-wire equality, per shape."""
|
||||
tokens = {
|
||||
row["value"]: row["effective"]
|
||||
for row in effort_ladder(shape.provider, shape.caps, shape.api_surface)
|
||||
}
|
||||
subsets = {knob: _effort_wire_subset(_wire_payload(shape, knob), shape) for knob in KNOB_VALUES}
|
||||
for a, b in itertools.combinations(KNOB_VALUES, 2):
|
||||
same_token = tokens[a] == tokens[b]
|
||||
same_wire = subsets[a] == subsets[b]
|
||||
assert same_token == same_wire, (
|
||||
f"{shape.id}: knobs {a!r}/{b!r} have "
|
||||
f"{'equal' if same_token else 'distinct'} tokens "
|
||||
f"({tokens[a]!r} vs {tokens[b]!r}) but "
|
||||
f"{'identical' if same_wire else 'different'} wire subsets "
|
||||
f"({subsets[a]} vs {subsets[b]})"
|
||||
)
|
||||
@@ -225,6 +225,22 @@ class TestRoles:
|
||||
assert resp.status_code == 200, resp.json()
|
||||
assert "model.skills.write" in resp.json()["permissions"]
|
||||
|
||||
def test_create_role_with_persona_permissions(self, client):
|
||||
"""``persona.{create,read,write}`` (migration 063) are enumerated in
|
||||
``_VALID_PERMISSIONS`` and pass role-create validation. Before the fix
|
||||
they 400'd — a custom role could never carry a persona grant."""
|
||||
resp = client.post(
|
||||
"/v1/api/admin/roles",
|
||||
json=_role_payload(
|
||||
name="personaeditor",
|
||||
permissions="read,persona.create,persona.read,persona.write",
|
||||
),
|
||||
)
|
||||
assert resp.status_code == 200, resp.json()
|
||||
perms = resp.json()["permissions"]
|
||||
for p in ("persona.create", "persona.read", "persona.write"):
|
||||
assert p in perms
|
||||
|
||||
def test_permission_sections_js_covers_valid_permissions(self):
|
||||
"""F-5: ``_PERMISSION_SECTIONS`` in governance.js mirrors
|
||||
``_VALID_PERMISSIONS`` in console/server.py. A new perm added
|
||||
@@ -355,6 +371,19 @@ class TestRoles:
|
||||
assert role["display_name"] == "Senior Analyst"
|
||||
assert role["permissions"] == "read,write,approve"
|
||||
|
||||
def test_update_role_accepts_persona_permissions(self, client):
|
||||
"""Editing a custom role to carry ``persona.*`` must validate (they were
|
||||
rejected before 063 added them to ``_VALID_PERMISSIONS``)."""
|
||||
create_resp = client.post("/v1/api/admin/roles", json=_role_payload())
|
||||
role_id = create_resp.json()["role_id"]
|
||||
resp = client.put(
|
||||
f"/v1/api/admin/roles/{role_id}",
|
||||
json={"permissions": "read,persona.read,persona.write"},
|
||||
)
|
||||
assert resp.status_code == 200, resp.json()
|
||||
perms = resp.json()["permissions"]
|
||||
assert "persona.read" in perms and "persona.write" in perms
|
||||
|
||||
def test_update_nonexistent_role(self, client):
|
||||
resp = client.put(
|
||||
"/v1/api/admin/roles/nonexistent",
|
||||
@@ -451,6 +480,20 @@ class TestRoleOverrides:
|
||||
assert "model.skills.write" in body["effective"]
|
||||
assert body["grants"] == ["model.skills.write"]
|
||||
|
||||
def test_overrides_grant_persona_write(self, client, storage):
|
||||
# persona.write is admin-default (063) but grantable to any builtin
|
||||
# role via the overrides layer — the endpoint must accept it, not 400
|
||||
# it as an unknown permission.
|
||||
_seed_builtin_admin(storage, "read,write,admin.roles")
|
||||
resp = client.put(
|
||||
"/v1/api/admin/roles/builtin-admin/overrides",
|
||||
json={"grant": ["persona.write"], "revoke": []},
|
||||
)
|
||||
assert resp.status_code == 200, resp.json()
|
||||
body = resp.json()
|
||||
assert "persona.write" in body["effective"]
|
||||
assert body["grants"] == ["persona.write"]
|
||||
|
||||
def test_overrides_replace_semantics(self, client, storage):
|
||||
_seed_builtin_admin(storage, "read,write,admin.roles")
|
||||
client.put(
|
||||
|
||||
@@ -208,6 +208,16 @@ class TestRolePermissionOverrides:
|
||||
db.set_role_overrides("r1", {"approve", "model.skills.write"}, {"write"})
|
||||
assert db.get_user_permissions("u1") == {"read", "approve", "model.skills.write"}
|
||||
|
||||
def test_get_user_permissions_applies_persona_write_overlay(self, db):
|
||||
# persona.write is admin-default (migration 063), but the override layer
|
||||
# can grant it to any NON-admin builtin role — the grant must flow
|
||||
# through get_user_permissions like any other overlay perm.
|
||||
db.create_role("r1", "editor", "Editor", "read,write", builtin=True, org_id="")
|
||||
db.create_user("u1", "alice", "Alice", "$2b$hash")
|
||||
db.assign_role("u1", "r1")
|
||||
db.set_role_overrides("r1", {"persona.write"}, set())
|
||||
assert db.get_user_permissions("u1") == {"read", "write", "persona.write"}
|
||||
|
||||
def test_get_user_permissions_ignores_overlay_on_custom_role(self, db):
|
||||
# Overrides only apply to builtin rows. A custom role with stray
|
||||
# override rows (defensive case — should never happen via the API)
|
||||
|
||||
@@ -28,8 +28,10 @@ from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from tests._helpers import wait_until as _wait_until
|
||||
from tests.test_session_manager import FakeStorage
|
||||
from turnstone.core.idle_nudge_watcher import IdleNudgeWatcher
|
||||
from turnstone.core import session_worker
|
||||
from turnstone.core.idle_nudge_watcher import IdleNudgeWatcher, wake_workstream_if_pending
|
||||
from turnstone.core.session import ChatSession
|
||||
from turnstone.core.session_manager import SessionManager
|
||||
from turnstone.core.trajectory import dicts_from_turns, turn_from_dict
|
||||
@@ -299,6 +301,72 @@ def test_idle_event_with_empty_queue_does_not_dispatch_wake(real_mgr, tmp_db):
|
||||
watcher.shutdown()
|
||||
|
||||
|
||||
def test_watch_fire_on_already_idle_session_drives_wake_send(real_mgr, tmp_db):
|
||||
"""A watch firing on an ALREADY-idle workstream sees no IDLE
|
||||
transition, so :class:`IdleNudgeWatcher` never re-checks the queue —
|
||||
the dispatch closure's ``wake_fn`` must drive the wake itself.
|
||||
|
||||
Boundary path under test (only the LLM stream is patched):
|
||||
dispatch closure (real, built by ``set_watch_runner``)
|
||||
→ NudgeQueue.enqueue (real)
|
||||
→ wake_fn → wake_workstream_if_pending (real)
|
||||
→ session_worker.send (real) → daemon thread
|
||||
→ ChatSession.deliver_wake_nudge_from_queue (real)
|
||||
→ ChatSession.send("") → watch_triggered system turn in history
|
||||
"""
|
||||
mgr, _adapter = real_mgr
|
||||
ws = mgr.create(user_id="u1", name="watch-wake-int", skill=None)
|
||||
assert ws.session is not None
|
||||
|
||||
captured: dict[str, Any] = {}
|
||||
|
||||
class _StubRunner:
|
||||
def set_dispatch_fn(self, ws_id: str, fn: Any) -> None:
|
||||
captured["fn"] = fn
|
||||
|
||||
# Production wiring shape (server.py): wake_fn closes over the
|
||||
# Workstream OBJECT — not its id — so eviction+restore id drift
|
||||
# can't strand the wake.
|
||||
ws.session.set_watch_runner(
|
||||
_StubRunner(), wake_fn=lambda: wake_workstream_if_pending(ws, trigger="watch-fire")
|
||||
)
|
||||
|
||||
with (
|
||||
patch.object(ws.session, "_create_stream_with_retry", return_value=iter([])),
|
||||
patch.object(
|
||||
ws.session,
|
||||
"_stream_response",
|
||||
return_value={"role": "assistant", "content": "ok"},
|
||||
),
|
||||
patch.object(ws.session, "_update_token_table"),
|
||||
patch.object(ws.session, "_print_status_line"),
|
||||
patch.object(ws.session, "_visible_memory_count", return_value=0),
|
||||
patch("turnstone.core.session.save_message"),
|
||||
):
|
||||
ws.session._title_generated = True
|
||||
# Idle all along — no worker, and no state transition coming.
|
||||
assert ws.state is WorkstreamState.IDLE
|
||||
|
||||
# Simulate the WatchRunner poll thread delivering a fire.
|
||||
captured["fn"]({"type": "watch_triggered", "text": "deploy finished: OK"}, "watch-1")
|
||||
|
||||
_wait_for_worker_done(ws)
|
||||
|
||||
# Queue drained by the wake — not parked until the next user message.
|
||||
assert len(ws.session._nudge_queue) == 0
|
||||
|
||||
msgs = dicts_from_turns(ws.session.messages)
|
||||
user_msgs = [m for m in msgs if m.get("role") == "user"]
|
||||
assert user_msgs, "expected a synthesized user message from the wake"
|
||||
assert user_msgs[-1]["content"] == ""
|
||||
assert user_msgs[-1].get("_source") == "system_nudge"
|
||||
sys_turns = [m for m in msgs if m.get("role") == "system"]
|
||||
assert any(
|
||||
m.get("_source") == "watch_triggered" and "deploy finished: OK" in m.get("content", "")
|
||||
for m in sys_turns
|
||||
), f"expected a watch_triggered system turn, got {sys_turns!r}"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def coord_mgr() -> tuple[SessionManager, _BuildRealSessionAdapter, FakeStorage]:
|
||||
"""Real coord-side SessionManager with the adapter's kind set to
|
||||
@@ -411,3 +479,120 @@ def test_coord_idle_with_active_children_emits_envelope_via_real_managers(coord_
|
||||
finally:
|
||||
watcher.shutdown()
|
||||
observer.shutdown()
|
||||
|
||||
|
||||
def test_coord_idle_emitted_from_worker_thread_still_wakes(coord_mgr, tmp_db):
|
||||
"""The production-shaped race the test above does NOT exercise: in
|
||||
production, IDLE is emitted from INSIDE the worker (``set_state``
|
||||
subscribers fire on the calling thread — the coord's send emits IDLE
|
||||
before its worker exits). The watcher's wake dispatch therefore
|
||||
lands on ``session_worker.send``'s reuse path while the
|
||||
transitioning worker still owns the flag, and no-ops. Without the
|
||||
ownership-clear backstop the ``idle_children`` nudge strands until
|
||||
the next user message — a coord that forgot ``wait_for_workstream``
|
||||
never revives.
|
||||
|
||||
Boundary path under test:
|
||||
worker thread: mgr.set_state(IDLE)
|
||||
→ observer enqueues (real) → watcher wake no-ops (worker owns flag)
|
||||
→ run() returns → session_worker._runner finally clears the flag
|
||||
→ _retry_pending_wake → wake_workstream_if_pending (real)
|
||||
→ wake daemon → deliver_wake_nudge_from_queue → send("")
|
||||
→ idle_children system turn in history
|
||||
"""
|
||||
from turnstone.console.coordinator_idle_observer import CoordinatorIdleObserver
|
||||
from turnstone.core.workstream import WorkstreamKind as _Kind
|
||||
|
||||
mgr, adapter, storage = coord_mgr
|
||||
observer = CoordinatorIdleObserver(mgr, storage)
|
||||
observer.start()
|
||||
watcher = IdleNudgeWatcher(mgr)
|
||||
watcher.start()
|
||||
|
||||
try:
|
||||
coord = mgr.create(user_id="u1", name="parent-coord-2", skill=None)
|
||||
assert coord.session is not None
|
||||
|
||||
storage.register_workstream(
|
||||
"child-x",
|
||||
user_id="u1",
|
||||
name="crawl-docs",
|
||||
kind=_Kind.INTERACTIVE,
|
||||
parent_ws_id=coord.id,
|
||||
state="running",
|
||||
)
|
||||
|
||||
coord.session.messages.append(turn_from_dict({"role": "user", "content": "spawn 1"}))
|
||||
coord.session.messages.append(turn_from_dict({"role": "assistant", "content": "ok"}))
|
||||
|
||||
with (
|
||||
patch.object(coord.session, "_create_stream_with_retry", return_value=iter([])),
|
||||
patch.object(
|
||||
coord.session,
|
||||
"_stream_response",
|
||||
return_value={"role": "assistant", "content": "ack"},
|
||||
),
|
||||
patch.object(coord.session, "_full_messages", return_value=[]),
|
||||
patch.object(coord.session, "_update_token_table"),
|
||||
patch.object(coord.session, "_print_status_line"),
|
||||
patch.object(coord.session, "_visible_memory_count", return_value=0),
|
||||
patch("turnstone.core.session.save_message"),
|
||||
):
|
||||
coord.session._title_generated = True
|
||||
|
||||
# Drive the IDLE transition from INSIDE a session_worker
|
||||
# worker, as production does.
|
||||
ok = session_worker.send(
|
||||
coord,
|
||||
enqueue=lambda: None,
|
||||
run=lambda: mgr.set_state(coord.id, WorkstreamState.IDLE),
|
||||
thread_name="coord-send-sim",
|
||||
)
|
||||
assert ok is True
|
||||
# Without the backstop the queue never drains (the watcher's
|
||||
# transition-time wake no-opped against the sim worker) and
|
||||
# this poll times out. Queue-empty implies the wake worker's
|
||||
# drain ran, so the follow-up flag poll waits for ITS exit.
|
||||
_wait_until(lambda: len(coord.session._nudge_queue) == 0)
|
||||
_wait_for_worker_done(coord)
|
||||
|
||||
# Queue drained by the wake, not waiting on the next user message.
|
||||
assert len(coord.session._nudge_queue) == 0
|
||||
msgs = dicts_from_turns(coord.session.messages)
|
||||
user_msgs = [m for m in msgs if m.get("role") == "user"]
|
||||
wake_msg = user_msgs[-1]
|
||||
assert wake_msg["content"] == ""
|
||||
assert wake_msg.get("_source") == "system_nudge"
|
||||
idle_turns = [
|
||||
m for m in msgs if m.get("role") == "system" and m["_source"] == "idle_children"
|
||||
]
|
||||
assert len(idle_turns) == 1
|
||||
assert "crawl-docs" in idle_turns[0]["content"]
|
||||
assert "wait_for_workstream" in idle_turns[0]["content"]
|
||||
finally:
|
||||
watcher.shutdown()
|
||||
observer.shutdown()
|
||||
|
||||
|
||||
def test_wake_delivery_contains_generation_cancelled(tmp_db):
|
||||
"""A close/force-cancel racing the wake turn raises
|
||||
``GenerationCancelled`` (a BaseException) out of ``send("")`` — the
|
||||
wake method must contain it: it IS the wake worker's ``run()``
|
||||
closure, and ``session_worker._runner`` catches only ``Exception``,
|
||||
so an escape would land in ``threading.excepthook`` as stderr noise
|
||||
on every close-vs-wake race."""
|
||||
from tests._helpers import make_chat_session
|
||||
from turnstone.core.session import GenerationCancelled
|
||||
|
||||
session = make_chat_session()
|
||||
session._nudge_queue.enqueue("idle_children", "kids waiting", "any")
|
||||
|
||||
def _cancelled_send(*_a: Any, **_k: Any) -> None:
|
||||
raise GenerationCancelled
|
||||
|
||||
session.send = _cancelled_send # type: ignore[method-assign]
|
||||
|
||||
session.deliver_wake_nudge_from_queue() # must not raise
|
||||
|
||||
assert session._wake_source_tag == ""
|
||||
assert session._wake_drained_reminders is None
|
||||
|
||||
@@ -9,13 +9,14 @@ module-level function to capture calls without spawning real threads.
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import logging
|
||||
import threading
|
||||
from typing import Any
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from turnstone.core.idle_nudge_watcher import IdleNudgeWatcher
|
||||
from turnstone.core.idle_nudge_watcher import IdleNudgeWatcher, wake_workstream_if_pending
|
||||
from turnstone.core.nudge_queue import NudgeQueue
|
||||
from turnstone.core.workstream import WorkstreamState
|
||||
|
||||
@@ -32,6 +33,7 @@ class _FakeSession:
|
||||
class _FakeWorkstream:
|
||||
def __init__(self, ws_id: str = "ws-test") -> None:
|
||||
self.id = ws_id
|
||||
self.state = WorkstreamState.IDLE
|
||||
self.session: _FakeSession | None = _FakeSession()
|
||||
self._lock = threading.Lock()
|
||||
self._worker_running = False
|
||||
@@ -163,3 +165,154 @@ class TestIdleNudgeWatcher:
|
||||
watcher.start()
|
||||
watcher.shutdown()
|
||||
watcher.shutdown() # no error
|
||||
|
||||
|
||||
class TestWakeWorkstreamIfPending:
|
||||
"""Direct tests for the shared wake gate.
|
||||
|
||||
The IDLE-transition path (via the watcher) is covered above; these
|
||||
pin the gates the watch dispatch closure relies on when it calls
|
||||
the helper directly, with no state event involved.
|
||||
"""
|
||||
|
||||
def test_wakes_idle_ws_with_pending_entry(self, fake_mgr_and_ws):
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
with patch("turnstone.core.session_worker.send", return_value=True) as mock_send:
|
||||
assert wake_workstream_if_pending(ws) is True
|
||||
assert mock_send.call_count == 1
|
||||
kwargs = mock_send.call_args.kwargs
|
||||
assert kwargs["enqueue"]() is None
|
||||
kwargs["run"]()
|
||||
assert ws.session.deliver_wake_nudge_from_queue_called == 1
|
||||
assert kwargs["thread_name"].startswith("wake-nudge-")
|
||||
|
||||
def test_skips_session_none(self, fake_mgr_and_ws):
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session = None
|
||||
with patch("turnstone.core.session_worker.send") as mock_send:
|
||||
assert wake_workstream_if_pending(ws) is False
|
||||
assert mock_send.call_count == 0
|
||||
|
||||
def test_skips_closed_ws(self, fake_mgr_and_ws):
|
||||
"""A workstream mid-``close()`` must not get a wake spawned on
|
||||
its torn-down session, even while its ``state`` field still
|
||||
reads IDLE (there is no CLOSED member — close uses the
|
||||
``_closed`` tombstone)."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
ws._closed = True
|
||||
with patch("turnstone.core.session_worker.send") as mock_send:
|
||||
assert wake_workstream_if_pending(ws) is False
|
||||
assert mock_send.call_count == 0
|
||||
|
||||
def test_skips_non_idle_states(self, fake_mgr_and_ws):
|
||||
"""Busy states imply a live worker that drains at its own seams;
|
||||
ERROR stays parked for the operator — neither gets a wake."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
with patch("turnstone.core.session_worker.send") as mock_send:
|
||||
for state in (
|
||||
WorkstreamState.RUNNING,
|
||||
WorkstreamState.THINKING,
|
||||
WorkstreamState.ATTENTION,
|
||||
WorkstreamState.ERROR,
|
||||
):
|
||||
ws.state = state
|
||||
assert wake_workstream_if_pending(ws) is False
|
||||
assert mock_send.call_count == 0
|
||||
|
||||
def test_skips_tool_only_entries(self, fake_mgr_and_ws):
|
||||
"""Tool-channel entries belong to the next tool-result seam — a
|
||||
synthetic empty user turn can't drain them, so no wake."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("tool_error", "check memories", "tool")
|
||||
with patch("turnstone.core.session_worker.send") as mock_send:
|
||||
assert wake_workstream_if_pending(ws) is False
|
||||
assert mock_send.call_count == 0
|
||||
|
||||
def test_refuses_non_nudgequeue_stub(self, fake_mgr_and_ws):
|
||||
"""The gate refuses on TYPE, not just presence: a mock session's
|
||||
auto-created ``_nudge_queue`` answers ``has_pending`` truthily
|
||||
while its ``deliver_wake_nudge_from_queue`` consumes nothing —
|
||||
with the worker-exit backstop re-running this gate after every
|
||||
exit, one worker on such a session would respawn wake workers
|
||||
forever (the storm that took down the full-suite CI run). Only
|
||||
a real :class:`NudgeQueue` carries the drain semantics the wake
|
||||
contract needs."""
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue = MagicMock() # truthy has_pending, no real drain
|
||||
with patch("turnstone.core.session_worker.send") as mock_send:
|
||||
assert wake_workstream_if_pending(ws) is False
|
||||
assert mock_send.call_count == 0
|
||||
|
||||
def test_dispatched_path_logs_trigger(self, fake_mgr_and_ws, caplog):
|
||||
"""A fresh spawn — ``send`` returns True without touching the
|
||||
passed ``enqueue`` — emits ``nudge_wake.dispatched`` tagged with
|
||||
the trigger label (structlog renders the event name + ``%s``
|
||||
placeholders into ``msg``; substring-match like the sibling
|
||||
nudge_queue tests)."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
with (
|
||||
patch("turnstone.core.session_worker.send", return_value=True) as mock_send,
|
||||
caplog.at_level(logging.INFO, logger="turnstone.core.idle_nudge_watcher"),
|
||||
):
|
||||
assert wake_workstream_if_pending(ws, trigger="idle-transition") is True
|
||||
assert mock_send.call_count == 1
|
||||
dispatched = [r for r in caplog.records if "nudge_wake.dispatched" in r.getMessage()]
|
||||
assert len(dispatched) == 1
|
||||
assert dispatched[0].levelno == logging.INFO
|
||||
assert "trigger=" in dispatched[0].getMessage()
|
||||
# The reuse-path drop line must not appear on a fresh spawn.
|
||||
assert not any("nudge_wake.deferred_worker_busy" in r.getMessage() for r in caplog.records)
|
||||
|
||||
def test_deferred_path_logs_worker_busy(self, fake_mgr_and_ws, caplog):
|
||||
"""The reuse path — ``send`` invokes the passed ``enqueue`` and
|
||||
returns True — emits ``nudge_wake.deferred_worker_busy`` instead
|
||||
of ``dispatched``. The entry stays owed to the owning worker's
|
||||
exit backstop; the return value is still True."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
|
||||
def _reuse_send(_ws: Any, *, enqueue: Any, run: Any, thread_name: Any) -> bool:
|
||||
# Mimic a live worker owning the workstream: send routes the
|
||||
# wake to the no-op enqueue rather than spawning a daemon.
|
||||
enqueue()
|
||||
return True
|
||||
|
||||
with (
|
||||
patch("turnstone.core.session_worker.send", side_effect=_reuse_send) as mock_send,
|
||||
caplog.at_level(logging.INFO, logger="turnstone.core.idle_nudge_watcher"),
|
||||
):
|
||||
assert wake_workstream_if_pending(ws, trigger="idle-transition") is True
|
||||
assert mock_send.call_count == 1
|
||||
deferred = [
|
||||
r for r in caplog.records if "nudge_wake.deferred_worker_busy" in r.getMessage()
|
||||
]
|
||||
assert len(deferred) == 1
|
||||
assert deferred[0].levelno == logging.INFO
|
||||
assert "trigger=" in deferred[0].getMessage()
|
||||
assert not any("nudge_wake.dispatched" in r.getMessage() for r in caplog.records)
|
||||
|
||||
def test_refused_path_logs_refusal(self, fake_mgr_and_ws, caplog):
|
||||
"""``send`` refusing outright — its authoritative under-lock
|
||||
``_closed`` re-check caught a teardown the gate's lockless peek
|
||||
missed — emits ``nudge_wake.refused``: a dropped wake must stay
|
||||
traceable to its trigger, not vanish silently."""
|
||||
_mgr, ws = fake_mgr_and_ws
|
||||
ws.session._nudge_queue.enqueue("watch_triggered", "output", "any")
|
||||
with (
|
||||
patch("turnstone.core.session_worker.send", return_value=False) as mock_send,
|
||||
caplog.at_level(logging.INFO, logger="turnstone.core.idle_nudge_watcher"),
|
||||
):
|
||||
assert wake_workstream_if_pending(ws, trigger="watch-fire") is False
|
||||
assert mock_send.call_count == 1
|
||||
refused = [r for r in caplog.records if "nudge_wake.refused" in r.getMessage()]
|
||||
assert len(refused) == 1
|
||||
assert refused[0].levelno == logging.INFO
|
||||
assert "trigger=" in refused[0].getMessage()
|
||||
assert not any("nudge_wake.dispatched" in r.getMessage() for r in caplog.records)
|
||||
assert not any("nudge_wake.deferred_worker_busy" in r.getMessage() for r in caplog.records)
|
||||
|
||||
@@ -15,6 +15,8 @@ from pathlib import Path
|
||||
|
||||
_ROOT = Path(__file__).resolve().parent.parent
|
||||
_INTERACTIVE = _ROOT / "turnstone/shared_static/interactive.js"
|
||||
_COMPOSER = _ROOT / "turnstone/shared_static/composer.js"
|
||||
_AUTH = _ROOT / "turnstone/shared_static/auth.js"
|
||||
_APP = _ROOT / "turnstone/ui/static/app.js"
|
||||
_UI_INDEX = _ROOT / "turnstone/ui/static/index.html"
|
||||
|
||||
@@ -245,3 +247,392 @@ def test_controller_terminal_dead_state() -> None:
|
||||
assert "base: base," in body, "the controller must expose its transport base"
|
||||
# Dead controllers don't reconnect on re-auth.
|
||||
assert "if (connected && !dead) pane._loadHistoryThenConnect(wsId);" in body
|
||||
|
||||
|
||||
def test_stream_pipeline_is_wedge_proof() -> None:
|
||||
"""Long-session hardening (perf audit P0): the SSE pipeline must not be
|
||||
able to permanently wedge the pane. ``onmessage`` guards BOTH the
|
||||
``JSON.parse`` and the ``handleEvent`` dispatch (an exception escaping it
|
||||
doesn't close the EventSource, so an unguarded throw left the streaming
|
||||
refs poisoned for the rest of the session), and ``stream_end`` resets the
|
||||
segment refs BEFORE the finalize render, with a plain-text fallback —
|
||||
with the old order a finalize throw skipped the clears and every later
|
||||
delta painted into the dead segment."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
assert "dropping malformed SSE frame" in body
|
||||
assert "handleEvent failed for" in body
|
||||
case = body.index('case "stream_end"')
|
||||
seg = body[case : body.index("break;", case)]
|
||||
clears = seg.index("this.currentAssistantBodyEl = null;")
|
||||
finalize = seg.index("streamingRenderFinalize(")
|
||||
assert clears < finalize, (
|
||||
"stream_end must clear segment refs BEFORE finalize — the old "
|
||||
"finalize-first order wedged all later assistant output on a throw."
|
||||
)
|
||||
assert "doneBodyEl.textContent = doneBuffer;" in seg
|
||||
|
||||
|
||||
def test_rebuild_quiesces_live_events_and_releases_agent_tracking() -> None:
|
||||
"""clear_ui / replay_truncated re-render race (perf audit P0): live SSE
|
||||
events painted between the history snapshot and ``replaceChildren()``
|
||||
were wiped with no redelivery, and streaming refs kept pointing at
|
||||
detached nodes. Pinned: the quiesce queue sits on the handleEvent hot
|
||||
path, both re-render triggers arm it, ``replayHistory`` resets the
|
||||
streaming refs and clears the agent-card/orphan maps (the detached-DOM
|
||||
retention leak), and the mid-stream guard covers the reasoning bubble."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
assert "this._replayQueue.events.push(evt);" in body
|
||||
assert body.count("this._beginReplayQuiesce(") >= 2, (
|
||||
"both clear_ui and replay_truncated must arm the quiesce"
|
||||
)
|
||||
assert "!this.currentAssistantEl && !this.currentReasoningEl" in body
|
||||
replay = body.index("replayHistory(messages) {")
|
||||
seg = body[replay : replay + 1600]
|
||||
for line in (
|
||||
"this._resetStreamingRefs();",
|
||||
"this._clearAgentTracking();",
|
||||
):
|
||||
assert line in seg, f"replayHistory must reset: {line!r}"
|
||||
assert "this._agentCards.clear();" in body
|
||||
# Review-hardened lifecycle: the card entry SURVIVES the terminal
|
||||
# tool_result (a late child event finding no Map entry would rebuild a
|
||||
# duplicate empty card beside the finished one), and transport-only
|
||||
# reconnects preserve the maps + any armed quiesce queue — clearing them
|
||||
# in disconnectSSE duplicated cards and dropped buffered orphan steps on
|
||||
# every transient stream blip. Full-reload cleanup lives in
|
||||
# _loadHistoryThenConnect; terminal cleanup in the factory's destroy().
|
||||
assert "this._agentCards.delete(callId);" not in body
|
||||
disc = body.index("disconnectSSE() {")
|
||||
disc_seg = body[disc : body.index("_loadHistoryThenConnect(wsId) {", disc)]
|
||||
assert "this._clearAgentTracking();" not in disc_seg
|
||||
assert "this._replayQueue = null;" not in disc_seg
|
||||
load = body.index("_loadHistoryThenConnect(wsId) {")
|
||||
load_seg = body[load : load + 2200]
|
||||
assert "this._clearAgentTracking();" in load_seg
|
||||
assert "this._replayQueue = null;" in load_seg
|
||||
# A mid-stream replay_truncated DEFERS the re-sync (flag consumed on the
|
||||
# idle edge) instead of dropping it — skipping left the lost-event gap
|
||||
# unrepaired for the rest of the session.
|
||||
assert "this._pendingTruncatedResync = true;" in body
|
||||
# The refetch FAILURE branch resets streaming refs too — it never reaches
|
||||
# replayHistory, and stale refs there streamed the retried generation's
|
||||
# first segment into a detached bubble.
|
||||
fail = body.index("Failure path never reaches replayHistory")
|
||||
assert "this._resetStreamingRefs();" in body[fail : fail + 400], (
|
||||
"the refetch failure branch must reset streaming refs"
|
||||
)
|
||||
|
||||
|
||||
def test_per_token_hot_path_avoids_container_scans() -> None:
|
||||
"""P1 (perf audit): per-token work must stay O(1) in transcript length.
|
||||
The thinking indicator is an instance ref (the class-selector miss walked
|
||||
the whole transcript on EVERY content/reasoning delta); near-bottom state
|
||||
comes from the passive scroll listener instead of a forced-layout
|
||||
geometry read per event; the scroll pin is rAF-coalesced; per-tool
|
||||
row/stream lookups resolve through the self-healing caches."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
stripped = _strip_comments(body)
|
||||
assert 'querySelector(".thinking-indicator")' not in stripped, (
|
||||
"thinking indicator must use the instance ref, not a container scan"
|
||||
)
|
||||
assert "this._thinkingEl" in body
|
||||
near = body.index("isNearBottom() {")
|
||||
assert "return this._nearBottom;" in body[near : near + 700]
|
||||
assert "passive: true" in body
|
||||
# The rAF pin re-checks the flag AT FIRE TIME (a user scroll landing in
|
||||
# the schedule→rAF window must win over a stale pin), with force
|
||||
# requests latched across the coalescing window; resizes re-derive the
|
||||
# flag via ResizeObserver since they move the bottom without a scroll.
|
||||
assert "this._scrollPinForce = false;" in body
|
||||
assert "ResizeObserver" in body
|
||||
for helper in ("_toolRow(callId) {", "_streamEl(callId) {"):
|
||||
assert helper in body, f"missing lookup-cache helper: {helper!r}"
|
||||
|
||||
|
||||
# -- Shared-workstream cross-user send gate -----------------------------------
|
||||
#
|
||||
# The UX complement to the server-side CrossUserInterjectionError (a 409): while
|
||||
# another participant's turn is in flight, this viewer's send button is disabled
|
||||
# so they can't interject under the initiator's credentials / be misattributed.
|
||||
# The wiring spans three modules; these string-presence guards catch the silent
|
||||
# one-line regression the way the rest of this file does (no JS test framework).
|
||||
|
||||
|
||||
def test_composer_exposes_hard_send_block() -> None:
|
||||
"""The composer has an independent hard-block axis, reconciled with busy,
|
||||
so a caller can disable send even in queueWhileBusy (queue) mode."""
|
||||
body = _COMPOSER.read_text(encoding="utf-8")
|
||||
assert "Composer.prototype.setSendBlocked = function" in body
|
||||
assert "Composer.prototype._reconcileDisabled = function" in body
|
||||
assert "this._sendBlocked = false;" in body
|
||||
# setBusy must route the disabled write through the reconciler (not clobber
|
||||
# the block with a direct sendBtn.disabled assignment).
|
||||
stripped = _strip_comments(body)
|
||||
setbusy = stripped.index("Composer.prototype.setBusy = function")
|
||||
setbusy_end = stripped.index("Composer.prototype._reconcileDisabled")
|
||||
assert "this._reconcileDisabled();" in stripped[setbusy:setbusy_end]
|
||||
assert "this.sendBtn.disabled =" not in stripped[setbusy:setbusy_end], (
|
||||
"setBusy must not write sendBtn.disabled directly — reconcile owns it"
|
||||
)
|
||||
|
||||
|
||||
def test_auth_retains_user_id_for_gate() -> None:
|
||||
"""whoami's opaque user_id is retained (separately from the display
|
||||
username) so the pane can compare it against the acting-user id."""
|
||||
body = _AUTH.read_text(encoding="utf-8")
|
||||
assert 'sessionStorage.setItem("ts.user_id", data.user_id);' in body
|
||||
assert 'sessionStorage.removeItem("ts.user_id");' in body
|
||||
|
||||
|
||||
def test_pane_gates_send_on_cross_user_busy() -> None:
|
||||
"""The pane tracks the acting user from state_change, compares it against
|
||||
the viewer's own id, and blocks send while another participant is busy."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
assert "_reconcileSendBlock() {" in body
|
||||
# tracks the acting user from the state_change event...
|
||||
assert "this._actingUserId = evt.acting_user_id;" in body
|
||||
assert "this._actingUserId = null;" in body # cleared when the turn settles
|
||||
# ...compares against the viewer's own id from /whoami...
|
||||
assert 'sessionStorage.getItem("ts.user_id")' in body
|
||||
assert "this._actingUserId !== me" in body
|
||||
# ...and drives the composer's hard block, re-run on every busy edge.
|
||||
assert "this.composer.setSendBlocked(" in body
|
||||
stripped = _strip_comments(body)
|
||||
setbusy = stripped.index("setBusy(b) {")
|
||||
assert "this._reconcileSendBlock();" in stripped[setbusy : setbusy + 600]
|
||||
|
||||
|
||||
def test_pane_handles_cross_user_409() -> None:
|
||||
"""The reactive fallback: a 409 (button not yet disabled) surfaces a clean
|
||||
message, not the generic 'Connection error' catch."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
assert "r.status === 409" in body
|
||||
assert 'status: "cross_user_interjection"' in body
|
||||
assert 'data.status === "cross_user_interjection"' in body
|
||||
|
||||
|
||||
def test_sync_approval_state_prunes_orphan_cycles() -> None:
|
||||
"""``_syncApprovalState`` prunes cycles whose block elements are no longer
|
||||
in the living DOM (``.isConnected === false``). This covers the rare case
|
||||
where an ``approve_request`` event is processed between a DOM wipe
|
||||
(``clear_ui`` / ``replay_truncated`` / ``replaceChildren``) and the
|
||||
refetch-restore — the cycle card lives in a detached subtree, the matching
|
||||
``approval_resolved`` never arrives, and the send button stays disabled
|
||||
forever without this guard. The pin guards against a future refactor that
|
||||
drops the orphan prune but doesn't otherwise break ``_syncApprovalState``."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
fn_start = body.index("_syncApprovalState() {")
|
||||
assert "entry.blockEls && !entry.blockEls.some((el) => el.isConnected)" in body, (
|
||||
"orphan pruning must check .isConnected on block elements"
|
||||
)
|
||||
tail = body[fn_start : body.index("_oldestCycleId()", fn_start)]
|
||||
assert "this.approvalCycles.delete(cid);" in tail, (
|
||||
"orphan pruning must delete the cycle from the Map"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# SSE overflow recovery + close-on-hide (fast-stream corruption fixes)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_stream_overflow_case_counts_and_rate_limits() -> None:
|
||||
"""The server closes an overflowed stream after an id-less
|
||||
``stream_overflow`` frame; the pane must count it (field
|
||||
instrumentation for the drop-vs-render-wedge diagnosis) and route it
|
||||
through the reconnect limiter so a persistently slow consumer trips
|
||||
the degraded catch-up instead of churning reconnect/replay cycles."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
assert 'case "stream_overflow":' in body
|
||||
assert "this._noteStreamOverflow();" in body
|
||||
assert "_streamHealth = { overflows: 0, renderThrows: 0, malformedFrames: 0 }" in body
|
||||
# Both wedge-class catch sites increment the render-throw counter,
|
||||
# and the malformed-frame drop counts too — the C-OVERDETERMINED
|
||||
# instrumentation that tells drops apart from wedges in the field.
|
||||
assert body.count("this._streamHealth.renderThrows += 1;") == 2
|
||||
assert "this._streamHealth.malformedFrames += 1;" in body
|
||||
assert "this._streamHealth.overflows += 1;" in body
|
||||
|
||||
|
||||
def test_degraded_catchup_stops_live_stream_and_retries() -> None:
|
||||
"""Degraded catch-up contract: close the stream FIRST (which also
|
||||
clears any earlier degraded timer — disconnectSSE owns that), show a
|
||||
plain-language status, then arm the retry timer with a doubling
|
||||
cooldown. The retry must defer to the show edge when the tab is
|
||||
hidden (reopening into a throttled tab would overflow again)."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
m = re.search(r"_enterDegradedCatchup\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert m is not None, "_enterDegradedCatchup method not found"
|
||||
method = m.group(1)
|
||||
# Order matters: disconnect before arming the timer, or the fresh
|
||||
# timer would be cancelled by its own disconnect.
|
||||
assert method.index("this.disconnectSSE()") < method.index("this._degradedTimer = setTimeout")
|
||||
assert "Connection is slow" in method, "degraded state must use plain language"
|
||||
assert "DEGRADED_COOLDOWN_MAX_MS" in method
|
||||
assert "document.hidden" in method
|
||||
# disconnectSSE owns the timer teardown (ws-switch / giveUp / destroy
|
||||
# all supersede a pending degraded retry through it).
|
||||
dis = re.search(r"disconnectSSE\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert dis is not None
|
||||
assert "clearTimeout(this._degradedTimer)" in dis.group(1)
|
||||
|
||||
|
||||
def test_visibilitychange_closes_on_hide_reconnects_on_show() -> None:
|
||||
"""Close-on-hide / replay-on-show: a hidden tab's throttled drain is
|
||||
the likeliest slow consumer behind server-side overflow (the old
|
||||
"PR-G closes those connections on hide" comment described a handler
|
||||
that never existed). The pane installs one visibilitychange
|
||||
listener, marks ITS OWN hide-closes via ``_hiddenDisconnect`` so a
|
||||
show edge never resurrects a deliberately-closed stream, and the
|
||||
factory's destroy removes the listener (it strongly references the
|
||||
pane)."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
assert 'document.addEventListener("visibilitychange", this._visHandler);' in body
|
||||
assert 'document.removeEventListener("visibilitychange", this._visHandler);' in body
|
||||
vis = re.search(r"_onVisibilityChange\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert vis is not None, "_onVisibilityChange method not found"
|
||||
method = vis.group(1)
|
||||
assert "this.disconnectSSE();" in method
|
||||
assert "this._hiddenDisconnect = true;" in method
|
||||
assert "this.connectSSE(this.wsId);" in method
|
||||
# Reconnect only consumes OUR hide-close marker.
|
||||
assert "else if (this._hiddenDisconnect)" in method
|
||||
# Teardown: the factory controller removes the listener on destroy.
|
||||
assert "pane._removeVisibilityHandler();" in body
|
||||
# The streaming buffers survive a hide-close: disconnectSSE stays
|
||||
# transport-only (no contentBuffer wipe) so the visible tail is
|
||||
# intact when the tab returns.
|
||||
dis = re.search(r"disconnectSSE\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert dis is not None
|
||||
assert "contentBuffer" not in dis.group(1)
|
||||
|
||||
|
||||
def test_no_global_sse_gap_detector() -> None:
|
||||
"""Live event ids are NOT strictly monotonic across concurrent
|
||||
tool+content emit (the fan-out runs outside the listeners lock), so
|
||||
a naive ``id !== lastEventId + 1`` gap check would false-positive.
|
||||
Recovery is server-signalled (``stream_overflow``) + reconnect
|
||||
replay instead. This tripwire pins the absence of the naive
|
||||
arithmetic — if gap detection is ever added, it must be scoped to
|
||||
the content stream only (content-vs-content never reorders)."""
|
||||
code = _strip_comments(_INTERACTIVE.read_text(encoding="utf-8"))
|
||||
assert not re.search(r"_lastEventId\s*[+\-]\s*1", code), (
|
||||
"found lastEventId +/- 1 arithmetic — a global gap detector "
|
||||
"false-positives on legal concurrent tool/content id inversion"
|
||||
)
|
||||
|
||||
|
||||
def test_overflow_helpers_extracted_to_shared_module() -> None:
|
||||
"""The storm-guard constants + the two pure helpers were extracted to the
|
||||
shared ``sse_overflow.js`` module (its own runtime probes live in
|
||||
``test_sse_overflow_js.py``) so the interactive and coordinator panes can't
|
||||
drift. Pin that the pane IMPORTS them rather than re-declaring a local
|
||||
copy: a stray local ``function overflowWindowTripped`` / ``const
|
||||
OVERFLOW_TRIP_COUNT`` would silently fork the trip math again."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
m = re.search(
|
||||
r"import \{([^}]*)\} from \"\./sse_overflow\.js\";",
|
||||
body,
|
||||
re.S,
|
||||
)
|
||||
assert m is not None, "interactive pane must import the shared overflow helpers"
|
||||
imported = m.group(1)
|
||||
for name in (
|
||||
"OVERFLOW_TRIP_COUNT",
|
||||
"OVERFLOW_TRIP_WINDOW_MS",
|
||||
"DEGRADED_COOLDOWN_BASE_MS",
|
||||
"DEGRADED_COOLDOWN_MAX_MS",
|
||||
"DEGRADED_COOLDOWN_RESET_MS",
|
||||
"overflowWindowTripped",
|
||||
"degradedCooldownStep",
|
||||
):
|
||||
assert name in imported, f"{name} must be imported from sse_overflow.js"
|
||||
# No local fork of the extracted definitions.
|
||||
assert not re.search(r"^function overflowWindowTripped\(", body, re.M), (
|
||||
"overflowWindowTripped must be imported, not re-declared locally"
|
||||
)
|
||||
assert not re.search(r"^function degradedCooldownStep\(", body, re.M), (
|
||||
"degradedCooldownStep must be imported, not re-declared locally"
|
||||
)
|
||||
assert not re.search(r"^const OVERFLOW_TRIP_COUNT\s*=", body, re.M), (
|
||||
"the trip constants must be imported, not re-declared locally"
|
||||
)
|
||||
|
||||
|
||||
def test_note_stream_overflow_does_not_reset_cooldown() -> None:
|
||||
"""The exact finding [0] bug shape must not regress: _noteStreamOverflow
|
||||
only counts + trips; it must NOT touch _degradedCooldownMs (the reset
|
||||
that defeated the ladder lived here). The ladder decision lives solely
|
||||
in _enterDegradedCatchup, keyed off _lastDegradedAt via
|
||||
degradedCooldownStep."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
note = re.search(r"_noteStreamOverflow\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert note is not None, "_noteStreamOverflow not found"
|
||||
assert "_degradedCooldownMs" not in note.group(1), (
|
||||
"_noteStreamOverflow must not write _degradedCooldownMs — that reset "
|
||||
"was the bug that stopped the ladder escalating"
|
||||
)
|
||||
enter = re.search(r"_enterDegradedCatchup\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert enter is not None
|
||||
assert "degradedCooldownStep(" in enter.group(1)
|
||||
assert "this._lastDegradedAt = now" in enter.group(1)
|
||||
|
||||
|
||||
def test_recover_beat_defers_reconnect_when_tab_hidden() -> None:
|
||||
"""Review round-2 finding [1]: the factory's transient-error recovery
|
||||
beat (recoverTimer) must NOT reopen an EventSource into a hidden tab —
|
||||
that re-creates the throttled slow-consumer overflow that close-on-hide
|
||||
exists to prevent. It guards on document.hidden and defers to the
|
||||
visibilitychange show edge (marking _hiddenDisconnect)."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
beat = re.search(r"recoverTimer = setTimeout\(\(\) => \{(.*?)\n \}, 5000\);", body, re.S)
|
||||
assert beat is not None, "recoverTimer setTimeout body not found"
|
||||
b = beat.group(1)
|
||||
assert "document.hidden" in b, "recovery beat must guard on document.hidden"
|
||||
assert "pane._hiddenDisconnect = true" in b, (
|
||||
"recovery beat must defer to the show edge when hidden"
|
||||
)
|
||||
# The hidden guard must precede the reconnect (connectSSE) so it can't fall
|
||||
# through to reopening the stream.
|
||||
assert b.index("document.hidden") < b.index("pane.connectSSE(pane.wsId)")
|
||||
|
||||
|
||||
def test_giveup_removes_visibility_handler() -> None:
|
||||
"""Review round-2 finding [3]: giveUp() (markDead) must detach the
|
||||
visibility handler and clear _hiddenDisconnect, or a tab hidden before
|
||||
the give-up resurrects the dead controller's stream on return (the show
|
||||
edge would connectSSE the closed ws and 404-reconnect it forever)."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
give = re.search(r"const giveUp = function \(\) \{(.*?)\n \};", body, re.S)
|
||||
assert give is not None, "giveUp function body not found"
|
||||
g = give.group(1)
|
||||
assert "pane._removeVisibilityHandler();" in g, (
|
||||
"giveUp must remove the visibility handler so a show edge can't resurrect a dead controller"
|
||||
)
|
||||
# _removeVisibilityHandler also clears _hiddenDisconnect (pinned in its body).
|
||||
rvh = re.search(r"_removeVisibilityHandler\(\)\s*\{(.*?)\n \}", body, re.S)
|
||||
assert rvh is not None
|
||||
assert "this._hiddenDisconnect = false" in rvh.group(1)
|
||||
|
||||
|
||||
def test_connectsse_defers_open_when_tab_hidden() -> None:
|
||||
"""PR #805 review (Copilot + R3): connectSSE is the single connect
|
||||
chokepoint and must not open an EventSource into a hidden tab. The
|
||||
fresh-connect path (_loadHistoryThenConnect) has no timer guard, so a
|
||||
first load in a background tab would otherwise open a throttled stream —
|
||||
the slow-consumer overflow this PR exists to prevent. The guard sits
|
||||
AFTER the visibilitychange-handler install (so the show edge can
|
||||
reconnect) and AFTER the wsId assignment (so it targets the right ws),
|
||||
and BEFORE `new EventSource` (so nothing opens)."""
|
||||
body = _INTERACTIVE.read_text(encoding="utf-8")
|
||||
start = body.index("connectSSE(wsId) {")
|
||||
open_at = body.index("new EventSource(evtUrl)", start)
|
||||
head = body[start:open_at] # connectSSE up to the EventSource open
|
||||
assert "if (document.hidden) {" in head, (
|
||||
"connectSSE must guard on document.hidden BEFORE opening the stream"
|
||||
)
|
||||
assert "this._hiddenDisconnect = true;" in head, (
|
||||
"the deferred connect must mark _hiddenDisconnect so the show edge reconnects"
|
||||
)
|
||||
assert head.index("this.wsId = wsId;") < head.index("if (document.hidden) {")
|
||||
assert head.index('addEventListener("visibilitychange"') < head.index("if (document.hidden) {")
|
||||
|
||||
@@ -476,6 +476,74 @@ class TestContextPreparation:
|
||||
assert "Conversation context:" in result[1]["content"]
|
||||
|
||||
|
||||
class TestArgBudget:
|
||||
"""The projected ``func_args`` and the conversation transcript share the
|
||||
judge model's context window; large arguments are honestly truncated to it
|
||||
rather than blind-capped."""
|
||||
|
||||
def test_positive_window_coerces_zero_and_non_int(self):
|
||||
from turnstone.core.judge import _DEFAULT_JUDGE_CONTEXT_WINDOW, _positive_window
|
||||
|
||||
assert _positive_window(50_000) == 50_000
|
||||
assert _positive_window(0, 40_000) == 40_000 # 0 falls through to next
|
||||
assert _positive_window(None, 0, 32_000) == 32_000 # None + 0 fall through
|
||||
assert _positive_window(-5, floor=1_000) == 1_000
|
||||
assert _positive_window(0) == _DEFAULT_JUDGE_CONTEXT_WINDOW # floor default
|
||||
|
||||
def test_honest_truncate_verbatim_when_it_fits(self):
|
||||
from turnstone.core.judge import honest_truncate
|
||||
|
||||
assert honest_truncate("short", 100) == "short"
|
||||
|
||||
def test_honest_truncate_reports_exact_omitted_count(self):
|
||||
from turnstone.core.judge import honest_truncate
|
||||
|
||||
out = honest_truncate("A" * 5000, 1000)
|
||||
assert out.startswith("A" * 1000)
|
||||
assert "4,000 of 5,000 chars omitted" in out
|
||||
|
||||
def test_arg_budget_scales_with_context_window_uncapped(self):
|
||||
"""The judge-prompt budget scales with the real window and is NOT
|
||||
ceilinged — a big-window judge gets a proportionally big budget so args
|
||||
lower whole; only a genuine overflow truncates."""
|
||||
from turnstone.core.judge import _ARG_CONTEXT_RATIO, _CHARS_PER_TOKEN
|
||||
|
||||
judge = _make_judge()
|
||||
judge._judge_context_window = 40_000
|
||||
small = judge.arg_budget_chars()
|
||||
judge._judge_context_window = 200_000
|
||||
big = judge.arg_budget_chars()
|
||||
assert small == int(40_000 * _ARG_CONTEXT_RATIO * _CHARS_PER_TOKEN)
|
||||
assert big == int(200_000 * _ARG_CONTEXT_RATIO * _CHARS_PER_TOKEN) # no ceiling
|
||||
|
||||
def test_verdict_record_copy_is_capped_by_oh_crap_backstop(self):
|
||||
"""The func_args stored on the verdict (persisted + streamed) is bounded
|
||||
by _VERDICT_ARG_CAP even when the args are enormous — the judge PROMPT
|
||||
is bounded separately by the window, not by this cap."""
|
||||
from turnstone.core.judge import _VERDICT_ARG_CAP, evaluate_heuristic
|
||||
|
||||
v = evaluate_heuristic("write_file", {"content": "Z" * 40_000}, "write_file", "c1")
|
||||
assert len(v.func_args) <= _VERDICT_ARG_CAP + 80 # payload + honest marker
|
||||
assert "chars omitted" in v.func_args
|
||||
|
||||
def test_large_args_shrink_the_history_they_share_the_window_with(self):
|
||||
"""A big write/edit must eat into the transcript budget, not push the
|
||||
prompt past the window."""
|
||||
judge = _make_judge()
|
||||
# One anchor user turn (the judge trims to the last user message
|
||||
# onward), then many assistant turns that compete for the budget.
|
||||
messages: list[dict[str, Any]] = [{"role": "user", "content": "anchor"}]
|
||||
messages += [{"role": "assistant", "content": "x" * 1000} for _ in range(50)]
|
||||
|
||||
small = judge._prepare_context(_make_item(func_args={"command": "ls"}), messages)
|
||||
big = judge._prepare_context(
|
||||
_make_item(func_name="write_file", func_args={"content": "Z" * 200_000}), messages
|
||||
)
|
||||
# Each included history turn renders one "ASSISTANT:" line; the
|
||||
# big-argument call fits strictly fewer of them.
|
||||
assert big[1]["content"].count("ASSISTANT:") < small[1]["content"].count("ASSISTANT:")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Confidence arbitration
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -875,6 +943,48 @@ class TestModelAliasResolution:
|
||||
assert judge._client_factory_args["api_key"] == "alias-key"
|
||||
assert judge._client_factory_args["provider_name"] == "openai"
|
||||
|
||||
def test_alias_window_comes_from_registry_config_not_provider_caps(self):
|
||||
"""The judge window must come from the registry's ModelConfig
|
||||
(cfg.context_window=50_000 here), NOT provider.get_capabilities(), which
|
||||
returns a static 200000 for every local model and would over-budget a
|
||||
small local judge into overflow."""
|
||||
alias_provider = _make_mock_provider()
|
||||
alias_provider.provider_name = "openai"
|
||||
# If the code (wrongly) consulted caps, it'd read this fictitious 200k.
|
||||
alias_provider.get_capabilities = MagicMock(return_value=MagicMock(context_window=200_000))
|
||||
alias_client = MagicMock(base_url="https://alias/v1", api_key="k")
|
||||
registry = self._make_alias_registry("judge-mini", alias_provider, alias_client, "local-9b")
|
||||
judge = IntentJudge(
|
||||
config=JudgeConfig(enabled=True, model="judge-mini"),
|
||||
session_provider=_make_mock_provider(),
|
||||
session_client=MagicMock(base_url="https://s/v1", api_key="s"),
|
||||
session_model="session-model",
|
||||
context_window=100_000,
|
||||
model_registry=registry,
|
||||
)
|
||||
assert judge._judge_context_window == 50_000
|
||||
|
||||
def test_alias_zero_context_window_falls_back_to_session(self):
|
||||
"""config.toml can hand back a ModelConfig with context_window=0 (that
|
||||
path lacks the DB loader's 0→inherit normalization); a 0 window would
|
||||
zero every budget and make honest_truncate drop everything, so it must
|
||||
fall back to the session window."""
|
||||
cfg = MagicMock()
|
||||
cfg.context_window = 0
|
||||
registry = MagicMock()
|
||||
registry.has_alias.side_effect = lambda a: a == "judge-mini"
|
||||
registry.resolve.return_value = (MagicMock(base_url="http://a", api_key="k"), "m", cfg)
|
||||
registry.get_provider.return_value = _make_mock_provider()
|
||||
judge = IntentJudge(
|
||||
config=JudgeConfig(enabled=True, model="judge-mini"),
|
||||
session_provider=_make_mock_provider(),
|
||||
session_client=MagicMock(base_url="http://s", api_key="s"),
|
||||
session_model="session-model",
|
||||
context_window=100_000,
|
||||
model_registry=registry,
|
||||
)
|
||||
assert judge._judge_context_window == 100_000 # session window, not 0
|
||||
|
||||
def test_unknown_alias_inherits_session_model(self):
|
||||
"""``judge.model`` is alias-only. A value that doesn't resolve
|
||||
through the registry inherits the session model (same path as
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user