diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 200b877023..ae42587f8e 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -64,10 +64,14 @@ updates: - ">= 7.0.0" # @cloudflare/vitest-pool-workers (0.22.0, the latest) peers on vitest # ^4.1.0 and runs the control-plane integration tests, so vitest 5 cannot - # run them. Drop this entry once pool-workers accepts vitest 5. + # run them. Drop this entry once pool-workers accepts vitest 5. The + # coverage providers peer on the exact vitest version, so hold them too. - dependency-name: vitest versions: - ">= 5.0.0" + - dependency-name: "@vitest/coverage-*" + versions: + - ">= 5.0.0" # Prettier tracks the version locked upstream (ColeMurray/background-agents), # not the latest. A newer formatter rewrites upstream-owned files, which then # conflict on every sync, and fails format:check on files a sync brings in diff --git a/.github/workflows/ci-python.yml b/.github/workflows/ci-python.yml index bc143f943c..ab970d6880 100644 --- a/.github/workflows/ci-python.yml +++ b/.github/workflows/ci-python.yml @@ -195,7 +195,8 @@ jobs: pip install -e ".[dev]" - name: Run tests - run: pytest tests/ -v + # Dump stacks and exit stalled tests before the job timeout prevents log upload. + run: pytest tests/ -v --durations=20 -o faulthandler_timeout=60 -o faulthandler_exit_on_timeout=true - name: Run Node.js tests run: node --test tests/*.test.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6341aaaa9e..fb9ecb448a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -6,6 +6,7 @@ on: paths: - ".env.example" - ".github/workflows/ci.yml" + - ".github/workflows/terraform.yml" - ".nvmrc" - ".prettierignore" - ".prettierrc" @@ -42,6 +43,7 @@ on: paths: - ".env.example" - ".github/workflows/ci.yml" + - ".github/workflows/terraform.yml" - ".nvmrc" - ".prettierignore" - ".prettierrc" diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml new file mode 100644 index 0000000000..85b95293ee --- /dev/null +++ b/.github/workflows/coverage.yml @@ -0,0 +1,349 @@ +# Keep the workflow unfiltered so the aggregate Coverage check can be required +# for merge without leaving unrelated pull requests pending. Each suite runs +# only when its inputs change; skipped suites still report success. +name: Coverage + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + changes: + name: Detect coverage inputs + runs-on: ubuntu-latest + timeout-minutes: 5 + outputs: + control-plane: ${{ steps.changes.outputs.control-plane }} + web: ${{ steps.changes.outputs.web }} + other-ts: ${{ steps.changes.outputs.other-ts }} + python: ${{ steps.changes.outputs.python }} + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + fetch-depth: 0 + - name: Check for changes affecting each coverage suite + id: changes + env: + # PRs compare from the merge base; pushes compare the complete push. + DIFF_RANGE: ${{ github.event_name == 'pull_request' && format('{0}...{1}', github.event.pull_request.base.sha, github.event.pull_request.head.sha) || format('{0}..{1}', github.event.before, github.sha) }} + run: | + # The coverage gate itself applies to every suite. + gate=( + .github/workflows/coverage.yml + .nvmrc + scripts/check-coverage.mjs + scripts/coverage-baseline.json + scripts/coverage-policy.ts + ) + # Every TypeScript suite installs the workspace and imports the built shared package. + typescript=( + package.json + package-lock.json + packages/shared + ) + # Inputs outside a suite's own packages are files its tests read or import. + # Keep these lists in sync with those reads. Package READMEs are not inputs. + detect() { + local suite=$1 + shift + # Unknown paths and unavailable comparisons run the suite. No rename + # detection: moving an input out of a watched directory must run it. + if git diff --quiet --no-renames "$DIFF_RANGE" -- "$@" ':(exclude,glob)packages/**/*.md'; then + echo "$suite=false" >> "$GITHUB_OUTPUT" + echo "No $suite coverage inputs changed: skipping." >> "$GITHUB_STEP_SUMMARY" + else + echo "$suite=true" >> "$GITHUB_OUTPUT" + fi + } + detect control-plane "${gate[@]}" "${typescript[@]}" \ + packages/control-plane \ + packages/opencomputer-infra \ + packages/sandbox-images \ + packages/vercel-infra \ + packages/sandbox-runtime/src/sandbox_runtime/runtime_manifest.json \ + terraform/d1/migrations \ + terraform/environments/production/workers-control-plane.tf \ + .env.example \ + docker-compose.yml \ + docs/CONTROL_PLANE_CONTAINER.md + detect web "${gate[@]}" "${typescript[@]}" \ + packages/web \ + packages/control-plane/src/session/batch-archive.ts \ + packages/control-plane/src/session/contracts.ts \ + packages/control-plane/src/session/runtime-client.ts + detect other-ts "${gate[@]}" "${typescript[@]}" \ + packages/github-bot \ + packages/linear-bot \ + packages/slack-bot \ + scripts/check-coverage.test.mjs \ + docs/AVAILABLE_MODELS.md \ + packages/docs/content/docs/models/choosing-a-model.mdx + detect python "${gate[@]}" \ + packages/modal-infra \ + packages/sandbox-runtime \ + packages/sandbox-images \ + packages/control-plane/src/image-builds/timeouts.ts \ + packages/shared/src/types/integrations.ts \ + terraform/modules/modal-app/scripts/deploy.sh + + # Control-plane coverage is CPU-bound on per-file Workers pool startup, so it is split across + # runners. Shards skip the full-suite floor; the merge job below enforces it on the merged report. + control-plane-coverage-shard: + name: Coverage (control-plane ${{ matrix.shard }}) + needs: changes + if: needs.changes.outputs.control-plane == 'true' + runs-on: ubuntu-latest + timeout-minutes: 10 + strategy: + fail-fast: false + matrix: + shard: ["1/8", "2/8", "3/8", "4/8", "5/8", "6/8", "7/8", "8/8"] + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-node@v7 + with: + node-version-file: .nvmrc + cache: npm + - name: Install and build shared + run: | + npm ci + npm run build -w @open-inspect/shared + - name: Run control-plane coverage shard + env: + COVERAGE_SHARD: "true" + SHARD: ${{ matrix.shard }} + run: >- + npm run test:coverage -w @open-inspect/control-plane -- + --shard="$SHARD" + --reporter=default --reporter=github-actions --reporter=blob + --coverage.reporter=json-summary + - name: Upload shard report + if: always() + uses: actions/upload-artifact@v7 + with: + name: control-plane-coverage-shard-${{ strategy.job-index }} + path: packages/control-plane/.vitest-reports/ + include-hidden-files: true + if-no-files-found: error + + control-plane-coverage: + name: Coverage (control-plane) + needs: control-plane-coverage-shard + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-node@v7 + with: + node-version-file: .nvmrc + cache: npm + - name: Install dependencies + run: npm ci + - uses: actions/download-artifact@v8 + with: + pattern: control-plane-coverage-shard-* + path: packages/control-plane/.vitest-reports/ + merge-multiple: true + - name: Merge and enforce control-plane coverage + working-directory: packages/control-plane + run: | + npx vitest run --merge-reports=.vitest-reports --config vitest.coverage.config.ts --coverage + node ../../scripts/check-coverage.mjs control-plane coverage/coverage-summary.json + - name: Upload control-plane coverage + if: always() + uses: actions/upload-artifact@v7 + with: + name: coverage-data-control-plane + path: packages/control-plane/coverage/ + if-no-files-found: error + + web-coverage: + name: Coverage (web) + needs: changes + if: needs.changes.outputs.web == 'true' + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-node@v7 + with: + node-version-file: .nvmrc + cache: npm + - name: Install and build shared + run: | + npm ci + npm run build -w @open-inspect/shared + - name: Enforce web coverage + run: | + npm run test:coverage -w @open-inspect/web + node scripts/check-coverage.mjs web packages/web/coverage/coverage-summary.json + - name: Upload web coverage + if: always() + uses: actions/upload-artifact@v7 + with: + name: coverage-data-web + path: packages/web/coverage/ + if-no-files-found: error + + other-ts-coverage: + name: Coverage (shared and bots) + needs: changes + if: needs.changes.outputs.other-ts == 'true' + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-node@v7 + with: + node-version-file: .nvmrc + cache: npm + - name: Install and build shared + run: | + npm ci + npm run build -w @open-inspect/shared + - name: Test the coverage gate + run: node --test scripts/check-coverage.test.mjs + - name: Enforce TypeScript coverage + run: | + for package in shared slack-bot linear-bot github-bot; do + npm run test:coverage -w "@open-inspect/$package" + node scripts/check-coverage.mjs "$package" "packages/$package/coverage/coverage-summary.json" + done + - name: Upload other TypeScript coverage + if: always() + uses: actions/upload-artifact@v7 + with: + name: coverage-data-ts + path: | + packages/shared/coverage/ + packages/slack-bot/coverage/ + packages/linear-bot/coverage/ + packages/github-bot/coverage/ + if-no-files-found: error + + python-coverage: + name: Coverage (Python) + needs: changes + if: needs.changes.outputs.python == 'true' + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-node@v7 + with: + node-version-file: .nvmrc + cache: npm + - uses: actions/setup-python@v7 + with: + python-version: "3.12" + - uses: astral-sh/setup-uv@v7 + with: + version: "0.9.7" + - name: Enforce Python statement and branch coverage + run: | + for package in modal-infra sandbox-runtime; do + uv run --frozen --project "packages/$package" --extra dev pytest "packages/$package/tests" --cov="packages/$package/src" --cov-branch --cov-report="json:packages/$package/coverage/coverage.json" + node scripts/check-coverage.mjs "$package" "packages/$package/coverage/coverage.json" + done + - name: Upload Python coverage + if: always() + uses: actions/upload-artifact@v7 + with: + name: coverage-data-python + path: | + packages/modal-infra/coverage/ + packages/sandbox-runtime/coverage/ + if-no-files-found: error + + coverage: + name: Coverage + if: ${{ always() }} + needs: + [ + changes, + control-plane-coverage-shard, + control-plane-coverage, + web-coverage, + other-ts-coverage, + python-coverage, + ] + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - name: Require successful coverage jobs + env: + CHANGES_RESULT: ${{ needs.changes.result }} + CP_CHANGED: ${{ needs.changes.outputs.control-plane }} + WEB_CHANGED: ${{ needs.changes.outputs.web }} + TS_CHANGED: ${{ needs.changes.outputs.other-ts }} + PYTHON_CHANGED: ${{ needs.changes.outputs.python }} + CP_SHARD_RESULT: ${{ needs.control-plane-coverage-shard.result }} + CP_RESULT: ${{ needs.control-plane-coverage.result }} + WEB_RESULT: ${{ needs.web-coverage.result }} + TS_RESULT: ${{ needs.other-ts-coverage.result }} + PYTHON_RESULT: ${{ needs.python-coverage.result }} + run: | + status=0 + # A job passes when it succeeded, or was skipped because none of its inputs changed. + require() { + local job=$1 result=$2 changed=$3 + if [ "$result" = success ] || { [ "$result" = skipped ] && [ "$changed" = false ]; }; then + return + fi + echo "::error::$job coverage job result: $result" + status=1 + } + require changes "$CHANGES_RESULT" true + require control-plane-shards "$CP_SHARD_RESULT" "$CP_CHANGED" + require control-plane "$CP_RESULT" "$CP_CHANGED" + require web "$WEB_RESULT" "$WEB_CHANGED" + require other-ts "$TS_RESULT" "$TS_CHANGED" + require python "$PYTHON_RESULT" "$PYTHON_CHANGED" + exit "$status" + - if: needs.changes.outputs.control-plane == 'true' + uses: actions/download-artifact@v8 + with: + name: coverage-data-control-plane + path: packages/control-plane/coverage/ + - if: needs.changes.outputs.web == 'true' + uses: actions/download-artifact@v8 + with: + name: coverage-data-web + path: packages/web/coverage/ + - if: needs.changes.outputs.other-ts == 'true' + uses: actions/download-artifact@v8 + with: + name: coverage-data-ts + path: packages/ + - if: needs.changes.outputs.python == 'true' + uses: actions/download-artifact@v8 + with: + name: coverage-data-python + path: packages/ + - name: Upload coverage reports + if: always() && contains(needs.changes.outputs.*, 'true') + uses: actions/upload-artifact@v7 + with: + name: production-coverage + path: packages/*/coverage/ + if-no-files-found: error diff --git a/.github/workflows/terraform.yml b/.github/workflows/terraform.yml index 7e6b49fab9..a49c5b3430 100644 --- a/.github/workflows/terraform.yml +++ b/.github/workflows/terraform.yml @@ -45,8 +45,6 @@ on: permissions: contents: read - pull-requests: write - issues: write concurrency: group: terraform-${{ github.ref }} @@ -90,6 +88,12 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 5 needs: [check-secrets] + outputs: + fmt: ${{ steps.fmt.outcome }} + init: ${{ steps.init.outcome }} + validate: ${{ steps.validate.outcome }} + test: ${{ steps.test.outcome }} + modal-test: ${{ steps.modal_test.outcome }} steps: - name: Checkout uses: actions/checkout@v7 @@ -129,50 +133,14 @@ jobs: terraform test working-directory: terraform/modules/modal-app - - name: Post Validation Results - if: always() && github.event_name == 'pull_request' - uses: actions/github-script@v9 - with: - script: | - const hasSecrets = '${{ needs.check-secrets.outputs.has-secrets }}' === 'true'; - const planNote = hasSecrets - ? '' - : '\n\n> **Note:** Terraform plan was skipped because secrets are not configured. This is expected for external contributors. See [docs/GETTING_STARTED.md](docs/GETTING_STARTED.md) for setup instructions.'; - - const output = `### Terraform Validation Results - - | Step | Status | - |------|--------| - | Format | ${{ steps.fmt.outcome == 'success' && '✅' || '⚠️' }} | - | Init | ${{ steps.init.outcome == 'success' && '✅' || '❌' }} | - | Validate | ${{ steps.validate.outcome == 'success' && '✅' || '❌' }} | - | Tests | ${{ steps.test.outcome == 'success' && '✅' || '❌' }} | - | Modal module tests | ${{ steps.modal_test.outcome == 'success' && '✅' || '❌' }} | - ${planNote} - - *Pushed by: @${{ github.actor }}, Action: \`${{ github.event_name }}\`*`; - - try { - await github.rest.issues.createComment({ - issue_number: context.issue.number, - owner: context.repo.owner, - repo: context.repo.repo, - body: output - }); - } catch (error) { - if (error.status === 403) { - core.notice('Could not post PR comment — GITHUB_TOKEN has read-only access (expected for fork PRs).'); - } else { - throw error; - } - } - plan: name: Plan runs-on: ubuntu-latest timeout-minutes: 15 needs: [validate, check-secrets] if: github.event_name == 'pull_request' && needs.check-secrets.outputs.has-secrets == 'true' + outputs: + outcome: ${{ steps.plan.outcome }} steps: - name: Checkout uses: actions/checkout@v7 @@ -198,6 +166,8 @@ jobs: uses: hashicorp/setup-terraform@v4 with: terraform_version: ${{ env.TF_VERSION }} + # Capture output as a file, not the wrapper's multiline step outputs. + terraform_wrapper: false - name: Terraform Init env: @@ -225,11 +195,14 @@ jobs: - name: Terraform Plan id: plan + shell: bash run: | + set -euo pipefail + # Plan text must not be interpreted as runner workflow commands either. + command_token="$(openssl rand -hex 32)" + echo "::stop-commands::$command_token" + trap 'printf "\n::%s::\n" "$command_token"' EXIT terraform plan -no-color -out=tfplan 2>&1 | tee plan_output.txt - echo "plan<> $GITHUB_OUTPUT - cat plan_output.txt | head -c 60000 >> $GITHUB_OUTPUT - echo "EOF" >> $GITHUB_OUTPUT working-directory: ${{ env.TF_WORKING_DIR }} continue-on-error: true env: @@ -265,6 +238,7 @@ jobs: TF_VAR_classification_anthropic_api_key: ${{ secrets.CLASSIFICATION_ANTHROPIC_API_KEY }} TF_VAR_classification_openai_api_key: ${{ secrets.CLASSIFICATION_OPENAI_API_KEY }} TF_VAR_classification_model: "${{ vars.CLASSIFICATION_MODEL || 'claude-haiku-4-5' }}" + TF_VAR_classification_reasoning_effort: "${{ vars.CLASSIFICATION_REASONING_EFFORT }}" TF_VAR_token_encryption_key: ${{ secrets.TOKEN_ENCRYPTION_KEY }} TF_VAR_repo_secrets_encryption_key: ${{ secrets.REPO_SECRETS_ENCRYPTION_KEY }} TF_VAR_provider_accounts_encryption_key: ${{ secrets.PROVIDER_ACCOUNTS_ENCRYPTION_KEY }} @@ -329,41 +303,127 @@ jobs: TF_VAR_docs_site_enabled: "${{ vars.DOCS_SITE_ENABLED || secrets.DOCS_SITE_ENABLED || 'false' }}" TF_VAR_docs_custom_domain: ${{ vars.DOCS_CUSTOM_DOMAIN || secrets.DOCS_CUSTOM_DOMAIN }} - - name: Post Plan Results - uses: actions/github-script@v9 - with: - script: | - const planOutcome = '${{ steps.plan.outcome }}'; - const plan = `${{ steps.plan.outputs.plan }}`; + - name: Prepare Plan Comment + id: prepare_comment + if: always() && (steps.plan.outcome == 'success' || steps.plan.outcome == 'failure') + env: + PLAN_OUTCOME: ${{ steps.plan.outcome }} + run: node "$GITHUB_WORKSPACE/scripts/terraform-plan-comment.mjs" plan_output.txt plan_comment.txt "$PLAN_OUTCOME" + working-directory: ${{ env.TF_WORKING_DIR }} - const truncatedPlan = plan.length > 60000 - ? plan.substring(0, 60000) + '\n... (truncated)' - : plan; + - name: Upload Plan Comment + if: steps.prepare_comment.conclusion == 'success' + uses: actions/upload-artifact@v7 + with: + name: terraform-plan-comment + path: ${{ env.TF_WORKING_DIR }}/plan_comment.txt + if-no-files-found: error + retention-days: 1 - const output = `### Terraform Plan Results + - name: Plan Status + if: always() && steps.plan.outcome == 'failure' + run: exit 1 - **Status:** ${planOutcome === 'success' ? '✅ Success' : '❌ Failed'} + comment: + name: Comment + runs-on: ubuntu-latest + timeout-minutes: 5 + needs: [validate, check-secrets, plan] + if: always() && !cancelled() && github.event_name == 'pull_request' + permissions: + pull-requests: write + # No checkout or execution of PR code. Artifacts are read only as comment data. + steps: + - name: Post Validation Results + uses: actions/github-script@v9 + env: + HAS_SECRETS: ${{ needs.check-secrets.outputs.has-secrets }} + CHECK_SECRETS_RESULT: ${{ needs.check-secrets.result }} + VALIDATION_RESULT: ${{ needs.validate.result }} + FORMAT_OUTCOME: ${{ needs.validate.outputs.fmt }} + INIT_OUTCOME: ${{ needs.validate.outputs.init }} + VALIDATE_OUTCOME: ${{ needs.validate.outputs.validate }} + TEST_OUTCOME: ${{ needs.validate.outputs.test }} + MODAL_TEST_OUTCOME: ${{ needs.validate.outputs.modal-test }} + with: + script: | + const validationResult = process.env.VALIDATION_RESULT; + const checkSecretsResult = process.env.CHECK_SECRETS_RESULT; + const statuses = { + Format: process.env.FORMAT_OUTCOME, + Init: process.env.INIT_OUTCOME, + Validate: process.env.VALIDATE_OUTCOME, + Tests: process.env.TEST_OUTCOME, + 'Modal module tests': process.env.MODAL_TEST_OUTCOME + }; + const rows = Object.entries(statuses).map(([step, outcome]) => + `| ${step} | ${outcome === 'success' ? 'Success' : outcome || (validationResult === 'skipped' ? 'skipped' : 'Not reported')} |` + ).join('\n'); + let planNote = ''; + if (checkSecretsResult !== 'success') { + planNote = '\n\n> **Warning:** Secret availability is unknown because the Check Secrets job did not succeed.'; + } else if (process.env.HAS_SECRETS === 'false') { + planNote = '\n\n> **Note:** Terraform plan was skipped because secrets are not configured. This is expected for external contributors. See [docs/GETTING_STARTED.md](docs/GETTING_STARTED.md) for setup instructions.'; + } else if (process.env.HAS_SECRETS !== 'true') { + planNote = '\n\n> **Warning:** Check Secrets completed without a valid secret-availability result.'; + } + const output = `### Terraform Validation Results -
Show Plan + **Validation job:** ${validationResult || 'unknown'} + **Check Secrets job:** ${checkSecretsResult || 'unknown'} - \`\`\`terraform - ${truncatedPlan} - \`\`\` + | Step | Status | + |------|--------| + ${rows} + ${planNote} -
+ *Pushed by: @${context.actor}, Action: \`${context.eventName}\`*`; - *Pushed by: @${{ github.actor }}*`; + try { + await github.rest.issues.createComment({ + issue_number: context.issue.number, + owner: context.repo.owner, + repo: context.repo.repo, + body: output + }); + } catch (error) { + if (error.status === 403) { + core.notice('Could not post PR comment: GITHUB_TOKEN has read-only access (expected for fork PRs).'); + } else { + throw error; + } + } - github.rest.issues.createComment({ - issue_number: context.issue.number, - owner: context.repo.owner, - repo: context.repo.repo, - body: output - }); + - name: Download Plan Comment + if: needs.plan.outputs.outcome == 'success' || needs.plan.outputs.outcome == 'failure' + uses: actions/download-artifact@v8 + with: + name: terraform-plan-comment + path: ${{ runner.temp }}/terraform-plan-comment - - name: Plan Status - if: steps.plan.outcome == 'failure' - run: exit 1 + - name: Post Plan Results + if: needs.plan.outputs.outcome == 'success' || needs.plan.outputs.outcome == 'failure' + uses: actions/github-script@v9 + env: + PLAN_COMMENT_PATH: ${{ runner.temp }}/terraform-plan-comment/plan_comment.txt + with: + script: | + const fs = require('node:fs'); + const body = fs.readFileSync(process.env.PLAN_COMMENT_PATH, 'utf8'); + try { + await github.rest.issues.createComment({ + issue_number: context.issue.number, + owner: context.repo.owner, + repo: context.repo.repo, + body + }); + } catch (error) { + if (error.status === 403) { + core.notice('Could not post PR comment: GITHUB_TOKEN has read-only access (expected for fork PRs).'); + } else { + throw error; + } + } apply: name: Apply @@ -465,6 +525,7 @@ jobs: TF_VAR_classification_anthropic_api_key: ${{ secrets.CLASSIFICATION_ANTHROPIC_API_KEY }} TF_VAR_classification_openai_api_key: ${{ secrets.CLASSIFICATION_OPENAI_API_KEY }} TF_VAR_classification_model: "${{ vars.CLASSIFICATION_MODEL || 'claude-haiku-4-5' }}" + TF_VAR_classification_reasoning_effort: "${{ vars.CLASSIFICATION_REASONING_EFFORT }}" TF_VAR_token_encryption_key: ${{ secrets.TOKEN_ENCRYPTION_KEY }} TF_VAR_repo_secrets_encryption_key: ${{ secrets.REPO_SECRETS_ENCRYPTION_KEY }} TF_VAR_provider_accounts_encryption_key: ${{ secrets.PROVIDER_ACCOUNTS_ENCRYPTION_KEY }} diff --git a/.gitignore b/.gitignore index 4073e2ef29..5a82801276 100644 --- a/.gitignore +++ b/.gitignore @@ -44,6 +44,9 @@ yarn-error.log* # Test coverage/ +.vitest-reports/ +.coverage +.coverage.* .nyc_output/ # Cloudflare diff --git a/.prettierignore b/.prettierignore index fd4ebeaafe..bddcc25dfa 100644 --- a/.prettierignore +++ b/.prettierignore @@ -30,3 +30,5 @@ packages/docs/.source/ *.d.ts *.min.js *.min.css +packages/sandbox-runtime/src/sandbox_runtime/memory_contract.json +packages/sandbox-runtime/src/sandbox_runtime/tools/_memory-contract.js diff --git a/CHANGELOG.md b/CHANGELOG.md index 526c87632d..1535fc358a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,243 +2,84 @@ New features, integrations, and notable improvements to Open-Inspect — newest first. -## Unreleased +## October 4, 2026 -### Changed +**GitHub model overrides.** Start a GitHub `@mention` with `!model` or `!reasoning`, the same flags +Slack uses, to pick the model and reasoning effort for that session. Upgrade the GitHub bot and +control plane together. See +[GitHub integration](docs/integrations/GITHUB.md#model-and-reasoning-overrides). -Team-owned session actions now require current owning-team membership in every `TEAMS_ENFORCEMENT` -mode, including for Owners and Administrators. Visibility still determines read access; collaborator -self-removal requires only read access. Sessions, automations, and environments cannot move between -teams or to/from the workspace; a team-owned session never becomes workspace-owned. Visibility and -collaborator controls remain available to authorized users. Historical `session.moved` audit events -remain readable. +## October 3, 2026 -## October 1, 2026 - -### Changed - -GitHub App sandbox credentials now reach only the session's repositories, including workspace-owned -sessions and members snapshotted from an environment. Team-owned sessions also intersect that set -with their team's current repository grants; an installation grant does not widen the credential -beyond the session's members. Sessions with no repositories or no remaining granted repositories -receive no token. Unresolved repository IDs and scopes exceeding GitHub's repository limit are -refused rather than falling back to installation-wide access. - -This breaks private submodule, repository-backed dependency, and sibling-clone setups unless those -repositories are included in the session's environment and, for team sessions, granted to its team. -Repository image builds receive a token for that repository alone; environment builds use only their -member repositories, intersected with the environment team's grants. Metadata and workspace-catalog -operations retain installation-wide access. GitLab still uses a deployment-wide PAT and does not -enforce repository-scoped credentials. - -Token cache keys cover the sorted, de-duplicated repository set, the process cache is bounded, and -overlapping refreshes share one mint per scope. Grant removal changes the next credential scope but -does not revoke already-issued tokens; sandbox helpers cache them until shortly before expiry. - -**Modal snapshot restores use brokered git credentials.** Restored sandboxes now fetch git -credentials from the control plane like fresh sessions, instead of receiving a token minted by -Modal. The control plane now sends the VCS host and clone username with every Modal create, restore, -and image-build request, so Modal no longer reads `SCM_PROVIDER` or needs GitHub App credentials. -Terraform no longer provisions Modal's `github-app` secret; you can delete the existing secret from -Modal after upgrading. - -### Added - -Team leads and workspace administrators can manage encrypted secrets from a team's Secrets tab. -Team-owned sessions receive global secrets, then team secrets, then environment or repository -secrets, with later scopes taking precedence. Environment image builds include the environment's -team secrets; repository-shared images do not. Secret mutation audits contain key names only. Team -secret changes atomically supersede affected environment images. After the database batch, detached, -best-effort rebuild scheduling is attempted for enabled team-owned environments; enumeration or -trigger failures may leave no rebuild request. Team-owned environment images require matching -session ownership. Team-only legacy OAuth refresh tokens do not enable managed authentication; API -keys remain usable. Team-secret read and decryption errors abort environment builds rather than -falling back to other secret scopes. - -**Team-owned environments.** The environment form offers team ownership, and team pages include an -Environments tab. Team environments are visible to their members and administrators, and controls -use server capabilities. Environment names are unique within each team. Sessions, including -inherited child targets, can use a team environment only when they belong to that team. Environment -secrets, settings, and image routes also require access to the owning environment. Workspace -environment management remains permission-based for custom roles, and existing environments retain -their ownership. The require-team creation setting also applies to new environments. Changing -environment secrets, settings, or images now also requires `environments.manage`, so custom roles -holding only `environments.secrets.manage`, `environments.settings.manage`, or -`environments.images.manage` lose those actions. Actorless bots see only workspace environments, -both in lists and by ID. - -**Team-owned automations.** The automation form offers team ownership, and team pages include an -Automations tab. Team automations are visible to their members and administrators, and controls use -server capabilities. Automation leads can manage team work and reassign departed executors; executor -reassignment verifies the candidate's launch permissions before writing and is audited as -`automation.executor_changed`. Selected environments must belong to the automation's team. -Executions require active membership, an unarchived team, and current repository grants, and their -sessions inherit the team's default visibility. Slack follow-ups use the persisted session's -collaboration decision, and automation history redacts inaccessible session metadata. Existing -automations retain their ownership and visibility. The require-team creation setting also applies to -new automations. These owned-resource checks apply in every session enforcement mode; source-control -token narrowing remains a separate change. - -### Removed - -Removed the team Activity tab and `GET /teams/:id/activity` endpoint. Team operations continue to be -recorded in the workspace audit log, available to viewers with `workspace.audit.read` and filterable -by team. No audit history is deleted. - -### Fixed - -The team directory and collaborator picker now show email addresses only to viewers with -`workspace.members.read` (Owners and Administrators in the built-in roles). Other viewers receive -names and avatars with no email address, and unnamed users have a neutral label with a short ID -suffix. This restriction applies in every team enforcement mode. - -Session navigation now defaults to **All my teams**, with the team selector available even for a -single membership. Composer team and visibility choices stay local, including automatic team -selection when new sessions require a team. Transient membership refresh failures retain loaded -data, and changing draft configuration retires the old warm session without starting a replacement -sandbox until the next prompt input or submission. Scope changes refresh lists without clearing -terminal access or per-session caches. Visibility changes require a changed selection and confirm -non-private child-session cascades. Workspace audit readers can filter by teams they do not belong -to. - -### Added - -Teams now have a Repositories tab. Members can view grants; team leads and workspace administrators -can grant all installation repositories or select named repositories, and remove grants. Team-scoped -repository catalogs and repository-bearing writes check these grants, with explicit missing-grant -errors and audited grant changes. Workspace-level session catalogs remain installation-wide. Grant -changes advance the team's grant version but do not yet narrow or revoke sandbox installation -tokens. - -Workspace-level skills, repository secrets, and image builds retain existing permissions on -repositories granted to no team. Team-owned repositories additionally require membership (lead -membership for repository secrets), or workspace Owner/Administrator access, in every enforcement -mode. Manual team-owned environment builds require access to the owning team. Hidden and missing -team requests now record identical denied authorization decisions without changing their 404 -responses. - -## September 30, 2026 - -### Added - -Teams now have a searchable directory with favorites, member lists, session overviews, and -visibility-filtered activity. Active users can browse team names and memberships; a team's work -remains restricted to members and administrators. Workspace audit readers can filter events by team. -Session details show the owning team and visibility, with server-authorized controls to move -sessions, change visibility, and manage private-session collaborators, including child-session -cascades. Archived team metadata and member lists remain visible only to team members and workspace -administrators. Team activity shows domain operations; HTTP authorization decisions remain in the -permission-gated workspace audit log. - -**Team-aware session discovery and creation.** Following the team and visibility APIs, the web app -now supports team selection and scoped session discovery. Inbox snapshot and paged reads accept -ownership, visibility, and workspace scope filters and return effective server capabilities for -roots and descendants. The current user's team response includes the require-team creation setting -without requiring settings-management permissions. Bot team selection and automation team ownership -remain later phases; repository-backed team sessions still require existing grants, with no grant -creation API or UI yet. - -### Fixed - -Allowed team directory, member, session, activity, and collaborator-candidate reads no longer add -authorization-decision rows to the audit log. Capability writes and membership departures remain -audited. Unauthorized cross-member removals are recorded as denied decisions. Live session -subscriptions now include team memberships when computing capabilities in every enforcement mode, -preserving team leads' move and visibility controls without adding reads to per-command -authorization in `off` or `shadow`. - -## September 29, 2026 - -### Added - -**Team-scoped session access.** Teams remain optional: existing sessions stay teamless workspace -rows, and **Settings > Teams > Require a team for new sessions** is off by default. Operators can -roll out `TEAMS_ENFORCEMENT=off|shadow|on` (`shadow` by default): `shadow` records would-be team and -ownership denials without blocking non-private sessions, while `on` enforces them. Private -visibility is restricted in every mode. Session item routes, lists and aggregates, live connections, -and sandbox access use the persisted session scope; Owners' private-session break-glass reads are -audited and do not make those sessions enumerable. Session creation and team moves check membership -and repository grants; team and visibility change APIs have landed, with discovery UI following in -the next entry. Visibility, scope, and collaborator mutations enforce the resolver in every mode and -cascades refuse inaccessible descendants. The require-team setting refuses teamless session creation -API requests; automation runs remain exempt until team ownership is supported. Repository grant -creation is not yet available, so missing grants refuse repository-backed team sessions with -`target_team_missing_grant`. Team grants do not yet narrow the shared source-control installation -token in sandboxes. See [Authentication and Authorization](docs/AUTH.md). +**Persistent memory.** Personal, repository, and environment memories carry facts and directives +between sessions. Agents in both harnesses search, read, and propose memories. Only personal facts +written in a root session that has stayed private and owner-only become active immediately; all +other agent writes wait for approval. Manage personal memories under **Settings > Memories** and +repository and environment memories under **Settings > Shared memories**; each session's sidebar +lists the memories it loaded. Requires D1 migration 0084 and a sandbox image rebuild. See +[Persistent session memory](docs/MEMORY.md). -## September 28, 2026 +## October 2, 2026 -### Added +**Analytics redesign.** The analytics page is now an overview plus Usage, Cost, Pull requests, and +People tabs. The range, scope, and tab are kept in the URL. A new **Session origins** breakdown +shows where sessions start and who they are attributed to. -**Claude Sonnet 5.5.** Adds `anthropic/claude-sonnet-5-5` to the model picker and integrations, with -adaptive thinking controls from low through max. Claude Agent SDK 0.2.161 bundles Claude Code -2.1.284, which supports the new model. +**Saved prompt drafts.** Unsent prompts survive page reloads. Each session and the new-session +composer keep their own draft. -OpenCode sessions using a connected ChatGPT subscription now report estimated model costs through -the existing session cost display and spending limit. These are API-price equivalents, not -additional subscription charges or an OpenAI invoice; estimates remain zero if catalog pricing is -unavailable. +**Teams.** Group members into teams that own sessions, environments, automations, and secrets, with +repository grants and Slack and Linear channel bindings managed from **Settings > Teams**. GitHub +work routes to teams by numeric repository ID: upgrade the GitHub bot and control plane together, +and reselect repositories on older GitHub event automations so they keep matching events. Sandbox +GitHub tokens now cover only the session's repositories. See +[Authentication and Authorization](docs/AUTH.md). -Workspace settings now includes Teams. Administrators can create teams, manage members and leads, -edit team defaults, and archive or restore teams. Team leads can manage their own teams where -permitted. +**Classifier reasoning effort.** Set `classification_reasoning_effort` to send a reasoning effort to +OpenAI classification models used by the Slack and Linear bots. Leaving it blank keeps the model +default. -## September 27, 2026 +## October 1, 2026 -### Changed +**Brokered credentials for Modal restores.** Restored Modal sandboxes now fetch git credentials from +the control plane like fresh ones, so Modal no longer needs the `github-app` secret. You can delete +it after upgrading. -Trace export now emits published schema 2: messages, events and usage are all oldest first, and -session trace byte-budget errors use `trace_budget_exceeded`. Single-session downloads export only -the requested session; `scope` on that route now returns 400. Whole runs remain available through -the paginated bulk export. +**Session page redesign.** The session sidebar is split into Info, Changes, Tasks, and Tools tabs, +with captured media under **Artifacts** in Info. Changed files open in the main column beside the +sidebar, which leaves room for split diffs. -### Added +## September 28, 2026 -Operators can select `modal-vm` deployment-wide for Docker-capable Modal sandboxes, with separate -prepared images and filesystem snapshot recovery. The existing `modal` backend remains the default; -switching backends does not migrate existing sessions or images. See -[Modal VM setup](docs/MODAL_DOCKER.md). +**Claude Sonnet 5.5.** Adds `anthropic/claude-sonnet-5-5` to the model picker and integrations, with +adaptive thinking controls from low through max. -The analytics dashboard now shows harness metrics, automation performance in automation and all -scopes, complete runs, and pull-request cost per merged PR by model and harness. +**ChatGPT subscription cost estimates.** OpenCode sessions using a connected ChatGPT subscription +now report API-equivalent cost estimates in the session cost display and spending limit. -The analytics dashboard now lets operators select human, agent, automation or all sessions and -compare token usage, cost by model and provider billing in the selected scope. +## September 27, 2026 -Analytics responses now include session token totals and cache hit ratio, pull-request cost by model -and harness, and the top 20 scoped runs in the dashboard snapshot. Run titles may be null. +**Trace export schema 2.** Messages, events, and usage are now exported oldest first, and +single-session downloads include only the requested session. See the +[trace export reference](docs/TRACE_EXPORT.md). -Session analytics API now accepts `scope=human|agent|automation|all` (default `human`) on the -dashboard, summary, timeseries, and breakdown routes, and supports `by=model`, `by=harness`, -`by=spawnSource`, `by=automation`, and `by=provider` breakdowns. Provider rows include the number of -sessions billed through a matching provider account. +**Docker-capable Modal sandboxes.** Operators can select the `modal-vm` backend deployment-wide for +Docker support; `modal` remains the default. See [Modal VM setup](docs/MODAL_DOCKER.md). -The [trace export reference](docs/TRACE_EXPORT.md) includes a JSON Schema and instructions for -manually downloading paginated runs through the web app. +**Scoped analytics.** The analytics dashboard and API can filter human, agent, automation, or all +sessions, with breakdowns by model, harness, provider, and automation, token totals, cache hit +ratio, and cost per merged PR. ## September 26, 2026 -### Added - -Bulk session export accepts `include` as a comma-separated list of `messages`, `events`, and -`usage`, so one session line can carry the prompt, the persisted timeline events, and per-step token -usage. Each session's included collections are read in one storage snapshot and share one 4 MiB byte -budget and one page cap, and any include limits the request to 5 sessions per page. Messages keep -their existing newest-first order; events and usage are listed in timeline order. +**Richer bulk session export.** Bulk export can now include each session's messages, timeline +events, and per-step token usage through the `include` parameter. ## September 25, 2026 -### Added - -Bulk session export now includes run identity, harness, model provider, repository membership, pull -request lifecycle, and projected token totals on session lines. Schema 1 consumers must ignore -unknown fields; `source` is unchanged and also appears as `spawnSource`. - -### Changed - -Bulk session export now requires `sessions.export` instead of `sessions.read`. Owners, -Administrators, and users granted the permission through a custom role may export; Viewers, Members, -and bot services cannot. +**Bulk export metadata and permission.** Exported session lines now include run identity, harness, +model provider, repositories, pull request lifecycle, and token totals. Bulk export now requires the +`sessions.export` permission, granted to Owners and Administrators by default. ## September 23, 2026 diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 41107aa101..3e9f8a33c3 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -77,6 +77,13 @@ options, including Radix's hidden native selects, while preserving accessibility label queries when selecting labeled inputs. Investigate slow operations before increasing test timeouts or adding retries. +For coverage commands, the recorded baseline, and test-reduction tradeoffs, see +[Coverage-Guided Test Reduction](docs/TEST_REDUCTION.md). Control-plane coverage combines the Node +and workerd suites; unit-only coverage is not the measure of retained integration coverage. The +unfiltered `Coverage` workflow enforces production-only TypeScript and separate Python +statement/branch floors. Repository administrators should require its `Coverage` check in main +rules. + ### Commit Messages Use clear, descriptive commit messages: diff --git a/README.md b/README.md index 55fb6e5827..b977c9ec48 100644 --- a/README.md +++ b/README.md @@ -24,32 +24,40 @@ Open-Inspect provides a hosted background coding agent that can: ## Security Model (Single-Tenant Only) > **Important**: This system is designed for **single-tenant deployment only**, where all users are -> trusted members of the same organization with access to the same repositories. +> trusted members of the same organization. Teams add internal access controls, not tenant +> isolation. ### How It Works -The system uses a shared GitHub App installation for git operations (clone, fetch, push). The -control plane mints short-lived installation tokens server-side and brokers them to sandboxes -through the git credential helper on demand. This means: - -- **Authorized users share the same GitHub App credentials** - The GitHub App must be installed on - your organization's repositories, and active users whose role permits repository use can access - any repo the App has access to -- **No per-user repository access validation** - The system does not verify that a user has - permission to access a specific repository before creating a session -- **GitHub users' OAuth tokens are used for PR creation** - For GitHub logins, PRs are created using - the user's GitHub OAuth token, ensuring proper attribution and that they can only create PRs on - repos they have write access to. Users who sign in another way (e.g. Google) carry no SCM token, - so their PRs fall back to the shared GitHub App bot - -### Token Architecture - -| Token Type | Purpose | Scope | -| ------------------ | -------------------------------------- | -------------------------------- | -| GitHub App Token | Brokered git clone/fetch/push auth | All repos where App is installed | -| User OAuth Token | Create PRs, user info | Repos user has access to | -| Sandbox Auth Token | Sandbox-to-control-plane session calls | Single session | -| WebSocket Token | Real-time session auth | Single session | +Teams own sessions, environments, and automations. A session's Workspace, Team, or Private +visibility is separate from its fixed ownership; resources cannot move between teams or to/from the +workspace. Team-owned session actions require current owning-team membership, even for workspace +Owners and Administrators. With enforcement on, deletion additionally requires `sessions.delete` and +being the session owner, an owning-team lead, or a workspace administrator; team-owned and private +sessions apply their action checks in every mode. + +Private sessions are readable by their owner and explicit collaborators. Workspace Owners have an +audited break-glass read path, not automatic collaboration or sandbox access. Team visibility's read +boundary requires `TEAMS_ENFORCEMENT=on`; the deployment default remains `shadow`. + +For a single-team deployment, create one team, add the users who need to act on its sessions, and +grant its repositories. Existing workspace-owned rows are not moved or hidden. Even with every user +in one team, being a Member does not let someone delete another member's session under enforcement +unless they satisfy the ownership/lead rule. Follow the +[first-team setup](docs/GETTING_STARTED.md#step-10-create-the-first-team-and-test-a-session) before +inviting users. + +The shared GitHub App installation bounds workspace repository reach. Sandbox git credentials are +limited to the session's persisted repositories and, for team-owned sessions, current owning-team +grants. Workspace permissions and resource access still apply; Open-Inspect does not compare a +user's personal GitHub repository permissions before creating a session. GitLab credentials remain a +deployment-wide PAT rather than a per-session token. + +Credentials are brokered on demand but cached on disk, so snapshots can retain them. Grant removal +does not immediately revoke issued tokens. See the canonical +[Authentication and Authorization guide](docs/AUTH.md) for access rules and +[credential boundaries](docs/AUTH.md#repository-and-credential-boundaries), including user versus +App identity and snapshot limitations. ### Why Single-Tenant Only diff --git a/docs/AUTH.md b/docs/AUTH.md index db16011670..e4b4945dd6 100644 --- a/docs/AUTH.md +++ b/docs/AUTH.md @@ -1,13 +1,14 @@ # Authentication and Authorization Open-Inspect uses authentication to establish who you are and workspace authorization to decide what -you can do. This guide explains the behavior users and workspace administrators will see. +you can do. This is the canonical guide to security, resource access, and credential boundaries; +other guides summarize these rules for their audiences. > **Important:** Open-Inspect is designed for a single trusted organization. A deployment is one -> workspace, and the source-control App installation defines the repositories available to that -> workspace. Roles control which Open-Inspect features a person can use; teams and session -> visibility further limit access to sessions. Neither is a replacement for source-control -> repository permissions. +> workspace. The GitHub App installation bounds GitHub repository reach; GitLab uses a deployment +> PAT. Roles control which Open-Inspect features a person can use; teams and session visibility +> further limit access to resources. Neither is a replacement for source-control repository +> permissions. --- @@ -59,6 +60,9 @@ Open-Inspect includes four built-in roles. | View image-build history | Yes | Yes | Yes | Yes | | Manage personal skill profiles | Yes | Yes | Yes | No | +These are workspace feature permissions, not unconditional resource access. Team membership, lead +authority, ownership, and session visibility add the checks described below. + ### Owner Owners administer the workspace but do not automatically collaborate in other people's private @@ -75,11 +79,10 @@ ownership, change who holds the Owner role, or suspend and restore an Owner. ### Member -Members can create and use sessions, collaborate in sessions visible to them (with current -owning-team membership for team-owned sessions, and private-session participation), use shared -repositories and environments, and create automations. They can manage and manually trigger -automations they own but cannot modify another person's automation or administer shared -configuration. They can view workspace analytics. +Members can create and use sessions, collaborate in sessions they can access, use shared +repositories and environments, and create, manage, and manually trigger automations they execute. +Team leads can also manage their team's membership, grants, secrets, and automations; leading a team +does not grant workspace-wide configuration permissions. Members can view workspace analytics. ### Viewer @@ -90,214 +93,231 @@ change shared configuration. ## Teams and Session Visibility -Teams are optional within a workspace. Existing and teamless sessions remain workspace rows with -`ownerTeamId: null`; creating a team does not move them into it. A team has members and leads, a -join policy (open or invite-only), and a default session visibility. Owners and Administrators can -create teams in **Settings > Teams**; the creator becomes the first lead. Team membership does not -replace the workspace role: a person still needs the relevant session permission in addition to any -team access. - -### Team Directory and Pages - -Every active workspace user can list active teams and read their member lists, even without -membership in those teams. The team directory supports search and favorites, and team pages show -team metadata and members. Archived teams and their member lists are available only to their members -and workspace Owners and Administrators. +Teams are optional. A team has members and leads, an open or invite-only join policy, and a default +session visibility (`team` or `workspace`). Owners and Administrators create teams in **Settings > +Teams**, and the creator becomes the first lead. Existing and teamless sessions stay workspace-owned +(`ownerTeamId: null`); creating a team does not move them. Team membership does not replace the +workspace role: a person still needs the relevant workspace permission. -The team directory and the session collaborator picker identify people by display name and avatar. -Email addresses are included only for viewers with `workspace.members.read` (Owners and -Administrators in the built-in roles); all other viewers receive `email: null`, including team leads -and session owners. An unnamed user is labeled with a short user ID suffix instead of an email -address or full ID. This privacy rule applies in every team enforcement mode. +Any active workspace user can join an open team; invite-only teams require a lead or workspace +Owner/Administrator to add members. Leads and workspace Owners/Administrators manage membership, +lead roles, team metadata, and archive/restore. Members can leave with **Remove** on their own +membership row. The last lead cannot leave, be removed, or be demoted. -A team's session overview is available to its members and workspace Owners and Administrators, with -session visibility checks applied on the server. Team pages do not expose an audit activity feed. -Team operations are still recorded in the workspace audit log behind `workspace.audit.read`; its -team filter includes teams the reader does not belong to. +### Team Directory and Pages -The sidebar context defaults to **All my teams**, which leaves session lists unfiltered by team -while preserving server visibility checks. Users with at least one active team can choose Workspace -(teamless rows), a team, or All my teams; Owners and Administrators can also choose All teams. Users -without active teams have no selector and keep unfiltered lists. A stored Workspace or active-team -choice is retained; unknown or archived selections fall back to All my teams. +Every active workspace user can list active teams and their members. Archived teams are visible only +to their members and workspace Owners and Administrators. Email addresses in the team directory and +collaborator picker are shown only to viewers with `workspace.members.read`. -The new-session composer's team and visibility are draft-local choices initialized from the sidebar -context. Changing them does not change the sidebar or command-menu recents. If a team is required -and the context does not name one, the composer selects the user's first active team locally. +A team's session overview is available to its members and workspace Owners and Administrators, still +subject to session visibility. Team tabs appear according to the viewer's capabilities. Team +operations are recorded in the workspace audit log, which `workspace.audit.read` holders can filter +by any team. Sidebar team selections only narrow which readable sessions are listed. The new-session +composer's team and visibility, by contrast, set the created session's ownership and audience. ### Session Visibility -Each session stores a visibility independently of its team: - -| Visibility | Who can read the session when team enforcement is on | -| ----------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `workspace` | Workspace users with session read permission, even if the session has a team. | -| `team` | Members of the owning team, plus workspace Owners and Administrators, with session read permission. Requires an owning team. | -| `private` | The session owner and explicit collaborators (who must be current owning-team members on team-owned sessions) with session read permission. A workspace Owner can also open it by ID under audited break-glass access; Administrators do not get this exception. | - -Private visibility is enforced in every enforcement mode. An Owner's break-glass read is audited, -does not cause the session to appear in their lists, and does not grant prompt or sandbox access. An -Owner with the required lifecycle or delete permission can manage a private session they opened by -ID, but must also be a current member of the owning team if the session is team-owned; being an -Administrator alone does not grant access. Actorless bot services cannot read private sessions; -user-backed integration requests still depend on the acting user's access. - -For a team-owned session, **every non-read action requires current membership in the owning team** -in `off`, `shadow`, and `on` modes. Session owners, team leads, workspace Owners, and Administrators -are not exempt. This includes prompting, sandbox access, lifecycle operations, deletion, visibility -changes, and collaborator management, in addition to the relevant workspace permissions and -action-specific rules. Visibility still controls read access: a readable workspace-visible session -or an explicit private-session collaborator grant does not grant team membership or authorize -non-read actions. The sole exception is a collaborator removing themselves, which requires only -session read access. - -Owners and Administrators must join the owning team before acting on its sessions; team membership -changes are audited. - -The **Mine** filter helps find sessions you created but does not define who may access them. A -session's owner is its creating workspace user, not its team. Explicit collaborators are an access -grant for private sessions; they still need the relevant workspace permission to read, prompt, or -use the sandbox. Runtime participants record who connected or contributed and may carry runtime -credentials; being a participant alone is not a visibility grant. Conversely, making someone a -collaborator does not turn them into a runtime participant. Removing a collaborator revokes their -private-session access on subsequent authorization checks. Adding collaborators or removing someone -else requires `manageCollaborators`: after the session read check, only the session owner or a -workspace Owner may do so, with current owning-team membership for team-owned sessions. A -collaborator may remove themselves with session read access alone; team membership, collaboration, -or lifecycle permission is not required for self-removal. - -The collaborator picker is available to session owners and workspace Owners after the session read -and collaborator-management checks. For a workspace-owned session it lists every active workspace -user; for a team-owned session it lists only active members of the owning team, and adding anyone -else is rejected with `not_team_member`. Selecting a collaborator is an explicit private-session -access grant, not a team membership or workspace role change. On a team-owned session the grant is -honored only while the collaborator remains a current member of the owning team: leaving or being -removed from the team ends their collaborator access on the next authorization check, even though -the collaborator record itself is kept. - -Session actions have additional rules after visibility: prompting requires collaboration permission, -sandbox use requires sandbox permission, and lifecycle operations require lifecycle permission. With -team enforcement on, or for team-owned sessions in any mode, deletion requires delete permission -**and** session ownership, a lead role in the owning team, or a workspace Owner/Administrator role. -Team leads do not gain access to private sessions simply by leading the team. Changing private -visibility is reserved for the session owner or a workspace Owner. These role and ownership rules -never bypass the current-membership requirement for team-owned actions. Private-session action rules -apply even while team enforcement is off or in shadow mode. +Each session stores a visibility independently of its owning team, so a team-owned session can be +team-visible, workspace-visible, or explicitly private. + +| Visibility | Who can read the session when team enforcement is on | +| ----------- | ---------------------------------------------------------------------------------------------- | +| `workspace` | Workspace users with session read permission, even if the session has a team. | +| `team` | Members of the owning team, plus workspace Owners and Administrators. Requires an owning team. | +| `private` | The session owner and explicit collaborators, plus audited Owner break-glass reads by ID. | + +All rows also require session read permission. + +- **Private sessions are enforced in every mode.** An Owner's break-glass read is audited, does not + list the session, and does not grant prompt or sandbox access unless the Owner becomes a + collaborator. With the relevant permissions, the Owner can still manage the session's lifecycle, + deletion, collaborators, and visibility (with owning-team membership if it is team-owned). + Administrators have no break-glass exception, and actorless bot services cannot read private + sessions. +- **Team-owned sessions require current team membership for every non-read action** in every mode, + including for session owners, team leads, Owners, and Administrators. Visibility grants reads + only. The one exception is a collaborator removing themselves, which needs only read access. +- **Collaborators** are a private-session access grant, not a team membership or role change. Only + the session owner or a workspace Owner can add or remove other collaborators. On a team-owned + session, collaborators must be current team members (others are rejected with `not_team_member`), + and leaving the team ends their access. +- **Session owners and participants differ.** A session's owner is its creator, not its team, and + can read it even when private (with session read permission). Runtime participants and the + **Mine** filter are attribution and discovery only; they do not grant or limit access. + +Each action also needs its own permission: prompting needs collaboration, sandbox use needs sandbox +access, and lifecycle operations need lifecycle permission. Deleting a team-owned session, or any +session when enforcement is `on`, also requires session ownership, owning-team lead status, or a +workspace Owner/Administrator role. Only the session owner or a workspace Owner can change private +visibility. Leading a team does not grant access to its private sessions. + +### Team Defaults and Migration + +A team's default visibility accepts only `team` or `workspace` and sets the audience of new +sessions, not their ownership. `private` remains an explicit per-session choice. Migration `0085` +changes existing private team defaults to `team` without changing existing sessions, and adds +triggers that reject private defaults. **Apply migration `0085` before deploying the matching +application code**; otherwise existing private defaults can fail validation. ### Creating Sessions -Session creation checks the selected repository or environment as well as the creator's workspace -permission. Supplying a team requires active membership in that team and a grant covering **every** -repository used by the session; archived teams cannot be selected. Without an explicit visibility, -team sessions use the team's default and teamless sessions default to `workspace`. `team` visibility -requires a team; `private` requires a workspace user owner. A teamless session may still be private. +Creating a team session requires active membership in that team and a grant covering **every** +session repository; archived teams cannot be selected. Without an explicit visibility, sessions from +any launch source (including Slack, Linear, and automations) use the owning team's current default, +and teamless sessions use `workspace`. `team` visibility requires a team. Agent-spawned children of +a team-owned session require the active prompt author to still be a team member and fail with +`not_member` otherwise; they never borrow the parent owner's membership. -Owners and Administrators can configure **Settings > Teams > Require a team for new sessions** -(`requireTeamOnCreate`). It is off by default. When enabled, new sessions must select a team; it -refuses session creation API requests without a team with `team_required`. The web app supports team -selection; team selection in bots is a later phase, so their teamless creation API requests are also -refused when the setting is enabled. Automation runs are exempt until automation team ownership is -supported. The setting does not migrate or hide existing `ownerTeamId: null` workspace rows. +**Settings > Teams > Require a team for new sessions** (`requireTeamOnCreate`, off by default) +requires a team for new sessions, environments, and automations, including bot-created sessions. +Requests without one fail with `team_required`. Existing workspace-owned resources are unaffected, +and existing workspace-owned automations still run and create workspace-owned sessions. Team leads and workspace Owners/Administrators manage repository grants in the team's Repositories -tab or through `/teams/:id/repository-grants`. Team members and workspace Owners/Administrators can -read the grants. A team can have either installation-wide access or named grants by SCM repository -ID, but not both. Creating a team does not grant repository access. Repository-backed team sessions -without covering grants are refused with `target_team_missing_grant`. Repository-less team sessions -do not need grants. Removing a grant advances the team's grant version and leaves existing -repository references intact; grants do not yet narrow or revoke sandbox installation tokens. - -Repository skills, repository secrets, and repository image builds remain workspace-level resources; -grants do not assign them to an owning team. They keep their existing permission checks when no team -grants the repository. Once any team grants it, callers must be current members of an active -granting team (leads for repository secrets), or be a workspace Owner or Administrator. Installation -grants count for every repository. Importing repository secrets into an environment checks the -source repository's workspace-level grant access as well as the destination owning team's coverage, -if the environment has an owning team. These checks apply in every `TEAMS_ENFORCEMENT` mode. - -Manual environment image builds instead follow the environment's owning team. For a team-owned -environment, the caller must be a current member of that team or a workspace Owner/Administrator, -and the active owning team must have grants covering every current repository in the environment. -Membership or grants in another team cannot replace that coverage, including for Owners and -Administrators. Workspace-level environment builds keep their existing checks. These rules apply in +tab. A team has either installation-wide access or named grants by repository ID, and no grants by +default. Repository-backed team sessions without covering grants fail with +`target_team_missing_grant`; repository-less team sessions need no grants. Removing a grant narrows +future GitHub sandbox credentials but does not revoke issued tokens; GitLab's deployment PAT is +never narrowed. + +Repository skills, secrets, and image builds remain workspace-level resources. Once any team grants +a repository, using them requires membership in an active granting team (lead for repository +secrets) or a workspace Owner/Administrator role, in every enforcement mode. An installation-wide +grant counts as granting every repository. + +### Environment Access + +Team-owned environments require owning-team membership or a workspace Owner/Administrator role to +read or use. Managing any environment, including its secrets, settings, and images, requires +`environments.manage` in addition to the specific feature permission, plus lead status or a +workspace Owner/Administrator role for team-owned environments. Manual image builds also require the +owning team's grants to cover every repository. Environment names are unique within their owning +team or the workspace. + +A team environment can launch only into sessions owned by the same team. A team's session catalog +can also include workspace environments whose repositories its grants cover. These checks apply in every `TEAMS_ENFORCEMENT` mode. +### Ownership and Discovery + Sessions, automations, and environments cannot move between teams or between a team and the -workspace. A session's owning team is fixed at creation: a team-owned session never becomes -workspace-owned. Changing its visibility to `workspace` changes who may read it, not its ownership -or the team membership required for non-read actions. - -Visibility changes can include descendants. A cascading visibility change refuses the entire request -if any included descendant is inaccessible or denies the requested action, including the -current-membership check for each team-owned descendant; it does not silently skip that descendant. -The web visibility control requires a changed selection and asks for confirmation when applying -non-private visibility to child sessions, since private descendants will receive that visibility -too. Team grants constrain repository selection, not the source-control App token already available -to a running sandbox. - -Session discovery and inbox filters compose on the server: `ownerFilter=started` matches the -creator, `participating` also includes explicit collaborators and users with persisted read state, -and `anyone` adds no ownership filter. `visibility=team|workspace|private` and repeated `teamIds[]` -narrow the readable rows. `scope=workspace` means teamless sessions, not workspace visibility; -`scope=all` is reserved for workspace Owners and Administrators and does not bypass visibility or -enumerate break-glass-only private sessions. Inbox `mine=true` remains creator-only and excludes -direct automation and GitHub-bot sessions, but retains eligible agent descendants. +workspace. Changing a team session's visibility to `workspace` changes who can read it, not its +ownership or the membership required to act on it. Visibility changes that include child sessions +refuse the whole request if any child is inaccessible. + +Session lists and filters (`ownerFilter`, `visibility`, `teamIds[]`, `scope`) only narrow the +sessions a user can already read. `scope=workspace` means teamless sessions, not workspace +visibility. `scope=all` is reserved for Owners and Administrators and does not list break-glass-only +private sessions. ### Enforcement and Access Paths -Operators set `TEAMS_ENFORCEMENT` to `off`, `shadow` (the default), or `on`: - -- `off`: legacy read visibility for non-private sessions and legacy actions for non-private - workspace-owned sessions. Private access and the full action resolver for team-owned sessions - remain enforced. -- `shadow`: continue those legacy reads and workspace-owned actions while auditing would-be denials. - Private access and the full action resolver for team-owned sessions remain enforced. -- `on`: enforce visibility, team membership, and action/ownership rules for sessions. - -No mode relaxes current owning-team membership for non-read actions, including for workspace Owners -and Administrators. The visibility and collaborator mutation routes always enforce the session -access resolver, including in `off` and `shadow` modes. Those modes do not relax these mutation -checks or the checks on descendants included in a cascading operation. Collaborator self-removal -remains read-only-authorized in every mode. - -The session boundary covers four paths, not just the session page: - -- **HTTP item routes** authorize by the persisted session row before serving snapshots, actions, - children, exports, or other session-specific data. A session hidden by visibility responds with a - non-enumerating `404` rather than confirming that its ID exists. -- **Lists and aggregates** filter by visibility before returning sessions in search, inbox, child - lists, bulk export, and analytics. Private sessions do not appear in an Owner's lists solely - because of break-glass access; administrative analytics can include a scope-filtered, unattributed - private cost total without exposing those sessions. -- **Durable Object connections** recheck subscription and commands against the current session row, - so a stale browser tab does not turn a previous grant into lasting access. A private break-glass - subscription requires an audit write. -- **Sandbox access** is a separate session action. Snapshot sandbox URLs and supported sandbox tools - are not granted just because a session can be read; a break-glass Owner cannot use another - person's private sandbox without becoming a collaborator. Session-bound sandbox credentials are - not general user visibility grants. - -New HTTP requests reflect role, membership, collaborator, and visibility changes on the next check. -Live browser connections are rechecked at least every five minutes, so an existing connection may -remain open for up to five minutes after access changes. Recreating the session is not required. +Operators set `TEAMS_ENFORCEMENT` to `off`, `shadow` (the default), or `on`. `off` keeps legacy read +access to non-private sessions and legacy actions on non-private workspace-owned sessions; `shadow` +does the same while auditing would-be denials; `on` enforces team visibility and action rules. +Terraform exposes this as `teams_enforcement`, resolved in CI from the repository variable, then the +same-named secret, then `shadow`. The AWS configuration sets `shadow` in its `config` map; change +`TEAMS_ENFORCEMENT` there instead. Deploying Teams alone does not enable `on`. + +No mode relaxes private-session access, current membership for team-owned actions, checks on +visibility and collaborator changes, or environment, automation, repository-grant, and team-secret +checks. + +Session access is enforced on four paths: + +- **HTTP item routes** authorize against the stored session and return a non-enumerating `404` for + hidden sessions. +- **Lists and aggregates**, including search, inbox, bulk export, and analytics, filter by + visibility before returning results. Administrative analytics can include an unattributed total + cost of private sessions without exposing the sessions. +- **Durable Object connections** recheck subscriptions and commands against the current session. +- **Sandbox access** is a separate action; being able to read a session does not grant its sandbox. + +New HTTP requests reflect role, membership, collaborator, and visibility changes immediately. Live +browser connections are rechecked at least every five minutes, so one may remain open for up to five +minutes after access changes. + +### Reviewing Shadow Denials + +Before switching `TEAMS_ENFORCEMENT` from `shadow` to `on`, review the requests `on` would have +denied. Shadow records are observation only and are written only in `shadow` mode: + +- HTTP item routes record `authorization.request_allowed` with `shadow_denied:`. +- Session lists, inbox reads, child lists, and bulk exports record one `shadow_denied:batch` row per + request, with the hidden-row count in `metadata_json.shadowDenialCount`. +- WebSocket reads record `session.shadow_denied` at most once per connection, session, and reason. + +Analytics aggregates are not observed, and audit writes are best effort. In the audit log, WebSocket +records show **Would deny**; HTTP records show **Allowed** with a `shadow_denied:*` reason. For +daily counts by path and reason, run this query against the D1 `authorization_audit_events` table, +replacing the start date with your shadow rollout date: + +```sql +WITH shadow AS ( + SELECT id, date(occurred_at / 1000, 'unixepoch') AS day, + CASE WHEN action = 'session.shadow_denied' THEN 'websocket' + WHEN reason_code = 'shadow_denied:batch' + AND json_extract(metadata_json, '$.httpMethod') = 'GET' + THEN 'http_list' + ELSE 'http_item' END AS seam, + reason_code, metadata_json + FROM authorization_audit_events + WHERE occurred_at >= unixepoch('2026-10-01') * 1000 + AND reason_code LIKE 'shadow_denied:%' + AND action IN ('authorization.request_allowed', 'session.shadow_denied') +), reasons AS ( + SELECT id, day, seam, substr(reason_code, 15) AS reason + FROM shadow WHERE reason_code != 'shadow_denied:batch' + UNION + SELECT id, day, seam, json_extract(metadata_json, '$.shadowDenialReason') AS reason + FROM shadow + WHERE reason_code = 'shadow_denied:batch' + AND json_extract(metadata_json, '$.shadowDenialReason') IS NOT NULL + UNION + SELECT s.id, s.day, s.seam, json_extract(d.value, '$.reason') AS reason + FROM shadow s, json_each(s.metadata_json, '$.shadowDenials') d + WHERE s.reason_code = 'shadow_denied:batch' + UNION + SELECT id, day, 'http_item', json_extract(metadata_json, '$.shadowReason') + FROM shadow + WHERE reason_code = 'shadow_denied:batch' + AND json_extract(metadata_json, '$.shadowReason') IS NOT NULL +) +SELECT day, seam, reason, COUNT(*) AS would_be_denied_requests +FROM reasons +GROUP BY day, seam, reason +ORDER BY day, seam, reason; +``` + +`not_member` results cover any viewer outside the owning team, including users who belong to no +team. Counts are affected requests or WebSocket connections, not hidden sessions or unique users. ## How Automation Access Works -Automation definitions and run history are visible workspace-wide to roles with automation read -access. Creating, changing, and manually triggering automations use ownership rules. +Automations have a fixed owning team (or workspace ownership) and a separate executor, initially the +creator. Reading a team automation requires team membership or a workspace Owner/Administrator role +in every enforcement mode. Creating any automation requires both `automations.create` and +`sessions.create`. -- Members can manage and manually trigger automations they own. -- Administrators and Owners can manage and manually trigger any automation. -- Viewers can inspect automations but cannot create, change, or run them. +- Executors and owning-team leads can manage and trigger eligible automations with the `own` + permissions; `any` permissions extend this to all eligible automations. +- Administrators and Owners can manage and manually trigger any automation, but manually running a + team automation still requires membership in its owning team. +- Viewers can inspect eligible automations but cannot create, change, or run them. -Automation ownership follows the signed-in account that created it, not a display name or external -provider username. +Team leads and workspace Owners/Administrators can reassign the executor to another authorized user; +reassignment is audited. See [executor reassignment](AUTOMATIONS.md#executor-reassignment). Reading +an automation does not grant access to its sessions, and run history redacts sessions the viewer +cannot read. ### Scheduled and Event Runs -Scheduled and event-driven runs execute under the automation owner's authority. At run time, the -owner must still be active and allowed to create sessions and use every selected repository or -environment. If those permissions have been removed, the run does not start. +Scheduled and event-driven runs execute under the executor's authority. At run time, the executor +must still be active and allowed to create sessions and use the selected targets. Team runs also +require current team membership and grants covering their repositories, and their sessions use the +team's default visibility; workspace automations create workspace-visible sessions. If authorization +fails, the run does not start. ### Manual Runs @@ -324,6 +344,35 @@ Some integrations also apply their own ingress rules. For example, the GitHub in require an allowed trigger user or sufficient repository collaborator access before it sends a request to Open-Inspect. +### Slack and Linear Bindings + +Team leads and workspace Owners/Administrators bind Slack channels and Linear teams in **Teams > +Channels**, so requests from them create sessions owned by that team. Each channel or Linear team +belongs to at most one Open-Inspect team. Bindings do not add members or grants, and changing one +does not reassign existing sessions. Each integration's `unboundChannels` setting in **Settings > +Integrations** either creates workspace sessions from unbound channels (`workspace`, the default) or +rejects them (`reject`). + +In every enforcement mode, actorless bot reads scoped to an unbound channel or Linear team see only +workspace-owned, non-private sessions, so unbinding immediately revokes access to that team's +sessions. Slack never posts private sessions, even when the acting user can read them, or posts +team-owned sessions to a channel not bound to that team; confirmed publication denials close the +thread without session content, while a follow-up refused for one user does not close it for others. +Linear withholds completion results if the issue has moved to another Linear team. Its actorless +reads, including completion reads, otherwise follow `TEAMS_ENFORCEMENT`: full Team-visibility +isolation requires `on`. See [Slack](integrations/SLACK.md) and [Linear](integrations/LINEAR.md). + +### GitHub Routing + +GitHub routes by numeric repository ID rather than channel bindings. Event automations run as their +executor for their owning team, which must still hold a grant for the repository. Mentions use the +linked PR session's team, then a team of the sender that holds a grant (broken by their most recent +session when several qualify), then workspace ownership when neither resolves. Routing never +bypasses session-creation checks or the require-team policy. Deprecated auto-review-on-open remains +workspace-owned; use a team-owned GitHub Event automation instead. See +[GitHub](integrations/GITHUB.md), including +[upgrade steps](integrations/GITHUB.md#upgrading-to-repository-id-routing) for existing deployments. + ## Suspension Suspending a member disables their workspace access without deleting their account or historical @@ -334,19 +383,45 @@ After suspension: - New browser and bot operations are denied. - Existing browser sign-in sessions are invalidated. - Live browser session connections close within five minutes. -- Scheduled and event-driven automations owned by the member no longer pass run authorization. +- Scheduled and event-driven automations using the member as executor no longer pass run + authorization. - Existing session history and authorship remain intact. -Suspension does not automatically stop a sandbox that is already executing. An Administrator or -Owner can manage that session separately. +Suspension does not automatically stop a sandbox that is already executing. A user with the required +session action access can manage it separately; Administrators and Owners still need owning-team +membership for team-owned session actions. ## Repository and Credential Boundaries -Open-Inspect uses a shared source-control App installation for clone, fetch, and push operations. -The App should be installed only on repositories intended for the workspace. Team repository grants -check which repositories may be chosen for a team session; they do **not** narrow the shared -installation token delivered to a sandbox. Token narrowing is a future phase, not a protection -provided by team visibility today. +Open-Inspect uses a shared GitHub App installation for GitHub clone, fetch, and push operations. The +App should be installed only on repositories intended for the workspace. GitHub sandbox tokens cover +only the session's repositories, further limited to the owning team's grants for team-owned +sessions. Private submodules, dependencies, and sibling clones must be included in the session +before it starts. Unresolvable repositories fail closed rather than falling back to +installation-wide access. + +Removing a grant does not revoke tokens already issued; they remain valid until expiry. GitLab uses +a deployment-wide PAT that is not scoped per session. Teams do not establish multi-tenant isolation. +See [Sandbox Repository Access](GETTING_STARTED.md#sandbox-repository-access). + +### Credential Delivery and Snapshots + +Sandboxes fetch git credentials on demand from the control plane through the `oi-git-credentials` +helper, rather than receiving a clone token at launch. The helper caches the credential on disk +(mode `0600`) until shortly before expiry, and a cached credential can be used without a new +authorization check. + +Modal snapshots capture the full sandbox filesystem without clearing that cache, so a snapshot can +contain and restore cached tokens, along with any credentials written by setup scripts or the agent. +On GitLab the cached credential is the deployment-wide PAT, which stays valid beyond the cache +lifetime. Treat snapshots as sensitive artifacts. + +Image builds receive `VCS_CLONE_TOKEN` instead, scoped on GitHub to the repository or the +environment's repositories (intersected with the owning team's grants); GitLab builds receive the +deployment PAT. Files written during a build can persist in prebuilt images; see +[Secrets and Prebuilt Images](SECRETS.md#secrets-and-prebuilt-images). + +### User Credentials and Secrets A user's role determines whether they may read or use workspace repositories, but Open-Inspect does not compare that role with the user's personal GitHub access for each repository. Linked GitHub diff --git a/docs/AUTOMATIONS.md b/docs/AUTOMATIONS.md index db43bf20a5..6dab4c8772 100644 --- a/docs/AUTOMATIONS.md +++ b/docs/AUTOMATIONS.md @@ -24,8 +24,22 @@ new Sentry issues, and recurring report generation. Navigate to **Automations** in the sidebar, then click **Create Automation**. +Creation requires both `automations.create` and `sessions.create`, plus permission to use the +selected targets. The creator becomes the initial executor. The web creation and template entry +points also require both permissions; automation creation permission alone is insufficient. + Start by choosing a **Trigger Type**. The rest of the form adjusts based on that choice. +Choose workspace ownership or an owning team; team pages also expose an **Automations** tab. The +creator must belong to the selected active team, and that team's grants must cover the selected +repositories, including repositories resolved from environment targets. Newly created teams have no +repository grants. When `requireTeamOnCreate` is enabled, new automation definitions must have a +team. Existing workspace automations are not migrated or blocked from running by that setting alone. + +Ownership is fixed at creation: an automation cannot move between teams or between a team and the +workspace. Its **executor**, initially the creator, is a separate user identity that can be +reassigned without changing ownership. + ### Required Fields | Field | Description | @@ -195,13 +209,19 @@ means separate deliveries for the same automation can run at the same time. Successful requests return JSON in this shape: ```json -{ "ok": true, "triggered": 1, "skipped": 0 } +{ "ok": true, "triggered": 1, "skipped": 0, "steered": 0, "invocationId": "3f2a…" } ``` -`triggered` is the number of automation runs started. +`triggered` is the number of automation runs started. `invocationId` identifies the firing this +request belongs to (for a repeated `idempotencyKey`, the original firing), or is `null` when nothing +was recorded. Read its status with the same API key at +`GET /webhooks/automation//invocations/`, which returns +`{ invocationId, status, runs: [{ id, status, sessionId }] }` and never session content. -`skipped` is the number of matching runs that were ignored because of duplicate delivery or -concurrency protection. +`skipped` includes runs not started because of duplicate delivery, concurrency protection, or +runtime authorization denial. Authorization denial does not pause an event-driven automation or +record a run-history invocation; restore the required executor/team/target access before sending +another delivery. ### Error Responses @@ -359,6 +379,36 @@ Examples: ## Managing Automations +Workspace automations are readable with `automations.read`; team automations additionally require +owning-team membership or a workspace Owner/Administrator role. Executors and owning-team leads can +manage and trigger eligible automations with the corresponding `own` permissions; `any` permissions +allow those actions across eligible automations. Built-in Owners and Administrators have the latter; +Viewers cannot manage or trigger. These resource checks apply even in `off` or `shadow` session +enforcement mode. + +Scheduled and event runs use the executor's authority. Each run checks that the execution user is +active, can create sessions and use its targets, and, for team automations, is a current member of +the active owning team. Current team grants must cover all repositories being launched. Generated +sessions inherit the automation's owning team and that team's default visibility; workspace runs +remain workspace-owned and workspace-visible. Environment targets must have the same ownership as +the automation; unlike the team session picker, a team automation cannot select a workspace-owned +environment. + +If a scheduled run is denied execution authorization, it is recorded as **Skipped** and the +automation is paused immediately, clearing its next run time without adding a failure strike. +Restore the required access or reassign the executor, then click **Resume**. Restoring access or +reassigning alone does not resume the schedule. Event-driven authorization denials skip the event +without pausing the automation. + +### Executor Reassignment + +A team lead or workspace Owner/Administrator with automation management access can use +`PATCH /automations/:id` with `{"userId":""}` to change the executor. Being the +executor alone does not permit reassignment. The replacement must be active, able to launch the +stored targets, and a member of the active owning team for a team automation. The change is audited +as `automation.executor_changed`. Reassignment allows recovery when the old executor loses access +without recreating the definition; it does not move ownership or rewrite existing sessions. + ### Pause and Resume **Pausing** an automation stops it from firing. Scheduled automations will not run on their cron, @@ -377,6 +427,11 @@ triggers follow the same concurrency rules as all other runs: if the automation **Concurrent runs** bound, the trigger is rejected. Trigger Now also works while the automation is paused, so you can verify a fix before resuming. +Manual runs use the requester's authority and linked source-control credentials, not the stored +executor's. In addition to trigger permission, the requester must pass runtime session/target checks +and be a current member of the active owning team for a team automation, even if they are a +workspace Owner or Administrator. + ### Edit You can change an automation's name, repository selection, branch, model, and instructions at any @@ -396,8 +451,17 @@ any sessions it created are preserved. ## Run History -Each automation's detail page shows a chronological list of runs — one row per firing — with status, -duration, and links to the underlying sessions. +Each automation's detail page shows a chronological list of runs — one row per recorded invocation — +with status, duration, and links to the underlying sessions. Generic event execution-authorization +denials do not create invocation records and do not appear in this history. + +GitHub repository-grant denials are an exception. Matching events without an owning-team grant for +the repository can record a sessionless **Unauthorized** invocation with reason `repo_not_granted` +before executor authorization, including when the executor cannot be resolved. A missing executor +alone does not create that record. These denials increment the event response's `skipped` count, do +not pause the automation, and neither add a failure strike nor reset existing failures. See +`packages/control-plane/src/automation/github-event-admission.ts` and +`packages/control-plane/src/db/github-automation-store.ts` for this pre-admission path. A single-repository firing renders as a flat row, exactly as before. A multi-repository firing renders as one expandable row summarizing its repositories (for example "10 repositories — 8 @@ -406,16 +470,20 @@ reason, and session link. ### Run Statuses -| Status | Meaning | -| ------------------- | ------------------------------------------------------------------------------------------------------ | -| **Starting** | A session is being created for this run. | -| **Running** | At least one session is actively executing. | -| **Completed** | Every session finished successfully. | -| **Failed** | Every session encountered an error. The failure reason is shown on the run. | -| **Partial failure** | A multi-repository run where some repositories completed and some failed. | -| **Skipped** | The run was skipped because a previous run was still active (see [Concurrent Runs](#concurrent-runs)). | - -Click **View session** on any run to jump to the full session with its output and artifacts. +| Status | Meaning | +| ------------------- | ------------------------------------------------------------------------------------------------------------------- | +| **Starting** | A session is being created for this run. | +| **Running** | At least one session is actively executing. | +| **Completed** | Every session finished successfully. | +| **Failed** | Every session encountered an error. The failure reason is shown on the run. | +| **Partial failure** | A multi-repository run where some repositories completed and some failed. | +| **Skipped** | A previous run was still active, or a scheduled firing was denied execution authorization. | +| **Unauthorized** | GitHub event admission found no owning-team grant for the repository (`repo_not_granted`); no session was launched. | + +When authorized, click **View session** to open the session with its output and artifacts. +Automation read access is not session read access: history redacts session IDs, titles, and artifact +summaries for sessions the viewer cannot read. Lists do not expose private sessions solely through +an Owner's break-glass privilege; a qualifying single-run read audits that access. --- @@ -423,11 +491,11 @@ Click **View session** on any run to jump to the full session with its output an Automations display one of three statuses: -| Status | Meaning | -| ------------ | ------------------------------------------------------------------------------------- | -| **Enabled** | Running normally and ready to respond to its trigger. | -| **Degraded** | Enabled but has recent consecutive failures. The failure count is shown on the badge. | -| **Paused** | Not firing. Either manually paused or auto-paused after repeated failures. | +| Status | Meaning | +| ------------ | --------------------------------------------------------------------------------------------------------------------------------------- | +| **Enabled** | Running normally and ready to respond to its trigger. | +| **Degraded** | Enabled but has recent consecutive failures. The failure count is shown on the badge. | +| **Paused** | Not firing. Manually paused, auto-paused after repeated failures, or immediately paused after scheduled execution authorization denial. | --- diff --git a/docs/CLAUDE_AGENT.md b/docs/CLAUDE_AGENT.md index f4e4bb364d..1827fd9609 100644 --- a/docs/CLAUDE_AGENT.md +++ b/docs/CLAUDE_AGENT.md @@ -18,7 +18,9 @@ sign in inside a sandbox. Every session runs on exactly one harness, chosen when the session is created and fixed for its lifetime (like the base branch). Child sessions inherit their parent's harness. Automations carry a -harness for the sessions they create. Bots and integrations create OpenCode sessions. +harness for the sessions they create. The Linear integration has a harness setting (global and per +repository, OpenCode by default); see [Linear sessions](#linear-sessions). Slack and GitHub sessions +run on OpenCode. | Harness | Models | Anthropic authentication | Notes | | ---------------- | --------------------- | ------------------------------------------ | ------------------------------------- | @@ -33,6 +35,29 @@ An **installation default** Anthropic account only takes effect on Claude Agent bot and automation sessions on OpenCode keep using the API key, so setting a default never breaks sessions that cannot use it. +### Linear sessions + +**Settings > Integrations > Linear** has an **Agent harness** setting, globally and per repository +override; a repository override wins, and unset means OpenCode. The setting is a preference that +follows the model: Linear resolves the model first (`model:*` label, user preference, repository or +global default, deployment `DEFAULT_MODEL`), then runs the session on the configured harness when it +can run that model and on OpenCode otherwise. So with Claude Agent selected, an Anthropic model runs +on Claude Agent and a `model:gpt-*` label (or a non-Anthropic default) runs on OpenCode instead of +failing. The Linear activity names the harness: +`Creating coding session on (agent: Claude Agent, model: …)`. To keep every Linear session +on Claude Agent, choose an Anthropic default model, keep any repository override's model Anthropic, +and turn off **Allow user model preferences** and **Allow model labels**. + +Saving a harness and an incompatible model at the same level (for example Claude Agent with an +OpenAI model) is rejected, and the settings form only lists models the harness can run. A repository +override that sets only one of the two can still combine with the other level's value; the fallback +above covers it. + +Linear sessions are unattended, so on Claude Agent they follow the **Automated authentication** +policy: with a default Claude account and that policy on **Use default**, Linear usage draws on the +connected subscription; otherwise it uses `ANTHROPIC_API_KEY`. Switching the setting affects only +new sessions; follow-ups on an existing issue session keep the harness it was created with. + --- ## Setup @@ -194,6 +219,9 @@ fix instead. groups the sub-agent's activity under it the same way. Background sub-agents would let the turn end before their work is done and deliver their findings on a later turn nobody reads; as a second guard, the harness ignores the result of any turn it did not submit. +- **Memory.** The child runs with `CLAUDE_CODE_DISABLE_AUTO_MEMORY=1`, so Claude Code's file-based + auto memory (the `memory/` directory under `CLAUDE_CONFIG_DIR` and its system-prompt section) is + off. Open-Inspect's memory tools are the agent's only memory system. - **Follow-ups queue.** Both harnesses hold follow-up prompts until the running turn completes. - **Image.** The sandbox image pins `claude-agent-sdk`, whose wheel bundles the `claude` binary. The runtime manifest's `harnessMinimumGeneration` controls which prepared images new Claude sessions diff --git a/docs/DEBUGGING_PLAYBOOK.md b/docs/DEBUGGING_PLAYBOOK.md index 8ada3c5c97..e2fe752d0c 100644 --- a/docs/DEBUGGING_PLAYBOOK.md +++ b/docs/DEBUGGING_PLAYBOOK.md @@ -437,7 +437,10 @@ Fencing revokes a generation for good: the sandbox token hash is blanked, the so exits cleanly, and the supervisor logs `supervisor.boot_cancelled` and stops the sandbox. A fenced `failed` row never becomes `ready`, even if a late `ready` arrives. A connect-watchdog failure fences only when the provider can stop the sandbox; otherwise the boot may still connect later, -which is logged as `sandbox.failed_reconnected` and resumes as `connecting`. +which is logged as `sandbox.failed_reconnected` and resumes as `connecting`. A fenced +connect-watchdog failure re-drives the pending prompt onto a replacement sandbox right away. If +those timeouts open the circuit breaker, or the provider stop fails (a fenced generation is replaced +only after a confirmed stop), that prompt fails instead. ### "Why did a sandbox spawn fail?" diff --git a/docs/GETTING_STARTED.md b/docs/GETTING_STARTED.md index ab4da2e4df..32c44acf8e 100644 --- a/docs/GETTING_STARTED.md +++ b/docs/GETTING_STARTED.md @@ -37,6 +37,34 @@ be resolved from it. --- +## Teams Enforcement + +Cloudflare production Terraform's `teams_enforcement` sets the control-plane `TEAMS_ENFORCEMENT` +value. Accepted values are `off`, `shadow`, and `on`, with **`shadow` as the default**. For a fresh +deployment, explicitly set `teams_enforcement = "on"` in `terraform.tfvars` (as in Step 5) so the +control plane runs with `TEAMS_ENFORCEMENT=on` and enforces non-private Team-visibility reads. AWS +production is separate and defaults to `shadow`: add `TEAMS_ENFORCEMENT = "on"` to its `config` map +in `terraform/environments/aws-production/terraform.tfvars` to opt in, then apply Terraform and +restart the service as described in [AWS Bring-Up](AWS_BRING_UP.md). Do not assume a deployed +instance has full team read isolation simply because it includes Teams. + +For an existing deployment, review its production `shadow_denied:*` authorization audit entries +before opting into `on`. Resolve unexpected would-deny decisions and verify membership, repository +grants, and bot bindings before changing the deployed value. A default flip remains gated on that +production audit review; this guide does not establish that the gate has passed. Runtime and +Terraform defaults remain `shadow`, independently of the fresh-deployment opt-in shown here. + +Private-session access and current owning-team membership for non-read session actions remain +enforced in every mode. Environment/automation ownership, repository grants, and team-secret checks +also remain enforced. See [Enforcement and Access Paths](AUTH.md#enforcement-and-access-paths). + +Separately, administrators can enable `requireTeamOnCreate` in **Settings > Teams**. It defaults off +and requires a team for new sessions, environments, and automation definitions, including teamless +bot session creation requests. It does not migrate existing workspace resources or prevent existing +workspace automations from running when their runtime authorization checks pass. New teams start +without repository grants; configure named or installation-wide grants before selecting +repositories. + ## Overview Open-Inspect uses Terraform to automate deployment across multiple cloud providers: @@ -54,9 +82,9 @@ Open-Inspect uses Terraform to automate deployment across multiple cloud provide **Your job**: Create accounts, gather credentials, and configure one file (`terraform.tfvars`). **Terraform's job**: Create all infrastructure and configure services. -**How this guide is organized**: Steps 1–9 are the minimal deploy path: one web platform, the +**How this guide is organized**: Steps 1–10 are the minimal deploy path: one web platform, the default Modal sandbox provider, a GitHub App for repository access and sign-in, and the Slack, -Linear, and GitHub bots turned off. The [Optional Sections](#optional-sections) after Step 9 cover +Linear, and GitHub bots turned off. The [Optional Sections](#optional-sections) after Step 10 cover alternative sandbox providers, Google login, Slack, Linear, the GitHub bot, a custom domain, CI/CD, branding, updating, and troubleshooting. @@ -309,6 +337,12 @@ For GitHub sign-in, you should also have: - **Client ID** (e.g., `Iv1.abc123...`) - **Client Secret** (e.g., `abc123...`) +The App private key belongs in the control plane, not in a Modal `github-app` secret or a session +secret. Fresh and restored sandboxes obtain scoped Git credentials from the control plane on demand. +The optional GitHub bot Worker still needs its own App credential bindings; Terraform supplies them +when enabled. Modal's `llm-api-keys` and `internal-api` secrets remain required and are provisioned +by Terraform. + --- ## Step 4: Generate Security Secrets @@ -450,6 +484,7 @@ anthropic_api_key = "" # served by classification_anthropic_api_key, falling back to anthropic_api_key. # classification_model = "claude-haiku-4-5" # e.g. "gpt-5.4-mini" to classify on OpenAI classification_openai_api_key = "" # Required when classification_model is an OpenAI id +# classification_reasoning_effort = "low" # OpenAI ids only; blank keeps the model default # Security Secrets (from Step 4) token_encryption_key = "your-generated-value" @@ -487,6 +522,11 @@ allowed_github_orgs = "" # Comma-separated orgs whose act # Explicitly opt into open access only if you want any authenticated user to be # able to sign in when all allowlists are empty. unsafe_allow_all_users = false + +# Fresh deployment opt-in: sets control-plane TEAMS_ENFORCEMENT=on. +# The Terraform/runtime default remains shadow. Existing deployments must review +# production shadow_denied audit entries before opting in. +teams_enforcement = "on" ``` > **Core path bot settings**: The snippet above deploys with `enable_slack_bot = false`, @@ -648,7 +688,8 @@ curl -I "$(terraform -chdir=terraform/environments/production output -raw web_ap ``` Visit your web app URL and sign in with each configured provider. New users get the Member role, -which cannot manage secrets, so the end-to-end session test comes after Step 9. +which cannot manage secrets, so the end-to-end session test comes after Owner and team setup in +Steps 9 and 10. --- @@ -706,14 +747,32 @@ The preflight row should report `"status":"no-op"` with the detail command exits non-zero and the `detail` field gives the reason. The control-plane `/health` endpoint reports service liveness, not Owner status. -### Test the Full Flow - -1. As the Owner, add a model credential: go to **Settings > Secrets**, select the repository used - for this test, and add the key for your model (e.g. `ANTHROPIC_API_KEY` for Claude). Skip this if - you set `anthropic_api_key` in `terraform.tfvars` and will use a Claude model. See - [Secrets Management](SECRETS.md). -2. Create a new session with a repository, selecting a model whose credential you added -3. Send a prompt and verify the sandbox starts +## Step 10: Create the First Team and Test a Session + +1. As the bootstrapped Owner, open **Settings > Teams > Create team**, enter a name and slug, and + create your first team. The creator becomes its first **Lead** automatically; Owner bootstrap + alone does not create a team or membership. +2. Open the team's settings. Choose **Join policy**: **Invite only** (the creation default) or + **Open** for workspace users to join themselves. Choose **Default visibility** for new sessions: + **Team**, **Workspace**, or **Private**. Use Team for team-readable work, and confirm the + deployed control plane has `TEAMS_ENFORCEMENT=on` before relying on that read boundary. +3. In **Teams > your team > Repositories**, grant the repositories the team needs. New teams have no + grants. Prefer named repositories; use **All installation repositories** only when that broad + access is intended. Include every repository required by a multi-repository environment. +4. Have the other operators sign in once, then add them in the team's member list. For an open team, + they can instead use **Join team** from Teams. Assign Lead to operators who need team-management + duties. Workspace Owner/Administrator status is not a substitute for joining the team to perform + non-read actions on its sessions. +5. Add a model credential in the team's **Secrets** tab, or use **Settings > Secrets** for a global + or repository key (e.g. `ANTHROPIC_API_KEY` for Claude). Skip this if a suitable deployment-wide + credential is already configured. See [Secrets Management](SECRETS.md). +6. Create a session with this team selected, a granted repository, the intended visibility, and a + model whose credential is configured. Send a prompt and verify sandbox startup, repository + checkout, and streamed results. Have another team member verify the intended collaboration. +7. If new sessions, environments, and automation definitions must be team-owned, enable **Require a + team for new sessions** in **Settings > Teams** after membership and grants are ready. This + separate `requireTeamOnCreate` policy also covers new environments and automation definitions; it + does not migrate existing workspace resources. --- @@ -1025,6 +1084,16 @@ In Slack, for each channel where you want the bot to respond: The bot only responds to @mentions in channels it has been invited to. +#### Choose Team Routing + +In Open-Inspect, bind each Slack channel in **Teams > your team > Channels**. Then open **Settings > +Integrations > Slack** and save **Unbound channels**: `workspace` (the default, **Create +workspace-level sessions**) or `reject` (**Reject requests until the channel is bound**). This is +the integration's workspace-wide `unboundChannels` setting, not an environment variable or Terraform +input. Slack DMs are personal conversations and bypass the unbound-channel rejection policy; session +creation still checks other policies, including `requireTeamOnCreate`. A workspace fallback does not +override a requirement to choose a team. See [Slack Integration](integrations/SLACK.md). + --- ## Linear Agent (Optional) @@ -1061,8 +1130,15 @@ https://open-inspect-linear-bot-{deployment_name}.YOUR-SUBDOMAIN.workers.dev/oau ``` A Linear workspace admin must approve the installation. After installation, the agent appears in -mention and assignment menus. Test it by mentioning the agent on an issue, then use **View Session** -to follow the corresponding Open-Inspect session. +mention and assignment menus. Before testing, bind the external Linear team in Open-Inspect's +**Teams > your team > Channels**. In **Settings > Integrations > Linear**, save **Unbound Linear +teams**: `workspace` (the default, **Create workspace-level sessions**) or `reject` (**Reject +requests until bound**). This workspace-wide integration setting is named `unboundChannels`; it is +not an environment variable or Terraform input. Workspace fallback still must satisfy other creation +policies, including `requireTeamOnCreate`. + +Test by mentioning the agent on an issue in the bound Linear team, then use **View Session** to +verify the corresponding Open-Inspect session has the intended owning team and repository. For upgrades, enable **Client credentials tokens** before deploying. No reinstall is expected for an eligible existing installation, but allow already-running sessions to finish before upgrading @@ -1209,6 +1285,7 @@ ENABLE_LINEAR_BOT LINEAR_CLIENT_ID # Access control and branding +TEAMS_ENFORCEMENT ALLOWED_USERS ALLOWED_EMAIL_DOMAINS ALLOWED_EMAILS @@ -1253,6 +1330,10 @@ workflow default exists. Existing secret-only deployments need no migration. If variable wins; delete it to return to the secret. An empty variable does not clear an existing secret. Values such as `false` and `0` are strings in Actions variables and are preserved. +For Teams, both Terraform plan and apply use the repository variable `TEAMS_ENFORCEMENT`, then the +same-named secret, then `shadow`. Set the variable to `on` when enabling non-private team read +enforcement through CI; a local `terraform.tfvars` choice alone does not configure the workflow. + Keep credentials in the **Secrets** tab: API tokens/keys, OAuth client secrets, signing secrets, private keys, encryption keys, and both `MODAL_TOKEN_ID` and `MODAL_TOKEN_SECRET`. Allowlist values may contain personal information; leave them in secrets if you prefer masking in workflow logs. @@ -1268,6 +1349,7 @@ Secrets for credentials: | `R2_MEDIA_LOCATION` | R2 location hint for the media bucket (defaults to `ENAM`) | | `R2_MEDIA_BUCKET_NAME` | Optional media bucket name override for a pre-created bucket | | `DEPLOYMENT_NAME` | Your deployment name | +| `TEAMS_ENFORCEMENT` | Session enforcement: `off`, `shadow` (default), or `on`; prefer a repository variable | | `R2_ACCESS_KEY_ID` | R2 access key ID | | `R2_SECRET_ACCESS_KEY` | R2 secret access key | | `WEB_PLATFORM` | `vercel` or `cloudflare` | @@ -1355,6 +1437,8 @@ also requires the `CLASSIFICATION_OPENAI_API_KEY` secret; an Anthropic value is `CLASSIFICATION_ANTHROPIC_API_KEY`, falling back to `ANTHROPIC_API_KEY`. To keep the classifier key out of Modal and OpenComputer sandboxes, set `CLASSIFICATION_ANTHROPIC_API_KEY` and leave `ANTHROPIC_API_KEY` unset; sandboxes then take model credentials from Open-Inspect's secret store. +The optional `CLASSIFICATION_REASONING_EFFORT` variable sets the reasoning effort an OpenAI +classifier requests (for example `low`); leave it unset to use the model's default. When enabling or upgrading the Linear bot, also enable **Client credentials tokens** on the OAuth application in **Linear Settings → API → Applications**. This provider-side setting is not managed diff --git a/docs/HOW_IT_WORKS.md b/docs/HOW_IT_WORKS.md index b4b4946173..1be1af838f 100644 --- a/docs/HOW_IT_WORKS.md +++ b/docs/HOW_IT_WORKS.md @@ -107,14 +107,24 @@ high performance even with hundreds of concurrent sessions. An **environment** is a named, reusable set of repositories — the thing you reach for when the same multi-repository workspace comes up again and again (a frontend + its API, a service + its shared library). Environments are managed under **Settings > Environments** and appear at the top of the -new-session picker. +new-session picker when accessible. They can be workspace-owned or team-owned, with names unique +within that ownership scope. Ownership is immutable. Team pages also list their own environments. + +Team environments require membership in the owning team or a workspace Owner/Administrator role, +plus the appropriate environment permission. Managing one requires a lead or workspace +Owner/Administrator and `environments.manage`; secrets, settings, and image mutations also require +their respective feature permissions. A team environment can launch only into a session owned by +that team. A team's session catalog can include eligible workspace environments too, provided its +repository grants cover their members. Unbound bot catalogs do not expose team environments. +`requireTeamOnCreate`, when enabled, also requires a team for new environment definitions. See +[Authentication and Authorization](AUTH.md#environment-access). An environment defines: - **An ordered repository list** (up to 10) with a base branch per repository; the first repository is the primary -- **Environment secrets** — sessions launched from the environment receive global secrets plus the - environment's secrets (repository secrets do not flow in; see +- **Environment secrets** — sessions receive global secrets, then their owning team's secrets (if + any), then the environment's secrets; later scopes win. Repository secrets do not flow in (see [Secrets Management](./SECRETS.md#which-secrets-a-session-receives)) - **Optional prebuilt images** — the whole environment (all clones + all setup scripts) is built ahead of time so sessions boot in seconds (see [Pre-Built Images](./IMAGE_PREBUILD.md)) @@ -668,10 +678,11 @@ same organization. ### Why Single-Tenant? -The system uses a shared GitHub App installation for all git operations. This means: +GitHub git operations use a shared GitHub App installation. This means: -- Any user can access any repository the GitHub App is installed on -- There's no per-user repository access validation +- The installation defines the workspace's maximum repository reach; workspace permissions and team + repository grants further constrain access +- Open-Inspect does not compare a user's personal GitHub repository permissions with that scope - The trust boundary is your organization, not individual users This follows @@ -680,31 +691,30 @@ was built for internal use where all employees have access to company repositori ### Token Architecture -| Token | Purpose | Scope | -| ------------------ | ------------------------------------------ | -------------------------------- | -| GitHub App Token | Mint brokered git credentials | All repos where App is installed | -| User OAuth Token | Create PRs, identify users | Repos the user has access to | -| Sandbox Auth Token | Authenticate sandbox → control plane calls | Single session | -| WebSocket Token | Authenticate client connections | Single session | -| Managed LLM Token | Short-lived OpenAI or xAI model access | Pinned session provider account | - -Session sandboxes, whether fresh, prebuilt-image, or restored from a snapshot, fetch git credentials -on demand through the control plane instead of relying on a token embedded in the environment or -remote URL. The helper authorizes HTTPS requests for the configured SCM host, preserving existing -setup/start hooks that clone other private repositories available to the installation. This protects -continuously running sessions and persistent resumes from expired embedded credentials. One-shot -image builds still receive `VCS_CLONE_TOKEN` because they have no session to broker through. +Session sandboxes fetch git credentials through the control plane and cache them on disk. GitHub +credentials cover the persisted session repositories, intersected with current owning-team grants +only for team-owned sessions. GitLab returns the deployment PAT without per-session narrowing. Modal +filesystem snapshots can retain the helper cache; brokerage is not a token-free snapshot guarantee, +and grant removal does not immediately revoke issued credentials. + +Image builds receive `VCS_CLONE_TOKEN` because they have no session broker. For GitHub it is scoped +to the build repositories, with current owning-team grants applied for team-owned environment +builds. For GitLab it is the deployment PAT, not a repository-scoped or single-use credential. See +the canonical [access and credential boundaries](AUTH.md#repository-and-credential-boundaries) for +dependency access, cache behavior, and provider limitations. ### Secrets -You can configure environment variables (API keys, credentials) at global, per-repository, or -per-environment scope. A session receives global secrets plus its **session target's** secrets: +You can configure environment variables (API keys, credentials) at global, team, repository, or +environment scope. Precedence is global, then owning team when present, then the **session +target's** secrets; later layers win collisions: - **Global secrets** apply to all sessions (e.g., `ANTHROPIC_API_KEY`, `DEEPSEEK_API_KEY`, `ZHIPU_API_KEY`, `OPENCODE_API_KEY`) -- **Repository secrets** apply to sessions launched from that repo (including all bot-created - sessions) and override global secrets with the same key; ad-hoc multi-repository sessions receive - each selected repository's secrets, with the primary winning collisions +- **Team secrets** apply to sessions owned by that team, overriding global values +- **Repository secrets** apply to repository-targeted sessions and override global and team secrets + with the same key; ad-hoc multi-repository sessions receive each selected repository's secrets, + with the primary winning collisions - **Environment secrets** apply to sessions launched from that environment — its repositories' repository secrets do not flow in - Stored encrypted (AES-256-GCM) in D1 database diff --git a/docs/IMAGE_PREBUILD.md b/docs/IMAGE_PREBUILD.md index df48742664..a515cd2dc1 100644 --- a/docs/IMAGE_PREBUILD.md +++ b/docs/IMAGE_PREBUILD.md @@ -64,6 +64,13 @@ the pre-built image automatically — no changes to your workflow needed. 3. Saving the environment triggers the first build immediately; you can also click the rebuild button on the environment row at any time +Saving the prebuild toggle and scheduling its build require environment management access. Manual +rebuilds additionally require `environments.images.manage`. For team environments, management +requires a team lead or workspace Owner/Administrator with `environments.manage`, and manual builds +require the active owning team's grants to cover every current repository. Another team's grants do +not substitute. These checks are independent of the session enforcement mode; repository images +remain workspace resources with their own grant checks. + ### What You'll See in the UI The Images settings page (and the environment rows under Settings > Environments) show the status of @@ -127,6 +134,9 @@ Builds also trigger immediately, outside the schedule, when: - You **change an environment's secrets** — this additionally retires the existing ready image before the rebuild, so rotated values can't keep serving from an old image (see [Secrets Management](SECRETS.md#secrets-and-prebuilt-images)) +- You **write or delete an owning team's secrets**: images for that team's environments are + invalidated, with best-effort rebuild scheduling for prebuild-enabled environments. Shared + repository images are unaffected because they never receive team secrets. - You click the **manual rebuild** button — next to the repository in Settings > Images, or on the environment row in Settings > Environments @@ -192,8 +202,10 @@ Queue and stores build state in D1; Node delivers it with the `jobs.db` poller a state in `global.db`. Both hosts run the same finalization handler and retry contract. A failing setup script fails the whole build, and for environment builds the error names the -repository. Build-time secrets are exactly what the scope's sessions get: global + repository -secrets for a repository scope, global + environment secrets for an environment scope +repository. Repository builds receive global + repository secrets, **never team secrets**, because +their images are shared across teams. Environment builds receive global + the environment's owning +team (if any) + environment secrets; member-repository secrets do not flow in. Later scopes win. A +workspace environment build has no team layer, even when a team session later launches from it ([session-target scoping](SECRETS.md#which-secrets-a-session-receives)). Everything your setup scripts install — dependencies, build artifacts, caches — is captured in the @@ -218,7 +230,13 @@ the Worker until Modal's provider-session endpoints are available. A session boots from the ready image when the image's fingerprint matches the session's own repository snapshot (same repositories, same order, same base branches — for a single-repository -session that means the default branch): +session that means the default branch). + +Team environment images additionally require the session to have the same owning team, since setup +scripts may have persisted that team's secrets in the image. This check does not depend on session +visibility or the enforcement rollout mode. + +When an eligible image is ready: 1. The sandbox starts from the saved image artifact (code + dependencies already present) 2. A fast git sync pulls any commits pushed since the image was built (per repository) @@ -231,8 +249,9 @@ If no matching ready image is available (disabled, first build hasn't finished, succeeded, a non-default branch was selected, or the environment was edited after the session was created), the session falls back to the normal startup flow automatically. A **failed rebuild does not retire the image it was replacing**: an older ready image that still matches keeps serving -sessions until a newer build succeeds. If the saved artifact itself fails to restore, the image is -marked failed and the session retries from the base image. Either way, you'll never be blocked from +sessions until a newer build succeeds, unless already invalidated (for example, by an environment or +owning-team secret change). If the saved artifact itself fails to restore, the image is marked +failed and the session retries from the base image. Either way, you'll never be blocked from starting a session. **Ad-hoc multi-repository sessions never use prebuilt images.** Picking "Multiple repositories" in diff --git a/docs/MEMORY.md b/docs/MEMORY.md new file mode 100644 index 0000000000..9d79a38e7c --- /dev/null +++ b/docs/MEMORY.md @@ -0,0 +1,286 @@ +# Persistent session memory + +Memory carries useful knowledge between sessions without changing repository files. It has no +embedding store or semantic search: the agent sees a compact fact catalog, discovers additional +facts with lexical `memory_search`, and reads relevant bodies with `memory_read`. + +## User experience + +| Surface | Behavior | +| -------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Settings → Memories | Manage your personal facts and directives, review proposals, edit, archive, restore, supersede, revert to an earlier revision and inspect history. | +| Settings → Shared memories | Select a repository or environment. Readers see its catalog; authorized maintainers can manage it. | +| Personal default | **Include my personal memories in new sessions** is initially enabled. It applies to web, integration-created and scheduled sessions, and is available to every session creator. | +| Session sidebar → Memories | Pinned memories grouped by repository, environment or personal scope; hover or focus a row for its revision, inclusion and estimate. Flags omitted records and subsequent edits/archives. | + +Personal memories can be included in **shared sessions**. Included content may appear in responses +and be visible to collaborators. This does not give collaborators access to the owner's personal +settings catalog. Opting out excludes personal context and also denies personal reads and writes +through memory tools. Changing the default affects **new sessions only**; it cannot erase text +already supplied to an agent or included in a conversation. + +### Facts, directives and approval + +| Record | How it loads | Agent-write policy | +| -------------------------------- | --------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| Personal fact | Title and description in the catalog; body read on demand | Active immediately only in a root session that has remained private and owner-only. Otherwise proposed for owner approval. | +| Personal directive | Full content | Proposed for owner approval. | +| Repository/environment fact | Catalog; body on demand | Proposed for maintainer approval. | +| Repository/environment directive | Full content | Proposed for maintainer approval. | + +Human-authored records are active immediately. Editing keeps the original record provenance and adds +a revision with the editor's identity. Rejected proposals are archived; restoring them returns them +to proposed status. Restoring an approved record returns it to active status. A replacement proposal +does **not** archive its predecessor until approval. Concurrent edits and decisions reject stale +revisions with HTTP 409. + +Sandbox tools currently authenticate a session, not an immutable prompt author. For this reason, +automatic personal facts are disabled permanently once a session is shared or gains a collaborator, +even if it is later made private again. Children never auto-save personal facts. A child created by +a different participant cannot write to the inherited owner's personal scope. + +## Architecture and implementation + +| Layer | Where | Responsibility | +| ---------------- | --------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Shared contracts | `packages/shared/src/types/memories.ts`, `memory-tools.ts` | Wire schemas (management DTOs, selection summaries, sandbox requests/responses), limits, scope helpers, the lifecycle transition table and the agent tool definitions. `memories.manage_own`. | +| Domain | `control-plane/src/memory/` | Partitions (identity only; records carry a separate display `scope`), selection and budget, rendering, initial status, DTO projection, `SessionMemorySelector` for new sessions and `SessionMemoryService` for agent operations. Services receive their stores through constructors; `*-factory.ts` modules and routes wire D1. | +| Access policy | `control-plane/src/authorization/memory-access.ts` | `MemoryManagementPolicy` for humans and `SessionMemoryAccessPolicy` for session principals, both returning typed decisions with stores injected; `memory-access-factory.ts` wires D1. | +| Stores | `control-plane/src/db/memory-records.ts`, `session-memory-selections.ts`, … | SQL only: revisioned records and lifecycle, pinned selections, preferences, lexical fact search and the commit-time agent write guard. Queries are built with the `sql` fragment template. | +| D1 | Migration `0084_memories.sql` | Records keyed by typed identity columns (`owner_user_id` — a foreign key to `users` — `repo_id` or `environment_id`, exactly one per `partition_type`) with a generated, indexed `partition_key`; repository names are display-only. Immutable revisions, per-user default, session manifest headers and ordered revision references. | +| Session creation | `routes/session-create.ts`, scheduler, child spawn | Select and pin the session's memories in the session insert's transaction (`Pinned`: resolved for roots, inherited by children). Schedulers use the execution owner's default. | +| Runtime boot | `sandbox-runtime/src/sandbox_runtime/memories.py` | Fetch the rendered memory with the session-bound token, clear stale restored content, then atomically write owner-readable `oi-memory.md` in the harness configuration directory. | +| Harness tools | `tools/_memory.js`, `harness/memory_tools.py` | Both harnesses build `memory_read`, `memory_search` and `memory_write` from the generated specs and forward arguments verbatim. OpenCode reads the file via `instructions`; Claude appends it. | +| Web | `web/src/hooks/use-memories.ts`, `components/settings/memories-settings/` | Typed queries and mutations (revision fencing via `If-Match`), owner/shared management pages and session diagnostics. | + +### Extending memory + +- **A new scope** (for example, team): add it to `MEMORY_SCOPE_TYPES` and `memoryScopeSchema`, then + follow the compiler — every switch over scopes and partitions is exhaustive (`partition.ts`, + `sources.ts`, `memory-access.ts`, the write guard, the shared scope helpers and the web settings + link). Storage needs one typed identity column (for example, `team_id` with its foreign key), + added to the `partition_key` expression and the check constraint; the indexes are unchanged. +- **A new lifecycle action or state:** add an entry to `MEMORY_TRANSITIONS`; the store, routes, DTO + capabilities and web action buttons all derive from it. +- **A new agent tool:** add it to `MEMORY_TOOLS` in `packages/shared/src/memory-tools.ts`, run + `npm run generate:memory-contract -w @open-inspect/shared`, and add a four-line OpenCode wrapper + in `tools/`. The Claude harness picks it up from the generated JSON. A shared test fails when the + generated artifacts are stale. +- **Selection semantics:** bump `MEMORY_SELECTION_VERSION`. It is provenance only; loaders never + branch on it, and rendering always uses the current format with its own hard limit, so existing + sessions keep booting. + +No project scope, embeddings, automatic memory search, repository writes or new infrastructure +services are required. Memory estimates are available on the session manifest; a broader +per-component context-snapshot feature is not introduced here. + +### Pinning and live reads + +Memory boot requires runtime v74 or later for either harness. The runtime/rebuild floors reject +pre-memory repository images. Restores remove old context and abandoned staging files before +fetching; new content is installed through a unique exclusive 0600 file and atomic replacement. + +- Directives and catalog entries use **pinned revisions** for the session's lifetime, including + sandbox restarts and restores. A child copies the same selection and personal owner, not the + spawning participant's personal catalog. +- `memory_read` returns the **current** fact body and provenance. A pinned archived record returns + only its archive notice (`archiveKind`, `archivedAt`, `archiveNote`). Proposals, unpinned + archives, and directives cannot be expanded by sandbox tools. +- A top-level session may directly read an active fact in its authorized scopes even if budget + limits omitted it. An inherited child cannot expand into unpinned personal records. +- Editing or archiving does not rewrite an existing session's injected text. New sessions resolve + from current active records. +- The selection hash covers the ordered record/revision/inclusion tuples and the personal selection + flag. User aliases can be merged without changing the pinned selection hash. + +### Limits and ordering + +| Limit | Value | +| ----------------------------------- | -------------------------------------------------------: | +| Title | 200 characters | +| Description | 10–420 characters | +| Directive body | 2,000 characters | +| Fact body | 20,000 characters | +| Directives per scope / overall | 6,000 / 12,000 body characters; at most 100 directives | +| Fact catalog | 24,000 title+description characters, at most 200 entries | +| Rendered boot context | 240,000 characters including labels/escaping | +| Management page | 50 records by default, at most 100 | +| Accepted agent writes per session | 20, including subsequently archived records | +| Pending agent proposals per session | 5 | + +Scope priority is environment, repositories in session order, then personal. Directives are oldest +first; facts are most recently updated first; IDs break timestamp ties. Records beyond a budget are +omitted whole and counted in aggregate; only selected items are persisted. Candidate queries are +bounded per scope/type and never fetch fact bodies. Token estimates use rendered text length / 4, +including labels and framing; they are estimates, not provider-measured token counts. + +## API + +All human memory responses are private/non-cacheable. Identity comes from authentication, never a +request body. Repository management uses existing repository permissions and team grants; +environment management uses existing ownership/management authorization. Personal management is +owner-only, including when the caller is another administrator. + +| Method and path | Purpose | +| --------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `GET /memories?scope=...&status=...&offset=...&limit=...` | Page through one scope (`personal`, `repository`, or `environment`) and status (`active`, `proposed`, or `archived`). Repository scope also takes `repoOwner` and `repoName`; environment takes `environmentId`. `nextOffset` is null at the last page. | +| `POST /memories` | Create a memory, optionally superseding an active one via `supersedesMemoryId`; the superseded record lists it in `supersededByMemoryIds`. | +| `GET /memories/:id` | Current record and server-calculated management capabilities. | +| `PATCH /memories/:id` | Revise content; `If-Match: ` is required (428 when missing, 409 when stale). | +| `GET /memories/:id/revisions` | Immutable revision history. | +| `POST /memories/:id/{action}` | `approve`, `reject`, `archive`, or `restore` with `If-Match`; only `archive` and `reject` accept an optional `archiveNote`. Transitions follow `MEMORY_TRANSITIONS`. | +| `POST /memories/preview` | Summarize the selection a new session would pin (items, sizes, omissions), without creating it. | +| `GET, PUT /memory-preferences` | Read/save the current user's personal inclusion default. | +| `GET /sessions/:id/memories` | The pinned selection summary with per-item drift, readable with the session. | +| `GET /sessions/:id/sandbox-memory` | The session's rendered memory (boot context), sandbox-bound. | +| `GET /sessions/:id/sandbox-memory/:memoryId` | Sandbox-bound live read. | +| `POST /sessions/:id/sandbox-memory` | Sandbox-bound agent write with scope/approval/quota checks. | +| `POST /sessions/:id/sandbox-memory/search` | Session-authorized lexical discovery of active current facts, including records outside the boot catalog. | + +Sandbox routes reject credentials belonging to another session and recheck current workspace/team +repository grants and environment ownership on every boot render, read, search and write. Agent +inserts additionally enforce, atomically, the facts about the writing session: an active owner, a +live (`created`/`active`) session that still reaches the target partition, personal auto-save +eligibility, quotas and the replacement predecessor. Settled sessions must be reactivated by a fresh +prompt before writing. Grant rules are deliberately not duplicated in that SQL guard: a grant +revoked in the milliseconds between the route check and the insert can leave a shared-scope +_proposal_, which every later read denies and a human must approve. Repository memories are keyed by +the stable repository ID, so a reused name never inherits them and a renamed repository keeps them. +Record-level management (read, edit, history, lifecycle actions) authorizes the stored partition +identity directly and never resolves a record through its display names. Browser responses for +previews and session status carry a selection summary only, never the personal owner or hash. + +New sessions never fail because of memory: `SessionMemorySelector` omits repositories or an +environment that the session principal cannot read from the selection, and session admission remains +`authorizeSessionTarget`'s job. Audits record record/revision/status/actor/session IDs, never memory +content or private archive-note text. Scope identifiers are retained after target deletion to +preserve historical manifests. There is no hard-delete endpoint. Restoring an approved memory is +allowed only when its entire replacement family has no active record. + +### Agent write destinations + +`memory_write` uses session-relative scopes. The control plane derives the personal owner, the sole +repository, or the associated environment from the authenticated session: + +```javascript +memory_write({ + scopeType: "repository", // Or "personal" / "environment" + memoryType: "fact", + title: "Integration test setup", + description: "Database preparation required before the integration suite.", + content: "Start Postgres and apply test migrations before running integration tests.", +}); +``` + +For a multi-repository session, add **both** `repoOwner` and `repoName` to select a member +repository. This applies to ad-hoc sets and environment-backed sessions alike. Omitting the selector +returns HTTP 400 with the available repository names; the server never silently defaults to the +primary repository. An explicit non-member, absent environment/repository, or repository without a +stable ID is denied. Sandbox writes do not accept `environmentId`; environment identity is always +derived from the session. Human management APIs still require explicit repository/environment +identities. + +Both harnesses forward the tool arguments unchanged to `POST /sessions/:id/sandbox-memory`; the +endpoint's request schema is the tool input schema (`sandboxMemoryWriteSchema`), so `scopeType` is +`"repository"`, `"environment"`, or `"personal"` and an explicit selector is the top-level +`repoOwner`/`repoName` pair. The server resolves a complete scope before performing the existing +current-access and commit-time checks. Inference does not change approval, opt-out, quotas, +replacement rules, or pinned context. + +## Local verification + +### Searching beyond the catalog + +Both harnesses expose `memory_search`. It discovers **current active facts**, not directives, +proposals, archives, or historical revision text. Search results contain IDs, revision IDs, scope +labels, titles and descriptions; call `memory_read` for the body. + +```javascript +memory_search({ query: "billing webhook deduplication", scopeType: "repository", limit: 10 }); +``` + +Omit `scopeType` to search permitted session scopes. Repository scope searches all attached +repositories; optionally supply both `repoOwner` and `repoName` to select one. Personal owner and +environment identity are derived from the session. Personal opt-out excludes personal results, and +inherited children can discover only their pinned personal memory IDs. Selected shared scopes are +authorized before the query and checked again before returning results. + +Queries are literal whitespace-separated keywords: every term must occur in the current title, +description or body. Matching uses SQLite's ASCII case folding; it provides no stemming, synonyms, +semantic similarity or wildcard/query-language syntax. SQL wildcard characters are escaped. Per-term +title matches score 5, description matches 3, and body matches 1; the strongest field match for each +term is summed. Updated time and memory ID break ties. SQL matches and ranks before limiting, across +the full applicable fact store rather than the latest 200 records. + +Queries are 2–256 characters with at most eight distinct terms. Results default to 10, max out at +20, and fit a 24,000-character serialized response budget. `hasMore` signals that the agent should +refine its query. There is no pagination or unrestricted browsing endpoint. Empty memory context +does not remove the search tool, and searching does not alter the session's pinned context. + +Search sits behind the `FactSearchIndex` port (`memory/fact-search.ts`): the session service builds +an engine-independent `FactQuery` from authorized partitions and shapes the bounded response, and +`LexicalFactIndex` answers it with one escaped-`LIKE` statement ranked in SQL by +`FACT_SEARCH_FIELDS`. A different engine (Postgres full-text search, embeddings) is another adapter; +no migration or new service is needed today. Body matching still scans text; bounded results do not +guarantee constant query cost. The opt-in Node SQLite benchmark records query plans and +rare/broad-query timings at 1,000 and 10,000 facts with representative and maximum-size bodies: + +```bash +MEMORY_SEARCH_BENCHMARK=1 MEMORY_SEARCH_BENCHMARK_OUTPUT=/tmp/memory-search-benchmark.json \ + npm test -w @open-inspect/control-plane -- \ + src/db/lexical-fact-index.test.ts --maxWorkers=1 +``` + +These local measurements are not deployed D1 latency or provider-canary proof. Indexed lexical +search and semantic retrieval remain separate follow-ups if corpus size or measured cost demands +them. + +### Test commands + +Use Node 24; build shared before dependent TypeScript checks. Run heavyweight checks sequentially. +No Cloudflare, Modal or model-provider credentials are needed for these tests. + +```bash +npm ci +npm run build -w @open-inspect/shared +npm test -w @open-inspect/shared -- --maxWorkers=1 +npm test -w @open-inspect/control-plane -- --maxWorkers=1 +npm run test:integration -w @open-inspect/control-plane -- \ + test/integration/memories.test.ts \ + test/integration/memories-routes.test.ts \ + test/integration/memories-access.test.ts \ + test/integration/memories-search.test.ts --maxWorkers=1 +npm test -w @open-inspect/web -- --maxWorkers=1 +npm run typecheck -w @open-inspect/control-plane -w @open-inspect/web +npm run lint:sql-portability + +cd packages/sandbox-runtime +uv sync --frozen --extra dev --python 3.12 +# A short disposable temp path avoids macOS Unix-socket path-length failures. +uv run --frozen --extra dev pytest tests --basetemp=/tmp/oi-memory-pytest +OPENCODE_TEST_BINARY=/path/to/opencode-1.18.29 \ + uv run --frozen --extra dev pytest tests/test_opencode_reasoning_contract.py -k memory_text +``` + +The optional binary test uses isolated configuration, synthetic credentials and a localhost fake +provider. It verifies actual OpenCode prompt/tool serialization, not a real model's behavior. + +## Rollout and failure behavior + +1. Apply migration 0084 with the existing D1/Node migration mechanism. +2. Deploy the control plane. Existing sessions without a manifest receive empty memory context; they + are not retroactively resolved. No new Durable Object binding is needed. +3. Rebuild/deploy the sandbox runtime image, then deploy the web app. Both harnesses must use the + updated runtime to gain the tools and boot phase. +4. In a disposable session, create a personal directive and fact, verify the preview and loaded + diagnostics, exercise read/write/approval, then repeat with personal inclusion disabled and with + a child session. Verify restore and archived-read behavior before broad rollout. + +An old control plane returning 404 produces empty memory, clearing stale files. Transient fetch +errors retry a bounded number of times. Unauthorized, malformed or exhausted fetches fail the +`memory` boot phase before the harness starts. A failed database transaction creates neither orphan +revisions nor successful domain audit records. Rollback application code without dropping the memory +tables; already-injected text cannot be revoked from a running conversation. + +Local tests are not evidence of deployed Cloudflare or Modal behavior. Deployment/provider canaries +remain a separate rollout gate. diff --git a/docs/MULTI_REPO_AUTOMATIONS.md b/docs/MULTI_REPO_AUTOMATIONS.md index 4f491616ca..f6fa46cfe5 100644 --- a/docs/MULTI_REPO_AUTOMATIONS.md +++ b/docs/MULTI_REPO_AUTOMATIONS.md @@ -15,16 +15,22 @@ where changes are needed. ```text automation ── repositories (0..10, the live selection) │ - └── invocation one per firing (schedule tick, Trigger Now, or event) + └── invocation one per recorded firing (schedule tick, Trigger Now, or event) │ carries the firing-scoped keys and skip reason └── runs (0..10) one per repository, each linked to one session └── session ordinary sandbox session; owns branch, artifacts, PR ``` -Every firing — single-repo, multi-repo, repo-less, or skipped — takes the same path: it records one -`automation_invocations` row, and unless it was skipped, one `automation_runs` child per repository. -There is no separate single-repo pipeline and no group-of-N special case; a single-repo firing is -simply an invocation with one run. +Recorded firings — single-repo, multi-repo, repo-less, or skipped — take the same path: one +`automation_invocations` row, and unless skipped, one `automation_runs` child per repository. There +is no separate single-repo pipeline and no group-of-N special case; a single-repo firing is simply +an invocation with one run. + +Generic event execution-authorization denials increment the response's `skipped` count without +creating an invocation. GitHub's earlier trigger-repository grant check is an exception: it can +record a sessionless `unauthorized` run with reason `repo_not_granted`, including before an executor +is resolved. Scheduled authorization denials instead record a skipped invocation and pause the +automation; see [Automations](AUTOMATIONS.md#managing-automations). **API ↔ UI vocabulary.** The API speaks `automation / repository / invocation / run / session`. The UI keeps its established "run" copy: the history section is still titled "Run History", the empty @@ -229,9 +235,10 @@ invocation aggregates links for display only; there is no cross-repository "mega ## API surface -- `GET /automations/:id/invocations` — the history endpoint: one entry per firing +- `GET /automations/:id/invocations` — the history endpoint: one entry per recorded invocation (`{invocations, total}`), each carrying its child `runs` with repository snapshots. `total` counts - invocations. + invocations. Generic event execution-authorization denials are not recorded; GitHub pre-admission + grant denials can appear as `unauthorized` instead. - `POST /automations/:id/trigger` — returns `201 {invocationId, runs}`; `409` when blocked by an active invocation. - Repository selection is written via `repositories: [{repoOwner, repoName, baseBranch?}]` on diff --git a/docs/SECRETS.md b/docs/SECRETS.md index c34a0f39c8..a787595a30 100644 --- a/docs/SECRETS.md +++ b/docs/SECRETS.md @@ -9,12 +9,9 @@ control plane instead of being injected into sandboxes. ## Quick Start -1. Open your Open-Inspect web app and go to **Settings** -2. Navigate to the scope you want: - - **Global or repository secrets**: the **Secrets** tab (selected by default) — use the scope - dropdown at the top to choose **All Repositories (Global)** or a specific repository - - **Environment secrets**: the **Environments** tab — open the environment and switch to its - **Secrets** tab +1. Open your Open-Inspect web app. +2. Open **Settings > Secrets** for global/repository secrets, an environment's **Secrets** tab under + **Settings > Environments**, or a team's **Secrets** tab for team secrets. 3. Click **Add secret**, enter a key and value, then click **Save** That's it — the next sandbox you launch from that scope will have the secret available as an @@ -27,30 +24,36 @@ environment variable. | Scope | Applies to | Use case | | --------------- | --------------------------------------- | ---------------------------------------------------------------------- | | **Global** | All sessions | API keys shared across projects (`ANTHROPIC_API_KEY`, `ZHIPU_API_KEY`) | +| **Team** | Sessions owned by that team | Credentials shared within the owning team | | **Repository** | Sessions launched from that repo | Repo-specific credentials (`STRIPE_SECRET_KEY`, `AWS_ACCESS_KEY_ID`) | | **Environment** | Sessions launched from that environment | Credentials curated for a multi-repository environment (see below) | Global and repository secrets are managed under **Settings > Secrets**; environment secrets are -managed on the **Secrets** tab of each environment under **Settings > Environments**. +managed on the **Secrets** tab of each environment under **Settings > Environments**. Team secrets +are managed on the team's **Secrets** tab by team leads and workspace Owners/Administrators. The +team-secret list/write/delete routes require a human user; bots and sandbox credentials cannot +manage them. Secret values are not returned, even to managers. -**Precedence**: Repository (or environment) secrets override global secrets with the same key. When -viewing a repository's secrets, inherited global keys are shown in a read-only section with a -"Global" badge. If you override a global key at the repo or environment level, the global entry -shows which scope overrode it. +**Precedence**: global, then owning team, then repository or environment; later scopes override the +same key. Workspace-owned sessions have no team layer. When viewing a repository's secrets, +inherited global keys are shown in a read-only section with a "Global" badge. If you override a +global key at the repo or environment level, the global entry shows which scope overrode it. ### Which secrets a session receives -A session receives **global secrets plus its session target's secrets** — the session target is -whatever you picked when creating the session: +A session receives **global secrets + owning-team secrets (if any) + session-target secrets**, in +that order. Team membership alone does not select secrets: the session's owning team does. The +session target is whatever you picked when creating the session: -- **Single repository** (web picker, Slack, GitHub, Linear): global + that repository's secrets. -- **Environment**: global + that **environment's** secrets only. The repositories inside the - environment do **not** contribute their repository secrets — environments are curated, so a key - added to a repository never silently lands in every environment containing it. To reuse a - repository secret, import it (below) or move it to global scope. -- **Ad-hoc multi-repository session** ("Multiple repositories" in the picker): global + each - selected repository's secrets. On key collisions the **primary repository** (first in the list) - wins. +- **Single repository**: global + owning team (if any) + that repository's secrets. +- **Environment**: global + owning team (if any) + that **environment's** secrets. Its repositories + do **not** contribute their repository secrets: environments are curated, so a key added to a + repository never silently lands in every environment containing it. To reuse a repository secret, + import it (below) or move it to global scope. +- **Ad-hoc multi-repository session** ("Multiple repositories" in the picker): global + owning team + (if any) + each selected repository's secrets. On key collisions the **primary repository** (first + in the list) wins. +- **No repository**: global + owning team (if any). The new-session picker states this disclosure for environment and multi-repository selections. @@ -61,6 +64,12 @@ environment: pick the source repository, select the keys, and the values are cop control-plane-side (never displayed). Imports are **copies** — if you later rotate the value on the repository, re-import it or update the environment secret directly. +Imports require destination environment management access and `environments.secrets.manage`. The +source repository must pass the workspace repository-grant check (a lead in an active granting team, +or a workspace Owner/Administrator, when the repository has grants). A team-owned destination also +requires its owning team's grant to cover the source repository; access through another team does +not substitute for that grant. Repository secrets remain workspace resources, not team-owned ones. + ### When to use global secrets Use global secrets for keys that every session needs regardless of which repository it runs against. @@ -119,14 +128,14 @@ Click the delete button next to any secret row and confirm. ## Limits -| Constraint | Limit | -| -------------------------------- | ------------------------------------------------------- | -| Max secrets per scope | 50 | -| Max key length | 256 characters | -| Max value size | 16 KB | -| Max total value size (per scope) | 64 KB | -| Max combined size per session | 128 KB (global + session target, after merging) | -| Key format | `[A-Za-z_][A-Za-z0-9_]*` (letters, digits, underscores) | +| Constraint | Limit | +| -------------------------------- | ------------------------------------------------------------- | +| Max secrets per scope | 50 | +| Max key length | 256 characters | +| Max value size | 16 KB | +| Max total value size (per scope) | 64 KB | +| Max combined size per session | 128 KB (global + owning team + session target, after merging) | +| Key format | `[A-Za-z_][A-Za-z0-9_]*` (letters, digits, underscores) | If the merged payload for a session (or an image build) exceeds the combined cap, the spawn fails with an error that attributes bytes per contributing scope so you know what to trim. This mostly @@ -148,6 +157,21 @@ If you try to save a reserved key, the UI will show a validation error. ## Security +### GitHub App credentials are not session secrets + +Fresh and restored sandboxes obtain scoped Git credentials on demand from the control plane. Keep +the GitHub App private key in the control plane; do not add it as a global, team, repository, or +environment secret. The optional GitHub bot Worker still needs its own App credential bindings. + +The legacy Modal `github-app` secret is optional and is no longer required for sandbox Git +credentials. This does not remove the required Modal `internal-api` secret (`MODAL_API_SECRET` and +`ALLOWED_CONTROL_PLANE_HOSTS`) or the `llm-api-keys` secret object. The latter may contain an empty +model key when sessions receive their model credentials from the control-plane secret store. +Terraform provisions these required Modal secrets. See +[Modal setup](../packages/modal-infra/README.md#prerequisites). + +### Stored secret protection + - Secrets are encrypted with **AES-256-GCM** before being stored in the database - Values are **never returned by the API** after saving — only key names are visible - Secrets are decrypted at sandbox creation time and injected as environment variables @@ -161,7 +185,7 @@ and requests short-lived access through `POST /sessions/:id/provider-auth/:provi Provider-account mode removes that provider's canonical API key from the sandbox environment so the runtime cannot bypass the selected subscription. API-key mode continues to use ordinary global, -repository, or environment secrets. +team, repository, or environment secrets. ### Legacy managed OAuth coexistence @@ -179,23 +203,36 @@ XAI_OAUTH_ACCESS_TOKEN XAI_OAUTH_ACCESS_TOKEN_EXPIRES_AT ``` +Team secrets are **not** a legacy OAuth broker source. The broker reads global and target scopes +(the primary repository or environment), not the team layer; putting a refresh token in team secrets +does not configure legacy subscription authentication. + Do not reuse the same rotating refresh token in both systems. Operators may remove legacy keys once the legacy-bound sessions that depend on them are no longer needed. ### Secrets and prebuilt images -Image builds (repository images and environment images) run your `.openinspect/setup.sh` with the -same secrets a session would get. Anything the script **persists to disk** — an `.npmrc`, a `.env` -file, a downloaded credential — is captured in the image and re-served to every session that boots -from it, even after you rotate the secret. Two guidelines: +Image builds run your `.openinspect/setup.sh` with secrets selected for the build scope: + +- **Repository images**: global + repository secrets, **never team secrets**, because these images + are shared across teams. +- **Environment images**: global + the environment's owning team (if any) + environment secrets; + member-repository secrets do not flow in. A team environment's image is selected only for sessions + owned by the same team. A workspace environment build has no team layer, even if a team session + later uses it. + +Anything the script **persists to disk**, such as an `.npmrc`, a `.env` file, or a downloaded +credential, is captured in the image and re-served to every session that boots from it, even after +you rotate the secret. Two guidelines: - **Avoid writing long-lived secrets to disk in `setup.sh`.** Read them from the environment at runtime (they are re-injected fresh on every session) instead of baking them into files. -- **Environment-secret changes invalidate prebuilt images automatically**: saving an environment's - secrets supersedes its existing ready image and triggers a rebuild, so a revoked value cannot keep - serving from an old image. Rotating **repository or global** secrets does _not_ invalidate images - — stale on-disk material persists until the next commit-triggered rebuild, which is another reason - to keep secrets out of the image filesystem. +- **Environment-secret changes invalidate that environment's images automatically. Team-secret + writes/deletes invalidate images of environments owned by that team**, with best-effort rebuild + scheduling for prebuild-enabled environments. Invalidated images are not used for new boots; this + does not erase files from already-running sandboxes or their snapshots. Rotating **repository or + global** secrets does _not_ invalidate images; stale on-disk material persists until the next + commit-triggered rebuild, which is another reason to keep secrets out of the image filesystem. Where the trust boundary sits: Open-Inspect's own build plumbing never persists a credential into an image. The build's callback token stays in process memory, and the clone token and scope secrets @@ -209,22 +246,22 @@ image as no less sensitive than the scope's secrets. ## Common Examples -| Key | Scope | Purpose | -| --------------------------------- | ------ | -------------------------------------------------------------------------------------------------------------------------------------- | -| `ANTHROPIC_API_KEY` | Global | Claude API access | -| `OPENAI_API_KEY` | Global | OpenAI API access when a session selects API-key mode | -| `XAI_API_KEY` | Global | xAI API access when a session selects API-key mode | -| `DEEPSEEK_API_KEY` | Global | DeepSeek API access | -| `ZHIPU_API_KEY` | Global | Z.AI Coding Plan GLM access | -| `OPENCODE_API_KEY` | Global | OpenCode Zen and OpenCode Go access | -| `OPENAI_API_KEY_FALLBACK` | Any | Spillover once the ChatGPT subscription reaches its ceiling ([guide](OPENAI_MODELS.md#spilling-over-before-the-subscription-runs-out)) | -| `OPENAI_SUBSCRIPTION_MAX_PERCENT` | Any | Share of a Codex rate-limit window sandboxes may consume (default 100) | -| `OPENAI_OAUTH_REFRESH_TOKEN` | Any | Legacy OpenAI Codex via ChatGPT subscription ([setup guide](OPENAI_MODELS.md)) | -| `OPENAI_OAUTH_ACCOUNT_ID` | Any | Legacy OpenAI Codex via ChatGPT subscription ([setup guide](OPENAI_MODELS.md)) | -| `XAI_OAUTH_REFRESH_TOKEN` | Any | Legacy SuperGrok access ([setup guide](GROK_MODELS.md)) | -| `DATABASE_URL` | Repo | Database connection string | -| `AWS_ACCESS_KEY_ID` | Repo | AWS credentials for a specific project | -| `STRIPE_SECRET_KEY` | Repo | Stripe API key for a specific project | +| Key | Scope | Purpose | +| --------------------------------- | ----------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| `ANTHROPIC_API_KEY` | Global | Claude API access | +| `OPENAI_API_KEY` | Global | OpenAI API access when a session selects API-key mode | +| `XAI_API_KEY` | Global | xAI API access when a session selects API-key mode | +| `DEEPSEEK_API_KEY` | Global | DeepSeek API access | +| `ZHIPU_API_KEY` | Global | Z.AI Coding Plan GLM access | +| `OPENCODE_API_KEY` | Global | OpenCode Zen and OpenCode Go access | +| `OPENAI_API_KEY_FALLBACK` | Global/repo/environment | Spillover once the ChatGPT subscription reaches its ceiling ([guide](OPENAI_MODELS.md#spilling-over-before-the-subscription-runs-out)) | +| `OPENAI_SUBSCRIPTION_MAX_PERCENT` | Global/repo/environment | Share of a Codex rate-limit window sandboxes may consume (default 100) | +| `OPENAI_OAUTH_REFRESH_TOKEN` | Global/repo/environment | Legacy OpenAI Codex via ChatGPT subscription ([setup guide](OPENAI_MODELS.md)) | +| `OPENAI_OAUTH_ACCOUNT_ID` | Global/repo/environment | Legacy OpenAI Codex via ChatGPT subscription ([setup guide](OPENAI_MODELS.md)) | +| `XAI_OAUTH_REFRESH_TOKEN` | Global/repo/environment | Legacy SuperGrok access ([setup guide](GROK_MODELS.md)) | +| `DATABASE_URL` | Repo | Database connection string | +| `AWS_ACCESS_KEY_ID` | Repo | AWS credentials for a specific project | +| `STRIPE_SECRET_KEY` | Repo | Stripe API key for a specific project | --- @@ -243,8 +280,8 @@ provider-account setup guidance in [OpenAI models](OPENAI_MODELS.md) or ### Secret not appearing in sandbox -1. Verify the secret is saved under the correct scope (global, the specific repo, or the - environment) +1. Verify the secret is saved under the correct scope (global, the owning team, the specific repo, + or the environment) 2. Check that the key isn't in the reserved keys list above 3. New secrets only apply to **new** sandboxes — restart your session to pick up changes 4. For sessions launched from an **environment**: repository secrets do not flow in. Add the key to diff --git a/docs/SETUP_GUIDE.md b/docs/SETUP_GUIDE.md index edf7da039d..4bcc3fe725 100644 --- a/docs/SETUP_GUIDE.md +++ b/docs/SETUP_GUIDE.md @@ -205,12 +205,23 @@ For full infrastructure setup, use: Critical notes before deploy: +- After [Owner bootstrap](GETTING_STARTED.md#step-9-bootstrap-the-workspace-owner), + [create the first team](GETTING_STARTED.md#step-10-create-the-first-team-and-test-a-session). The + creator becomes Lead; choose membership policy and default visibility, grant repositories, + join/add operators, and create a team-owned test session. +- Fresh deployments should explicitly set `teams_enforcement = "on"` in Terraform to deploy + `TEAMS_ENFORCEMENT=on`. Runtime and Terraform defaults remain `shadow`. Existing deployments must + review production `shadow_denied:*` authorization audit entries before opting in; no completed + default-flip audit gate is established by this guide. - Build workers before running Terraform apply. - Build `@open-inspect/shared` first. - Use two-phase Terraform deploy for DO/service bindings. - For Modal deployments, eagerly build the Sandbox image with `uv run python deploy.py --build-sandbox-image`, then deploy with `uv run modal deploy deploy.py` (not `src/app.py`). +- Modal's legacy `github-app` secret is not required for fresh or restored sandbox Git credentials; + keep the App private key in the control plane (and the GitHub bot's own bindings when enabled). + Retain the required Modal `llm-api-keys` and `internal-api` secret objects. - Existing sessions keep their pinned authentication. Remove legacy OAuth keys only after dependent legacy-bound sessions are no longer needed. diff --git a/docs/TEST_REDUCTION.md b/docs/TEST_REDUCTION.md new file mode 100644 index 0000000000..35c4bc470e --- /dev/null +++ b/docs/TEST_REDUCTION.md @@ -0,0 +1,195 @@ +# Coverage-Guided Test Reduction + +## Results + +These figures record the initial reduction measured against `d343cac`, before later changes from +`main` were merged. Subsequent merge resolutions retain newly introduced regression tests rather +than restoring the removed redundant suites, so these are historical counts, not current-main +totals. + +Removed **3,414 test cases** and **278 complete test files** across eight packages. The measured +suites decreased from 14,587 to 11,173 cases, a **23.4% reduction**. Counts include four unchanged +skipped tests. + +The limit is a maximum **three-percentage-point decrease in each coverage metric, per package**, not +an aggregate average that could hide a large loss in one package. The historical comparisons below +were recomputed from saved counters with identical production-only exclusions on both sides. Every +statement, branch, function, and line denominator remains unchanged within that comparison. + +| Package | Before | After | Removed | Largest Coverage Drop | +| ------------------------- | -----: | ----: | ------: | --------------------: | +| control-plane, both hosts | 7,916 | 5,841 | 2,075 | 2.34 pp | +| web | 2,708 | 1,881 | 827 | 2.40 pp | +| shared | 1,093 | 903 | 190 | 2.28 pp | +| slack-bot | 512 | 397 | 115 | 2.27 pp | +| linear-bot | 267 | 228 | 39 | 2.04 pp | +| github-bot | 146 | 134 | 12 | 0.57 pp | +| sandbox-runtime | 1,404 | 1,278 | 126 | 0.90 pp | +| modal-infra | 541 | 511 | 30 | 0.00 pp | + +Other suites, including docs, sandbox-images, native Node tooling tests, and Terraform contracts, +were not reduced and are not included in these totals. + +### TypeScript Coverage + +Each cell shows baseline coverage followed by coverage after removal, in percent. + +| Package | Statements | Branches | Functions | Lines | +| ------------- | -------------- | -------------- | -------------- | -------------- | +| control-plane | 92.45 -> 90.78 | 84.82 -> 82.48 | 96.95 -> 96.09 | 94.11 -> 92.63 | +| web | 75.79 -> 73.39 | 74.55 -> 72.17 | 75.28 -> 73.05 | 76.84 -> 74.59 | +| shared | 93.63 -> 91.91 | 84.02 -> 81.94 | 91.90 -> 89.62 | 94.62 -> 92.93 | +| slack-bot | 90.24 -> 88.71 | 81.09 -> 78.82 | 95.46 -> 93.65 | 90.41 -> 89.47 | +| linear-bot | 86.61 -> 85.18 | 74.06 -> 72.19 | 90.47 -> 88.43 | 87.82 -> 86.48 | +| github-bot | 93.65 -> 93.65 | 88.57 -> 88.00 | 93.75 -> 93.75 | 94.69 -> 94.69 | + +### Python Coverage + +Coverage.py reports executable lines as statements and does not provide Vitest-style function +coverage. Statement and branch percentages were compared separately, not just the combined score. + +| Package | Statements / Lines | Branches | Combined | +| --------------- | ------------------ | -------------- | -------------- | +| sandbox-runtime | 89.83 -> 89.49 | 81.36 -> 80.46 | 87.87 -> 87.40 | +| modal-infra | 96.20 -> 96.20 | 89.92 -> 89.92 | 95.15 -> 95.15 | + +## Selection + +Per-suite coverage was compared using original source locations, then candidate deletions were +evaluated cumulatively. A point is redundant only while another retained suite still covers it; +individually redundant files are not necessarily redundant when removed together. Python test +contexts also identified duplicated cases within suites and redundant parameter combinations. + +The reduction favors removing mocked SQL, handler delegation, helper, and orchestration tests when +retained integration or higher-level behavioral tests exercise the implementation. It keeps: + +- All 140 control-plane workerd integration files, using real D1 and Durable Object storage. +- All Node-host and storage conformance suites. +- Web component and hook integration suites. +- Core authentication, signature, cookie identity, migration, architecture, and type contracts. +- Focused sandbox manager signature, spawn-admission, and late-provider-result race regressions. +- Real local-process, Git, shell, socket, and tool tests in the sandbox runtime. + +Coverage overlap is not assertion equivalence. Some isolated input permutations, error wording, +provider error-classification matrices, and interleavings no longer have dedicated assertions. The +retained integration tests cover their broader behavior, but this reduction does not claim to +preserve every old assertion or to be a mathematically optimal minimum test set. + +## Reproducing Coverage + +Build shared first and run heavyweight checks sequentially. `scripts/coverage-baseline.json` stores +the normalized baseline counters and the three-point budget. `scripts/coverage-policy.ts` derives +floors from those exact counters, rounding up to two decimals rather than loosening the budget. The +Vitest configs and report checker share this policy. Python statement and branch floors are enforced +independently; a high combined score cannot hide a branch regression. + +`.github/workflows/coverage.yml` runs the entire policy on every PR targeting `main` and every push +to `main`, without path filters or `continue-on-error`. Control-plane, web, other TypeScript, and +Python coverage run in parallel jobs. Control-plane coverage is further split into Vitest shards +(`COVERAGE_SHARD=true` skips the per-shard floor); `Coverage (control-plane)` merges their blob +reports and enforces the full-suite floor on the merged result. The final `Coverage` check fails if +any job fails and uploads the combined coverage artifact. The workflow tests its gate, including CLI +failure on low Python branch coverage. + +Repository rules are separate from workflow files: an administrator must add `Coverage` as a +required status check in the main ruleset. The authenticated integration cannot administer rules +(HTTP 403), and the current effective main rules have no required-status-check rule. This external +setting remains outstanding; the workflow alone is not claimed to make GitHub merges conditional. + +```bash +npm run build -w @open-inspect/shared +npm run test:coverage -w @open-inspect/control-plane +npm run test:coverage -w @open-inspect/web +npm run test:coverage -w @open-inspect/shared +npm run test:coverage -w @open-inspect/slack-bot +npm run test:coverage -w @open-inspect/linear-bot +npm run test:coverage -w @open-inspect/github-bot + +uv run --frozen --project packages/modal-infra --extra dev pytest packages/modal-infra/tests --cov=packages/modal-infra/src --cov-branch --cov-report=json:packages/modal-infra/coverage/coverage.json +node scripts/check-coverage.mjs modal-infra packages/modal-infra/coverage/coverage.json +uv run --frozen --project packages/sandbox-runtime --extra dev pytest packages/sandbox-runtime/tests --cov=packages/sandbox-runtime/src --cov-branch --cov-report=json:packages/sandbox-runtime/coverage/coverage.json +node scripts/check-coverage.mjs sandbox-runtime packages/sandbox-runtime/coverage/coverage.json +npm run test:coverage-gate +``` + +TypeScript JSON summaries are written to each package's `coverage/coverage-summary.json`. For +Python, both statement and branch percentages in the JSON `totals` are checked, rather than treating +`percent_covered` as line coverage. Missing branch counters fail closed. + +The control-plane coverage command now runs both Node and workerd projects in one Istanbul report. +V8 coverage cannot run inside workerd because Workers lack its inspector API. Both control-plane +measurements used the same Istanbul provider. Other packages retain V8. Both sides of the normalized +comparison exclude `.test`/`.spec` implementations, named test helpers/support/fixtures, declaration +files, and the existing `src/index.ts` entrypoint exclusions. Shared additionally excludes its +test-only `src/triggers/testing.ts`. No generic `*helper*` pattern excludes production credential +helpers. Python reports still measure only production `src/` files, so their baselines are +unchanged. + +Vitest 4 already excluded discovered `.test.tsx` suites before this correction: the original web +report contains zero `.test.tsx` or `.test.ts` files. The actual normalization removes two web +fixtures, three control-plane helpers, and one helper each from shared, Slack, and Linear. Counter +filtering, not averages of per-file percentages, produces the recorded baselines. + +## Merge Validation + +After merging `main` at `452b0b9`, six modify/delete conflicts were resolved by retaining focused +upstream regressions for canonical automation owners, bounded D1 parameters, linked-session privacy, +synchronous archive failures, executor audit events, and environment selection equality. The old +redundant cases remain removed, and all upstream integration additions are retained. + +The following are historical validation percentages from that merge, before helper/fixture +normalization, not a new before/after benchmark. + +| Package | Passed | Skipped | Statements | Branches | Functions | Lines | +| ------------------------- | -----: | ------: | ---------: | -------: | --------: | ----: | +| control-plane, both hosts | 6,002 | 1 | 90.92 | 82.66 | 95.86 | 92.72 | +| web | 1,965 | 0 | 74.35 | 73.05 | 74.15 | 75.54 | +| shared | 909 | 0 | 92.05 | 81.91 | 89.83 | 93.07 | +| sandbox-runtime | 1,279 | 3 | 89.50 | 80.49 | N/A | 89.50 | + +Sandbox-runtime combined coverage is 87.42%. The eleven focused conflict-resolution cases also pass +independently. Repository typechecks, ESLint, and formatting checks were rerun for the merge. + +## Security Review Follow-Up + +Three focused security checks were restored without restoring the removed suites: single-session +export refuses a readable session without `sessions.export` before loading its trace; the public +prompt route rejects caller-provided `authorId` before runtime dispatch; and failed +repository-scoped credential minting remains unavailable without retrying with broader credentials. +The route checks use real D1/DO sessions and verified browser credentials. The planner check injects +only the mint failure and retains the real scope resolver and planning path. + +## Deep Review Follow-Up + +Compact tests now preserve the contracts aggregate coverage cannot establish: + +- Real D1 environment membership/current-team grant intersection and TOCTOU rejection, including + unavailable credentials without broad-auth retry. +- Real SQLite post-encryption generation, status, fence, and provider-reference predicates, with + successor URLs/secrets unchanged and successful controls. +- Shutdown-handler-before-ACK and no-ACK-on-failure for both critical shutdown event types. +- Incoming bridge ACK routing through the real forwarder, including boot-time passthrough. +- Gated same-repository PR conflicts, claim release after failure, and independent-repository + concurrency through the real service and claims object. + +Production-only runs at `6f4c32f` pass all floors. These validation results are separate from the +historical fixed-denominator comparison above. + +| Package | Passed | Skipped | Statements | Branches | Functions | Lines | +| ------------------------- | -----: | ------: | ---------: | -------: | --------: | ----: | +| control-plane, both hosts | 6,038 | 1 | 90.97 | 82.74 | 96.21 | 92.77 | +| web | 1,965 | 0 | 74.15 | 73.04 | 73.83 | 75.37 | +| shared | 909 | 0 | 92.10 | 82.04 | 89.89 | 93.13 | +| slack-bot | 397 | 0 | 88.71 | 78.95 | 93.65 | 89.47 | +| linear-bot | 228 | 0 | 85.18 | 72.19 | 88.43 | 86.48 | +| github-bot | 134 | 0 | 93.65 | 88.00 | 93.75 | 94.69 | +| modal-infra | 511 | 0 | 96.20 | 89.92 | N/A | 96.20 | +| sandbox-runtime | 1,285 | 3 | 89.56 | 80.70 | N/A | 89.56 | + +## Additional Main Merge + +After merging `main` at `5abc1fb`, eight test conflicts were resolved by retaining the new analytics +source/user attribution, Slack channel/team boundary, and channel-binding audit assertions. The +older redundant cases remain removed. All 57 focused resolution cases, all eight full coverage +suites, and the unchanged coverage floors pass. No baseline or coverage budget was reset for the +upstream features. The required-check administration setting remains a separate external step. diff --git a/docs/integrations/GITHUB.md b/docs/integrations/GITHUB.md index edd67cfe13..4f0683850d 100644 --- a/docs/integrations/GITHUB.md +++ b/docs/integrations/GITHUB.md @@ -43,6 +43,10 @@ App bot through the PR reviewer picker. Use auto-review or `@mention` comments i ## Automatic PR Reviews +**Auto-review PR changes** is the deprecated, workspace-owned automatic review path. It still works +and does not use mention team routing. For team-owned event-driven reviews, use a GitHub Event +automation owned by the intended team instead. Deprecation does not disable the existing setting. + ### When It Runs When **Auto-review PR changes** is enabled, Open-Inspect starts a review session when a non-draft PR @@ -67,7 +71,8 @@ Auto-review is skipped when: Request a review or mention the bot to review it anyway. A PR opened, or pushed to, by the GitHub App itself is the App acting rather than a third party -asking it to act, so it bypasses both caller gates and is reviewed. +asking it to act, so it bypasses both caller gates and is reviewed: the App's login does not need to +be listed in **Allowed Trigger Users**, and no repository permission check runs for it. Draft PR events are skipped, but marking a draft ready emits `ready_for_review` and starts the automatic review path. @@ -126,6 +131,26 @@ a GitHub reply: Open-Inspect strips the bot mention before sending the request to the agent. The rest of the comment becomes the prompt. +### Model and Reasoning Overrides + +Start the request with `!model` or `!reasoning` to override the configured model or reasoning effort +for the session that comment starts: + +```text +@my-app[bot] !model openai/gpt-5.6-sol !reasoning high investigate the flaky test +``` + +The flags use the same syntax as Slack: each accepts a space or colon before its value (such as +`!model:anthropic/claude-sonnet-4-6` or `!reasoning:max`), and any flags must appear together at the +start of the request, right after the bot mention. Flags later in the comment are treated as part of +the request. Models must be enabled under **Settings > Models**, and the reasoning value must be +supported by the selected model. A model flag without `!reasoning` keeps the configured reasoning +effort when the new model supports it, and otherwise uses the model's default. + +If a flag is invalid, the bot replies with a PR comment explaining why and does not start a session. +Flags work in PR conversation comments and inline review threads. Auto-reviews and review requests +always use the configured model. + ### Inline Review Threads When you mention the bot in a PR review thread, Open-Inspect includes the file path and diff context @@ -145,6 +170,30 @@ when it needs context. Comment-triggered actions only run on pull requests. Mentions on ordinary GitHub issues are ignored. Comments from the bot itself are also ignored so the bot does not respond to its own output. +### Team Routing for Mentions + +Before creating a mention session, the bot asks the control plane for ownership using the numeric +GitHub repository ID and PR number. Routing follows this order: + +1. If the PR is linked to an existing Open-Inspect session, use that session's owning team (or + workspace ownership). This is only a routing hint: it neither resumes nor grants access to that + session, including a private one. +2. Otherwise, resolve the sender's linked GitHub identity and current teams with a repository or + installation grant covering the triggering repository. Exactly one eligible team wins. +3. If multiple teams qualify, use the sender's most recent session on that repository in a still + eligible team. If none resolves the ambiguity, use workspace ownership. Missing identity or no + eligible team also falls back to workspace ownership. + +A route lookup failure stops the request rather than guessing workspace ownership. GitHub trigger +gates still run, and session creation separately checks membership, team state, and repository +grants. A configured default environment is used only when ownership is compatible, it includes the +trigger repository, and the sender passes its repository checks; otherwise the target falls back to +the triggering repository. Workspace fallback can still be refused if the workspace requires team +ownership for new sessions. + +The runtime default for `TEAMS_ENFORCEMENT` remains `shadow`, not `on`; routing a session to a team +does not imply full team-read isolation in that mode. + --- ## What You See @@ -194,6 +243,20 @@ empty, no one can trigger direct bot workflows for that scope. These settings do not gate GitHub event automations. Automations are matched separately by their repository, event type, enabled state, and trigger conditions. +Event automations match the numeric GitHub repository ID, not just `owner/name`, and retain the +automation's saved owning team rather than using the mention sender's team. A team-owned automation +needs a current grant covering that repository at admission and launch. Losing the grant prevents a +run; restoring it allows a later event to run, not a replay of previously skipped events. Repository +renames retain identity through the numeric ID, while run targets use the event's current owner and +name. These checks are independent of the deprecated workspace auto-review setting. + +### Upgrading to Repository-ID Routing + +Upgrade the GitHub bot and control plane together. The control plane rejects event envelopes without +a numeric repository ID, so an older bot's events are not routed. Automations saved before +repository IDs were recorded no longer match events by name; open each one, reselect its +repositories, and save to resolve their IDs. + ### Models and Instructions | Setting | What it controls | @@ -204,7 +267,9 @@ repository, event type, enabled state, and trigger conditions. | Repository Overrides | Per-repository overrides for model, reasoning, instructions, and behavior | Repository overrides take priority over global defaults for the repository they apply to. If neither -a repository override nor global default sets a model, sessions use the deployment default model. +a repository override nor global default sets a model, sessions use the deployment default model. A +[`!model` or `!reasoning` flag](#model-and-reasoning-overrides) at the start of a mention overrides +both for that one session. ### Commit Signing @@ -285,8 +350,10 @@ Important limitations: ### Bot Behavior - Auto-review skips draft PRs and follow-up events on already-approved PRs. It runs again for - qualifying pushes, reopen events, and draft-to-ready transitions; a PR the GitHub App opened is - reviewed, bypassing the caller gates; manual `@mention` triggers remain separate. + qualifying pushes, reopen events, and draft-to-ready transitions. A PR the GitHub App opened is + reviewed, bypassing the caller gates; that self-review can only comment unless a separate reviewer + App submits it. Manual `@mention` triggers still pass the repository and user gates, then team + routing and creation checks. - With PR Feedback Autofix enabled, an Autofix push can trigger another review. Autofix's per-PR attempt limit bounds that feedback loop, but each iteration still consumes review compute. - The bot ignores bot-authored comments, ordinary issue comments, and comments that do not mention @@ -319,14 +386,14 @@ list. ### Auto-review did not run -Auto-review runs for non-draft PRs when opened, reopened, synchronized by a push, or marked ready. -It is skipped for draft PRs, disabled repositories, event senders who are not allowed to trigger the -bot (the GitHub App's own PRs and pushes bypass that gate), and follow-up events on PRs that already -carry a standing approval. On such a head the `open-inspect` status shows "Skipped — PR already -approved" when it was pending or absent; a terminal status already there (for example a finished -review) is left unchanged, and when the status could not be read or written the bot reviewed the PR -instead. Check the latest event's sender, the PR's approvals, and the configured repository scope -when an update does not start a review. +The deprecated workspace auto-review runs for non-draft PRs when opened, reopened, synchronized by a +push, or marked ready. It is skipped for draft PRs, disabled repositories, event senders who are not +allowed to trigger the bot (the GitHub App's own PRs and pushes bypass that gate), and follow-up +events on PRs that already carry a standing approval. On such a head the `open-inspect` status shows +"Skipped — PR already approved" when it was pending or absent; a terminal status already there (for +example a finished review) is left unchanged, and when the status could not be read or written the +bot reviewed the PR instead. Check the latest event's sender, the PR's approvals, and the configured +repository scope when an update does not start a review. ### A mention did not start a session @@ -343,7 +410,8 @@ after the request was accepted. Open the Open-Inspect web app to inspect the ses ### The wrong model or instructions were used Check **Settings > Integrations > GitHub**. Repository overrides take priority over global defaults. -Changes apply to new GitHub-triggered sessions. +Changes apply to new GitHub-triggered sessions. A `!model` or `!reasoning` flag at the start of the +mention overrides both for that session. ### The bot is active in too many repositories diff --git a/docs/integrations/LINEAR.md b/docs/integrations/LINEAR.md index e897437ac4..10ee8c0e99 100644 --- a/docs/integrations/LINEAR.md +++ b/docs/integrations/LINEAR.md @@ -100,6 +100,40 @@ details. If the resolved repo is outside the selected Linear scope, Linear shows an error and no session starts. +### Team Bindings and Existing Mappings + +A team lead or workspace administrator binds an external Linear team ID to an Open-Inspect team in +that team's **Channels** tab. **Primary** marks the main binding for that provider; **Source** also +routes new sessions to the owning team. An external Linear team can be bound to only one +Open-Inspect team, and each Open-Inspect team has at most one primary binding per provider. Enter +the Linear team ID, not its name or issue-key prefix; this is separate from Slack channel discovery. + +Ownership bindings and legacy target mappings have different meanings. The KV configuration +`config:team-repos` maps a Linear team to repositories or environments; `config:project-repos` maps +a project to a target. Neither grants membership or repository access, nor assigns Open-Inspect team +ownership. The teams database migration creates an empty binding table and does not translate those +KV mappings. To adopt team ownership, explicitly create the Linear binding, add the required +Open-Inspect members and repository grants, and retain target mappings only for target selection. +Existing sessions are not reassigned by adding a binding. + +Linear's `unboundChannels` policy defaults to `workspace`, so an unbound Linear team starts +workspace-owned sessions. `reject` instead asks for a binding and creates no session. A binding +lookup failure also stops launch rather than falling back to workspace ownership. + +Repository, environment, and resolved integration-setting lookups carry the external Linear team +scope. Actor-bearing catalogs require team membership or workspace-admin access; creating a +team-owned session checks actual membership and target grants. Scoped lookup failures do not fall +back to a workspace-wide catalog. A mapping cannot make an out-of-scope target accessible. + +Some discovery and callback reads have no human actor. Those are narrowly authorized Linear-bot +service reads, still carrying the Linear team scope; they do not impersonate a team member or give +the app permission to start arbitrary sessions. Automation-created delegations use the installed app +user as the acting identity for session creation. + +`TEAMS_ENFORCEMENT` still defaults to `shadow`, not `on`. In particular, team-read denials can be +audited rather than blocked in that mode. Do not treat scoped callback reads as a guarantee of full +cross-team output isolation in shadow mode; private-session access remains enforced. + --- ## What Linear Shows @@ -115,6 +149,14 @@ starts. When a session starts, Linear receives a **View Session** link. If the agent opens a pull request, Linear receives a **Pull Request** link when the session finishes. +Before posting completion content, the bot verifies the issue's current Linear team. If it differs +from the recorded launch team, or cannot be verified, results are withheld. The bot reads session +events and artifacts without a human actor, scoped to that verified Linear team. A denied or failed +read produces a generic results-withheld error instead of session content or a PR link. Legacy +callbacks recover the launch team from a matching issue-session mapping when possible; otherwise +they use the verified current team for the scoped read. Enforcement of team visibility on those +reads depends on the rollout mode described above. + Open the web session for live output, logs, artifacts, and file changes. For a human-initiated session, Open-Inspect moves an unstarted issue to the team's lowest-position `started` workflow state only after the initial prompt reaches a live sandbox. It leaves automation-initiated, @@ -130,15 +172,16 @@ remain the responsibility of Linear's GitHub integration and the team's PR autom Open the web app and go to **Settings > Integrations > Linear** to configure the Linear Agent. -| Setting | What it controls | -| ------------------------------ | ----------------------------------------------------------------- | -| Default model and effort | Model and reasoning depth for Linear-started sessions | -| Repository Scope | Whether Linear can run in all accessible repos or selected repos | -| Issue Session Instructions | Extra guidance appended to Linear issue prompts | -| Allow user model preferences | Whether admin-managed user preferences can override the model | -| Allow model labels (`model:*`) | Whether Linear issue labels can choose the model | -| Tool progress activities | Whether Linear shows intermediate file and command activity | -| Repository Overrides | Per-repository defaults for model, reasoning, and Linear behavior | +| Setting | What it controls | +| ------------------------------ | ---------------------------------------------------------------- | +| Agent harness | OpenCode (default) or Claude Agent for new Linear sessions | +| Default model and effort | Model and reasoning depth for Linear-started sessions | +| Repository Scope | Whether Linear can run in all accessible repos or selected repos | +| Issue Session Instructions | Extra guidance appended to Linear issue prompts | +| Allow user model preferences | Whether admin-managed user preferences can override the model | +| Allow model labels (`model:*`) | Whether Linear issue labels can choose the model | +| Tool progress activities | Whether Linear shows intermediate file and command activity | +| Repository Overrides | Per-repository harness, model, reasoning, and Linear behavior | If no Linear settings are configured, all accessible repositories are in scope, user preferences and model labels are allowed, and tool progress is enabled. @@ -150,6 +193,12 @@ Model selection uses this priority, highest to lowest: 3. Repository override or global Linear default. 4. Deployment default model. +The session then runs on the configured **Agent harness** when that harness can run the resolved +model, and on OpenCode otherwise: Claude Agent runs Anthropic models only, so a `model:gpt-*` label +or a non-Anthropic default runs on OpenCode. The "Creating coding session" activity names the +harness. On Claude Agent, Linear sessions follow the provider's **Automated authentication** policy +and may use a connected Claude account; see [Claude Agent](../CLAUDE_AGENT.md#linear-sessions). + Linear user preferences are currently admin/API-managed, not set from a self-service Linear screen. --- @@ -228,7 +277,8 @@ the rejected API request once. A reinstall is not normally required. ### The wrong model was used Check **Settings > Integrations > Linear**. Repository overrides, user preferences, and `model:*` -labels can affect model selection. Changes apply to new Linear-started sessions. +labels can affect model selection. A non-Anthropic model runs on OpenCode even when the harness is +Claude Agent. Changes apply to new Linear-started sessions. ### The wrong repository was used diff --git a/docs/integrations/SLACK.md b/docs/integrations/SLACK.md index 936b1ba950..43df6307cb 100644 --- a/docs/integrations/SLACK.md +++ b/docs/integrations/SLACK.md @@ -71,7 +71,7 @@ repository name when the request could apply to more than one repo: @Open-Inspect update the billing docs in acme/api ``` -Open-Inspect chooses from repositories and environments available to this deployment, using the +Open-Inspect chooses from repositories and environments available in the channel's scope, using the message, Slack channel context, and recent thread context. A configured [routing-rule keyword](#routing-rules) takes precedence, followed by a single channel association. Otherwise the classifier chooses the best target for the request, including **No repository** when @@ -219,6 +219,14 @@ Open-Inspect keeps the Slack thread connected to the session for about 7 days. I that mapping expires, or if you reply outside the thread, the bot may start target selection again and create a new session. +An existing mapping is not replaced merely because a follow-up fails. If the session is missing or +no longer publishable in this channel, the bot reports that the thread is closed; start a new +request in a new thread. A follow-up 404 triggers a separate publication check: the mapping is +marked closed only when that check confirms denial. An actor-specific 403 leaves the mapping usable +by other authorized people, and an actor-concealed 404 does not close it for everyone when channel +publication is still allowed. Binding changes are checked on each mapped follow-up. A closed mapping +can reopen if its original team binding matches again and publication is allowed. + For follow-ups, Open-Inspect includes up to ten recent thread messages posted after the preceding prompt and strictly before the new request. Replies that arrive while Slack history is being fetched are not exposed to the earlier turn; they remain eligible for a later follow-up. Earlier messages @@ -280,9 +288,9 @@ these preferences. ## Optional Agent Notifications -Interactive Slack sessions (DMs and `@mentions`) always get their normal thread replies and -completion messages. Agent notifications are separate: they let an agent post an extra message to a -Slack channel when you explicitly ask for it: +Interactive Slack sessions (DMs and `@mentions`) get normal thread replies and completion messages +only while publication is allowed. Agent notifications are separate: they let an agent post an extra +message to a Slack channel when you explicitly ask for it: ```text When you finish, post a short summary to #eng-updates. @@ -296,9 +304,12 @@ To use this workflow: 4. Optional: add repository overrides to inherit, force on, or force off agent notifications for specific repositories. -Channel membership controls where these extra posts can go. Invite the bot to a channel to make it -available; remove it from a channel to remove access. Slack may still reject missing, archived, -inaccessible, or rate-limited targets. +Bot membership is necessary, but is not the only publication check. Private-session output is +refused, as is output to a channel bound to a different owning team, even for a workspace-visible +session. These checks also cover managed completions and generated media. They use current session +visibility and channel bindings, not just the binding at launch. An unbound destination is not +automatically a cross-team refusal; do not treat channel bindings as an outbound allowlist. Slack +may still reject missing, archived, inaccessible, or rate-limited targets. Changes apply to new sessions. If you turn notifications on and an existing session cannot post to Slack, start a new session. Turning notifications off blocks future notification attempts. @@ -379,21 +390,28 @@ condition to filter by content. See automation's instructions; without an explicit instruction it will answer every message it is woken for. A run that opened a pull request or produced other artifacts always posts, and interactive `@mention` sessions never decline — a person is waiting on a visible answer there. -- Every reply in a thread continues the same session — during the run and after it finishes — for up - to 7 days after the thread's first trigger, like replying in an `@mention` thread. The reply is - routed to that session as a follow-up prompt (re-spawned from a snapshot if it had gone idle), - gets its own 👀 reaction and in-thread response, and does **not** need to match the trigger's text - condition — conditions gate new runs, not replies that continue a thread. A reply more than 7 days - after the first trigger starts a fresh run. +- Authorized replies in a thread continue the same session, during the run and after it finishes, + for up to 7 days after the thread's first trigger, like replying in an `@mention` thread. The + reply is routed to that session as a follow-up prompt (re-spawned from a snapshot if it had gone + idle), gets its own 👀 reaction and in-thread response, and does **not** need to match the + trigger's text condition — conditions gate new runs, not replies that continue a thread. A reply + more than 7 days after the first trigger starts a fresh run. + +The automation scheduler checks each reply author's session collaboration access; a rejected author +does not start a replacement run for that automation. This is separate from interactive +mapped-thread handling: if no steerable session exists or enqueueing fails, the scheduler may +re-evaluate the reply as a new trigger. Runs retain the automation's saved owning team, not the +channel's interactive routing choice. Publication still checks current visibility and channel +bindings, so admission does not guarantee a reply can be posted. ### Threat model Channel triggers widen who can start a coding session, so weigh the following before configuring them: -- **Any member of a watched channel can trigger a run** simply by posting a matching message. Treat - every watched channel as a list of people authorized to start sessions against the automation's - repository. +- **Any member of a watched channel can supply a matching trigger** unless conditions restrict them. + Treat watched channels as sources of untrusted requests to the automation's executor. Execution + membership and grant checks do not make the triggering message trustworthy. - **Prefer an allowlist.** Add a **Slack User** condition (`include`) so only specific people can trigger the automation, and keep watched channels small and trusted. - **Message text reaches the agent.** The triggering message becomes part of the prompt. Scope the @@ -409,16 +427,39 @@ them: These notes are most useful for workspace admins deciding where the Slack bot should be available. +### Team Channel Bindings + +A team lead or workspace administrator can bind a Slack channel in the team's **Channels** tab. +**Primary** marks the team's main Slack binding; **Source** adds another channel that routes new +interactive sessions to the same team. Both kinds determine ownership, not a repository or a default +notification destination. Each channel can belong to only one team, and each team can have only one +primary binding per provider. The bot must already be in the channel; externally shared Slack +Connect channels cannot be bound. + +The Slack integration's `unboundChannels` policy is `workspace` by default: an unbound channel +starts workspace-owned sessions. With `reject`, new interactive requests in unbound channels are +refused until a binding is added. Unbound DMs remain workspace-scoped under either policy. A failed +binding lookup stops the request rather than silently falling back to workspace ownership. + +Classification and target dropdowns use a live channel- and actor-scoped catalog. Bound-team +catalogs require team membership or workspace-admin access and are filtered by repository grants and +eligible environments. Starting a team-owned session checks the actor's actual team membership and +grants for the selected target; Slack channel membership alone does not enroll that person in the +Open-Inspect team. Picker submissions recheck the binding, so an old picker cannot transfer a +request to another team. Repository routing rules and channel associations select a target within +this scope; they do not grant access. + +`TEAMS_ENFORCEMENT` still defaults to `shadow`, not `on`. Do not assume full team-read isolation in +that mode. Team creation checks, scoped catalogs, live Slack follow-up channel checks, private +access, and Slack publication gates are not a promise that every team read is enforced. + - Slack bot tokens stay server-side. They are not sent to sandboxes. - Slack requests are verified before Open-Inspect acts on them. -- Slack-created sessions use deployment-level repository access. The repositories shown in Slack are - the repositories accessible to the configured GitHub App or SCM installation, not a per-Slack-user - GitHub permission list. -- Slack identity linking is best-effort and is not used to approve repository access. To restrict - what Slack sessions can touch, limit the GitHub App installation to selected repositories and - invite the Slack bot only into trusted channels. +- The source-control installation is the outer repository boundary, not a per-Slack-user GitHub + permission list. Team bindings, memberships, and repository grants further constrain team work. +- Invite the bot only into trusted channels; identity linking does not itself grant team membership. - Bot messages are ignored so the Slack bot does not respond to itself. -- Agent notifications use Slack channel membership as the access boundary. +- Agent notifications require bot membership and the publication checks described above. - Accepted notification text is sanitized and shortened to fit Slack block limits; extremely large raw inputs are rejected. diff --git a/docs/plans/managed-skills.md b/docs/plans/managed-skills.md index 2be04e646f..e252ab8c0d 100644 --- a/docs/plans/managed-skills.md +++ b/docs/plans/managed-skills.md @@ -5,6 +5,14 @@ Proposed design for V1. This document intentionally distinguishes product-visible version control, which is deferred, from immutable internal revisions, which are required for reproducible sessions. +**Historical baseline note (2026-10-02):** The authorization and environment-naming descriptions +below record the baseline when this proposal was written, not current product behavior. The product +now has workspace roles, Teams and membership checks, team-owned sessions/environments/automations, +team secrets, and repository grants. Environment names are unique within an owning team or the +workspace. Repository skills remain workspace resources with permission and grant checks, not +team-owned resources. See [Authentication and Authorization](../AUTH.md) for current access rules; +the baseline and V1 decisions below are retained as historical design context. + ## Summary Open-Inspect should let admitted users create and edit reusable agent skills in the web application, @@ -232,9 +240,7 @@ create-session request carries a discriminated choice, never an ambiguous nullab ```ts type SessionSkillSelection = - | { mode: "all" } - | { mode: "none" } - | { mode: "profile"; profileId: string }; + { mode: "all" } | { mode: "none" } | { mode: "profile"; profileId: string }; ``` Bot, automation, Slack, Linear, and GitHub-created sessions use `{ mode: "all" }` unless their diff --git a/eslint.config.js b/eslint.config.js index 8c9db139d8..b4acc8a096 100644 --- a/eslint.config.js +++ b/eslint.config.js @@ -459,6 +459,19 @@ export default tseslint.config( ], }, }, + { + files: ["packages/web/src/**/*.tsx"], + rules: { + "no-restricted-syntax": [ + "error", + { + selector: 'JSXOpeningElement[name.name="select"]', + message: + "Use Select / SelectTrigger / SelectContent / SelectItem from @/components/ui/select instead of a native { const parsed = sessionVisibilitySchema.safeParse(value); - if (parsed.success) { - setSelection(parsed.data); - setFailure(null); - setConfirmChildren(false); - } + if (parsed.success && parsed.data !== selected) + setConfirm({ target: parsed.data, includeChildren: false, chooseChildren: true }); }} > @@ -136,64 +169,87 @@ export function SessionVisibilityControl({ {selected === "team" && ownerTeamId && ( )} - {failure && {failure.message}} + {visibilityFailure && ( + {visibilityFailure.error.message} + )}
- + {pending && Updating...}
- + { + if (!open) setConfirm(null); + }} + > - Change child session visibility? + + {confirm?.includeChildren + ? "Change child session visibility?" + : "Change session visibility?"} + - This will change this session and any child sessions to {selected} visibility. Any - private child sessions will change to {selected} visibility. + {confirm?.includeChildren + ? `This will change this session and any child sessions to ${confirm.target} visibility. Any private child sessions will change to ${confirm.target} visibility.` + : `This will change this session to ${confirm?.target} visibility.`} + {confirm?.chooseChildren && ( + + )} + {confirm?.target === "team" && ownerTeamId && ( + + )} Cancel - void changeVisibility(true)}> + { + if (confirm) void changeVisibility(confirm.target, confirm.includeChildren); + }} + > Change visibility - {includeChildren && - failure instanceof SessionScopeError && - failure.canRetryWithoutChildren && ( + {visibilityFailure?.includedChildren && + visibilityFailure.error instanceof SessionScopeError && + visibilityFailure.error.canRetryWithoutChildren && ( diff --git a/packages/web/src/components/settings/appearance-settings.test.tsx b/packages/web/src/components/settings/appearance-settings.test.tsx new file mode 100644 index 0000000000..9fca61268a --- /dev/null +++ b/packages/web/src/components/settings/appearance-settings.test.tsx @@ -0,0 +1,53 @@ +// @vitest-environment jsdom +/// + +import { cleanup, render, screen, waitFor } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import * as matchers from "@testing-library/jest-dom/matchers"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { AppearanceSettings } from "./appearance-settings"; + +expect.extend(matchers); + +vi.mock("next-themes", () => ({ + useTheme: () => ({ theme: "system", setTheme: vi.fn() }), +})); + +beforeEach(() => localStorage.clear()); +afterEach(cleanup); + +describe("AppearanceSettings", () => { + it("changes and persists the light and dark themes through their labeled triggers", async () => { + const user = userEvent.setup(); + const { unmount } = render(); + expect(screen.getByLabelText("Light theme")).toHaveTextContent("Atom One Light"); + expect(screen.getByLabelText("Dark theme")).toHaveTextContent("Atom One Dark"); + + await user.click(screen.getByLabelText("Light theme")); + await user.click(await screen.findByRole("option", { name: "GitHub" })); + await user.click(screen.getByLabelText("Dark theme")); + await user.click(await screen.findByRole("option", { name: "GitHub Dark" })); + + expect(screen.getByLabelText("Light theme")).toHaveTextContent("GitHub"); + expect(screen.getByLabelText("Dark theme")).toHaveTextContent("GitHub Dark"); + unmount(); + render(); + expect(screen.getByLabelText("Light theme")).toHaveTextContent("GitHub"); + expect(screen.getByLabelText("Dark theme")).toHaveTextContent("GitHub Dark"); + }); + + it("supports keyboard selection and returns focus to the trigger", async () => { + const user = userEvent.setup(); + render(); + const trigger = screen.getByRole("combobox", { name: "Light theme" }); + trigger.focus(); + await user.keyboard("{Enter}"); + await waitFor(() => + expect(screen.getByRole("option", { name: "Atom One Light" })).toHaveFocus() + ); + await user.keyboard("{ArrowDown}{Enter}"); + expect(trigger).toHaveTextContent("GitHub"); + await waitFor(() => expect(trigger).toHaveFocus()); + expect(screen.queryByRole("listbox")).not.toBeInTheDocument(); + }); +}); diff --git a/packages/web/src/components/settings/appearance-settings.tsx b/packages/web/src/components/settings/appearance-settings.tsx index d33237e78e..dedf5550f0 100644 --- a/packages/web/src/components/settings/appearance-settings.tsx +++ b/packages/web/src/components/settings/appearance-settings.tsx @@ -1,6 +1,7 @@ "use client"; import { useTheme } from "next-themes"; +import { useId } from "react"; import { useSyntaxHighlightPreferences, LIGHT_THEMES, @@ -9,6 +10,13 @@ import { type SyntaxHighlightThemeDefinition, } from "@/hooks/use-syntax-highlight-preferences"; import { ToggleGroup, ToggleGroupItem } from "@/components/ui/toggle-group"; +import { + Select, + SelectTrigger, + SelectValue, + SelectContent, + SelectItem, +} from "@/components/ui/select"; import { SunIcon, MoonIcon, MonitorIcon } from "@/components/ui/icons"; const COLOR_SCHEME_OPTIONS: { value: ColorSchemeMode; label: string; icon: typeof SunIcon }[] = [ @@ -30,24 +38,27 @@ function ThemeRow({ themes: SyntaxHighlightThemeDefinition[]; onChange: (id: string) => void; }) { + const id = useId(); return (
- {label} +

{description}

- +
); } diff --git a/packages/web/src/components/settings/audit-log-settings.test.tsx b/packages/web/src/components/settings/audit-log-settings.test.tsx index fe6a86a1ac..6e878572ad 100644 --- a/packages/web/src/components/settings/audit-log-settings.test.tsx +++ b/packages/web/src/components/settings/audit-log-settings.test.tsx @@ -27,6 +27,8 @@ const filters = vi.hoisted(() => ({ teams: vi.fn(), memberships: vi.fn(), allowed: true, + teamsLoading: false, + teamsError: null as Error | null, })); vi.mock("@/hooks/use-audit-events", () => ({ useAuditEvents: (...args: unknown[]) => { @@ -46,8 +48,8 @@ vi.mock("@/hooks/use-teams", () => ({ { id: "team_two", name: "Engineering", archivedAt: null }, { id: "team_archived", name: "Archived team", archivedAt: 1 }, ], - loading: false, - error: null, + loading: filters.teamsLoading, + error: filters.teamsError, }; }, useMeTeams: () => { @@ -123,6 +125,8 @@ function renderSingle(event: Record) { beforeEach(() => { filters.allowed = true; + filters.teamsLoading = false; + filters.teamsError = null; filters.audit.mockReset(); filters.teams.mockReset(); filters.memberships.mockReset(); @@ -145,20 +149,41 @@ afterEach(cleanup); describe("AuditLogSettings", () => { it("lets a workspace Owner filter audit events by a team they are not a member of", async () => { + const user = userEvent.setup(); render(); expect(filters.teams).toHaveBeenLastCalledWith(true); expect(filters.memberships).not.toHaveBeenCalled(); - expect(screen.getByRole("option", { name: "Engineering" })).toBeInTheDocument(); - await userEvent.selectOptions(screen.getByRole("combobox", { name: "Team" }), "team_two"); + await user.click(screen.getByLabelText("Team")); + await user.click(await screen.findByRole("option", { name: "Engineering" })); expect(filters.audit).toHaveBeenLastCalledWith({ teamId: "team_two", enabled: true }); + expect(screen.getByRole("combobox", { name: "Team" })).toHaveTextContent("Engineering"); + + await user.click(screen.getByLabelText("Team")); + await user.click(await screen.findByRole("option", { name: "All teams" })); + expect(filters.audit).toHaveBeenLastCalledWith({ teamId: undefined, enabled: true }); + expect(screen.getByRole("combobox", { name: "Team" })).toHaveTextContent("All teams"); }); it("includes archived teams in the audit filter", async () => { + const user = userEvent.setup(); render(); - await userEvent.selectOptions(screen.getByRole("combobox", { name: "Team" }), "team_archived"); + await user.click(screen.getByRole("combobox", { name: "Team" })); + await user.click(await screen.findByRole("option", { name: "Archived team" })); expect(filters.audit).toHaveBeenLastCalledWith({ teamId: "team_archived", enabled: true }); }); + it("disables the team filter while teams load or fail to load", () => { + filters.teamsLoading = true; + const { rerender } = render(); + expect(screen.getByLabelText("Team")).toBeDisabled(); + + filters.teamsLoading = false; + filters.teamsError = new Error("failed"); + rerender(); + expect(screen.getByLabelText("Team")).toBeDisabled(); + expect(screen.getByRole("status")).toHaveTextContent("Unable to load team filters."); + }); + it("withholds the feed and filters without the existing audit permission", () => { filters.allowed = false; render(); @@ -199,6 +224,8 @@ describe("AuditLogSettings", () => { ["team.grant_removed", "Team repository grant removed"], ["team.secret_set", "Team secret set"], ["team.secret_deleted", "Team secret deleted"], + ["team.binding_added", "Team channel binding added"], + ["team.binding_removed", "Team channel binding removed"], ["automation.executor_changed", "Automation executor changed"], ])("labels %s as an operation in the workspace audit viewer", (action, label) => { const article = renderSingle(createEvent("applied", { action })); @@ -212,6 +239,25 @@ describe("AuditLogSettings", () => { expect(article.getByText("Applied")).toBeInTheDocument(); }); + it.each(["applied", "no_op", "denied", "rejected"] as const)( + "renders a shadow denial as informational Would deny despite stored result %s", + (operationResult) => { + const article = renderSingle( + createEvent(operationResult, { + action: "session.shadow_denied", + resourceType: "session", + reasonCode: "shadow_denied:not_member", + metadata: { before: {}, requested: {}, after: {}, channel: "ws" }, + }) + ); + expect(article.getByText("Session read shadow observation")).toBeInTheDocument(); + expect(article.getByText("Would deny")).toHaveClass("bg-info-muted", "text-info"); + expect(article.queryByText("Denied")).not.toBeInTheDocument(); + expect(article.getByText("shadow_denied:not_member")).toBeInTheDocument(); + expect(article.queryByText("HTTP response")).not.toBeInTheDocument(); + } + ); + it("renders outcomes, stable summaries, timestamps, and expandable structured details", async () => { hook.events = [ createEvent("applied"), @@ -345,12 +391,15 @@ describe("AuditLogSettings", () => { expect(scrollIntoView).toHaveBeenCalledWith({ block: "start" }); }); - it("explains what an authorization decision does and does not prove", () => { + it("explains how decisions and shadow observations differ from operation outcomes", () => { render(); expect( screen.getByText(/They do not confirm that the requested change took effect/) ).toBeInTheDocument(); + expect( + screen.getByText(/Would deny describe hypothetical denials, not enforced denials/) + ).toBeInTheDocument(); }); it.each([ diff --git a/packages/web/src/components/settings/audit-log-settings.tsx b/packages/web/src/components/settings/audit-log-settings.tsx index 7070522e0c..6b269589d9 100644 --- a/packages/web/src/components/settings/audit-log-settings.tsx +++ b/packages/web/src/components/settings/audit-log-settings.tsx @@ -6,12 +6,20 @@ import { interpretAuditEvent, type AuditEvent, type AuditEventInterpretation, + type AuditObservationAction, type AuditOperationAction, type AuditPrincipalKind, type AuditOperationResult, } from "@open-inspect/shared/types/audit-events"; import { Badge } from "@/components/ui/badge"; import { Button } from "@/components/ui/button"; +import { + Select, + SelectTrigger, + SelectValue, + SelectContent, + SelectItem, +} from "@/components/ui/select"; import { useAuditEvents } from "@/hooks/use-audit-events"; import { useTeams } from "@/hooks/use-teams"; import { useCurrentUserAuthorization } from "@/hooks/use-current-user-authorization"; @@ -23,6 +31,8 @@ interface BadgeTreatment { className: string; } +const ALL_TEAMS_VALUE = "all-teams"; + const OPERATION_OUTCOMES: Record = { applied: { label: "Applied", className: "bg-success-muted text-success" }, no_op: { label: "No change", className: "bg-muted text-muted-foreground" }, @@ -36,6 +46,10 @@ const AUTHORIZATION_DECISIONS: Record<"allowed" | "denied", BadgeTreatment> = { denied: { label: "Denied", className: "bg-destructive-muted text-destructive" }, }; +const OBSERVATIONS: Record<"would_deny", BadgeTreatment> = { + would_deny: { label: "Would deny", className: "bg-info-muted text-info" }, +}; + // The client cannot say what an unrecognized action's stored result means. const UNRECOGNIZED: BadgeTreatment = { label: "Unrecognized", @@ -43,6 +57,13 @@ const UNRECOGNIZED: BadgeTreatment = { }; const OPERATION_LABELS: Record = { + "memory.created": "Memory created", + "memory.revised": "Memory revised", + "memory.archived": "Memory archived", + "memory.restored": "Memory restored", + "memory.approved": "Memory approved", + "memory.rejected": "Memory rejected", + "memory.superseded": "Memory superseded", "session.private_break_glass": "Private session break-glass read", "session.visibility_changed": "Session visibility changed", "session.moved": "Session moved", @@ -66,12 +87,19 @@ const OPERATION_LABELS: Record = { "team.grant_removed": "Team repository grant removed", "team.secret_set": "Team secret set", "team.secret_deleted": "Team secret deleted", + "team.binding_added": "Team channel binding added", + "team.binding_removed": "Team channel binding removed", "automation.executor_changed": "Automation executor changed", }; +const OBSERVATION_LABELS: Record = { + "session.shadow_denied": "Session read shadow observation", +}; + const ACTION_LABELS = new Map([ [AUTHORIZATION_DECISION_ACTIONS.allowed, "Authorization allowed"], [AUTHORIZATION_DECISION_ACTIONS.denied, "Authorization denied"], + ...Object.entries(OBSERVATION_LABELS), ...Object.entries(OPERATION_LABELS), ]); @@ -83,6 +111,8 @@ function badgeTreatment(interpretation: AuditEventInterpretation): BadgeTreatmen switch (interpretation.kind) { case "authorization_decision": return AUTHORIZATION_DECISIONS[interpretation.decision]; + case "observation": + return OBSERVATIONS[interpretation.observation]; case "operation": return OPERATION_OUTCOMES[interpretation.result]; case "unknown": @@ -231,32 +261,38 @@ export function AuditLogSettings() { Audit log

- Review workspace operations and authorization decisions. Events are shown newest first. + Review workspace operations, authorization decisions, and observations. Events are shown + newest first.

Authorization decisions record whether a request was allowed or denied and the HTTP response it returned. They do not confirm that the requested change took effect. Applied, No change, - and Rejected are recorded only by the operation that made or refused the change. + and Rejected are recorded only by the operation that made or refused the change. Shadow + observations marked Would deny describe hypothetical denials, not enforced denials or + operation outcomes.

- setTeamId(value === ALL_TEAMS_VALUE ? "" : value)} disabled={teamsLoading || !!teamsError} - className="min-w-48 max-w-full rounded border border-border bg-background px-2 py-2 text-sm disabled:opacity-50" > - - {teams.map((team) => ( - - ))} - + + + + + All teams + {teams.map((team) => ( + + {team.name} + + ))} + + {teamsError && (

Unable to load team filters. diff --git a/packages/web/src/components/settings/environment-access.test.ts b/packages/web/src/components/settings/environment-access.test.ts deleted file mode 100644 index 8e4a0302ea..0000000000 --- a/packages/web/src/components/settings/environment-access.test.ts +++ /dev/null @@ -1,64 +0,0 @@ -import { describe, expect, it } from "vitest"; -import type { Environment } from "@open-inspect/shared/types/environments"; -import { environmentAccess, type EnvironmentFeatureGrants } from "./environment-access"; - -const allGrants: EnvironmentFeatureGrants = { - manageSecrets: true, - manageRepoSecrets: true, - manageSettings: true, - manageImages: true, - readImages: true, - readSettings: true, -}; - -function environment(capabilities?: Environment["capabilities"]): Environment { - return { - id: "env-1", - name: "Stack", - description: null, - prebuildEnabled: true, - createdAt: 1, - updatedAt: 1, - repositories: [], - capabilities, - }; -} - -describe("environmentAccess", () => { - it("grants nothing without server capabilities, whatever the feature grants", () => { - expect(environmentAccess(environment(), allGrants)).toEqual({ - canManage: false, - canEditSecrets: false, - canImportRepoSecrets: false, - canEditOverrides: false, - canViewImage: false, - canRebuild: false, - tabs: [], - }); - }); - - it("lets readers view secrets and overrides without editing them", () => { - const access = environmentAccess( - environment({ canRead: true, canManage: false, canUse: true }), - allGrants - ); - expect(access.tabs).toEqual(["secrets", "overrides"]); - expect(access).toMatchObject({ canEditSecrets: false, canEditOverrides: false }); - expect(access.canViewImage).toBe(true); - }); - - it("requires the feature grant alongside environment management", () => { - const access = environmentAccess( - environment({ canRead: true, canManage: true, canUse: true }), - { ...allGrants, manageSecrets: false, manageImages: false } - ); - expect(access.tabs).toEqual(["configuration", "overrides"]); - expect(access).toMatchObject({ - canManage: true, - canEditSecrets: false, - canImportRepoSecrets: false, - canRebuild: false, - canEditOverrides: true, - }); - }); -}); diff --git a/packages/web/src/components/settings/environment-form.test.tsx b/packages/web/src/components/settings/environment-form.test.tsx index 527dc0abec..418f7aaeb5 100644 --- a/packages/web/src/components/settings/environment-form.test.tsx +++ b/packages/web/src/components/settings/environment-form.test.tsx @@ -177,9 +177,69 @@ describe("EnvironmentForm", () => { fireEvent.submit(container.querySelector("form")!); expect(onSubmit).toHaveBeenCalledTimes(1); expect(onSubmit.mock.calls[0][0]).not.toHaveProperty("teamId"); - expect(onSubmit.mock.calls[0][0].repositories).toEqual([ - { repoOwner: "acme", repoName: "web", baseBranch: "main" }, - ]); + }); + + it("omits an unchanged repository selection from edit submissions", async () => { + mocks.reposValue = [repo("Acme", "Web", 1), repo("acme", "api", 2)]; + const onSubmit = vi.fn(); + const user = userEvent.setup(); + render( + + ); + + await user.clear(screen.getByLabelText("Name")); + await user.type(screen.getByLabelText("Name"), "renamed"); + await user.click(screen.getByRole("button", { name: /save environment/i })); + + expect(onSubmit).toHaveBeenCalledWith({ + name: "renamed", + description: null, + prebuildEnabled: false, + }); + }); + + it("sends the full selection when an edit changes only one base branch", async () => { + mocks.reposValue = [repo("group/subgroup", "web", 1), repo("acme", "api", 2)]; + const onSubmit = vi.fn(); + const user = userEvent.setup(); + render( + + ); + + const webRow = screen.getByTitle("group/subgroup/web").closest("div") as HTMLElement; + await user.click(within(webRow).getByRole("button", { name: "main" })); + await user.click(screen.getByRole("option", { name: "develop" })); + await user.click(screen.getByRole("button", { name: /save environment/i })); + + expect(onSubmit).toHaveBeenCalledWith( + expect.objectContaining({ + repositories: [ + { repoOwner: "group/subgroup", repoName: "web", baseBranch: "develop" }, + { repoOwner: "acme", repoName: "api", baseBranch: "main" }, + ], + }) + ); }); it("preserves a nested owner namespace when saving", async () => { @@ -188,7 +248,7 @@ describe("EnvironmentForm", () => { const user = userEvent.setup(); render( { /> ); - await user.click(screen.getByRole("button", { name: /save environment/i })); + await user.click(screen.getByRole("button", { name: /create environment/i })); expect(onSubmit).toHaveBeenCalledWith( expect.objectContaining({ diff --git a/packages/web/src/components/settings/environment-form.tsx b/packages/web/src/components/settings/environment-form.tsx index e204beea3d..c96bed0e70 100644 --- a/packages/web/src/components/settings/environment-form.tsx +++ b/packages/web/src/components/settings/environment-form.tsx @@ -29,7 +29,8 @@ export interface EnvironmentFormValues { name: string; description: string | null; prebuildEnabled: boolean; - repositories: RepositoryInput[]; + /** Omitted when an edit leaves the selection unchanged, so it is not revalidated as a replacement. */ + repositories?: RepositoryInput[]; } /** @@ -126,20 +127,33 @@ export function EnvironmentForm({ const handleSubmit = (e: React.FormEvent) => { e.preventDefault(); if (!canSubmit) return; + const repositories = selectedKeys.map((key) => { + const entry: RepositoryInput = parseRepositoryFullName(key) ?? { + repoOwner: "", + repoName: "", + }; + const branch = branchByKey[key]?.trim(); + if (branch) entry.baseBranch = branch; + return entry; + }); + const initialRepositories = initialValues?.repositories ?? []; + const repositoriesUnchanged = + mode === "edit" && + repositories.length === initialRepositories.length && + repositories.every( + (repository, index) => + selectedKeys[index] === + repositorySelectionKey( + initialRepositories[index].repoOwner, + initialRepositories[index].repoName + ) && (repository.baseBranch ?? "") === initialRepositories[index].baseBranch + ); onSubmit({ ...(mode === "create" ? { teamId } : {}), name: name.trim(), description: description.trim() ? description.trim() : null, prebuildEnabled, - repositories: selectedKeys.map((key) => { - const entry: RepositoryInput = parseRepositoryFullName(key) ?? { - repoOwner: "", - repoName: "", - }; - const branch = branchByKey[key]?.trim(); - if (branch) entry.baseBranch = branch; - return entry; - }), + ...(repositoriesUnchanged ? {} : { repositories }), }); }; diff --git a/packages/web/src/components/settings/integrations/github-auto-review-deprecation-notice.tsx b/packages/web/src/components/settings/integrations/github-auto-review-deprecation-notice.tsx new file mode 100644 index 0000000000..3c3f145c76 --- /dev/null +++ b/packages/web/src/components/settings/integrations/github-auto-review-deprecation-notice.tsx @@ -0,0 +1,20 @@ +"use client"; + +import Link from "next/link"; +import { useAutomationScope } from "@/hooks/use-automation-scope"; + +export function GitHubAutoReviewDeprecationNotice({ id }: { id: string }) { + const { navigation } = useAutomationScope(); + return ( +

+ Replace this deprecated setting with a team-owned automation using the{" "} + + Review new PRs + {" "} + template. Auto-review continues to create workspace-owned sessions. +

+ ); +} diff --git a/packages/web/src/components/settings/integrations/github-global-settings-section.tsx b/packages/web/src/components/settings/integrations/github-global-settings-section.tsx index 0b467ffa45..1a7a8c54eb 100644 --- a/packages/web/src/components/settings/integrations/github-global-settings-section.tsx +++ b/packages/web/src/components/settings/integrations/github-global-settings-section.tsx @@ -28,6 +28,7 @@ import { AlertDialogTitle, } from "@/components/ui/alert-dialog"; import { GitHubAutofixSettingsFields } from "./github-autofix-settings-fields"; +import { GitHubAutoReviewDeprecationNotice } from "./github-auto-review-deprecation-notice"; import { IntegrationSettingsMessage, IntegrationSettingsSection, @@ -220,26 +221,31 @@ export function GlobalSettingsSection({ }} /> - +
+ + +

Repository Scope

diff --git a/packages/web/src/components/settings/integrations/github-integration-settings.test.tsx b/packages/web/src/components/settings/integrations/github-integration-settings.test.tsx index 2142cd2c17..2883afac74 100644 --- a/packages/web/src/components/settings/integrations/github-integration-settings.test.tsx +++ b/packages/web/src/components/settings/integrations/github-integration-settings.test.tsx @@ -13,6 +13,11 @@ import { } from "@open-inspect/shared/types/integrations"; import { GitHubIntegrationSettings } from "./github-integration-settings"; +let search = ""; +vi.mock("next/navigation", () => ({ + useSearchParams: () => new URLSearchParams(search), +})); + vi.mock("@/hooks/use-current-user-authorization", () => ({ useCurrentUserAuthorization: () => ({ hasPermission: () => true }), })); @@ -130,6 +135,7 @@ beforeAll(() => { }); beforeEach(() => { + search = ""; fetchMock.mockReset(); toastSuccess.mockReset(); toastError.mockReset(); @@ -144,6 +150,79 @@ afterEach(() => { }); describe("GitHubIntegrationSettings", () => { + it("marks global auto-review deprecated with an accessible team-owned template replacement", () => { + setupSWR({ global: null }); + + render(); + + const toggle = screen.getByRole("switch", { name: /auto-review pr changes/i }); + const label = toggle.closest("label")!; + expect(label).toHaveAttribute("for", toggle.id); + expect(within(label).getByText("Deprecated")).toBeInTheDocument(); + expect(within(label).queryByRole("link")).not.toBeInTheDocument(); + expect(toggle).toHaveAccessibleDescription(/team-owned automation/); + expect(toggle).toHaveAccessibleDescription(/workspace-owned sessions/); + expect(toggle).toHaveAttribute("aria-checked", "true"); + expect(screen.getByRole("link", { name: "Review new PRs" })).toHaveAttribute( + "href", + "/automations/new?template=review-new-prs&requireTeam=true" + ); + }); + + it("preserves the current team scope in both replacement links", () => { + search = "teamId=team_engineering"; + setupSWR({ + global: null, + repos: [{ repo: "acme/web", settings: {} }], + availableRepos: [repo("acme/web")], + }); + + render(); + + const links = screen.getAllByRole("link", { name: "Review new PRs" }); + expect(links).toHaveLength(2); + for (const link of links) { + expect(link).toHaveAttribute( + "href", + "/automations/new?template=review-new-prs&teamId=team_engineering&requireTeam=true" + ); + } + }); + + it.each([undefined, false, true])( + "shows the deprecation notice and template for repo auto-review=%s without changing controls", + (autoReviewOnOpen) => { + setupSWR({ + global: { defaults: { autoReviewOnOpen: true } }, + repos: [{ repo: "acme/web", settings: { autoReviewOnOpen } }], + availableRepos: [repo("acme/web")], + }); + + render(); + + const controls = autoReviewControls(repoOverrideRow("acme/web")); + expect(within(controls).getByText("Deprecated")).toBeInTheDocument(); + expect(within(controls).getByRole("link", { name: "Review new PRs" })).toHaveAttribute( + "href", + "/automations/new?template=review-new-prs&requireTeam=true" + ); + expect(within(controls).getByRole("combobox")).toHaveAccessibleDescription( + /team-owned automation/ + ); + if (autoReviewOnOpen === undefined) { + expect(within(controls).queryByRole("switch")).not.toBeInTheDocument(); + expect(within(controls).getByRole("combobox")).toHaveTextContent("Use global default"); + } else { + const toggle = within(controls).getByRole("switch", { + name: autoReviewOnOpen ? "Enabled" : "Disabled", + }); + expect(toggle.closest("label")).not.toBeNull(); + expect(toggle).toHaveAttribute("aria-checked", String(autoReviewOnOpen)); + expect(toggle).toHaveAccessibleDescription(/workspace-owned sessions/); + } + } + ); + it("starts integration content at heading level two", () => { setupSWR({ global: null }); diff --git a/packages/web/src/components/settings/integrations/github-repo-overrides-section.tsx b/packages/web/src/components/settings/integrations/github-repo-overrides-section.tsx index f65e4ff9a4..ab5012c282 100644 --- a/packages/web/src/components/settings/integrations/github-repo-overrides-section.tsx +++ b/packages/web/src/components/settings/integrations/github-repo-overrides-section.tsx @@ -1,6 +1,6 @@ "use client"; -import { useState } from "react"; +import { useId, useState } from "react"; import { mutate } from "swr"; import { toast } from "sonner"; import { @@ -32,6 +32,7 @@ import { SelectValue, } from "@/components/ui/select"; import { GitHubAutofixSettingsFields } from "./github-autofix-settings-fields"; +import { GitHubAutoReviewDeprecationNotice } from "./github-auto-review-deprecation-notice"; const REPO_SETTINGS_KEY = "/api/integration-settings/github/repos"; @@ -140,6 +141,7 @@ function RepoOverrideRow({ defaultAutoReviewOnOpen: boolean; defaultAutofix: ResolvedGitHubAutofixSettings; }) { + const autoReviewNoticeId = useId(); const [model, setModel] = useState(entry.settings.model ?? ""); const [effort, setEffort] = useState(entry.settings.reasoningEffort ?? ""); const [triggerUserMode, setTriggerUserMode] = useState<"global" | "override">( @@ -327,10 +329,13 @@ function RepoOverrideRow({
-

Auto-review PR changes

+

+ Auto-review PR changes + Deprecated +

onChange(isValidHarness(next) ? next : undefined)} + > + + + + + {inheritLabel && {inheritLabel}} + {HARNESS_IDS.map((harness) => ( + + {getHarnessLabel(harness)} + + ))} + + + ); +} diff --git a/packages/web/src/components/settings/integrations/linear-integration-settings.test.tsx b/packages/web/src/components/settings/integrations/linear-integration-settings.test.tsx new file mode 100644 index 0000000000..5020500fc6 --- /dev/null +++ b/packages/web/src/components/settings/integrations/linear-integration-settings.test.tsx @@ -0,0 +1,366 @@ +// @vitest-environment jsdom +/// + +import { act, cleanup, fireEvent, render, screen, waitFor, within } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import * as matchers from "@testing-library/jest-dom/matchers"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import type { + LinearBotSettings, + LinearGlobalConfig, +} from "@open-inspect/shared/types/integrations"; +import type { ModelCategory } from "@open-inspect/shared/models"; +import { browserApiFetch } from "@/lib/browser-api-fetch"; +import { LinearIntegrationSettings } from "./linear-integration-settings"; + +expect.extend(matchers); + +const { authorization, modelOptions, useSWRMock, mutateMock, toastSuccess, toastError } = + vi.hoisted(() => ({ + authorization: { canManageGlobal: true }, + modelOptions: [ + { + category: "Anthropic", + models: [{ id: "anthropic/claude-sonnet-4-6", name: "Claude Sonnet 4.6", description: "" }], + }, + { + category: "OpenAI", + models: [{ id: "openai/gpt-6-sol", name: "GPT 6 Sol", description: "" }], + }, + ] satisfies ModelCategory[], + useSWRMock: vi.fn(), + mutateMock: vi.fn(), + toastSuccess: vi.fn(), + toastError: vi.fn(), + })); + +vi.mock("@/hooks/use-current-user-authorization", () => ({ + useCurrentUserAuthorization: () => ({ + hasPermission: (permission: string) => + permission !== "integrations.manage" || authorization.canManageGlobal, + }), +})); +vi.mock("@/hooks/use-enabled-models", () => ({ + useEnabledModels: () => ({ enabledModelOptions: modelOptions }), +})); +vi.mock("@/lib/browser-api-fetch", () => ({ browserApiFetch: vi.fn() })); +vi.mock("swr", () => ({ default: useSWRMock, mutate: mutateMock })); +vi.mock("sonner", () => ({ toast: { success: toastSuccess, error: toastError } })); + +const globalKey = "/api/integration-settings/linear"; +const repoKey = "/api/integration-settings/linear/repos"; + +function setupSWR(opts: { + settings?: LinearGlobalConfig | null; + overrides?: { repo: string; settings: LinearBotSettings }[]; +}) { + useSWRMock.mockImplementation((key: string) => { + if (key === globalKey) { + return { + data: opts.settings === undefined ? undefined : { settings: opts.settings }, + isLoading: false, + }; + } + if (key === repoKey) { + return { data: { repos: opts.overrides ?? [] }, isLoading: false }; + } + return { data: { repos: [] }, isLoading: false }; + }); +} + +beforeEach(() => { + vi.resetAllMocks(); + authorization.canManageGlobal = true; +}); +afterEach(cleanup); + +describe("LinearIntegrationSettings unbound policy", () => { + it("closes the open policy menu when permission is revoked without changing the value", async () => { + const user = userEvent.setup(); + setupSWR({ settings: null }); + const view = render(); + const policy = screen.getByRole("combobox", { name: "Unbound Linear teams" }); + await user.click(policy); + expect(screen.getByRole("listbox")).toBeInTheDocument(); + expect(screen.getByRole("option", { name: "Reject requests until bound" })).toBeInTheDocument(); + + authorization.canManageGlobal = false; + view.rerender(); + await waitFor(() => expect(screen.queryByRole("listbox")).not.toBeInTheDocument()); + expect(screen.queryAllByRole("option")).toHaveLength(0); + expect(policy).toBeDisabled(); + expect(policy).toHaveAttribute("disabled"); + expect(policy).toHaveTextContent("Create workspace-level sessions"); + + authorization.canManageGlobal = true; + view.rerender(); + expect(policy).toBeEnabled(); + expect(policy).toHaveTextContent("Create workspace-level sessions"); + expect(screen.getByRole("button", { name: "Save" })).toBeDisabled(); + expect(browserApiFetch).not.toHaveBeenCalled(); + }); + + it.each([null, { defaults: {} }])("uses the shared default for unset policy: %j", (settings) => { + setupSWR({ settings }); + render(); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveTextContent( + "Create workspace-level sessions" + ); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveAttribute( + "aria-describedby", + "linear-unbound-channels-help" + ); + expect(screen.getByRole("button", { name: "Save" })).toBeDisabled(); + }); + + it("hydrates a saved policy when settings arrive and resyncs clean revalidation", () => { + setupSWR({}); + const view = render(); + setupSWR({ settings: { defaults: { unboundChannels: "reject" } } }); + view.rerender(); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveTextContent( + "Reject requests until bound" + ); + setupSWR({ settings: null }); + view.rerender(); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveTextContent( + "Create workspace-level sessions" + ); + }); + + it("preserves dirty edits during revalidation and saves the policy in global defaults", async () => { + const user = userEvent.setup(); + setupSWR({ settings: null }); + const view = render(); + const policy = screen.getByRole("combobox", { name: "Unbound Linear teams" }); + await user.click(policy); + await user.click(await screen.findByRole("option", { name: "Reject requests until bound" })); + expect(screen.getByRole("button", { name: "Save" })).toBeEnabled(); + setupSWR({ settings: { defaults: { unboundChannels: "workspace" } } }); + view.rerender(); + expect(policy).toHaveTextContent("Reject requests until bound"); + let finish!: (response: Response) => void; + vi.mocked(browserApiFetch).mockReturnValueOnce( + new Promise((resolve) => { + finish = resolve; + }) + ); + mutateMock.mockImplementation((_key: string, data: { settings: LinearGlobalConfig }) => { + setupSWR({ settings: data.settings }); + }); + fireEvent.click(screen.getByRole("button", { name: "Save" })); + expect(policy).toBeDisabled(); + expect(policy).toHaveAttribute("disabled"); + expect(screen.getByRole("textbox", { name: "Issue Session Instructions" })).toBeDisabled(); + expect(screen.getByRole("checkbox", { name: "Allow user model preferences" })).toBeDisabled(); + expect(screen.getByRole("radio", { name: /All repositories/ })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Saving..." })).toBeDisabled(); + const saved = { + settings: { + defaults: { + allowUserPreferenceOverride: true, + allowLabelModelOverride: true, + emitToolProgressActivities: true, + unboundChannels: "reject", + }, + }, + }; + expect(browserApiFetch).toHaveBeenCalledWith(globalKey, { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(saved), + }); + await act(async () => finish(Response.json({ ok: true }))); + await waitFor(() => expect(mutateMock).toHaveBeenCalledWith(globalKey, saved)); + view.rerender(); + expect(policy).toHaveTextContent("Reject requests until bound"); + expect(screen.getByRole("button", { name: "Save" })).toBeDisabled(); + expect(toastSuccess).toHaveBeenCalledWith("Settings saved."); + }); + + it("deletes global settings on reset and restores the shared policy default", async () => { + const user = userEvent.setup(); + setupSWR({ settings: { defaults: { unboundChannels: "reject" } } }); + vi.mocked(browserApiFetch).mockResolvedValueOnce(new Response(null, { status: 204 })); + mutateMock.mockImplementation((_key: string, data: { settings: null }) => { + setupSWR({ settings: data.settings }); + }); + const view = render(); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveTextContent( + "Reject requests until bound" + ); + await user.click(screen.getByRole("button", { name: "Reset to defaults" })); + await user.click(screen.getByRole("button", { name: "Reset" })); + await waitFor(() => expect(mutateMock).toHaveBeenCalledWith(globalKey, { settings: null })); + view.rerender(); + expect(browserApiFetch).toHaveBeenCalledWith(globalKey, { method: "DELETE" }); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveTextContent( + "Create workspace-level sessions" + ); + expect(screen.getByRole("button", { name: "Save" })).toBeDisabled(); + expect(toastSuccess).toHaveBeenCalledWith("Settings reset to defaults."); + }); + + it.each(["save", "reset"])("retains the policy if %s is refused", async (operation) => { + const user = userEvent.setup(); + setupSWR({ settings: { defaults: { unboundChannels: "workspace" } } }); + vi.mocked(browserApiFetch).mockResolvedValueOnce( + Response.json({ error: "Forbidden" }, { status: 403 }) + ); + render(); + await user.click(screen.getByRole("combobox", { name: "Unbound Linear teams" })); + await user.click(await screen.findByRole("option", { name: "Reject requests until bound" })); + if (operation === "reset") { + await user.click(screen.getByRole("button", { name: "Reset to defaults" })); + await user.click(screen.getByRole("button", { name: "Reset" })); + } else { + await user.click(screen.getByRole("button", { name: "Save" })); + } + await waitFor(() => expect(toastError).toHaveBeenCalledWith("Forbidden")); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveTextContent( + "Reject requests until bound" + ); + expect(screen.getByRole("button", { name: "Save" })).toBeEnabled(); + expect(mutateMock).not.toHaveBeenCalled(); + }); + + it("disables policy/save/reset without global integration-management permission", async () => { + const user = userEvent.setup(); + setupSWR({ settings: { defaults: { unboundChannels: "reject" } } }); + const view = render(); + await user.click(screen.getByRole("combobox", { name: "Unbound Linear teams" })); + await user.click( + await screen.findByRole("option", { name: "Create workspace-level sessions" }) + ); + authorization.canManageGlobal = false; + view.rerender(); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toBeDisabled(); + expect(screen.getByRole("combobox", { name: "Unbound Linear teams" })).toHaveAttribute( + "disabled" + ); + expect(screen.getByRole("button", { name: "Save" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Reset to defaults" })).toBeDisabled(); + await user.click(screen.getByRole("combobox", { name: "Unbound Linear teams" })); + expect(screen.queryByRole("listbox")).not.toBeInTheDocument(); + await user.click(screen.getByRole("button", { name: "Save" })); + await user.click(screen.getByRole("button", { name: "Reset to defaults" })); + expect(browserApiFetch).not.toHaveBeenCalled(); + }); + + it("disables an open reset confirmation when global permission is revoked", async () => { + const user = userEvent.setup(); + setupSWR({ settings: { defaults: { unboundChannels: "reject" } } }); + const view = render(); + await user.click(screen.getByRole("button", { name: "Reset to defaults" })); + authorization.canManageGlobal = false; + view.rerender(); + expect(screen.getByRole("button", { name: "Reset" })).toBeDisabled(); + await user.click(screen.getByRole("button", { name: "Reset" })); + expect(browserApiFetch).not.toHaveBeenCalled(); + }); + + it("does not expose or serialize the global policy in repository overrides", async () => { + setupSWR({ + settings: { defaults: { unboundChannels: "reject" } }, + overrides: [{ repo: "group/subgroup/web", settings: {} }], + }); + vi.mocked(browserApiFetch).mockResolvedValueOnce(Response.json({ ok: true })); + render(); + const row = screen.getByText("group/subgroup/web").parentElement!; + expect(within(row).queryByRole("combobox", { name: "Unbound Linear teams" })).toBeNull(); + fireEvent.click(within(row).getByRole("checkbox", { name: "Tool updates" })); + fireEvent.click(within(row).getByRole("button", { name: "Save" })); + await waitFor(() => expect(mutateMock).toHaveBeenCalledWith(repoKey)); + expect(browserApiFetch).toHaveBeenCalledWith(`${repoKey}/group%2Fsubgroup/web`, { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + settings: { + allowUserPreferenceOverride: true, + allowLabelModelOverride: true, + emitToolProgressActivities: false, + }, + }), + }); + }); +}); + +describe("LinearIntegrationSettings harness", () => { + async function optionNames(user: ReturnType, trigger: HTMLElement) { + await user.click(trigger); + const names = screen.getAllByRole("option").map((option) => option.textContent); + await user.keyboard("{Escape}"); + return names; + } + + it("offers only models the selected harness can run and saves the global harness", async () => { + const user = userEvent.setup(); + setupSWR({ settings: { defaults: { model: "openai/gpt-6-sol" } } }); + vi.mocked(browserApiFetch).mockResolvedValueOnce(Response.json({ ok: true })); + render(); + const harness = screen.getByRole("combobox", { name: "Agent harness" }); + const model = screen.getByRole("combobox", { name: "Default model" }); + expect(harness).toHaveTextContent("OpenCode"); + expect(await optionNames(user, model)).toContain("GPT 6 Sol"); + + await user.click(harness); + await user.click(await screen.findByRole("option", { name: "Claude Agent" })); + + expect(model).toHaveTextContent("Use system default"); + expect(await optionNames(user, model)).toEqual(["Use system default", "Claude Sonnet 4.6"]); + await user.click(screen.getByRole("button", { name: "Save" })); + await waitFor(() => expect(browserApiFetch).toHaveBeenCalled()); + const body = JSON.parse(String(vi.mocked(browserApiFetch).mock.calls[0][1]?.body)); + expect(body.settings.defaults).toMatchObject({ harness: "claude" }); + expect(body.settings.defaults).not.toHaveProperty("model"); + }); + + it("filters repository models by the inherited harness and saves an explicit override", async () => { + const user = userEvent.setup(); + setupSWR({ + settings: { defaults: { harness: "claude" } }, + overrides: [{ repo: "acme/web", settings: {} }], + }); + vi.mocked(browserApiFetch).mockResolvedValueOnce(Response.json({ ok: true })); + render(); + const row = screen.getByText("acme/web").parentElement!; + const [harness, model] = within(row).getAllByRole("combobox"); + expect(harness).toHaveTextContent("Inherit (Claude Agent)"); + expect(await optionNames(user, model)).toEqual(["Claude Sonnet 4.6"]); + + await user.click(harness); + await user.click(await screen.findByRole("option", { name: "OpenCode" })); + await user.click(model); + await user.click(await screen.findByRole("option", { name: "GPT 6 Sol" })); + await user.click(within(row).getByRole("button", { name: "Save" })); + + await waitFor(() => expect(mutateMock).toHaveBeenCalledWith(repoKey)); + const body = JSON.parse(String(vi.mocked(browserApiFetch).mock.calls[0][1]?.body)); + expect(body.settings).toMatchObject({ harness: "opencode", model: "openai/gpt-6-sol" }); + }); + + const fallbackNotice = "Sessions using this default model will fall back to OpenCode."; + + it("warns when a repository harness clashes with the inherited global model", () => { + setupSWR({ + settings: { defaults: { model: "openai/gpt-5.4" } }, + overrides: [{ repo: "acme/web", settings: { harness: "claude" } }], + }); + render(); + const row = screen.getByText("acme/web").parentElement!; + + expect(row).toHaveTextContent( + `Model "openai/gpt-5.4" cannot run on the Claude Agent harness. ${fallbackNotice}` + ); + }); + + it("does not warn when a stale global model normalizes to one the harness can run", () => { + setupSWR({ + settings: { defaults: { model: "openai/gpt-5" } }, + overrides: [{ repo: "acme/web", settings: { harness: "claude" } }], + }); + render(); + + expect(screen.queryByText(/will fall back to OpenCode/)).toBeNull(); + }); +}); diff --git a/packages/web/src/components/settings/integrations/linear-integration-settings.tsx b/packages/web/src/components/settings/integrations/linear-integration-settings.tsx index 91c53664d5..70e5f4d5ec 100644 --- a/packages/web/src/components/settings/integrations/linear-integration-settings.tsx +++ b/packages/web/src/components/settings/integrations/linear-integration-settings.tsx @@ -8,17 +8,29 @@ import { parseRepositoryFullName, } from "@open-inspect/shared/types/repositories"; import type { EnrichedRepository } from "@open-inspect/shared/types/repository-catalog"; -import type { - LinearBotSettings, - LinearGlobalConfig, +import { + DEFAULT_LINEAR_UNBOUND_CHANNELS, + type LinearBotGlobalSettings, + type LinearBotSettings, + type LinearGlobalConfig, } from "@open-inspect/shared/types/integrations"; import { MODEL_REASONING_CONFIG, + getValidModelOrDefault, isValidReasoningEffort, type ModelCategory, type ValidModel, } from "@open-inspect/shared/models"; +import { + DEFAULT_HARNESS, + checkHarnessCompatibility, + getHarnessLabel, + getValidHarnessOrDefault, + harnessSupportsModel, + type HarnessId, +} from "@open-inspect/shared/harnesses"; import { useEnabledModels } from "@/hooks/use-enabled-models"; +import { filterModelOptionsForHarness } from "@/lib/session-harness"; import { browserApiFetch } from "@/lib/browser-api-fetch"; import { IntegrationSettingsSkeleton } from "./integration-settings-skeleton"; import { SettingsCardSection } from "../settings-card-section"; @@ -46,6 +58,7 @@ import { AlertDialogTitle, } from "@/components/ui/alert-dialog"; import { ModelReasoningDefaultsFields } from "./model-reasoning-defaults-fields"; +import { HarnessSelect } from "./harness-select"; import { useCurrentUserAuthorization } from "@/hooks/use-current-user-authorization"; const GLOBAL_SETTINGS_KEY = "/api/integration-settings/linear"; @@ -118,6 +131,7 @@ export function LinearIntegrationSettings() {
@@ -125,13 +139,15 @@ export function LinearIntegrationSettings() {
@@ -141,13 +157,18 @@ export function LinearIntegrationSettings() { function GlobalSettingsSection({ settings, + canManageGlobal, availableRepos, enabledModelOptions, }: { settings: LinearGlobalConfig | null | undefined; + canManageGlobal: boolean; availableRepos: EnrichedRepository[]; enabledModelOptions: ModelCategory[]; }) { + const [harness, setHarness] = useState( + getValidHarnessOrDefault(settings?.defaults?.harness) + ); const [model, setModel] = useState(settings?.defaults?.model ?? ""); const [effort, setEffort] = useState(settings?.defaults?.reasoningEffort ?? ""); const [enabledRepos, setEnabledRepos] = useState(settings?.enabledRepos ?? []); @@ -166,37 +187,38 @@ function GlobalSettingsSection({ const [issueSessionInstructions, setIssueSessionInstructions] = useState( settings?.defaults?.issueSessionInstructions ?? "" ); + const [unboundChannels, setUnboundChannels] = useState<"workspace" | "reject">( + settings?.defaults?.unboundChannels ?? DEFAULT_LINEAR_UNBOUND_CHANNELS + ); const [saving, setSaving] = useState(false); const [error, setError] = useState(""); const [dirty, setDirty] = useState(false); - const [initialized, setInitialized] = useState(false); const [showResetDialog, setShowResetDialog] = useState(false); useEffect(() => { - if (settings !== undefined && !initialized) { - if (settings) { - setModel(settings.defaults?.model ?? ""); - setEffort(settings.defaults?.reasoningEffort ?? ""); - setEnabledRepos(settings.enabledRepos ?? []); - setRepoScopeMode(settings.enabledRepos === undefined ? "all" : "selected"); - setAllowUserPreferenceOverride(settings.defaults?.allowUserPreferenceOverride ?? true); - setAllowLabelModelOverride(settings.defaults?.allowLabelModelOverride ?? true); - setEmitToolProgressActivities(settings.defaults?.emitToolProgressActivities ?? true); - setIssueSessionInstructions(settings.defaults?.issueSessionInstructions ?? ""); - } - setInitialized(true); - } - }, [settings, initialized]); + if (settings === undefined || dirty || saving) return; + setHarness(getValidHarnessOrDefault(settings?.defaults?.harness)); + setModel(settings?.defaults?.model ?? ""); + setEffort(settings?.defaults?.reasoningEffort ?? ""); + setEnabledRepos(settings?.enabledRepos ?? []); + setRepoScopeMode(settings?.enabledRepos == null ? "all" : "selected"); + setAllowUserPreferenceOverride(settings?.defaults?.allowUserPreferenceOverride ?? true); + setAllowLabelModelOverride(settings?.defaults?.allowLabelModelOverride ?? true); + setEmitToolProgressActivities(settings?.defaults?.emitToolProgressActivities ?? true); + setIssueSessionInstructions(settings?.defaults?.issueSessionInstructions ?? ""); + setUnboundChannels(settings?.defaults?.unboundChannels ?? DEFAULT_LINEAR_UNBOUND_CHANNELS); + }, [settings, dirty, saving]); const isConfigured = settings !== null && settings !== undefined; const resetNotice = - "Reset all Linear settings to defaults? This enables both label/user model overrides."; + "Reset all Linear settings to defaults? New sessions run on OpenCode, both label/user model overrides are enabled, and the default policy for unbound Linear teams is restored."; const handleReset = () => { setShowResetDialog(true); }; const handleConfirmReset = async () => { + if (!canManageGlobal || saving) return; setSaving(true); setError(""); @@ -204,7 +226,8 @@ function GlobalSettingsSection({ const res = await browserApiFetch(GLOBAL_SETTINGS_KEY, { method: "DELETE" }); if (res.ok) { - mutate(GLOBAL_SETTINGS_KEY); + mutate(GLOBAL_SETTINGS_KEY, { settings: null }); + setHarness(DEFAULT_HARNESS); setModel(""); setEffort(""); setEnabledRepos([]); @@ -213,6 +236,7 @@ function GlobalSettingsSection({ setAllowLabelModelOverride(true); setEmitToolProgressActivities(true); setIssueSessionInstructions(""); + setUnboundChannels(DEFAULT_LINEAR_UNBOUND_CHANNELS); setDirty(false); toast.success("Settings reset to defaults."); } else { @@ -227,15 +251,18 @@ function GlobalSettingsSection({ }; const handleSave = async () => { + if (!canManageGlobal || saving || !dirty) return; setSaving(true); setError(""); - const defaults: LinearBotSettings = { + const defaults: LinearBotGlobalSettings = { allowUserPreferenceOverride, allowLabelModelOverride, emitToolProgressActivities, + unboundChannels, }; + if (harness !== DEFAULT_HARNESS) defaults.harness = harness; if (model) defaults.model = model; if (effort) defaults.reasoningEffort = effort; if (issueSessionInstructions) defaults.issueSessionInstructions = issueSessionInstructions; @@ -253,7 +280,7 @@ function GlobalSettingsSection({ }); if (res.ok) { - mutate(GLOBAL_SETTINGS_KEY); + mutate(GLOBAL_SETTINGS_KEY, { settings: body }); toast.success("Settings saved."); setDirty(false); } else { @@ -281,161 +308,227 @@ function GlobalSettingsSection({ title="Defaults & Scope" description="Global model, fallback behavior, and repository scope." > - {error && } - - { - setModel(nextModel); - setEffort(nextEffort); - setDirty(true); - setError(""); - }} - /> - -
- -