diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a9daf5378b..0b684cf201 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,6 +2,8 @@ name: CI on: pull_request: + branches: + - main push: branches: - main diff --git a/.github/workflows/inventory-ci.yml b/.github/workflows/inventory-ci.yml new file mode 100644 index 0000000000..0d2e2ed90a --- /dev/null +++ b/.github/workflows/inventory-ci.yml @@ -0,0 +1,140 @@ +name: Inventory CI + +on: + pull_request: + branches: + - inventory-descriptions + push: + branches: + - inventory-descriptions + +permissions: + contents: read + +concurrency: + group: inventory-ci-${{ github.ref }} + cancel-in-progress: true + +env: + DO_NOT_TRACK: "1" + EXECUTOR_DISABLE_ANALYTICS: "1" + EXECUTOR_DISABLE_INTEGRATIONS_FETCH: "1" + +jobs: + checks: + name: Inventory ${{ matrix.name }} + runs-on: ubuntu-24.04 + timeout-minutes: 15 + strategy: + fail-fast: false + matrix: + include: + - name: Format + command: bun run format:check + - name: Lint + command: bun run lint + - name: Typecheck + command: bunx turbo run typecheck --filter='./packages/core/*' --filter='./packages/hosts/*' --filter=@executor-js/host-selfhost --concurrency=2 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 24 + - uses: oven-sh/setup-bun@v2 + with: + bun-version-file: package.json + - uses: actions/cache@v4 + with: + path: ~/.bun/install/cache + key: ${{ runner.os }}-${{ runner.arch }}-inventory-bun-${{ hashFiles('bun.lock', 'package.json', 'patches/**') }} + restore-keys: | + ${{ runner.os }}-${{ runner.arch }}-inventory-bun- + - run: bun install --frozen-lockfile + - run: bun run check:patches + - run: ${{ matrix.command }} + + tests: + name: Inventory ${{ matrix.name }} + runs-on: ubuntu-24.04 + timeout-minutes: 15 + strategy: + fail-fast: false + matrix: + include: + - name: Core and hosts tests + browsers: true + # Forward worker limits, the browser path and telemetry opt-outs. + command: bunx turbo run test --filter='./packages/core/*' --filter='./packages/hosts/*' --concurrency=1 --env-mode=loose + - name: Selfhost tests + browsers: false + command: bun run --cwd apps/host-selfhost test + env: + VITEST_MAX_WORKERS: "2" + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 24 + - uses: oven-sh/setup-bun@v2 + with: + bun-version-file: package.json + - uses: actions/cache@v4 + with: + path: ~/.bun/install/cache + key: ${{ runner.os }}-${{ runner.arch }}-inventory-bun-${{ hashFiles('bun.lock', 'package.json', 'patches/**') }} + restore-keys: | + ${{ runner.os }}-${{ runner.arch }}-inventory-bun- + - run: bun install --frozen-lockfile + - run: bun run check:patches + - uses: actions/cache@v4 + if: matrix.browsers + with: + path: ~/.cache/ms-playwright + key: ${{ runner.os }}-${{ runner.arch }}-inventory-playwright-${{ hashFiles('bun.lock') }} + # The MCP Apps shell suite drives a real Chromium. Use the workspace's + # pinned Playwright and install its Linux libraries on the hosted runner. + - name: Install Chromium + if: matrix.browsers + working-directory: e2e + run: | + bunx playwright install --with-deps chromium + echo "PLAYWRIGHT_CHROMIUM_EXECUTABLE_PATH=$(bun -e 'import { chromium } from "playwright"; console.log(chromium.executablePath())')" >> "$GITHUB_ENV" + - run: ${{ matrix.command }} + + passthrough: + name: Inventory Passthrough e2e + runs-on: ubuntu-24.04 + timeout-minutes: 15 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 24 + - uses: oven-sh/setup-bun@v2 + with: + bun-version-file: package.json + - uses: actions/cache@v4 + with: + path: ~/.bun/install/cache + key: ${{ runner.os }}-${{ runner.arch }}-inventory-bun-${{ hashFiles('bun.lock', 'package.json', 'patches/**') }} + restore-keys: | + ${{ runner.os }}-${{ runner.arch }}-inventory-bun- + - run: bun install --frozen-lockfile + - run: bun run check:patches + - uses: actions/cache@v4 + with: + path: ~/.cache/ms-playwright + key: ${{ runner.os }}-${{ runner.arch }}-inventory-playwright-${{ hashFiles('bun.lock') }} + - name: Install Chromium + working-directory: e2e + run: bunx playwright install --with-deps chromium + - name: Run passthrough journey with credential assertion + working-directory: e2e + run: bunx vitest run --project selfhost scenarios/mcp-passthrough.test.ts + - name: Upload passthrough evidence + if: always() + uses: actions/upload-artifact@v4 + with: + name: inventory-passthrough-e2e + path: e2e/runs/selfhost/ + retention-days: 7 diff --git a/.github/workflows/pkg-pr-new.yml b/.github/workflows/pkg-pr-new.yml index 1966249d59..dfa6938c33 100644 --- a/.github/workflows/pkg-pr-new.yml +++ b/.github/workflows/pkg-pr-new.yml @@ -2,6 +2,8 @@ name: Publish (pkg.pr.new) on: pull_request: + branches: + - main permissions: contents: read diff --git a/.github/workflows/preview.yml b/.github/workflows/preview.yml index 2409fd4f23..3891764f5e 100644 --- a/.github/workflows/preview.yml +++ b/.github/workflows/preview.yml @@ -13,6 +13,8 @@ name: Preview on: pull_request: + branches: + - main types: [opened, reopened, synchronize, closed] permissions: diff --git a/apps/host-selfhost/scripts/bench-mcp-session.ts b/apps/host-selfhost/scripts/bench-mcp-session.ts new file mode 100644 index 0000000000..ebf2dcd9fc --- /dev/null +++ b/apps/host-selfhost/scripts/bench-mcp-session.ts @@ -0,0 +1,167 @@ +import { mkdtempSync, readFileSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { resolve, join } from "node:path"; +import { createHash } from "node:crypto"; +import { Effect, Layer, Tracer } from "effect"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; +import { AuthTemplateSlug, ConnectionName, IntegrationSlug } from "@executor-js/sdk"; +import { makeMcpBuildServer, makeScopedExecutor, PluginsProvider } from "@executor-js/api/server"; +import { createSelfHostDb, SelfHostDb } from "../src/db/self-host-db"; +import { SelfHostExecutionStackLayer, SelfHostScopedExecutorSeams } from "../src/execution"; +import type { SelfHostPlugins } from "../src/plugins"; +import executorConfig from "../executor.config"; + +const dataDir = mkdtempSync(join(tmpdir(), "executor-session-bench-")); +process.env.EXECUTOR_DATA_DIR = dataDir; +const db = await createSelfHostDb({ path: join(dataDir, "data.db") }); +const dbLayer = Layer.succeed(SelfHostDb)(db); +const samples = Number(process.env.BENCH_SAMPLES ?? 25); +const phases: Record = {}; +const record = (name: string, ms: number) => (phases[name] ??= []).push(ms); +const tracer = Tracer.make({ + span(options) { + const span = new Tracer.NativeSpan(options); + const end = span.end.bind(span); + span.end = (time, exit) => { + record(options.name, Number(time - options.startTime) / 1e6); + end(time, exit); + }; + return span; + }, +}); +const pluginsLayer = Layer.succeed(PluginsProvider)({ + plugins(context) { + const start = performance.now(); + const plugins = executorConfig.plugins({ + activeToolkitSlug: + context?.mcpResource?.kind === "toolkit" ? context.mcpResource.slug : undefined, + }); + record("plugins.factory", performance.now() - start); + return plugins; + }, +}); +const stack = SelfHostExecutionStackLayer.pipe(Layer.provide(dbLayer)); +const build = makeMcpBuildServer(Layer.merge(stack, pluginsLayer)); +const principal = { + accountId: "bench", + organizationId: "bench", + organizationName: "Bench", + email: "bench@example.test", + name: "Bench", + avatarUrl: null, + roles: ["user"], + orgRoleModel: "organization" as const, +}; +const digest = (value: unknown) => createHash("sha256").update(JSON.stringify(value)).digest("hex"); +const hashes: Record = {}; +const checkHash = (name: string, value: unknown) => { + const hash = digest(value); + if (hashes[name] !== undefined && hashes[name] !== hash) { + throw new Error(`Benchmark result changed between samples: ${name}`); + } + hashes[name] = hash; +}; +const run = (effect: Effect.Effect) => Effect.runPromise(effect); +try { + const seed = await run( + makeScopedExecutor("bench", "bench", "Bench").pipe( + Effect.provide(SelfHostScopedExecutorSeams), + Effect.provide(dbLayer), + ), + ); + for (const slug of ["cloudflare_api", "vercel_api"]) { + const fixture = slug === "cloudflare_api" ? "cloudflare" : "vercel"; + await run( + seed.openapi.addSpec({ + spec: { + kind: "blob", + value: readFileSync( + resolve(import.meta.dir, `../../../packages/plugins/openapi/fixtures/${fixture}.json`), + "utf8", + ), + }, + slug, + }), + ); + await run( + seed.connections.create({ + owner: "org", + integration: IntegrationSlug.make(slug), + name: ConnectionName.make("main"), + template: AuthTemplateSlug.make("apiKey"), + value: "fixture-token", + }), + ); + } + const tools = await run(seed.tools.list({ includeAnnotations: false })); + await run(seed.close()); + for (let i = 0; i < samples; i++) { + const start = performance.now(); + const built = await run( + build(principal, { mode: "passthrough", artifactsEnabled: false }).pipe( + Effect.withTracer(tracer), + ), + ); + record("server.build", performance.now() - start); + const [ct, st] = InMemoryTransport.createLinkedPair(); + const client = new Client({ name: "fixture-bench", version: "1" }, { capabilities: {} }); + await built.mcpServer.connect(st); + try { + let start = performance.now(); + await client.connect(ct); + record("client.initialize", performance.now() - start); + start = performance.now(); + const listed = await client.listTools(); + record("client.tools_list", performance.now() - start); + checkHash("tools", listed); + checkHash("instructions", client.getInstructions()); + for (const [name, args] of [ + ["filtered", { query: "dns record", integration: "cloudflare_api", limit: 3 }], + ["full", { query: "dns record", integration: "cloudflare_api", limit: 3, detail: "full" }], + ] as const) { + start = performance.now(); + const result = await client.callTool({ name: "search", arguments: args }); + record(`client.search.${name}`, performance.now() - start); + if (result.isError) throw new Error(JSON.stringify(result)); + checkHash(name, result); + } + } finally { + await client.close(); + await built.mcpServer.close(); + await run(built.engine.shutdown); + if (built.executor) await run(built.executor.close()); + } + } + const summary = Object.fromEntries( + Object.entries(phases).map(([name, times]) => { + const sorted = [...times].sort((a, b) => a - b); + return [ + name, + { + n: times.length, + p50: sorted[Math.ceil(times.length * 0.5) - 1], + p95: sorted[Math.ceil(times.length * 0.95) - 1], + }, + ]; + }), + ); + const result = { + catalogTools: tools.filter((t) => !t.static).length, + samples, + hashes, + summary, + phases, + }; + writeFileSync( + process.env.BENCH_OUTPUT ?? join(tmpdir(), "executor-mcp-session-bench.json"), + JSON.stringify(result, null, 2), + ); + console.log(JSON.stringify({ catalogTools: result.catalogTools, hashes, summary }, null, 2)); +} finally { + await db.close(); + rmSync(dataDir, { recursive: true, force: true }); +} +// The plugin transport pool has process timers. All benchmark-owned handles +// are closed above; finish the standalone measurement process. +process.exit(0); diff --git a/apps/host-selfhost/scripts/bench-tool-usage.ts b/apps/host-selfhost/scripts/bench-tool-usage.ts new file mode 100644 index 0000000000..e5e25b78d5 --- /dev/null +++ b/apps/host-selfhost/scripts/bench-tool-usage.ts @@ -0,0 +1,100 @@ +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createClient } from "@libsql/client"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { observeToolUsageServer } from "../src/mcp/tool-usage"; +import { makeToolUsageRecorder } from "../src/mcp/tool-usage-store"; + +const dir = mkdtempSync(join(tmpdir(), "executor-usage-bench-")); +const samples = Number(process.env.BENCH_SAMPLES ?? 5000); +const quantile = (values: number[], q: number) => + [...values].sort((a, b) => a - b)[Math.ceil(values.length * q) - 1]!; +const reports = []; +try { + for (const bytes of [1024, 27 * 1024, 1024 * 1024]) { + const phases = []; + for (const enabled of [false, true]) { + const db = createClient({ url: `file:${join(dir, `${bytes}-${enabled}.db`)}` }); + await db.execute("PRAGMA journal_mode = WAL"); + await db.execute("PRAGMA synchronous = NORMAL"); + const recorder = makeToolUsageRecorder(db); + await recorder.ready; + const server = new McpServer({ name: "usage-bench", version: "1" }); + const result = { content: [{ type: "text" as const, text: "x".repeat(bytes) }] }; + server.registerTool("search", {}, () => result); + if (enabled) + observeToolUsageServer( + server, + recorder.memberHash("bench-org", "bench-member")!, + recorder.record, + ); + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair(); + // InMemoryTransport normally skips wire serialization. Include the same + // JSON serialization cost in both phases, as HTTP transport does. + const send = serverTransport.send.bind(serverTransport); + serverTransport.send = async (message, options) => { + JSON.stringify(message); + await send(message, options); + }; + const client = new Client({ name: "benchmark", version: "1" }); + await server.connect(serverTransport); + await client.connect(clientTransport); + for (let i = 0; i < 1000; i++) { + await client.callTool({ name: "search" }); + if ((i + 1) % 256 === 0) await recorder.flush(); + } + await recorder.flush(); + const values = []; + let flushMs = 0; + for (let i = 0; i < samples; i++) { + const start = performance.now(); + await client.callTool({ name: "search" }); + values.push(performance.now() - start); + if (enabled && (i + 1) % 256 === 0) { + const flushStart = performance.now(); + await recorder.flush(); + flushMs += performance.now() - flushStart; + } + } + const flushStart = performance.now(); + await recorder.close(); + flushMs += performance.now() - flushStart; + await client.close(); + await server.close(); + phases.push({ + enabled, + samples, + p50Ms: quantile(values, 0.5), + p95Ms: quantile(values, 0.95), + flushMsPerCall: flushMs / samples, + lost: (await db.execute("SELECT dropped_events FROM executor_tool_usage_state")).rows[0]! + .dropped_events, + }); + db.close(); + } + reports.push({ + responseContentBytes: bytes, + phases, + overheadP50Ms: phases[1]!.p50Ms - phases[0]!.p50Ms, + overheadP95Ms: phases[1]!.p95Ms - phases[0]!.p95Ms, + }); + } + console.log( + JSON.stringify( + { + runtime: Bun.version, + samples, + method: + "warm SDK in-memory MCP calls plus HTTP-equivalent JSON serialization; nearest-rank percentiles; WAL SQLite; flush each 256 calls outside request timing", + reports, + }, + null, + 2, + ), + ); +} finally { + rmSync(dir, { recursive: true, force: true }); +} diff --git a/apps/host-selfhost/scripts/tool-usage-summary.ts b/apps/host-selfhost/scripts/tool-usage-summary.ts new file mode 100644 index 0000000000..056c6a5286 --- /dev/null +++ b/apps/host-selfhost/scripts/tool-usage-summary.ts @@ -0,0 +1,74 @@ +import { Database } from "bun:sqlite"; +import { resolve } from "node:path"; +import { TOOL_USAGE_MAX_EVENTS, TOOL_USAGE_RETENTION_MS } from "../src/mcp/tool-usage-store"; +import { + toolUsageSummaryQuery, + integrationUsageSummaryQuery, + type ToolUsageWindow, +} from "../src/mcp/tool-usage-summary"; + +const args = process.argv.slice(2); +const option = (name: string, fallback?: string): string | undefined => { + const index = args.indexOf(name); + if (index < 0) return fallback; + if (!args[index + 1] || args[index + 1]!.startsWith("--")) + throw new Error(`Missing ${name} value`); + return args[index + 1]; +}; +if (args.includes("--help")) { + console.log( + "bun run apps/host-selfhost/scripts/tool-usage-summary.ts --db PATH [--from ISO] [--to ISO] [--traffic agent|benchmark|monitor|all] [--limit 20]", + ); + process.exit(0); +} +const dbPath = option("--db"); +if (!dbPath) throw new Error("--db PATH is required"); +const toMs = Date.parse(option("--to", new Date().toISOString())!); +const fromMs = Date.parse( + option("--from", new Date(toMs - TOOL_USAGE_RETENTION_MS).toISOString())!, +); +const traffic = option("--traffic", "agent"); +const limit = Number(option("--limit", "20")); +if (!Number.isFinite(fromMs) || !Number.isFinite(toMs) || fromMs >= toMs) + throw new Error("Invalid time window"); +if (!["agent", "benchmark", "monitor", "all"].includes(traffic!)) + throw new Error("Invalid traffic class"); +if (!Number.isInteger(limit) || limit < 1 || limit > 1000) + throw new Error("--limit must be 1..1000"); +const window = { + fromMs, + toMs, + trafficClass: traffic as ToolUsageWindow["trafficClass"], + limit, +}; +const query = toolUsageSummaryQuery(window); +const integrationQuery = integrationUsageSummaryQuery(window); +const db = new Database(resolve(dbPath), { readonly: true }); +try { + console.log( + JSON.stringify( + { + from: new Date(fromMs).toISOString(), + to: new Date(toMs).toISOString(), + traffic, + retentionDays: TOOL_USAGE_RETENTION_MS / (24 * 60 * 60 * 1000), + maxEvents: TOOL_USAGE_MAX_EVENTS, + coverage: db + .query(`SELECT COUNT(*) AS retained_calls, MIN(timestamp_ms) AS oldest_timestamp_ms, + MAX(timestamp_ms) AS newest_timestamp_ms FROM executor_tool_usage`) + .get(), + losses: db + .query( + "SELECT dropped_events, write_failures FROM executor_tool_usage_state WHERE id = 1", + ) + .get(), + tools: db.query(query.sql).all(fromMs, toMs, traffic!, traffic!, limit), + integrations: db.query(integrationQuery.sql).all(fromMs, toMs, traffic!, traffic!, limit), + }, + null, + 2, + ), + ); +} finally { + db.close(); +} diff --git a/apps/host-selfhost/scripts/tool-usage.md b/apps/host-selfhost/scripts/tool-usage.md new file mode 100644 index 0000000000..175353b2e5 --- /dev/null +++ b/apps/host-selfhost/scripts/tool-usage.md @@ -0,0 +1,92 @@ +# Tool usage metrics + +Self-host records authenticated `search`, `invoke`, `integrations`, `skills`, +and `execute` MCP calls. Every repeated call counts, including validation +errors and denied invokes. An `execute` call records one row per distinct +connected tool the sandbox attempted, capped at 32 targets in first-call order +(or one row with no target when it attempted none). Each target's status is the +most severe outcome of its repeated calls: `blocked`, then `error`, then `ok`. +A script or transport failure sets every attributed row to `error`. Successful +calls before a script failure retain their target. Duration and response bytes +are those of the whole execution. Older engines with only successful `toolPaths` +retain their existing attribution and execution status. Approval pauses defer +recording until the logical execution completes. Model and browser resumes +remain attributed to the original `execute`; repeated pauses, concurrent +resumes, and replayed results do not add execution rows. This does not instrument other hosts, +unauthenticated requests, or calls rejected before MCP tool dispatch. + +Each event has the call-start timestamp in milliseconds, an HMAC-SHA256 member +pseudonym scoped to its organization, MCP tool name, canonical invoke target, +integration slug, traffic class, status, elapsed milliseconds, and response +bytes. Discovery calls have no target or integration. Malformed invoke targets +have neither. The address must contain identifiers and be at most 512 bytes. +The local pseudonymization salt persists in the metrics state table. Member +IDs, email addresses, session IDs, headers, arguments, results, and error text +are never stored. Execute telemetry reads only validated target identifiers +and the fixed `ok`, `error`, and `blocked` outcomes. Code and logs are never +stored. Keep the database within its existing private access boundary. + +Response bytes count the complete serialized JSON-RPC response in UTF-8, before +HTTP/SSE framing or compression. For paused executions, bytes count only the +terminal response. Duration starts at the original `execute` and ends after +the terminal transport send, including approval wait time. Timestamp, member +pseudonym, and traffic class remain those of the original call. +The fixed hidden-tool refusal is counted as `blocked`; this includes unknown +addresses because the existing response deliberately combines both cases. +Terminal transport failures and abandoned active calls count as `error` +(zero response bytes when no response was produced). A paused session closed +without a terminal outcome counts as observation loss. Results and policies +keep their existing behavior. + +The observer uses public SDK transport callbacks. It does not inspect SDK +private fields or change search, schema, catalog refresh, or encrypted secrets. +PRs #17 and #18 and the refresh work for issue #8 can merge in either order. + +The `executor_tool_usage` table shares the existing libSQL client with the +host. The schema migration preserves rows, IDs, the AUTOINCREMENT high-water +value (including deleted IDs), the salt, and loss counters in one write batch. +No second write connection opens. Writes run in batches once per second, +with at most 1,024 queued events. Each MCP transport retains at most 1,024 +active tracked calls, 1,024 paused executions, and 1,024 in-flight resume request +IDs. Each paused entry holds only the original call metadata and a validated +opaque execution ID; the execution ID is never stored in usage rows. +Retention runs at startup, each flush, and hourly while idle: seven days and +at most 100,000 rows. SQLite reuses deleted space; the table reaches a bounded +high-water size. A shutdown drains the queue. A crash can lose up to one second +of queued metrics; a busy or unavailable database can lose more. Queue overflow +and failed batches increment durable loss counters on the next successful +flush. An initialization failure disables metrics for that process and emits a +fixed warning. Never interpret a partial window as a complete historical rank. + +Agents default to traffic class `agent`. Benchmarks and probes can send +`x-executor-traffic-class: benchmark` or `monitor`. Only these fixed values +survive; all other values become `agent`. This is a caller-supplied traffic +label, not an authenticated client identity. + +Read the summary locally with database read permission: + +```sh +bun run apps/host-selfhost/scripts/tool-usage-summary.ts \ + --db /path/to/executor.db \ + --from 2026-10-01T00:00:00Z --to 2026-10-08T00:00:00Z \ + --traffic agent --limit 20 +``` + +The CLI opens SQLite read-only. It emits JSON with top tools and integrations ranked by calls, +status counts, nearest-rank p50/p95 duration, and total response bytes for the +half-open time window. It also reports retention limits, oldest/newest retained +timestamps, and cumulative loss counters. It omits member pseudonyms. The +default window is seven days; `--traffic all` includes probes. Integration ranks combine invoke and execute targets by integration slug to rank adapters. Only retained events can be summarized. + +Measure request overhead locally: + +```sh +BENCH_SAMPLES=5000 bun run apps/host-selfhost/scripts/bench-tool-usage.ts +``` + +The benchmark compares warm real SDK calls with observation disabled/enabled, +using the same response sizes and HTTP-equivalent JSON serialization. It uses +a disposable WAL SQLite database, nearest-rank percentiles, and batches of 256. +Request latency excludes the background flush; amortized flush time is reported +separately. Small timing differences can include JIT and scheduling noise. +It does not measure network latency or production database contention. diff --git a/apps/host-selfhost/src/account/better-auth-account-provider.ts b/apps/host-selfhost/src/account/better-auth-account-provider.ts index 885ed74eee..b17c27f337 100644 --- a/apps/host-selfhost/src/account/better-auth-account-provider.ts +++ b/apps/host-selfhost/src/account/better-auth-account-provider.ts @@ -1,9 +1,10 @@ import { Effect, Layer } from "effect"; import { AccountProvider, type AccountHeaders } from "@executor-js/api/server"; -import { AccountError, AccountUnauthorized } from "@executor-js/api"; +import { AccountError, AccountNoOrganization, AccountUnauthorized } from "@executor-js/api"; import { BetterAuth } from "../auth/better-auth"; +import { findInstanceMembership } from "../auth/membership"; // --------------------------------------------------------------------------- // Self-host AccountProvider — implements the provider-neutral account surface @@ -37,7 +38,8 @@ const orgRole = (slug: string | undefined): "owner" | "admin" | "member" => export const betterAuthAccountProvider: Layer.Layer = Layer.effect(AccountProvider)( Effect.gen(function* () { - const { auth, organizationId, organizationName, organizationSlug } = yield* BetterAuth; + const betterAuth = yield* BetterAuth; + const { auth, organizationId, organizationName, organizationSlug } = betterAuth; const getSession = (headers: AccountHeaders) => Effect.tryPromise({ @@ -45,6 +47,26 @@ export const betterAuthAccountProvider: Layer.Layer new AccountError({ message: "Failed to resolve session" }), }).pipe(Effect.orElseSucceed(() => null)); + // The account plane acts as the calling user with their own headers, so + // Better Auth answers for any user with a live session — including one + // whose membership was removed. Every self-serve route first requires a + // CURRENT member row in the instance org (../auth/membership.ts): no + // session is 401, a session without membership is 403. The member + // management routes below are gated by the organization plugin itself, + // which refuses callers who are not members of the org. + const requireMember = (headers: AccountHeaders) => + Effect.gen(function* () { + const resolved = yield* getSession(headers); + if (!resolved) return yield* new AccountUnauthorized(); + const membership = yield* findInstanceMembership( + betterAuth, + resolved.user.id, + resolved.session.activeOrganizationId ?? organizationId, + ); + if (!membership) return yield* new AccountNoOrganization(); + return resolved; + }); + // Run a Better Auth api call, mapping any rejection to a neutral // AccountError with a stable, user-facing message. const call = (message: string, run: () => Promise) => @@ -53,8 +75,13 @@ export const betterAuthAccountProvider: Layer.Layer Effect.gen(function* () { - const resolved = yield* getSession(headers); - if (!resolved) return yield* new AccountUnauthorized(); + // `me` declares no NoOrganization in its contract: a removed + // member is simply no longer signed in to this instance (401). + const resolved = yield* requireMember(headers).pipe( + Effect.catchTag("AccountNoOrganization", () => + Effect.fail(new AccountUnauthorized()), + ), + ); return { user: { id: resolved.user.id, @@ -71,9 +98,12 @@ export const betterAuthAccountProvider: Layer.Layer - call("Failed to list API keys", () => - auth.api.listApiKeys({ headers: toHeaders(headers) }), - ).pipe( + requireMember(headers).pipe( + Effect.andThen( + call("Failed to list API keys", () => + auth.api.listApiKeys({ headers: toHeaders(headers) }), + ), + ), Effect.map((result) => ({ apiKeys: result.apiKeys.map((key) => ({ id: key.id, @@ -87,9 +117,12 @@ export const betterAuthAccountProvider: Layer.Layer - call("Failed to create API key", () => - auth.api.createApiKey({ body: { name }, headers: toHeaders(headers) }), - ).pipe( + requireMember(headers).pipe( + Effect.andThen( + call("Failed to create API key", () => + auth.api.createApiKey({ body: { name }, headers: toHeaders(headers) }), + ), + ), Effect.map((key) => ({ id: key.id, name: key.name ?? name, @@ -102,9 +135,14 @@ export const betterAuthAccountProvider: Layer.Layer - call("Failed to revoke API key", () => - auth.api.deleteApiKey({ body: { keyId: apiKeyId }, headers: toHeaders(headers) }), - ).pipe(Effect.as({ success: true })), + requireMember(headers).pipe( + Effect.andThen( + call("Failed to revoke API key", () => + auth.api.deleteApiKey({ body: { keyId: apiKeyId }, headers: toHeaders(headers) }), + ), + ), + Effect.as({ success: true }), + ), // Better Auth has no organization-OWNED key concept: every key it // issues belongs to the user who created it. Rather than inventing one diff --git a/apps/host-selfhost/src/auth/bearer-shape.test.ts b/apps/host-selfhost/src/auth/bearer-shape.test.ts new file mode 100644 index 0000000000..46b0d17a04 --- /dev/null +++ b/apps/host-selfhost/src/auth/bearer-shape.test.ts @@ -0,0 +1,43 @@ +import { describe, expect, it } from "@effect/vitest"; + +import { bearerShapeMemoFor, bearerTokenOf, makeBearerShapeMemo } from "./bearer-shape"; + +describe("bearer shape memo", () => { + it("remembers and forgets API key bearers without keeping the token", () => { + const memo = makeBearerShapeMemo(); + expect(memo.isApiKey("key-one")).toBe(false); + memo.rememberApiKey("key-one"); + expect(memo.isApiKey("key-one")).toBe(true); + expect(memo.isApiKey("key-two")).toBe(false); + memo.forget("key-one"); + expect(memo.isApiKey("key-one")).toBe(false); + expect(memo.size()).toBe(0); + }); + + it("drops the oldest entry past its bound", () => { + const memo = makeBearerShapeMemo(); + for (let i = 0; i < 1024; i++) memo.rememberApiKey(`key-${i}`); + expect(memo.size()).toBe(1024); + memo.rememberApiKey("key-new"); + expect(memo.size()).toBe(1024); + expect(memo.isApiKey("key-0")).toBe(false); + expect(memo.isApiKey("key-1")).toBe(true); + expect(memo.isApiKey("key-new")).toBe(true); + }); + + it("is shared per Better Auth instance", () => { + const first = {}; + const second = {}; + bearerShapeMemoFor(first).rememberApiKey("key"); + expect(bearerShapeMemoFor(first).isApiKey("key")).toBe(true); + expect(bearerShapeMemoFor(second).isApiKey("key")).toBe(false); + }); + + it("reads a bearer token case-insensitively and ignores other schemes", () => { + expect(bearerTokenOf(new Headers({ authorization: "Bearer abc" }))).toBe("abc"); + expect(bearerTokenOf(new Headers({ authorization: "bearer abc " }))).toBe("abc"); + expect(bearerTokenOf(new Headers({ authorization: "Basic abc" }))).toBeUndefined(); + expect(bearerTokenOf(new Headers({ authorization: "Bearer " }))).toBeUndefined(); + expect(bearerTokenOf(new Headers())).toBeUndefined(); + }); +}); diff --git a/apps/host-selfhost/src/auth/bearer-shape.ts b/apps/host-selfhost/src/auth/bearer-shape.ts new file mode 100644 index 0000000000..dd6f0e16be --- /dev/null +++ b/apps/host-selfhost/src/auth/bearer-shape.ts @@ -0,0 +1,71 @@ +import { createHash } from "node:crypto"; + +// --------------------------------------------------------------------------- +// Which credential shape a bearer token last resolved as. +// +// Every MCP request authenticates from scratch, and a bearer can be one of +// three things: an mcp() OAuth access token, a bearer session token, or an API +// key presented as a bearer. The resolver tries them in that order, so an API +// key pays for two database lookups that can never match it before the one +// that does. The portal in front of production opens a new session for every +// tool call, so that waste repeats three to four times per call. +// +// This memo records, per Better Auth instance, the digests of tokens that last +// resolved as API keys so the next request can go straight to the API key +// lookup. It is a routing hint, never a credential cache: the key itself is +// still verified against the database on every request, so a revoked, expired +// or disabled key fails exactly as before — and is forgotten on that failure. +// Digests only, bounded, oldest first. +// --------------------------------------------------------------------------- + +const MAX_REMEMBERED = 1024; + +export interface BearerShapeMemo { + /** Whether `token` last resolved as an API key. */ + readonly isApiKey: (token: string) => boolean; + readonly rememberApiKey: (token: string) => void; + readonly forget: (token: string) => void; + readonly size: () => number; +} + +const digest = (token: string): string => createHash("sha256").update(token).digest("hex"); + +export const makeBearerShapeMemo = (): BearerShapeMemo => { + const remembered = new Set(); + return { + isApiKey: (token) => remembered.has(digest(token)), + rememberApiKey: (token) => { + const key = digest(token); + if (remembered.has(key)) return; + if (remembered.size >= MAX_REMEMBERED) { + const oldest = remembered.values().next().value; + if (oldest !== undefined) remembered.delete(oldest); + } + remembered.add(key); + }, + forget: (token) => { + remembered.delete(digest(token)); + }, + size: () => remembered.size, + }; +}; + +const memos = new WeakMap(); + +/** The memo for one Better Auth instance (the MCP auth seam and the identity + * seam share it, so a key learned by either is a hint for both). */ +export const bearerShapeMemoFor = (auth: object): BearerShapeMemo => { + const existing = memos.get(auth); + if (existing) return existing; + const created = makeBearerShapeMemo(); + memos.set(auth, created); + return created; +}; + +export const bearerTokenOf = (headers: Headers): string | undefined => { + const authorization = headers.get("authorization"); + if (!authorization) return undefined; + return authorization.toLowerCase().startsWith("bearer ") + ? authorization.slice(7).trim() || undefined + : undefined; +}; diff --git a/apps/host-selfhost/src/auth/identity.ts b/apps/host-selfhost/src/auth/identity.ts index a637b12a6a..785b8cdd90 100644 --- a/apps/host-selfhost/src/auth/identity.ts +++ b/apps/host-selfhost/src/auth/identity.ts @@ -1,9 +1,10 @@ import { Effect, Layer } from "effect"; -import { IdentityProvider, Unauthorized } from "@executor-js/api/server"; +import { IdentityProvider, NoOrganization, Unauthorized } from "@executor-js/api/server"; -import { isPrivileged } from "../admin/require-admin"; +import { bearerShapeMemoFor, bearerTokenOf } from "./bearer-shape"; import { BetterAuth, type BetterAuthHandle } from "./better-auth"; +import { findInstanceMembership, type InstanceOrgRole } from "./membership"; // --------------------------------------------------------------------------- // The self-host identity seam — the production implementation of the shared @@ -20,34 +21,23 @@ import { BetterAuth, type BetterAuthHandle } from "./better-auth"; // identities tests inject live in `src/testing/test-app.ts`. // --------------------------------------------------------------------------- -const bearerToken = (headers: Headers): string | undefined => { - const authorization = headers.get("authorization"); - if (!authorization) return undefined; - return authorization.toLowerCase().startsWith("bearer ") - ? authorization.slice(7).trim() || undefined - : undefined; -}; - /** - * Resolve workspace-write authority from the caller's current membership in - * the self-host instance organization. Both ordinary API/MCP requests and the - * browser-decision adapter use this exact lookup so a role change takes effect - * at the mutation decision, without trusting the global Better Auth user role. - * Lookup failures fail closed to member authority. + * Require the caller's CURRENT membership in the self-host instance + * organization and resolve workspace-write authority from it. Ordinary + * API/MCP requests and the browser-decision adapter use this exact lookup, so + * removing a member denies their next request and a role change takes effect + * at the next mutation decision, without trusting the global Better Auth user + * role. A missing row — and a lookup failure — fails `NoOrganization` (403). */ -export const resolveSelfHostOrgRole = ( +export const requireInstanceMembership = ( betterAuth: BetterAuthHandle, - headers: Headers | Record, + userId: string, organizationId: string, -): Effect.Effect<"admin" | "member"> => - Effect.tryPromise(() => - betterAuth.auth.api.getActiveMemberRole({ - headers, - query: { organizationId }, - }), - ).pipe( - Effect.orElseSucceed(() => null), - Effect.map((membership) => (membership && isPrivileged(membership.role) ? "admin" : "member")), +): Effect.Effect => + findInstanceMembership(betterAuth, userId, organizationId).pipe( + Effect.flatMap((membership) => + membership ? Effect.succeed(membership.role) : Effect.fail(new NoOrganization()), + ), ); // --------------------------------------------------------------------------- @@ -59,6 +49,9 @@ export const resolveSelfHostOrgRole = ( // resolution fails we retry with the Bearer value as x-api-key, which (with // enableSessionForAPIKeys) mints the owner's session. This is what lets a // generated API key authenticate the API + MCP endpoint as a Bearer token. +// A bearer that last resolved this way is remembered (./bearer-shape) and +// tried as an API key first next time, skipping the session lookup that +// cannot match it; the key is still verified on every request. // Single-org instance, so organizationName is the boot-cached org name. // --------------------------------------------------------------------------- @@ -67,44 +60,62 @@ export const betterAuthIdentityLayer: Layer.Layer>; + const sessionFor = (headers: Headers): Effect.Effect => + Effect.promise(() => auth.api.getSession({ headers })).pipe( + Effect.withSpan("selfhost.identity.session"), + ); + const apiKeySessionFor = (token: string): Effect.Effect => + Effect.tryPromise({ + try: () => auth.api.getSession({ headers: { "x-api-key": token } }), + catch: () => "api-key session lookup failed", + }).pipe( + Effect.orElseSucceed(() => null), + Effect.tap((resolved) => + Effect.sync(() => { + if (resolved) bearerShapes.rememberApiKey(token); + else bearerShapes.forget(token); + }), + ), + Effect.withSpan("selfhost.identity.api_key_session"), + ); + return IdentityProvider.of({ authenticate: (request) => Effect.gen(function* () { - let resolved = yield* Effect.promise(() => - auth.api.getSession({ headers: request.headers }), - ); - // The credential shape that resolved the session — the SAME headers - // are what the membership-role lookup below must present. - let sessionHeaders: Headers | Record = request.headers; - if (!resolved) { - const token = bearerToken(request.headers); - if (token) { - const apiKeyHeaders = { "x-api-key": token }; - resolved = yield* Effect.tryPromise({ - try: () => auth.api.getSession({ headers: apiKeyHeaders }), - catch: () => "api-key session lookup failed", - }).pipe(Effect.orElseSucceed(() => null)); - sessionHeaders = apiKeyHeaders; - } - } + const token = bearerTokenOf(request.headers); + // A bearer that last resolved as an API key goes straight to the + // key lookup. Should the key no longer resolve, the memo forgets + // it and the full order below runs unchanged. + let resolved: Resolved = + token !== undefined && bearerShapes.isApiKey(token) + ? yield* apiKeySessionFor(token) + : null; + if (!resolved) resolved = yield* sessionFor(request.headers); + if (!resolved && token !== undefined) resolved = yield* apiKeySessionFor(token); // No session resolved from any credential shape -> unauthenticated. // The middleware's failure strategy renders this as a 401. if (!resolved) return yield* new Unauthorized(); // Single-org instance: every authenticated user belongs to the one // seeded org. Cookie/bearer-session logins are pinned to it by the // session hook; API-key-minted sessions carry no active org, so we - // default to the seeded org rather than rejecting with NoOrganization. + // default to the seeded org. Whether the user STILL belongs to it + // is decided below, on every request. const resolvedOrganizationId = resolved.session.activeOrganizationId ?? organizationId; - // The workspace role, resolved against the INSTANCE org exactly as - // the admin gate does (require-admin.ts): the explicit - // `organizationId` query keeps a caller-controlled active org from - // answering for an org they own elsewhere. FAIL CLOSED to "member" - // — an infra fault demotes rather than escalates. - const orgRole = yield* resolveSelfHostOrgRole( + // The credential names a user; the member row names a member. A + // user with no row in the INSTANCE org (removed, or never added) is + // refused with NoOrganization (403) — a removed member loses API + // and MCP access on their next request, whichever credential they + // hold. The workspace role comes from the same row, resolved + // against the instance org exactly as the admin gate does + // (require-admin.ts). Lookup failures fail closed to a refusal. + const orgRole = yield* requireInstanceMembership( betterAuth, - sessionHeaders, + resolved.user.id, resolvedOrganizationId, - ); + ).pipe(Effect.withSpan("selfhost.identity.org_role")); return { kind: "member" as const, accountId: resolved.user.id, diff --git a/apps/host-selfhost/src/auth/member-revocation.test.ts b/apps/host-selfhost/src/auth/member-revocation.test.ts new file mode 100644 index 0000000000..f06184aa4e --- /dev/null +++ b/apps/host-selfhost/src/auth/member-revocation.test.ts @@ -0,0 +1,391 @@ +// Membership is required on every authenticated request. Removing a member +// from the instance organization denies their very next request on the API +// plane, the account plane, the browser-approval plane, an OPEN MCP session +// and a NEW MCP session — for every credential shape self-host accepts: API +// key bearer, bearer session token, session cookie, and mcp() OAuth access +// token. Re-adding the member restores access on the next request, and a role +// change (admin -> member) takes effect on the next request. +// +// The first test is the release gate written by the security review of PR #21 +// (carlsonhouse-mcp #39), kept verbatim in its six denial assertions. +import { mkdtempSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { afterAll, expect, test } from "@effect/vitest"; + +import { mintInviteCode } from "../testing/mint-invite"; + +process.env.EXECUTOR_DATA_DIR = mkdtempSync(join(tmpdir(), "eh-member-revocation-")); +process.env.BETTER_AUTH_SECRET = "member-revocation-secret-0123456789-abcdefghij-klmnop"; +process.env.EXECUTOR_BOOTSTRAP_ADMIN_EMAIL = "admin@revocation.test"; +process.env.EXECUTOR_BOOTSTRAP_ADMIN_PASSWORD = "admin-pass-123456"; + +// Built from `makeSelfHostApp` (not `makeSelfHostApiHandler`) so the test can +// re-add a removed member through Better Auth's server-only `addMember`. +const { makeSelfHostApp } = await import("../app"); +const app = await makeSelfHostApp(); +const web = app.toWebHandler(); +const handler = web.handler; +afterAll(async () => { + await web.dispose(); + await app.closeDb(); +}); + +const BASE = "http://localhost:4788"; +const DENIED = [401, 403]; + +interface Identity { + readonly token: string; + readonly cookie: string; +} + +const identityFrom = (res: Response): Identity => ({ + token: res.headers.get("set-auth-token") ?? "", + cookie: res.headers.get("set-cookie")?.split(";", 1)[0] ?? "", +}); + +const signUp = async (email: string): Promise => { + const inviteCode = await mintInviteCode(handler); + const res = await handler( + new Request(`${BASE}/api/auth/sign-up/email`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ email, password: "password-12345678", name: email, inviteCode }), + }), + ); + expect(res.status).toBe(200); + return identityFrom(res); +}; + +const signInBootstrap = async (): Promise => { + const res = await handler( + new Request(`${BASE}/api/auth/sign-in/email`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + email: process.env.EXECUTOR_BOOTSTRAP_ADMIN_EMAIL, + password: process.env.EXECUTOR_BOOTSTRAP_ADMIN_PASSWORD, + }), + }), + ); + expect(res.status).toBe(200); + return identityFrom(res); +}; + +const withAuth = (token: string, path: string, init?: RequestInit) => + handler( + new Request(`${BASE}${path}`, { + ...init, + headers: { + ...Object.fromEntries(new Headers(init?.headers)), + authorization: `Bearer ${token}`, + }, + }), + ); + +const withCookie = (cookie: string, path: string, init?: RequestInit) => + handler( + new Request(`${BASE}${path}`, { + ...init, + headers: { ...Object.fromEntries(new Headers(init?.headers)), cookie }, + }), + ); + +const status = async (pending: Response | Promise): Promise => { + const res = await pending; + await res.text(); + return res.status; +}; + +/** Identity-seam probe on the API plane (execution-stack middleware). */ +const apiProbe = (token: string) => status(withAuth(token, "/api/policies")); + +/** Org-write gate: POST /api/policies with owner=org needs orgRole admin. */ +const orgWriteProbe = (token: string, pattern: string) => + status( + withAuth(token, "/api/policies", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ owner: "org", pattern, action: "block" }), + }), + ); + +const createKey = async (token: string, name: string) => { + const res = await withAuth(token, "/api/account/api-keys", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ name }), + }); + expect(res.status).toBe(200); + return (await res.json()) as { id: string; value: string }; +}; + +const mcp = (headers: Record, body: unknown, sessionId?: string) => + handler( + new Request(`${BASE}/mcp`, { + method: "POST", + headers: { + ...headers, + "content-type": "application/json", + accept: "application/json, text/event-stream", + ...(sessionId ? { "mcp-session-id": sessionId } : {}), + }, + body: JSON.stringify(body), + }), + ); + +const bearer = (token: string): Record => ({ authorization: `Bearer ${token}` }); + +const initBody = { + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2025-06-18", + capabilities: {}, + clientInfo: { name: "member-revocation", version: "1" }, + }, +}; + +const initSession = async (headers: Record): Promise => { + const res = await mcp(headers, initBody); + expect(res.status).toBe(200); + const sessionId = res.headers.get("mcp-session-id") ?? ""; + expect(sessionId).not.toBe(""); + await res.text(); + await status(mcp(headers, { jsonrpc: "2.0", method: "notifications/initialized" }, sessionId)); + return sessionId; +}; + +const listTools = (headers: Record, sessionId: string) => + status(mcp(headers, { jsonrpc: "2.0", id: 2, method: "tools/list" }, sessionId)); + +const newSession = (headers: Record) => status(mcp(headers, initBody)); + +const members = async (adminToken: string) => { + const res = await withAuth(adminToken, "/api/account/members"); + expect(res.status).toBe(200); + return (await res.json()) as { + members: ReadonlyArray<{ id: string; userId: string; email: string; role: string }>; + }; +}; + +const memberRow = async (adminToken: string, email: string) => { + const row = (await members(adminToken)).members.find((m) => m.email === email); + expect(row).toBeDefined(); + return row!; +}; + +const removeMember = async (adminToken: string, membershipId: string) => { + const res = await withAuth( + adminToken, + `/api/account/members/${encodeURIComponent(membershipId)}`, + { + method: "DELETE", + }, + ); + expect(await status(res)).toBe(200); +}; + +const setRole = async (adminToken: string, membershipId: string, roleSlug: "admin" | "member") => { + const res = await withAuth( + adminToken, + `/api/account/members/${encodeURIComponent(membershipId)}/role`, + { + method: "PATCH", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ roleSlug }), + }, + ); + expect(await status(res)).toBe(200); +}; + +/** Re-add a removed user through Better Auth's server-only route (no session). */ +const reAddMember = async (userId: string) => { + await app.betterAuth.auth.api.addMember({ + body: { userId, organizationId: app.betterAuth.organizationId, role: "member" }, + }); +}; + +const b64url = (buf: Uint8Array): string => + btoa(String.fromCharCode(...buf)) + .replaceAll("+", "-") + .replaceAll("/", "_") + .replaceAll("=", ""); + +const json = async (res: Response) => (await res.json()) as Record; + +/** A real mcp() OAuth access token: register -> authorize -> consent -> token. */ +const oauthToken = async (cookie: string): Promise => { + const reg = await handler( + new Request(`${BASE}/api/auth/mcp/register`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + client_name: "member-revocation", + redirect_uris: ["http://localhost:9999/callback"], + token_endpoint_auth_method: "none", + grant_types: ["authorization_code"], + response_types: ["code"], + }), + }), + ); + expect([200, 201]).toContain(reg.status); + const clientId = String((await json(reg)).client_id); + const verifier = b64url(crypto.getRandomValues(new Uint8Array(32))); + const codeChallenge = b64url( + new Uint8Array(await crypto.subtle.digest("SHA-256", new TextEncoder().encode(verifier))), + ); + const authorizeUrl = new URL(`${BASE}/api/auth/mcp/authorize`); + authorizeUrl.search = new URLSearchParams({ + response_type: "code", + client_id: clientId, + redirect_uri: "http://localhost:9999/callback", + code_challenge: codeChallenge, + code_challenge_method: "S256", + scope: "openid", + }).toString(); + const authorize = await handler( + new Request(authorizeUrl, { headers: { cookie }, redirect: "manual" }), + ); + expect(authorize.status).toBe(302); + const consentRedirect = new URL(authorize.headers.get("location") ?? "", BASE); + const consentCode = consentRedirect.searchParams.get("consent_code") ?? ""; + expect(consentCode).not.toBe(""); + const consent = await handler( + new Request(`${BASE}/api/auth/oauth2/consent`, { + method: "POST", + headers: { "content-type": "application/json", cookie }, + body: JSON.stringify({ accept: true, consent_code: consentCode }), + }), + ); + expect(consent.status).toBe(200); + const code = + new URL(String((await json(consent)).redirectURI ?? "")).searchParams.get("code") ?? ""; + const token = await handler( + new Request(`${BASE}/api/auth/mcp/token`, { + method: "POST", + headers: { "content-type": "application/x-www-form-urlencoded" }, + body: new URLSearchParams({ + grant_type: "authorization_code", + code, + redirect_uri: "http://localhost:9999/callback", + client_id: clientId, + code_verifier: verifier, + }).toString(), + }), + ); + expect(token.status).toBe(200); + return String((await json(token)).access_token); +}; + +// --------------------------------------------------------------------------- + +test("release gate: removed members lose API and existing/new MCP access immediately", async () => { + const admin = await signInBootstrap(); + const email = "removed-release-gate@revocation.test"; + const member = await signUp(email); + const key = await createKey(member.token, "release gate"); + const row = await memberRow(admin.token, email); + const credentials = [ + { label: "apiKey", token: key.value, sid: await initSession(bearer(key.value)) }, + { label: "session", token: member.token, sid: await initSession(bearer(member.token)) }, + ]; + for (const credential of credentials) { + expect(await apiProbe(credential.token)).toBe(200); + expect(await listTools(bearer(credential.token), credential.sid)).toBe(200); + } + await removeMember(admin.token, row.id); + for (const credential of credentials) { + const apiStatus = await apiProbe(credential.token); + const existingStatus = await listTools(bearer(credential.token), credential.sid); + const freshStatus = await newSession(bearer(credential.token)); + expect.soft(DENIED, `${credential.label}: API access after removal`).toContain(apiStatus); + expect + .soft(DENIED, `${credential.label}: existing MCP access after removal`) + .toContain(existingStatus); + expect.soft(DENIED, `${credential.label}: new MCP access after removal`).toContain(freshStatus); + } + // Re-adding the member restores access on the next request, with the SAME + // credentials: nothing was revoked, membership alone decides. + await reAddMember(row.userId); + for (const credential of credentials) { + expect(await apiProbe(credential.token)).toBe(200); + expect(await newSession(bearer(credential.token))).toBe(200); + } +}); + +test("session cookie: removal denies the API, account, approval and MCP planes; re-add restores", async () => { + const admin = await signInBootstrap(); + const email = "cookie@revocation.test"; + const member = await signUp(email); + const row = await memberRow(admin.token, email); + const headers = { cookie: member.cookie }; + const sid = await initSession(headers); + + expect(await status(withCookie(member.cookie, "/api/policies"))).toBe(200); + expect(await status(withCookie(member.cookie, "/api/account/me"))).toBe(200); + expect(await status(withCookie(member.cookie, "/api/account/api-keys"))).toBe(200); + expect(await status(withCookie(member.cookie, "/api/mcp-sessions/none/paused"))).toBe(404); + expect(await listTools(headers, sid)).toBe(200); + + await removeMember(admin.token, row.id); + + expect(DENIED).toContain(await status(withCookie(member.cookie, "/api/policies"))); + expect(await status(withCookie(member.cookie, "/api/account/me"))).toBe(401); + expect(await status(withCookie(member.cookie, "/api/account/api-keys"))).toBe(403); + expect( + await status( + withCookie(member.cookie, "/api/account/api-keys", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ name: "after removal" }), + }), + ), + ).toBe(403); + expect(await status(withCookie(member.cookie, "/api/mcp-sessions/none/paused"))).toBe(403); + expect(DENIED).toContain(await listTools(headers, sid)); + expect(DENIED).toContain(await newSession(headers)); + + await reAddMember(row.userId); + expect(await status(withCookie(member.cookie, "/api/policies"))).toBe(200); + expect(await status(withCookie(member.cookie, "/api/account/me"))).toBe(200); + expect(await newSession(headers)).toBe(200); +}); + +test("OAuth bearer: removal denies the open and new MCP sessions; re-add restores", async () => { + const admin = await signInBootstrap(); + const email = "oauth@revocation.test"; + const member = await signUp(email); + const row = await memberRow(admin.token, email); + const headers = bearer(await oauthToken(member.cookie)); + const sid = await initSession(headers); + expect(await listTools(headers, sid)).toBe(200); + + await removeMember(admin.token, row.id); + expect(DENIED).toContain(await listTools(headers, sid)); + expect(DENIED).toContain(await newSession(headers)); + + await reAddMember(row.userId); + expect(await newSession(headers)).toBe(200); +}); + +test("role change admin -> member takes effect on the next request for every credential", async () => { + const admin = await signInBootstrap(); + const email = "downgrade@revocation.test"; + const member = await signUp(email); + const key = await createKey(member.token, "downgrade"); + const row = await memberRow(admin.token, email); + + expect(await orgWriteProbe(member.token, "downgrade-before.*")).toBe(403); + await setRole(admin.token, row.id, "admin"); + expect(await orgWriteProbe(member.token, "downgrade-session.*")).toBe(200); + expect(await orgWriteProbe(key.value, "downgrade-key.*")).toBe(200); + await setRole(admin.token, row.id, "member"); + expect(await orgWriteProbe(member.token, "downgrade-after-session.*")).toBe(403); + expect(await orgWriteProbe(key.value, "downgrade-after-key.*")).toBe(403); + // Still a member: reads keep working. + expect(await apiProbe(member.token)).toBe(200); + expect(await apiProbe(key.value)).toBe(200); +}); diff --git a/apps/host-selfhost/src/auth/membership.ts b/apps/host-selfhost/src/auth/membership.ts new file mode 100644 index 0000000000..4ee4877c63 --- /dev/null +++ b/apps/host-selfhost/src/auth/membership.ts @@ -0,0 +1,57 @@ +import { Effect } from "effect"; + +import { isPrivileged } from "../admin/require-admin"; +import type { BetterAuthHandle } from "./better-auth"; + +// --------------------------------------------------------------------------- +// The one membership read every authenticated self-host request makes. +// +// A credential (session cookie, bearer session token, API key, mcp() OAuth +// access token) proves WHO the caller is. It does not prove the caller still +// belongs to the instance organization: removing a member deletes the `member` +// row and nothing else, so every session, key and token the removed user holds +// keeps resolving to a user. Membership is therefore read from the row on +// every request, after the credential resolves, and a missing row denies the +// request. There is no cache: a removed member is refused on the very next +// request, and re-adding them restores access on the next request too. +// +// This is the same `(userId, organizationId)` row the organization plugin's +// `getActiveMemberRole` endpoint answers from, read through Better Auth's own +// adapter so the OAuth seam (which holds no session headers) can share it. One +// direct read per request, shared by the API, MCP, account and approval planes. +// --------------------------------------------------------------------------- + +export type InstanceOrgRole = "admin" | "member"; + +export interface InstanceMembership { + /** Workspace-write authority, from the row's role (`owner`/`admin` -> admin). */ + readonly role: InstanceOrgRole; +} + +/** + * The caller's current membership in the instance organization, or `null` + * when there is none. A lookup failure also answers `null`: every caller + * treats `null` as a denial, so an infrastructure fault refuses the request + * rather than admitting it. + */ +export const findInstanceMembership = ( + betterAuth: BetterAuthHandle, + userId: string, + organizationId: string, +): Effect.Effect => + Effect.tryPromise(async () => { + const context = await betterAuth.auth.$context; + return context.adapter.findOne<{ readonly role?: string | null }>({ + model: "member", + where: [ + { field: "userId", value: userId }, + { field: "organizationId", value: organizationId }, + ], + }); + }).pipe( + Effect.orElseSucceed(() => null), + Effect.map((row): InstanceMembership | null => + row ? { role: row.role != null && isPrivileged(row.role) ? "admin" : "member" } : null, + ), + Effect.withSpan("selfhost.identity.membership"), + ); diff --git a/apps/host-selfhost/src/execution.ts b/apps/host-selfhost/src/execution.ts index 8dab577b93..3ddf21a1b9 100644 --- a/apps/host-selfhost/src/execution.ts +++ b/apps/host-selfhost/src/execution.ts @@ -56,6 +56,8 @@ export const SelfHostHostConfig: Layer.Layer = Layer.sync(HostConfig webBaseUrl: config.webBaseUrl, oauthCallbackPath: "/api/oauth/callback", toolsSyncTtlMs: config.toolsSyncTtlMs, + // The server and database outlive requests, so refresh can finish after a read. + toolsSyncGraceMs: 0, onIntegrationChange: (event) => selfHostAnalytics.record( event.kind === "added" ? "integration_added" : "integration_removed", diff --git a/apps/host-selfhost/src/executor-config.test.ts b/apps/host-selfhost/src/executor-config.test.ts index 576b93820c..b99d6c3087 100644 --- a/apps/host-selfhost/src/executor-config.test.ts +++ b/apps/host-selfhost/src/executor-config.test.ts @@ -2,6 +2,9 @@ import { afterEach, beforeEach, expect, test } from "@effect/vitest"; import { loadConfig } from "./config"; import executorConfig from "../executor.config"; +import { Effect } from "effect"; +import { HostConfig } from "@executor-js/api/server"; +import { SelfHostHostConfig } from "./execution"; const ENV_NAME = "EXECUTOR_ALLOW_STDIO_MCP"; const SECRET_ENV_NAME = "EXECUTOR_SECRET_KEY"; @@ -12,6 +15,15 @@ const originalTtl = process.env[TTL_ENV_NAME]; const RATE_LIMIT_ENV_NAME = "EXECUTOR_DISABLE_AUTH_RATE_LIMIT"; const originalRateLimit = process.env[RATE_LIMIT_ENV_NAME]; +test("self-host serves persisted catalogs without waiting for remote refresh", async () => { + const config = await Effect.runPromise( + Effect.gen(function* () { + return yield* HostConfig; + }).pipe(Effect.provide(SelfHostHostConfig)), + ); + expect(config.toolsSyncGraceMs).toBe(0); +}); + beforeEach(() => { process.env[SECRET_ENV_NAME] = originalSecret ?? "executor-config-test-secret"; }); diff --git a/apps/host-selfhost/src/mcp/auth.ts b/apps/host-selfhost/src/mcp/auth.ts index 967d4255d3..a910207a98 100644 --- a/apps/host-selfhost/src/mcp/auth.ts +++ b/apps/host-selfhost/src/mcp/auth.ts @@ -11,8 +11,9 @@ import { type Principal, } from "@executor-js/host-mcp"; -import { isPrivileged } from "../admin/require-admin"; +import { bearerShapeMemoFor, bearerTokenOf } from "../auth/bearer-shape"; import { BetterAuth } from "../auth/better-auth"; +import { findInstanceMembership } from "../auth/membership"; import { MCP_ORIGINAL_PATH_HEADER, mcpResourcePathFromOriginalPath } from "./org-path"; // --------------------------------------------------------------------------- @@ -157,7 +158,8 @@ export const selfHostMcpAuth: Layer.Layer auth.$context); - /** Enrich a bare OAuth `userId` into the full provider-neutral principal. */ + /** Enrich a bare OAuth `userId` into the full provider-neutral principal. + * `null` when the user is gone OR is no longer a member of the instance + * org: an OAuth token outlives a removed membership, so the member row + * is read on every request (../auth/membership.ts) and its absence + * refuses the token — the caller renders the 401 challenge. */ const principalFromUserId = (userId: string): Effect.Effect => Effect.gen(function* () { const user = yield* Effect.promise(() => context.internalAdapter.findUserById(userId)); @@ -209,21 +215,10 @@ export const selfHostMcpAuth: Layer.Layer - context.adapter.findOne<{ readonly role?: string | null }>({ - model: "member", - where: [ - { field: "userId", value: userId }, - { field: "organizationId", value: organizationId }, - ], - }), - ).pipe(Effect.orElseSucceed(() => null)); - const orgRole = - membership?.role != null && isPrivileged(membership.role) - ? ("admin" as const) - : ("member" as const); + // answers the same question against the same table). No row — or a + // lookup failure — refuses rather than admits. + const membership = yield* findInstanceMembership(betterAuth, userId, organizationId); + if (!membership) return null; return { accountId: user.id, // Single-org self-host: OAuth tokens carry no active org, so pin to @@ -236,7 +231,7 @@ export const selfHostMcpAuth: Layer.Layer auth.api.getMcpSession({ headers: request.headers }), - ); + ).pipe(Effect.withSpan("selfhost.auth.oauth_session")); if (!session) return null; // GOTCHA: getMcpSession does NOT validate accessTokenExpiresAt — an // expired token still resolves. Reject it here. @@ -260,6 +255,7 @@ export const selfHostMcpAuth: Layer.Layer => fallback.authenticate(request).pipe( + Effect.withSpan("selfhost.auth.identity"), Effect.map((principal) => (isPlatformPrincipal(principal) ? null : principal)), Effect.catchTags({ Unauthorized: () => Effect.succeed(null), @@ -270,12 +266,19 @@ export const selfHostMcpAuth: Layer.Layer { + const token = bearerTokenOf(request.headers); + return token !== undefined && bearerShapes.isApiKey(token); + }; + const authenticate = (request: Request): Effect.Effect => - (hasBearer(request) + (hasBearer(request) && !isRememberedApiKey(request) ? authenticateOAuthBearer(request).pipe( Effect.flatMap((principal) => principal ? Effect.succeed(principal) : authenticateSession(request), diff --git a/apps/host-selfhost/src/mcp/index.ts b/apps/host-selfhost/src/mcp/index.ts index c8a8683227..1c777b21ce 100644 --- a/apps/host-selfhost/src/mcp/index.ts +++ b/apps/host-selfhost/src/mcp/index.ts @@ -9,7 +9,7 @@ import type { } from "@executor-js/host-mcp"; import { BetterAuth, type BetterAuthHandle } from "../auth/better-auth"; -import { resolveSelfHostOrgRole } from "../auth/identity"; +import { requireInstanceMembership } from "../auth/identity"; import type { SelfHostDbHandle } from "../db/self-host-db"; import type { SelfHostConfig } from "../config"; import { selfHostMcpAuth } from "./auth"; @@ -87,21 +87,27 @@ type BetterAuthSession = NonNullable< const principalFromSession = ( resolved: BetterAuthSession, betterAuth: BetterAuthHandle, - headers: Headers, -): Effect.Effect => { +): Effect.Effect => { const organizationId = resolved.session.activeOrganizationId ?? betterAuth.organizationId; - return resolveSelfHostOrgRole(betterAuth, headers, organizationId).pipe( - Effect.map((orgRole) => ({ - accountId: resolved.user.id, - organizationId, - organizationName: betterAuth.organizationName, - email: resolved.user.email, - name: resolved.user.name ?? null, - avatarUrl: resolved.user.image ?? null, - roles: parseRoles(resolved.user.role ?? null), - orgRoleModel: "organization" as const, - orgRole, - })), + // A session names a user; the member row names a member. A removed member's + // session must not reach a paused execution or record an approval decision. + return requireInstanceMembership(betterAuth, resolved.user.id, organizationId).pipe( + Effect.catchTag("NoOrganization", () => Effect.succeed(null)), + Effect.map((orgRole) => + orgRole === null + ? null + : ({ + accountId: resolved.user.id, + organizationId, + organizationName: betterAuth.organizationName, + email: resolved.user.email, + name: resolved.user.name ?? null, + avatarUrl: resolved.user.image ?? null, + roles: parseRoles(resolved.user.role ?? null), + orgRoleModel: "organization" as const, + orgRole, + } satisfies Principal), + ), ); }; @@ -126,9 +132,8 @@ const makeApprovalHandler = }).pipe(Effect.orElseSucceed(() => null)), ); if (!session) return jsonResponse({ error: "Unauthorized" }, 401); - const principal = await Effect.runPromise( - principalFromSession(session, betterAuth, request.headers), - ); + const principal = await Effect.runPromise(principalFromSession(session, betterAuth)); + if (!principal) return jsonResponse({ error: "Forbidden" }, 403); return ( (await store.handlePausedRequest(request, principal)) ?? diff --git a/apps/host-selfhost/src/mcp/mcp.test.ts b/apps/host-selfhost/src/mcp/mcp.test.ts index 60f07d2988..17fb563e60 100644 --- a/apps/host-selfhost/src/mcp/mcp.test.ts +++ b/apps/host-selfhost/src/mcp/mcp.test.ts @@ -355,3 +355,104 @@ test("a browser approval uses the bootstrap admin's demoted membership at the si }); } }); + +const INITIALIZE = { + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2025-06-18", + capabilities: {}, + clientInfo: { name: "t", version: "1" }, + }, +} as const; + +const streamRequest = (token: string, sessionId: string) => + handler( + new Request(`${BASE}/mcp`, { + headers: { + authorization: `Bearer ${token}`, + accept: "text/event-stream", + "mcp-session-id": sessionId, + }, + }), + ); + +test("a bearer API key authenticates every MCP request on its own and stops at revocation", async () => { + const admin = await signInBootstrap(); + const created = await account(admin.token, "/api/account/api-keys", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ name: "session setup" }), + }); + expect(created.status).toBe(200); + const key = (await created.json()) as { id: string; value: string }; + + // The first request learns that this bearer is an API key; the next ones + // take the direct path and must resolve exactly the same. + await initSession(key.value); + const sessionId = await initSession(key.value); + const listed = await mcp(key.value, { jsonrpc: "2.0", id: 2, method: "tools/list" }, sessionId); + expect(listed.status).toBe(200); + await listed.text(); + + // A bearer that is neither a session nor a key is still refused. + const bogus = await mcp(`${key.value}x`, INITIALIZE); + expect(bogus.status).toBe(401); + await bogus.text(); + + // Revocation takes effect on the very next request, on the open session and + // on a new one alike: the key is verified every time. + const revoked = await account( + admin.token, + `/api/account/api-keys/${encodeURIComponent(key.id)}`, + { + method: "DELETE", + }, + ); + expect(revoked.status).toBe(200); + const onSession = await mcp( + key.value, + { jsonrpc: "2.0", id: 3, method: "tools/list" }, + sessionId, + ); + expect(onSession.status).toBe(401); + await onSession.text(); + const fresh = await mcp(key.value, INITIALIZE); + expect(fresh.status).toBe(401); + await fresh.text(); +}); + +test("GET /mcp is 405 for a session whose client can receive no server-initiated message", async () => { + const admin = await signInBootstrap(); + const sessionId = await initSession(admin.token); + + const stream = await streamRequest(admin.token, sessionId); + expect(stream.status).toBe(405); + expect(stream.headers.get("allow")).toBe("POST, DELETE"); + const body = (await stream.json()) as { error: { code: number } }; + expect(body.error.code).toBe(-32000); + + // The session is unaffected. + const listed = await mcp(admin.token, { jsonrpc: "2.0", id: 2, method: "tools/list" }, sessionId); + expect(listed.status).toBe(200); + await listed.text(); +}); + +test("GET /mcp still opens the stream for a client that declared elicitation", async () => { + const admin = await signInBootstrap(); + const res = await mcp(admin.token, { + ...INITIALIZE, + params: { ...INITIALIZE.params, capabilities: { elicitation: {} } }, + }); + expect(res.status).toBe(200); + const sessionId = res.headers.get("mcp-session-id") ?? ""; + expect(sessionId).not.toBe(""); + await res.text(); + await mcp(admin.token, { jsonrpc: "2.0", method: "notifications/initialized" }, sessionId); + + const stream = await streamRequest(admin.token, sessionId); + expect(stream.status).toBe(200); + expect(stream.headers.get("content-type")).toContain("text/event-stream"); + await stream.body?.cancel(); +}); diff --git a/apps/host-selfhost/src/mcp/session-initialization.test.ts b/apps/host-selfhost/src/mcp/session-initialization.test.ts new file mode 100644 index 0000000000..2e51bb9b2e --- /dev/null +++ b/apps/host-selfhost/src/mcp/session-initialization.test.ts @@ -0,0 +1,136 @@ +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterAll, beforeAll, expect, test } from "@effect/vitest"; +import { Effect, Layer } from "effect"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; +import { makeMcpBuildServer, makeScopedExecutor } from "@executor-js/api/server"; +import { AuthTemplateSlug, ConnectionName, IntegrationSlug, type Executor } from "@executor-js/sdk"; +import type { Principal } from "@executor-js/host-mcp"; +import { createSelfHostDb, SelfHostDb, type SelfHostDbHandle } from "../db/self-host-db"; +import { SelfHostExecutionStackLayer, SelfHostScopedExecutorSeams } from "../execution"; +import type { SelfHostPlugins } from "../plugins"; + +const dataDir = mkdtempSync(join(tmpdir(), "executor-session-init-")); +const originalDataDir = process.env.EXECUTOR_DATA_DIR; +process.env.EXECUTOR_DATA_DIR = dataDir; +let db: SelfHostDbHandle; +let seed: Executor; +const principal: Principal = { + accountId: "alice", + organizationId: "session-test", + organizationName: "Session test", + email: "alice@example.test", + name: "Alice", + avatarUrl: null, + roles: ["user"], + orgRoleModel: "organization", +}; +const spec = (operationId: string) => + JSON.stringify({ + openapi: "3.0.0", + info: { title: "Session fixture", version: "1" }, + servers: [{ url: "https://fixture.example.test" }], + paths: { "/read": { get: { operationId, responses: { "200": { description: "ok" } } } } }, + }); + +beforeAll(async () => { + db = await createSelfHostDb({ path: join(dataDir, "data.db") }); + seed = await Effect.runPromise( + makeScopedExecutor( + principal.accountId, + principal.organizationId, + principal.organizationName, + ).pipe(Effect.provide(SelfHostScopedExecutorSeams), Effect.provideService(SelfHostDb, db)), + ); + await Effect.runPromise( + seed.openapi.addSpec({ spec: { kind: "blob", value: spec("readFirst") }, slug: "fixture" }), + ); + await Effect.runPromise( + seed.connections.create({ + owner: "org", + name: ConnectionName.make("shared"), + integration: IntegrationSlug.make("fixture"), + template: AuthTemplateSlug.make("none"), + value: "", + }), + ); +}); + +afterAll(async () => { + if (seed) await Effect.runPromise(seed.close()); + await db?.close(); + if (originalDataDir === undefined) delete process.env.EXECUTOR_DATA_DIR; + else process.env.EXECUTOR_DATA_DIR = originalDataDir; + rmSync(dataDir, { recursive: true, force: true }); +}); + +const withSession = async (binding: Principal, check: (client: Client) => Promise) => { + const built = await Effect.runPromise( + makeMcpBuildServer( + SelfHostExecutionStackLayer.pipe(Layer.provide(Layer.succeed(SelfHostDb)(db))), + )(binding, { mode: "passthrough", artifactsEnabled: false }), + ); + const [ct, st] = InMemoryTransport.createLinkedPair(); + const client = new Client({ name: "session-test", version: "1" }, { capabilities: {} }); + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: MCP test cleanup must run after client assertion failures + try { + await built.mcpServer.connect(st); + await client.connect(ct); + await check(client); + } finally { + await client.close(); + await built.mcpServer.close(); + await Effect.runPromise(built.engine.shutdown); + await Effect.runPromise(built.executor!.close()); + } +}; + +const searchIds = async (client: Client): Promise => { + const result = await client.callTool({ + name: "search", + arguments: { query: "read", integration: "fixture", limit: 10 }, + }); + expect(result.isError).not.toBe(true); + const output = result.structuredContent as { items: readonly { id: string }[] }; + return output.items.map((item) => item.id); +}; + +test("session key reuse keeps policy revocation, catalog changes, and tenant binding current", async () => { + const first = "tools.fixture.org.shared.read.readFirst"; + await withSession(principal, async (client) => { + expect(await searchIds(client)).toEqual([first]); + const block = await Effect.runPromise( + seed.policies.create({ + owner: "org", + pattern: "fixture.org.shared.read.readFirst", + action: "block", + }), + ); + expect(await searchIds(client)).toEqual([]); + await withSession(principal, async (fresh) => { + expect(await searchIds(fresh)).toEqual([]); + }); + await Effect.runPromise(seed.policies.remove({ owner: "org", id: block.id })); + expect(await searchIds(client)).toEqual([first]); + await Effect.runPromise( + seed.openapi.updateSpec(IntegrationSlug.make("fixture"), { + spec: { kind: "blob", value: spec("readNext") }, + }), + ); + const next = "tools.fixture.org.shared.read.readNext"; + expect(await searchIds(client)).toEqual([next]); + await withSession(principal, async (fresh) => { + expect(await searchIds(fresh)).toEqual([next]); + }); + await withSession({ ...principal, organizationId: "other-tenant" }, async (other) => { + expect(await searchIds(other)).toEqual([]); + }); + await Effect.runPromise(seed.integrations.remove(IntegrationSlug.make("fixture"))); + expect(await searchIds(client)).toEqual([]); + await withSession(principal, async (fresh) => { + expect(await searchIds(fresh)).toEqual([]); + }); + }); +}); diff --git a/apps/host-selfhost/src/mcp/session-store.ts b/apps/host-selfhost/src/mcp/session-store.ts index f9d29b1f05..f0738c3064 100644 --- a/apps/host-selfhost/src/mcp/session-store.ts +++ b/apps/host-selfhost/src/mcp/session-store.ts @@ -1,4 +1,4 @@ -import { Layer } from "effect"; +import { Effect, Layer } from "effect"; import { makeConsoleMcpErrorReporter, makeMcpBuildServer } from "@executor-js/api/server"; import type { McpErrorReporter } from "@executor-js/host-mcp"; @@ -12,6 +12,8 @@ import { selfHostAnalytics } from "../analytics"; import { ErrorCaptureLive } from "../observability"; import { SelfHostDb, type SelfHostDbHandle } from "../db/self-host-db"; import { SelfHostExecutionStackLayer } from "../execution"; +import { makeToolUsageRecorder } from "./tool-usage-store"; +import { observeToolUsageServer } from "./tool-usage"; // --------------------------------------------------------------------------- // Self-host McpSessionStore wiring. The store body (Maps, dispatch, ownership, @@ -37,23 +39,41 @@ export const makeSelfHostMcpSessionStore = ( db: SelfHostDbHandle, webBaseUrl?: string, sessionIdleTtlMs?: number, -): InMemoryMcpSessionStore => - makeInMemoryMcpSessionStore( - makeMcpBuildServer( - SelfHostExecutionStackLayer.pipe(Layer.provide(Layer.succeed(SelfHostDb)(db))), - { - loadAppShellHtml: loadMcpAppsShellHtml, - smokeRenderArtifact, - // Artifact operations on the MCP plane come from an agent's tools. - onArtifactUsage: (action) => - selfHostAnalytics.record(`artifact_${action}`, { via: "agent" }), - }, - ), +): InMemoryMcpSessionStore => { + const usage = makeToolUsageRecorder(db.client); + const build = makeMcpBuildServer( + SelfHostExecutionStackLayer.pipe(Layer.provide(Layer.succeed(SelfHostDb)(db))), + { + loadAppShellHtml: loadMcpAppsShellHtml, + smokeRenderArtifact, + onArtifactUsage: (action) => selfHostAnalytics.record(`artifact_${action}`, { via: "agent" }), + }, + ); + const store = makeInMemoryMcpSessionStore( + (principal, options) => + Effect.promise(() => usage.ready).pipe( + Effect.flatMap(() => build(principal, options)), + Effect.tap(({ mcpServer }) => + Effect.sync(() => { + const memberHash = usage.memberHash(principal.organizationId, principal.accountId); + if (memberHash !== null) + observeToolUsageServer(mcpServer, memberHash, usage.record, usage.drop); + }), + ), + ), { ...(webBaseUrl === undefined ? {} : { webBaseUrl }), ...(sessionIdleTtlMs === undefined ? {} : { sessionIdleTtlMs }), }, ); + return { + ...store, + close: async () => { + await store.close(); + await usage.close(); + }, + }; +}; /** The `McpSessionStore` envelope seam over a freshly built in-process store. */ export const selfHostMcpSessions = inMemoryMcpSessionsLayer; diff --git a/apps/host-selfhost/src/mcp/tool-usage-selfhost.test.ts b/apps/host-selfhost/src/mcp/tool-usage-selfhost.test.ts new file mode 100644 index 0000000000..44ffae7067 --- /dev/null +++ b/apps/host-selfhost/src/mcp/tool-usage-selfhost.test.ts @@ -0,0 +1,528 @@ +import { execFileSync } from "node:child_process"; +import { mkdtempSync, rmSync } from "node:fs"; +import { createServer, type Server } from "node:http"; +import { tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { afterAll, expect, it } from "@effect/vitest"; +import { createClient } from "@libsql/client"; +import { Effect, Schema } from "effect"; +import { AuthTemplateSlug, ConnectionName, IntegrationSlug } from "@executor-js/sdk"; +import { makeScopedExecutor } from "@executor-js/api/server"; +import { createSelfHostDb, SelfHostDb } from "../db/self-host-db"; +import { SelfHostScopedExecutorSeams } from "../execution"; +import type { SelfHostPlugins } from "../plugins"; + +const dir = mkdtempSync(join(tmpdir(), "executor-usage-http-")); +process.env.EXECUTOR_DATA_DIR = dir; +process.env.BETTER_AUTH_SECRET = "usage-test-secret-0123456789-abcdefghij-klmnop"; +process.env.EXECUTOR_BOOTSTRAP_ADMIN_EMAIL = "admin@usage.test"; +process.env.EXECUTOR_BOOTSTRAP_ADMIN_PASSWORD = "admin-pass-123456"; +// The sandbox test calls a loopback OpenAPI upstream through the outbound guard. +const originalAllowLocalNetwork = process.env.EXECUTOR_ALLOW_LOCAL_NETWORK; +process.env.EXECUTOR_ALLOW_LOCAL_NETWORK = "true"; + +const decodeResponse = Schema.decodeUnknownSync( + Schema.Struct({ + result: Schema.optional(Schema.Unknown), + error: Schema.optional(Schema.Unknown), + }), +); +const decodeJson = Schema.decodeUnknownSync(Schema.fromJsonString(Schema.Unknown)); +const decodeSummary = Schema.decodeUnknownSync( + Schema.Struct({ + tools: Schema.Array( + Schema.Struct({ + mcp_tool: Schema.String, + target_tool: Schema.NullOr(Schema.String), + integration_slug: Schema.NullOr(Schema.String), + calls: Schema.Number, + ok: Schema.Number, + blocked: Schema.Number, + error: Schema.Number, + }), + ), + integrations: Schema.Array( + Schema.Struct({ integration_slug: Schema.String, calls: Schema.Number }), + ), + losses: Schema.Struct({ dropped_events: Schema.Number }), + }), +); +const decodeOrganization = Schema.decodeUnknownSync( + Schema.Struct({ organization: Schema.Struct({ id: Schema.String }) }), +); +const decodeExecute = Schema.decodeUnknownSync( + Schema.Struct({ + result: Schema.Struct({ + structuredContent: Schema.Struct({ + status: Schema.String, + executionId: Schema.optional(Schema.String), + approvalUrl: Schema.optional(Schema.String), + toolPaths: Schema.optional(Schema.Array(Schema.String)), + toolCalls: Schema.optional( + Schema.Array( + Schema.Struct({ + path: Schema.String, + status: Schema.Literals(["ok", "error", "blocked"]), + }), + ), + ), + }), + }), + }), +); + +const dbPath = join(dir, "usage.db"); +const BASE = "http://localhost:4788"; + +// Real HTTP results pass through the OpenAPI plugin and sandbox invoker. +let flakyCalls = 0; +const target: Server = createServer((request, response) => { + const failed = request.url === "/fail" || (request.url === "/flaky" && ++flakyCalls % 2 === 0); + response.statusCode = failed ? 400 : 200; + response.setHeader("content-type", "application/json"); + response.end(JSON.stringify({ secret: "upstream-secret" })); +}); +const targetUrl = await new Promise((done) => { + target.listen(0, "127.0.0.1", () => { + const address = target.address(); + done(typeof address === "object" && address ? `http://127.0.0.1:${address.port}` : ""); + }); +}); +afterAll(async () => { + if (originalAllowLocalNetwork === undefined) delete process.env.EXECUTOR_ALLOW_LOCAL_NETWORK; + else process.env.EXECUTOR_ALLOW_LOCAL_NETWORK = originalAllowLocalNetwork; + await new Promise((done) => target.close(() => done())); +}); + +const fixtureSpec = JSON.stringify({ + openapi: "3.0.0", + info: { title: "Usage fixture", version: "1" }, + servers: [{ url: targetUrl }], + paths: { + "/read": { + get: { operationId: "readFirst", responses: { "200": { description: "ok" } } }, + }, + "/fail": { + get: { operationId: "readFailed", responses: { "400": { description: "error" } } }, + }, + "/blocked": { + get: { operationId: "readBlocked", responses: { "200": { description: "ok" } } }, + }, + "/flaky": { + get: { + operationId: "readFlaky", + responses: { "200": { description: "ok" }, "400": { description: "error" } }, + }, + }, + "/other": { + get: { operationId: "readOther", responses: { "200": { description: "ok" } } }, + }, + }, +}); + +/** Register an org-owned connection on its own DB handle; WAL makes it visible to the app. */ +const addOrgIntegration = async ( + organizationId: string, + path = dbPath, + approval = false, +): Promise => { + const seedDb = await createSelfHostDb({ + path, + namespace: "executor_selfhost", + version: "1.0.0", + }); + await Effect.runPromise( + Effect.gen(function* () { + const admin = yield* makeScopedExecutor("seed", organizationId, "Default"); + yield* admin.openapi.addSpec({ + spec: { kind: "blob", value: fixtureSpec }, + slug: "fixture", + baseUrl: "", + }); + yield* admin.policies.create({ + owner: "org", + pattern: "fixture.org.shared.blocked.readBlocked", + action: "block", + }); + if (approval) { + for (const pattern of [ + "fixture.org.shared.other.readOther", + "fixture.org.shared.flaky.readFlaky", + ]) + yield* admin.policies.create({ owner: "org", pattern, action: "require_approval" }); + } + yield* admin.connections.create({ + owner: "org", + name: ConnectionName.make("shared"), + integration: IntegrationSlug.make("fixture"), + template: AuthTemplateSlug.make("none"), + value: "", + }); + }).pipe( + Effect.provide(SelfHostScopedExecutorSeams), + Effect.provideService(SelfHostDb, seedDb), + Effect.scoped, + ), + ); + await seedDb.close(); +}; + +const readSummary = () => { + const output = execFileSync( + "bun", + ["run", resolve("scripts/tool-usage-summary.ts"), "--db", dbPath], + { + encoding: "utf8", + }, + ); + return { output, summary: decodeSummary(decodeJson(output)) }; +}; + +const openSession = async ( + handler: (request: Request) => Promise, + token: string, + query: string, +) => { + const headers = { + authorization: `Bearer ${token}`, + "content-type": "application/json", + accept: "application/json, text/event-stream", + }; + const init = await handler( + new Request(`${BASE}/mcp?${query}`, { + method: "POST", + headers, + body: JSON.stringify({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2025-03-26", + capabilities: {}, + clientInfo: { name: "usage-test", version: "1" }, + }, + }), + }), + ); + expect(init.status).toBe(200); + await init.text(); + const sessionHeaders = { ...headers, "mcp-session-id": init.headers.get("mcp-session-id")! }; + return async (id: number, name: string, args: object) => { + const response = await handler( + new Request(`${BASE}/mcp`, { + method: "POST", + headers: sessionHeaders, + body: JSON.stringify({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name, arguments: args }, + }), + }), + ); + expect(response.status).toBe(200); + return decodeResponse(await response.json()); + }; +}; + +it("attributes real HTTP execution outcomes once across model and browser approval resumes", async () => { + const { makeSelfHostApiHandler } = await import("../app"); + const pauseDbPath = join(dir, "pause.db"); + const app = await makeSelfHostApiHandler({ dbPath: pauseDbPath }); + const approvalStatuses: number[] = []; + const browserSessions: string[] = []; + const replays: unknown[] = []; + const replayOutcomes: unknown[] = []; + const expected: { mcp_tool: string; target_tool: string; status: string }[] = []; + const addExpected = (path: string, status: string) => + expected.push({ mcp_tool: "execute", target_tool: `tools.fixture.org.shared.${path}`, status }); + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: flush telemetry and close the app even when an assertion fails + try { + const login = await app.handler( + new Request(`${BASE}/api/auth/sign-in/email`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ email: "admin@usage.test", password: "admin-pass-123456" }), + }), + ); + expect(login.status).toBe(200); + const token = login.headers.get("set-auth-token")!; + const me = await app.handler( + new Request(`${BASE}/api/account/me`, { + headers: { authorization: `Bearer ${token}` }, + }), + ); + await addOrgIntegration(decodeOrganization(await me.json()).organization.id, pauseDbPath, true); + let id = 2; + for (const mode of ["model", "browser"] as const) { + const call = await openSession(app.handler, token, `artifacts=0&elicitation_mode=${mode}`); + for (const action of ["accept", "decline", "cancel"] as const) { + const paused = decodeExecute( + await call(id++, "execute", { + code: [ + "await tools.search({ query: 'argument-secret' });", + "await tools.fixture.org.shared.read.readFirst({});", + "await tools.fixture.org.shared.other.readOther({});", + "await tools.fixture.org.shared.fail.readFailed({});", + "await tools.fixture.org.shared.blocked.readBlocked({});", + "console.log('log-secret'); return 'code-secret';", + ].join("\n"), + }), + ).result.structuredContent; + expect(paused.status).toBe( + mode === "model" ? "waiting_for_interaction" : "user_approval_required", + ); + expect(paused.toolCalls).toBeUndefined(); + const executionId = paused.executionId!; + if (mode === "browser") { + const sessionId = new URL(paused.approvalUrl!).searchParams.get("mcp_session_id")!; + browserSessions.push(sessionId); + const approval = await app.handler( + new Request(`${BASE}/api/mcp-sessions/${sessionId}/executions/${executionId}/resume`, { + method: "POST", + headers: { authorization: `Bearer ${token}`, "content-type": "application/json" }, + body: JSON.stringify({ action, content: { secret: "approval-secret" } }), + }), + ); + approvalStatuses.push(approval.status); + } + const resumed = decodeExecute(await call(id++, "resume", { executionId, action })).result + .structuredContent; + expect(resumed.status).toBe(action === "accept" ? "completed" : "error"); + expect(resumed.toolCalls).toEqual([ + { path: "fixture.org.shared.read.readFirst", status: "ok" }, + { + path: "fixture.org.shared.other.readOther", + status: action === "accept" ? "ok" : "blocked", + }, + ...(action === "accept" + ? [ + { path: "fixture.org.shared.fail.readFailed", status: "error" }, + { path: "fixture.org.shared.blocked.readBlocked", status: "blocked" }, + ] + : []), + ]); + addExpected("read.readFirst", action === "accept" ? "ok" : "error"); + addExpected("other.readOther", action === "accept" ? "ok" : "error"); + if (action === "accept") { + addExpected("fail.readFailed", "error"); + addExpected("blocked.readBlocked", "blocked"); + } + // Cached resume responses must not count another logical execution. + if (mode === "model") { + replays.push( + decodeExecute(await call(id++, "resume", { executionId, action })).result + .structuredContent, + ); + replayOutcomes.push(resumed); + } + } + } + expect(browserSessions.every(Boolean)).toBe(true); + expect(approvalStatuses).toEqual([200, 200, 200]); + expect(replays).toEqual(replayOutcomes); + const call = await openSession(app.handler, token, "artifacts=0&elicitation_mode=model"); + const initial = decodeExecute( + await call(id++, "execute", { + code: [ + "await tools.fixture.org.shared.read.readFirst({});", + "await tools.fixture.org.shared.other.readOther({});", + "await tools.fixture.org.shared.flaky.readFlaky({});", + "throw new Error('script-secret');", + ].join("\n"), + }), + ).result.structuredContent; + const firstId = initial.executionId!; + const next = decodeExecute( + await call(id++, "resume", { executionId: firstId, action: "accept" }), + ).result.structuredContent; + expect(next.status).toBe("waiting_for_interaction"); + expect(next.executionId).not.toBe(firstId); + expect( + decodeExecute(await call(id++, "resume", { executionId: firstId, action: "accept" })).result + .structuredContent, + ).toEqual(next); + const results = await Promise.all([ + call(id++, "resume", { executionId: next.executionId!, action: "accept" }), + call(id++, "resume", { executionId: next.executionId!, action: "accept" }), + ]); + expect(decodeExecute(results[0]).result.structuredContent.status).toBe("error"); + expect(decodeExecute(results[1]).result.structuredContent).toEqual( + decodeExecute(results[0]).result.structuredContent, + ); + for (const path of ["read.readFirst", "other.readOther", "flaky.readFlaky"]) + addExpected(path, "error"); + + // Native mode takes engine.execute, with no pause or resume boundary. + const inline = await openSession(app.handler, token, "artifacts=0&elicitation_mode=native"); + const final = decodeExecute( + await inline(id++, "execute", { + code: [ + "await tools.fixture.org.shared.read.readFirst({});", + "await tools.fixture.org.shared.fail.readFailed({});", + "await tools.fixture.org.shared.blocked.readBlocked({});", + "console.log('log-secret'); return 'code-secret';", + ].join("\n"), + }), + ).result.structuredContent; + expect(final.status).toBe("completed"); + addExpected("read.readFirst", "ok"); + addExpected("fail.readFailed", "error"); + addExpected("blocked.readBlocked", "blocked"); + } finally { + await app.dispose(); + } + const metricsDb = createClient({ url: `file:${pauseDbPath}` }); + const rows = await metricsDb.execute("SELECT * FROM executor_tool_usage ORDER BY id"); + metricsDb.close(); + expect( + rows.rows.map((row) => ({ + mcp_tool: row.mcp_tool, + target_tool: row.target_tool, + status: row.status, + })), + ).toEqual(expected); + expect( + rows.rows.every((row) => Number(row.response_bytes) > 0 && Number(row.duration_ms) >= 0), + ).toBe(true); + expect(JSON.stringify(rows.rows)).not.toMatch( + /argument-secret|approval-secret|code-secret|upstream-secret|script-secret|log-secret|admin@usage.test|admin-pass|exec_|toolCalls|toolPaths/, + ); +}); + +it("records authenticated HTTP passthrough and sandbox execute calls and exposes a read-only CLI summary", async () => { + const { makeSelfHostApiHandler } = await import("../app"); + const app = await makeSelfHostApiHandler({ dbPath }); + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: always close the full app before reading its SQLite file + try { + const login = await app.handler( + new Request(`${BASE}/api/auth/sign-in/email`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ email: "admin@usage.test", password: "admin-pass-123456" }), + }), + ); + expect(login.status).toBe(200); + const token = login.headers.get("set-auth-token")!; + const me = await app.handler( + new Request(`${BASE}/api/account/me`, { headers: { authorization: `Bearer ${token}` } }), + ); + expect(me.status).toBe(200); + await addOrgIntegration(decodeOrganization(await me.json()).organization.id); + + const call = await openSession(app.handler, token, "mode=passthrough&artifacts=0"); + await call(2, "search", { query: "argument-secret" }); + await call(3, "search", { query: "argument-secret" }); + const denied = await call(4, "invoke", { + tool: "tools.sample.org.test.read", + arguments: { secret: "invoke-secret" }, + }); + expect(JSON.stringify(denied)).toContain("Tool not found or blocked by policy"); + await call(5, "integrations", {}); + await call(6, "skills", {}); + + const execute = await openSession(app.handler, token, "artifacts=0"); + const executed = decodeExecute( + await execute(7, "execute", { + code: [ + "const first = await tools.fixture.org.shared.read.readFirst({});", + "const other = await tools.fixture.org.shared.other.readOther({});", + "await tools.fixture.org.shared.read.readFirst({});", + 'await tools.search({ query: "code-secret" });', + 'return { first, other, note: "code-secret" };', + ].join("\n"), + }), + ); + expect(executed.result.structuredContent.status).toBe("completed"); + expect(executed.result.structuredContent.toolPaths).toEqual([ + "fixture.org.shared.read.readFirst", + "fixture.org.shared.other.readOther", + ]); + expect(executed.result.structuredContent.toolCalls).toEqual([ + { path: "fixture.org.shared.read.readFirst", status: "ok" }, + { path: "fixture.org.shared.other.readOther", status: "ok" }, + ]); + const none = decodeExecute(await execute(8, "execute", { code: "return 6 * 7;" })); + expect(none.result.structuredContent).not.toHaveProperty("toolPaths"); + expect(none.result.structuredContent).not.toHaveProperty("toolCalls"); + const mixed = decodeExecute( + await execute(9, "execute", { + code: [ + "await tools.fixture.org.shared.read.readFirst({});", + "await tools.fixture.org.shared.fail.readFailed({});", + "await tools.fixture.org.shared.blocked.readBlocked({});", + "await tools.fixture.org.shared.flaky.readFlaky({});", + "await tools.fixture.org.shared.flaky.readFlaky({});", + "return 42;", + ].join("\n"), + }), + ); + expect(mixed.result.structuredContent.status).toBe("completed"); + expect(mixed.result.structuredContent.toolPaths).toEqual([ + "fixture.org.shared.read.readFirst", + "fixture.org.shared.flaky.readFlaky", + ]); + expect(mixed.result.structuredContent.toolCalls).toEqual([ + { path: "fixture.org.shared.read.readFirst", status: "ok" }, + { path: "fixture.org.shared.fail.readFailed", status: "error" }, + { path: "fixture.org.shared.blocked.readBlocked", status: "blocked" }, + { path: "fixture.org.shared.flaky.readFlaky", status: "error" }, + ]); + const failed = decodeExecute( + await execute(10, "execute", { + code: "await tools.fixture.org.shared.read.readFirst({}); console.log('log-secret'); throw new Error('script-secret');", + }), + ); + expect(failed.result.structuredContent.status).toBe("error"); + expect(failed.result.structuredContent.toolPaths).toEqual([ + "fixture.org.shared.read.readFirst", + ]); + expect(failed.result.structuredContent.toolCalls).toEqual([ + { path: "fixture.org.shared.read.readFirst", status: "ok" }, + ]); + } finally { + await app.dispose(); + } + const { output, summary } = readSummary(); + expect(summary.tools.find((tool) => tool.mcp_tool === "search")!.calls).toBe(2); + expect(summary.tools.find((tool) => tool.mcp_tool === "invoke")!.blocked).toBe(1); + const executes = summary.tools + .filter((tool) => tool.mcp_tool === "execute") + .map((tool) => [ + tool.target_tool, + tool.integration_slug, + tool.calls, + tool.ok, + tool.error, + tool.blocked, + ]); + expect(executes).toEqual( + expect.arrayContaining([ + ["tools.fixture.org.shared.read.readFirst", "fixture", 3, 2, 1, 0], + ["tools.fixture.org.shared.other.readOther", "fixture", 1, 1, 0, 0], + ["tools.fixture.org.shared.fail.readFailed", "fixture", 1, 0, 1, 0], + ["tools.fixture.org.shared.flaky.readFlaky", "fixture", 1, 0, 1, 0], + ["tools.fixture.org.shared.blocked.readBlocked", "fixture", 1, 0, 0, 1], + [null, null, 1, 1, 0, 0], + ]), + ); + expect(executes).toHaveLength(6); + // Execute targets rank beside the denied passthrough invoke, by integration. + expect(summary.integrations).toEqual([ + { integration_slug: "fixture", calls: 7 }, + { integration_slug: "sample", calls: 1 }, + ]); + expect(summary.tools.reduce((count, tool) => count + tool.calls, 0)).toBe(13); + expect(summary.losses.dropped_events).toBe(0); + expect(output).not.toMatch( + /argument-secret|invoke-secret|code-secret|upstream-secret|admin@usage.test|admin-pass/, + ); + const metricsDb = createClient({ url: `file:${dbPath}` }); + const rows = await metricsDb.execute("SELECT * FROM executor_tool_usage"); + metricsDb.close(); + expect(rows.rows).toHaveLength(13); + expect(JSON.stringify(rows.rows)).not.toMatch( + /argument-secret|invoke-secret|code-secret|upstream-secret|script-secret|log-secret|admin@usage.test|admin-pass|toolCalls|toolPaths/, + ); + rmSync(dir, { recursive: true, force: true }); +}); diff --git a/apps/host-selfhost/src/mcp/tool-usage-store.test.ts b/apps/host-selfhost/src/mcp/tool-usage-store.test.ts new file mode 100644 index 0000000000..ad73e0176a --- /dev/null +++ b/apps/host-selfhost/src/mcp/tool-usage-store.test.ts @@ -0,0 +1,425 @@ +import { afterEach, describe, expect, it } from "@effect/vitest"; +import { createClient, type Client } from "@libsql/client"; +import { + hashUsageMember, + initializeToolUsage, + makeToolUsageRecorder, + TOOL_USAGE_MAX_PENDING, + TOOL_USAGE_RETENTION_MS, + usageInsert, + usageRetention, + type ToolUsageEvent, +} from "./tool-usage-store"; +import { toolUsageSummaryQuery, integrationUsageSummaryQuery } from "./tool-usage-summary"; + +const clients: Client[] = []; +const stores: ReturnType[] = []; +const database = () => { + const client = createClient({ url: ":memory:" }); + clients.push(client); + return client; +}; +afterEach(async () => { + await Promise.all(stores.splice(0).map((store) => store.close())); + for (const client of clients.splice(0)) client.close(); +}); +const event = (overrides: Partial = {}): ToolUsageEvent => ({ + timestampMs: Date.now(), + memberHash: "a".repeat(64), + mcpTool: "invoke", + targetTool: "tools.sample.org.test.read", + integrationSlug: "sample", + trafficClass: "agent", + status: "ok", + durationMs: 10, + responseBytes: 123, + ...overrides, +}); + +describe("tool usage storage", () => { + it("stores only the metadata contract, with stable, tenant-specific member hashes", async () => { + const client = database(); + const store = makeToolUsageRecorder(client); + stores.push(store); + await store.ready; + const memberHash = store.memberHash("org-identity", "account-identity")!; + expect(memberHash).toMatch(/^[a-f0-9]{64}$/); + const hostileInput = { + ...event({ memberHash }), + arguments: { secret: "payload-secret" }, + results: { token: "result-secret" }, + headers: { authorization: "bearer-secret" }, + }; + store.record(hostileInput); + store.record(hostileInput); + await store.flush(); + const rows = await client.execute("SELECT * FROM executor_tool_usage"); + expect(rows.rows).toHaveLength(2); + expect(Object.keys(rows.rows[0]!).sort()).toEqual( + [ + "id", + "timestamp_ms", + "member_hash", + "mcp_tool", + "target_tool", + "integration_slug", + "traffic_class", + "status", + "duration_ms", + "response_bytes", + ].sort(), + ); + expect(JSON.stringify(rows.rows)).not.toMatch( + /payload-secret|result-secret|bearer-secret|org-identity|account-identity|arguments|results|headers/, + ); + const salt = await initializeToolUsage(client); + expect(hashUsageMember(salt, "org-identity", "account-identity")).toBe(memberHash); + expect(hashUsageMember(salt, "other-org", "account-identity")).not.toBe(memberHash); + }); + + it("prunes old calls, caps rows, and preserves retention boundary calls", async () => { + const client = database(); + await initializeToolUsage(client); + const now = Date.now(); + await client.batch( + [ + usageInsert(event({ timestampMs: now - TOOL_USAGE_RETENTION_MS - 1 })), + usageInsert(event({ timestampMs: now - TOOL_USAGE_RETENTION_MS })), + usageInsert(event({ timestampMs: now, status: "error" })), + usageInsert(event({ timestampMs: now, status: "blocked" })), + ...usageRetention(now, 3), + ], + "write", + ); + let rows = await client.execute( + "SELECT timestamp_ms, status FROM executor_tool_usage ORDER BY id", + ); + expect(rows.rows.map((row) => row.status)).toEqual(["ok", "error", "blocked"]); + expect(rows.rows[0]!.timestamp_ms).toBe(now - TOOL_USAGE_RETENTION_MS); + await client.batch(usageRetention(now, 2), "write"); + rows = await client.execute("SELECT status FROM executor_tool_usage ORDER BY id"); + expect(rows.rows.map((row) => row.status)).toEqual(["error", "blocked"]); + }); + + it("reports overflow instead of keeping an unbounded queue and flushes at close", async () => { + const client = database(); + const store = makeToolUsageRecorder(client); + await store.ready; + for (let i = 0; i < TOOL_USAGE_MAX_PENDING + 3; i++) store.record(event()); + await store.close(); + expect((await client.execute("SELECT COUNT(*) AS n FROM executor_tool_usage")).rows[0]!.n).toBe( + TOOL_USAGE_MAX_PENDING, + ); + expect( + (await client.execute("SELECT dropped_events FROM executor_tool_usage_state")).rows[0]! + .dropped_events, + ).toBe(3); + }); + + it("counts failed writes without logging their errors or retaining lost payloads", async () => { + const client = database(); + const store = makeToolUsageRecorder(client); + stores.push(store); + await store.ready; + store.record(event()); + await client.execute("DROP TABLE executor_tool_usage"); + await store.flush(); + await initializeToolUsage(client); + await store.flush(); + expect( + (await client.execute("SELECT dropped_events, write_failures FROM executor_tool_usage_state")) + .rows[0], + ).toMatchObject({ dropped_events: 1, write_failures: 1 }); + expect((await client.execute("SELECT COUNT(*) AS n FROM executor_tool_usage")).rows[0]!.n).toBe( + 0, + ); + }); + + it("does not fail serving when telemetry initialization fails", async () => { + const client = database(); + client.close(); + const store = makeToolUsageRecorder(client); + stores.push(store); + await store.ready; + expect(store.memberHash("org", "member")).toBeNull(); + expect(() => store.record(event())).not.toThrow(); + await store.flush(); + }); +}); + +const legacyTable = `CREATE TABLE executor_tool_usage ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + timestamp_ms INTEGER NOT NULL, + member_hash TEXT NOT NULL, + mcp_tool TEXT NOT NULL CHECK (mcp_tool IN ('search','invoke','integrations','skills')), + target_tool TEXT CHECK (length(target_tool) <= 512), + integration_slug TEXT CHECK (length(integration_slug) <= 64), + traffic_class TEXT NOT NULL CHECK (traffic_class IN ('agent','benchmark','monitor')), + status TEXT NOT NULL CHECK (status IN ('ok','error','blocked')), + duration_ms REAL NOT NULL CHECK (duration_ms >= 0), + response_bytes INTEGER NOT NULL CHECK (response_bytes >= 0) +)`; + +describe("tool usage schema migration", () => { + it("rebuilds a pre-execute table once, keeping ids, rows, index and state", async () => { + const client = database(); + const now = Date.now(); + await client.batch( + [ + legacyTable, + "CREATE INDEX IF NOT EXISTS executor_tool_usage_time ON executor_tool_usage(timestamp_ms)", + `CREATE TABLE executor_tool_usage_state ( + id INTEGER PRIMARY KEY CHECK (id = 1), salt TEXT NOT NULL, + dropped_events INTEGER NOT NULL DEFAULT 0, write_failures INTEGER NOT NULL DEFAULT 0 + )`, + "INSERT INTO executor_tool_usage_state (id, salt, dropped_events, write_failures) VALUES (1, 'legacy-salt', 4, 2)", + usageInsert( + event({ timestampMs: now, mcpTool: "search", targetTool: null, integrationSlug: null }), + ), + usageInsert(event({ timestampMs: now, durationMs: 5 })), + usageInsert(event({ timestampMs: now, status: "blocked" })), + "DELETE FROM executor_tool_usage WHERE id = 2", + ], + "write", + ); + await expect( + client.execute(usageInsert(event({ timestampMs: now, mcpTool: "execute" }))), + ).rejects.toThrow(); + + const salt = await initializeToolUsage(client); + expect(salt).toBe("legacy-salt"); + const rows = await client.execute( + "SELECT id, mcp_tool, target_tool, status, duration_ms FROM executor_tool_usage ORDER BY id", + ); + expect(rows.rows.map((row) => [row.id, row.mcp_tool, row.target_tool, row.status])).toEqual([ + [1, "search", null, "ok"], + [3, "invoke", "tools.sample.org.test.read", "blocked"], + ]); + const state = await client.execute("SELECT * FROM executor_tool_usage_state"); + expect(state.rows[0]).toMatchObject({ + salt: "legacy-salt", + dropped_events: 4, + write_failures: 2, + }); + const schema = await client.execute( + "SELECT type, name FROM sqlite_master WHERE name LIKE 'executor_tool_usage%' ORDER BY name", + ); + expect(schema.rows.map((row) => [row.type, row.name])).toEqual([ + ["table", "executor_tool_usage"], + ["table", "executor_tool_usage_state"], + ["index", "executor_tool_usage_time"], + ]); + + await client.execute(usageInsert(event({ timestampMs: now, mcpTool: "execute" }))); + const inserted = await client.execute( + "SELECT id, mcp_tool FROM executor_tool_usage ORDER BY id", + ); + expect(inserted.rows.map((row) => [row.id, row.mcp_tool])).toEqual([ + [1, "search"], + [3, "invoke"], + [4, "execute"], + ]); + + const definition = await client.execute( + "SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'executor_tool_usage'", + ); + expect(await initializeToolUsage(client)).toBe("legacy-salt"); + const again = await client.execute( + "SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'executor_tool_usage'", + ); + expect(again.rows[0]!.sql).toBe(definition.rows[0]!.sql); + expect((await client.execute("SELECT COUNT(*) AS n FROM executor_tool_usage")).rows[0]!.n).toBe( + 3, + ); + expect( + ( + await client.execute( + "SELECT name FROM sqlite_master WHERE name = 'executor_tool_usage_new'", + ) + ).rows, + ).toHaveLength(0); + }); + + for (const deletion of ["highest", "hole", "all"] as const) { + it(`preserves the sequence high-water value with ${deletion} IDs deleted`, async () => { + const client = database(); + await client.batch( + [ + legacyTable, + usageInsert(event()), + usageInsert(event()), + usageInsert(event()), + `DELETE FROM executor_tool_usage WHERE ${deletion === "highest" ? "id = 3" : deletion === "hole" ? "id = 2" : "1 = 1"}`, + ], + "write", + ); + const before = await client.execute("SELECT * FROM executor_tool_usage ORDER BY id"); + const salt = await initializeToolUsage(client); + expect((await client.execute("SELECT * FROM executor_tool_usage ORDER BY id")).rows).toEqual( + before.rows, + ); + expect( + (await client.execute("SELECT seq FROM sqlite_sequence WHERE name = 'executor_tool_usage'")) + .rows[0]!.seq, + ).toBe(3); + expect(await initializeToolUsage(client)).toBe(salt); + await client.execute(usageInsert(event({ mcpTool: "execute" }))); + expect( + (await client.execute("SELECT MAX(id) AS id FROM executor_tool_usage")).rows[0]!.id, + ).toBe(4); + }); + } + + it("migrates an empty legacy table that never allocated an ID", async () => { + const client = database(); + await client.execute(legacyTable); + await initializeToolUsage(client); + await client.execute(usageInsert(event({ mcpTool: "execute" }))); + expect((await client.execute("SELECT id FROM executor_tool_usage")).rows[0]!.id).toBe(1); + }); + + it("rolls back a failed migration and can retry without losing sequence or state", async () => { + const client = database(); + await client.batch( + [ + legacyTable, + usageInsert(event()), + usageInsert(event()), + "DELETE FROM executor_tool_usage WHERE id = 2", + // Deliberately conflict with the index creation after the table swap. + "CREATE TABLE executor_tool_usage_time (id INTEGER)", + ], + "write", + ); + await expect(initializeToolUsage(client)).rejects.toThrow(); + expect( + (await client.execute("SELECT sql FROM sqlite_master WHERE name = 'executor_tool_usage'")) + .rows[0]!.sql, + ).not.toContain("'execute'"); + expect( + (await client.execute("SELECT seq FROM sqlite_sequence WHERE name = 'executor_tool_usage'")) + .rows[0]!.seq, + ).toBe(2); + expect( + ( + await client.execute( + "SELECT name FROM sqlite_master WHERE name = 'executor_tool_usage_new'", + ) + ).rows, + ).toHaveLength(0); + await client.execute("DROP TABLE executor_tool_usage_time"); + await initializeToolUsage(client); + await client.execute(usageInsert(event({ mcpTool: "execute" }))); + expect( + (await client.execute("SELECT MAX(id) AS id FROM executor_tool_usage")).rows[0]!.id, + ).toBe(3); + }); + + it("counts execute rows beside invoke in the integration ranking", async () => { + const client = database(); + await initializeToolUsage(client); + const now = Date.now(); + await client.batch( + [ + usageInsert(event({ timestampMs: now, durationMs: 1 })), + usageInsert(event({ timestampMs: now, mcpTool: "execute", durationMs: 3 })), + usageInsert( + event({ timestampMs: now, mcpTool: "execute", targetTool: null, integrationSlug: null }), + ), + usageInsert( + event({ timestampMs: now, mcpTool: "search", targetTool: null, integrationSlug: null }), + ), + ], + "write", + ); + const integrations = await client.execute( + integrationUsageSummaryQuery({ fromMs: now, toMs: now + 1, trafficClass: "agent", limit: 5 }), + ); + expect(integrations.rows).toHaveLength(1); + expect(integrations.rows[0]).toMatchObject({ integration_slug: "sample", calls: 2, p95_ms: 3 }); + const tools = await client.execute( + toolUsageSummaryQuery({ fromMs: now, toMs: now + 1, trafficClass: "agent", limit: 5 }), + ); + expect(tools.rows.map((row) => [row.mcp_tool, row.target_tool, row.calls])).toEqual([ + ["execute", null, 1], + ["execute", "tools.sample.org.test.read", 1], + ["invoke", "tools.sample.org.test.read", 1], + ["search", null, 1], + ]); + }); +}); + +describe("tool usage summary", () => { + it("ranks repeated calls and calculates nearest-rank percentiles for a half-open window", async () => { + const client = database(); + await initializeToolUsage(client); + const now = Date.now(); + await client.batch( + [ + ...Array.from({ length: 20 }, (_, index) => + usageInsert( + event({ + timestampMs: now, + durationMs: index + 1, + status: index === 0 ? "blocked" : index === 1 ? "error" : "ok", + }), + ), + ), + usageInsert( + event({ + timestampMs: now, + targetTool: "tools.other.org.test.read", + integrationSlug: "other", + durationMs: 7, + }), + ), + usageInsert(event({ timestampMs: now - 1, durationMs: 999 })), + usageInsert(event({ timestampMs: now + 1, durationMs: 999 })), + usageInsert(event({ timestampMs: now, durationMs: 999, trafficClass: "benchmark" })), + ], + "write", + ); + const result = await client.execute( + toolUsageSummaryQuery({ fromMs: now, toMs: now + 1, trafficClass: "agent", limit: 20 }), + ); + expect(result.rows).toHaveLength(2); + expect(result.rows[0]).toMatchObject({ + calls: 20, + ok: 18, + error: 1, + blocked: 1, + p50_ms: 10, + p95_ms: 19, + response_bytes: 2460, + }); + expect(result.rows[1]).toMatchObject({ calls: 1, p50_ms: 7, p95_ms: 7 }); + const all = await client.execute( + toolUsageSummaryQuery({ fromMs: now, toMs: now + 1, trafficClass: "all", limit: 1 }), + ); + expect(all.rows[0]).toMatchObject({ calls: 21, p50_ms: 11, p95_ms: 20 }); + const empty = await client.execute( + toolUsageSummaryQuery({ fromMs: now + 2, toMs: now + 3, trafficClass: "agent", limit: 20 }), + ); + expect(empty.rows).toHaveLength(0); + await client.execute( + usageInsert( + event({ timestampMs: now, targetTool: "tools.sample.org.test.write", durationMs: 50 }), + ), + ); + const integrations = await client.execute( + integrationUsageSummaryQuery({ + fromMs: now, + toMs: now + 1, + trafficClass: "agent", + limit: 20, + }), + ); + expect(integrations.rows[0]).toMatchObject({ + integration_slug: "sample", + calls: 21, + p50_ms: 11, + p95_ms: 20, + }); + expect(integrations.rows[1]).toMatchObject({ integration_slug: "other", calls: 1, p50_ms: 7 }); + }); +}); diff --git a/apps/host-selfhost/src/mcp/tool-usage-store.ts b/apps/host-selfhost/src/mcp/tool-usage-store.ts new file mode 100644 index 0000000000..7e0d6bac47 --- /dev/null +++ b/apps/host-selfhost/src/mcp/tool-usage-store.ts @@ -0,0 +1,224 @@ +import { createHmac } from "node:crypto"; +import type { Client, InStatement } from "@libsql/client"; + +export const TOOL_USAGE_RETENTION_MS = 7 * 24 * 60 * 60 * 1000; +export const TOOL_USAGE_MAX_EVENTS = 100_000; +export const TOOL_USAGE_MAX_PENDING = 1024; + +export interface ToolUsageEvent { + readonly timestampMs: number; + readonly memberHash: string; + readonly mcpTool: "search" | "invoke" | "integrations" | "skills" | "execute"; + readonly targetTool: string | null; + readonly integrationSlug: string | null; + readonly trafficClass: "agent" | "benchmark" | "monitor"; + readonly status: "ok" | "error" | "blocked"; + readonly durationMs: number; + readonly responseBytes: number; +} + +export const usageInsert = (event: ToolUsageEvent): InStatement => ({ + sql: `INSERT INTO executor_tool_usage + (timestamp_ms, member_hash, mcp_tool, target_tool, integration_slug, traffic_class, + status, duration_ms, response_bytes) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)`, + // Explicit projection: extra input properties can never reach storage. + args: [ + event.timestampMs, + event.memberHash, + event.mcpTool, + event.targetTool, + event.integrationSlug, + event.trafficClass, + event.status, + event.durationMs, + event.responseBytes, + ], +}); + +export const usageRetention = (now: number, maxEvents = TOOL_USAGE_MAX_EVENTS): InStatement[] => [ + { + sql: "DELETE FROM executor_tool_usage WHERE timestamp_ms < ?", + args: [now - TOOL_USAGE_RETENTION_MS], + }, + { + sql: `DELETE FROM executor_tool_usage WHERE id IN + (SELECT id FROM executor_tool_usage ORDER BY id DESC LIMIT -1 OFFSET ?)`, + args: [maxEvents], + }, +]; + +const USAGE_COLUMNS = + "id, timestamp_ms, member_hash, mcp_tool, target_tool, integration_slug, traffic_class, status, duration_ms, response_bytes"; + +const usageTableDefinition = (name: string): string => `CREATE TABLE IF NOT EXISTS ${name} ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + timestamp_ms INTEGER NOT NULL, + member_hash TEXT NOT NULL, + mcp_tool TEXT NOT NULL CHECK (mcp_tool IN ('search','invoke','integrations','skills','execute')), + target_tool TEXT CHECK (length(target_tool) <= 512), + integration_slug TEXT CHECK (length(integration_slug) <= 64), + traffic_class TEXT NOT NULL CHECK (traffic_class IN ('agent','benchmark','monitor')), + status TEXT NOT NULL CHECK (status IN ('ok','error','blocked')), + duration_ms REAL NOT NULL CHECK (duration_ms >= 0), + response_bytes INTEGER NOT NULL CHECK (response_bytes >= 0) + )`; + +/** + * One-time rebuild for tables created before `execute` joined the `mcp_tool` + * CHECK. SQLite cannot alter a CHECK in place, so copy every row (ids kept) + * into a table with the current definition and swap it in. The state table + * and its salt are untouched. Runs inside the initialization write batch. + */ +const usageMigration = async (client: Client): Promise => { + const existing = await client.execute({ + sql: "SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'executor_tool_usage'", + args: [], + }); + const definition = existing.rows[0]?.sql; + if (typeof definition !== "string" || definition.includes("'execute'")) return []; + return [ + usageTableDefinition("executor_tool_usage_new"), + `INSERT INTO executor_tool_usage_new (${USAGE_COLUMNS}) + SELECT ${USAGE_COLUMNS} FROM executor_tool_usage ORDER BY id`, + // Preserve AUTOINCREMENT history, including a deleted highest id or an empty table. + `INSERT INTO sqlite_sequence (name, seq) + SELECT 'executor_tool_usage_new', seq FROM sqlite_sequence + WHERE name = 'executor_tool_usage' AND NOT EXISTS + (SELECT 1 FROM sqlite_sequence WHERE name = 'executor_tool_usage_new')`, + `UPDATE sqlite_sequence SET seq = MAX(seq, COALESCE( + (SELECT seq FROM sqlite_sequence WHERE name = 'executor_tool_usage'), 0)) + WHERE name = 'executor_tool_usage_new'`, + "DROP TABLE executor_tool_usage", + "ALTER TABLE executor_tool_usage_new RENAME TO executor_tool_usage", + ]; +}; + +export const initializeToolUsage = async (client: Client): Promise => { + await client.batch( + [ + ...(await usageMigration(client)), + usageTableDefinition("executor_tool_usage"), + "CREATE INDEX IF NOT EXISTS executor_tool_usage_time ON executor_tool_usage(timestamp_ms)", + `CREATE TABLE IF NOT EXISTS executor_tool_usage_state ( + id INTEGER PRIMARY KEY CHECK (id = 1), salt TEXT NOT NULL, + dropped_events INTEGER NOT NULL DEFAULT 0, write_failures INTEGER NOT NULL DEFAULT 0 + )`, + "INSERT OR IGNORE INTO executor_tool_usage_state (id, salt) VALUES (1, lower(hex(randomblob(32))))", + ...usageRetention(Date.now()), + ], + "write", + ); + const result = await client.execute("SELECT salt FROM executor_tool_usage_state WHERE id = 1"); + return String(result.rows[0]!.salt); +}; + +/** This salt is a local pseudonymization key, not a service credential. */ +export const hashUsageMember = (salt: string, organizationId: string, accountId: string): string => + createHmac("sha256", salt) + .update(JSON.stringify([organizationId, accountId])) + .digest("hex"); + +/** Bounded queue on the shared DB client; no awaited writes in MCP dispatch. */ +export const makeToolUsageRecorder = (client: Client) => { + let salt: string | null = null; + let pending: ToolUsageEvent[] = []; + let dropped = 0; + let writeFailures = 0; + let flushing: Promise | null = null; + let closed = false; + const ready = (async () => { + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: telemetry setup must not prevent MCP serving + try { + salt = await initializeToolUsage(client); + } catch { + console.warn("[executor:tool-usage] storage unavailable; metrics disabled"); + } + })(); + + const record = (event: ToolUsageEvent): void => { + if (closed || salt === null) return; + if (pending.length >= TOOL_USAGE_MAX_PENDING) { + dropped++; + return; + } + // Copy only fields from the storage contract, never retain caller objects. + pending.push({ + timestampMs: event.timestampMs, + memberHash: event.memberHash, + mcpTool: event.mcpTool, + targetTool: event.targetTool, + integrationSlug: event.integrationSlug, + trafficClass: event.trafficClass, + status: event.status, + durationMs: event.durationMs, + responseBytes: event.responseBytes, + }); + }; + + const flush = (): Promise => { + if (flushing) return flushing; + flushing = (async () => { + await ready; + if (salt === null) return; + const batch = pending; + pending = []; + const droppedCount = dropped; + const failureCount = writeFailures; + dropped = 0; + writeFailures = 0; + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: a failed metrics transaction must not fail tool calls or shutdown + try { + await client.batch( + [ + ...batch.map(usageInsert), + ...usageRetention(Date.now()), + { + sql: `UPDATE executor_tool_usage_state SET dropped_events = dropped_events + ?, + write_failures = write_failures + ? WHERE id = 1`, + args: [droppedCount, failureCount], + }, + ], + "write", + ); + } catch { + dropped += droppedCount + batch.length; + writeFailures += failureCount + 1; + console.warn("[executor:tool-usage] metrics write failed; events dropped"); + } + })().finally(() => { + flushing = null; + }); + return flushing; + }; + // Flush only when needed. Idle pruning still runs once per hour. + const flushTimer = setInterval(() => { + if (pending.length || dropped || writeFailures) void flush(); + }, 1000); + const retentionTimer = setInterval( + () => { + void flush(); + }, + 60 * 60 * 1000, + ); + flushTimer.unref(); + retentionTimer.unref(); + return { + ready, + record, + drop: () => { + if (!closed && salt !== null) dropped++; + }, + memberHash: (organizationId: string, accountId: string): string | null => + salt === null ? null : hashUsageMember(salt, organizationId, accountId), + flush, + close: async () => { + closed = true; + clearInterval(flushTimer); + clearInterval(retentionTimer); + await flushing; + await flush(); + }, + }; +}; + +export type ToolUsageRecorder = ReturnType; diff --git a/apps/host-selfhost/src/mcp/tool-usage-summary.ts b/apps/host-selfhost/src/mcp/tool-usage-summary.ts new file mode 100644 index 0000000000..de2efbf105 --- /dev/null +++ b/apps/host-selfhost/src/mcp/tool-usage-summary.ts @@ -0,0 +1,46 @@ +import type { InStatement } from "@libsql/client"; + +export interface ToolUsageWindow { + readonly fromMs: number; + readonly toMs: number; + readonly trafficClass: "agent" | "benchmark" | "monitor" | "all"; + readonly limit: number; +} + +const summaryQuery = ( + window: ToolUsageWindow, + dimension: "tool" | "integration", +): Exclude => { + const columns = + dimension === "tool" ? "mcp_tool, target_tool, integration_slug" : "integration_slug"; + const filter = + dimension === "integration" + ? "AND mcp_tool IN ('invoke','execute') AND integration_slug IS NOT NULL" + : ""; + return { + sql: `WITH ranked AS ( + SELECT ${columns}, status, duration_ms, response_bytes, + ROW_NUMBER() OVER (PARTITION BY ${columns} ORDER BY duration_ms) AS rank, + COUNT(*) OVER (PARTITION BY ${columns}) AS calls + FROM executor_tool_usage + WHERE timestamp_ms >= ? AND timestamp_ms < ? AND (? = 'all' OR traffic_class = ?) ${filter} + ) + SELECT ${columns}, MAX(calls) AS calls, + SUM(status = 'ok') AS ok, SUM(status = 'error') AS error, SUM(status = 'blocked') AS blocked, + MAX(CASE WHEN rank = (calls + 1) / 2 THEN duration_ms END) AS p50_ms, + MAX(CASE WHEN rank = (calls * 95 + 99) / 100 THEN duration_ms END) AS p95_ms, + SUM(response_bytes) AS response_bytes + FROM ranked GROUP BY ${columns} + ORDER BY calls DESC, ${columns} LIMIT ?`, + args: [window.fromMs, window.toMs, window.trafficClass, window.trafficClass, window.limit], + }; +}; + +/** Nearest-rank percentiles over retained calls in [fromMs, toMs). */ +export const toolUsageSummaryQuery = (window: ToolUsageWindow): Exclude => + summaryQuery(window, "tool"); + +/** Combine all attempted invoke and sandbox execute targets for each integration. */ +export const integrationUsageSummaryQuery = ( + window: ToolUsageWindow, +): Exclude => summaryQuery(window, "integration"); diff --git a/apps/host-selfhost/src/mcp/tool-usage.test.ts b/apps/host-selfhost/src/mcp/tool-usage.test.ts new file mode 100644 index 0000000000..75dd49b404 --- /dev/null +++ b/apps/host-selfhost/src/mcp/tool-usage.test.ts @@ -0,0 +1,643 @@ +import { describe, expect, it } from "@effect/vitest"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { CallToolRequestSchema, type JSONRPCMessage } from "@modelcontextprotocol/sdk/types.js"; +import type { Transport } from "@modelcontextprotocol/sdk/shared/transport.js"; +import { + observeToolUsageServer, + observeToolUsageTransport, + USAGE_EXECUTE_MAX_TARGETS, + usageExecuteTargets, + usageTarget, + usageStatus, +} from "./tool-usage"; +import type { ToolUsageEvent } from "./tool-usage-store"; + +const memberHash = "b".repeat(64); + +describe("tool usage MCP boundary", () => { + it("counts real SDK calls, validation failures and denials without changing results", async () => { + const server = new McpServer({ name: "usage-test", version: "1" }); + const events: ToolUsageEvent[] = []; + observeToolUsageServer(server, memberHash, (event) => events.push(event)); + const ok = { content: [{ type: "text" as const, text: "result-secret" }] }; + server.registerTool("search", { inputSchema: CallToolRequestSchema }, () => ok); + server.registerTool("invoke", {}, () => ({ + isError: true, + content: [ + { + type: "text" as const, + text: "Tool not found or blocked by policy. Search for an available tool.", + }, + ], + })); + server.registerTool("integrations", {}, () => ok); + server.registerTool("skills", {}, () => ok); + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair(); + const client = new Client({ name: "client-secret", version: "1" }); + await server.connect(serverTransport); + await client.connect(clientTransport); + await client.listTools(); + const reply = await client.callTool({ + name: "search", + arguments: { method: "tools/call", params: { name: "argument-secret" } }, + }); + expect(reply).toEqual(ok); + await client.callTool({ + name: "search", + arguments: { method: "tools/call", params: { name: "argument-secret" } }, + }); + const invalid = await client.callTool({ name: "search", arguments: { method: 7 } }); + expect(invalid.isError).toBe(true); + await client.callTool({ + name: "invoke", + arguments: { tool: "tools.sample.org.test.read", arguments: { secret: "invoke-secret" } }, + }); + await client.callTool({ name: "integrations" }); + await client.callTool({ name: "skills" }); + await client.close(); + await server.close(); + expect(events.map((event) => event.status)).toEqual([ + "ok", + "ok", + "error", + "blocked", + "ok", + "ok", + ]); + expect(events[3]).toMatchObject({ + targetTool: "tools.sample.org.test.read", + integrationSlug: "sample", + memberHash, + }); + expect(events.every((event) => event.responseBytes > 0 && event.durationMs >= 0)).toBe(true); + expect(JSON.stringify(events)).not.toMatch( + /argument-secret|result-secret|invoke-secret|client-secret/, + ); + }); + + it("keeps concurrent IDs separate, classifies traffic and counts abandoned calls", async () => { + const events: ToolUsageEvent[] = []; + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: async () => {}, + }; + observeToolUsageTransport(transport, memberHash, (event) => events.push(event)); + await transport.start(); + const request = (id: number): JSONRPCMessage => ({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name: "skills", arguments: { name: "payload-secret" } }, + }); + transport.onmessage!(request(1), { + requestInfo: { + headers: { "x-executor-traffic-class": "benchmark", authorization: "bearer-secret" }, + }, + }); + transport.onmessage!(request(2)); + transport.onmessage!(request(3)); + await transport.send({ + jsonrpc: "2.0", + id: 2, + result: { content: [{ type: "text", text: "é" }] }, + }); + await transport.send({ + jsonrpc: "2.0", + id: 1, + error: { code: -32603, message: "error-secret" }, + }); + transport.onclose!(); + expect(events.map((event) => [event.trafficClass, event.status])).toEqual([ + ["agent", "ok"], + ["benchmark", "error"], + ["agent", "error"], + ]); + expect(events[0]!.responseBytes).toBe( + Buffer.byteLength( + JSON.stringify({ + jsonrpc: "2.0", + id: 2, + result: { content: [{ type: "text", text: "é" }] }, + }), + ), + ); + expect(events[2]!.responseBytes).toBe(0); + expect(JSON.stringify(events)).not.toMatch(/payload-secret|bearer-secret|error-secret/); + }); + + it("preserves transport send failures and records an error", async () => { + const events: ToolUsageEvent[] = []; + // oxlint-disable-next-line executor/no-error-constructor -- boundary: test the SDK transport rejection contract + const failure = new TypeError("send failed"); + const transport: Transport = { + start: async () => {}, + close: async () => {}, + // oxlint-disable-next-line executor/no-promise-reject -- boundary: simulate a rejected native SDK send + send: () => Promise.reject(failure), + }; + observeToolUsageTransport(transport, memberHash, (event) => events.push(event)); + await transport.start(); + transport.onmessage!({ + jsonrpc: "2.0", + id: 1, + method: "tools/call", + params: { name: "search" }, + }); + await expect(transport.send({ jsonrpc: "2.0", id: 1, result: { content: [] } })).rejects.toBe( + failure, + ); + expect(events).toHaveLength(1); + expect(events[0]!.status).toBe("error"); + }); + + it("counts active observation overflow as loss and recognizes policy error codes", async () => { + let lost = 0; + const events: ToolUsageEvent[] = []; + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: async () => {}, + }; + observeToolUsageTransport( + transport, + memberHash, + (event) => events.push(event), + () => { + lost++; + }, + ); + await transport.start(); + for (let id = 0; id < 1025; id++) { + transport.onmessage!({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name: "skills" }, + }); + } + for (let id = 0; id < 1025; id++) + await transport.send({ jsonrpc: "2.0", id, result: { content: [] } }); + expect(lost).toBe(1); + expect(events).toHaveLength(1024); + expect(events.every((event) => event.status === "ok")).toBe(true); + expect( + usageStatus({ + jsonrpc: "2.0", + id: 1, + result: { + isError: true, + structuredContent: { error: { code: "tool_blocked", message: "error-secret" } }, + }, + }), + ).toBe("blocked"); + }); + + it("links a reentrant resume to execute, retains original traffic and ignores invalid and replayed resumes", async () => { + const events: ToolUsageEvent[] = []; + const executionId = "exec_00000000-0000-4000-8000-000000000001"; + const request = (id: number, name: string, args: object = {}): JSONRPCMessage => ({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name, arguments: args }, + }); + const result = (id: number, structuredContent: object): JSONRPCMessage => ({ + jsonrpc: "2.0", + id, + result: { content: [], structuredContent }, + }); + let eventsAfterInvalid: readonly ToolUsageEvent[] = []; + const terminal = { + status: "completed", + toolCalls: [ + { path: "sample.org.test.read", status: "ok" }, + { path: "sample.org.test.write", status: "blocked" }, + ], + result: "result-secret", + logs: ["log-secret"], + }; + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: async (message) => { + if ("id" in message && message.id === 1) { + // The client answers immediately, before the pause send resolves. + transport.onmessage!( + request(2, "resume", { + executionId, + action: "bad", + content: { secret: "argument-secret" }, + }), + ); + await transport.send({ + jsonrpc: "2.0", + id: 2, + error: { code: -32602, message: "error-secret" }, + }); + eventsAfterInvalid = [...events]; + transport.onmessage!(request(3, "resume", { executionId, action: "accept" }), { + requestInfo: { headers: { "x-executor-traffic-class": "benchmark" } }, + }); + await transport.send(result(3, terminal)); + } + }, + }; + observeToolUsageTransport(transport, memberHash, (event) => events.push(event)); + await transport.start(); + transport.onmessage!(request(1, "execute", { code: "code-secret" }), { + requestInfo: { headers: { "x-executor-traffic-class": "monitor" } }, + }); + await transport.send( + result(1, { + status: "waiting_for_interaction", + executionId, + interaction: { args: { secret: "interaction-secret" } }, + }), + ); + transport.onmessage!(request(4, "resume", { executionId, action: "accept" })); + await transport.send(result(4, terminal)); + transport.onmessage!( + request(5, "resume", { executionId: "argument-secret", action: "accept" }), + ); + await transport.send(result(5, terminal)); + transport.onclose!(); + expect(eventsAfterInvalid).toEqual([]); + expect( + events.map((event) => [event.mcpTool, event.targetTool, event.status, event.trafficClass]), + ).toEqual([ + ["execute", "tools.sample.org.test.read", "ok", "monitor"], + ["execute", "tools.sample.org.test.write", "blocked", "monitor"], + ]); + expect( + events.every( + (event) => event.responseBytes === Buffer.byteLength(JSON.stringify(result(3, terminal))), + ), + ).toBe(true); + expect(JSON.stringify(events)).not.toMatch( + /code-secret|argument-secret|interaction-secret|result-secret|log-secret|error-secret|exec_/, + ); + }); + + it("bounds paused observations and reports unfinished executions as losses", async () => { + const events: ToolUsageEvent[] = []; + let lost = 0; + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: async () => {}, + }; + observeToolUsageTransport( + transport, + memberHash, + (event) => events.push(event), + () => lost++, + ); + await transport.start(); + for (let id = 0; id < 1025; id++) { + transport.onmessage!({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name: "execute", arguments: { code: "code-secret" } }, + }); + await transport.send({ + jsonrpc: "2.0", + id, + result: { + structuredContent: { + status: "user_approval_required", + executionId: `exec_00000000-0000-4000-8000-${id.toString(16).padStart(12, "0")}`, + approvalUrl: "url-secret", + }, + }, + }); + } + expect(lost).toBe(1); + expect(events).toEqual([]); + // A different session cannot attribute an execution it never observed. + const unrelated: Transport = { + start: async () => {}, + close: async () => {}, + send: async () => {}, + }; + observeToolUsageTransport(unrelated, memberHash, (event) => events.push(event)); + await unrelated.start(); + unrelated.onmessage!({ + jsonrpc: "2.0", + id: 1, + method: "tools/call", + params: { + name: "resume", + arguments: { executionId: "exec_00000000-0000-4000-8000-000000000001" }, + }, + }); + await unrelated.send({ + jsonrpc: "2.0", + id: 1, + result: { structuredContent: { status: "completed", toolPaths: ["sample.org.test.read"] } }, + }); + transport.onclose!(); + expect(lost).toBe(1025); + expect(events).toEqual([]); + transport.onclose!(); + expect(lost).toBe(1025); + }); + + it("preserves terminal resume send failures and records the targets once as execute errors", async () => { + const events: ToolUsageEvent[] = []; + const executionId = "exec_00000000-0000-4000-8000-000000000001"; + // oxlint-disable-next-line executor/no-error-constructor -- boundary: exercise the native transport rejection contract + const failure = new TypeError("transport-secret"); + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: (message) => { + if ("id" in message && message.id === 2) { + // oxlint-disable-next-line executor/no-promise-reject -- boundary: simulate terminal response delivery failure + return Promise.reject(failure); + } + return Promise.resolve(); + }, + }; + observeToolUsageTransport(transport, memberHash, (event) => events.push(event)); + await transport.start(); + transport.onmessage!({ + jsonrpc: "2.0", + id: 1, + method: "tools/call", + params: { name: "execute" }, + }); + await transport.send({ + jsonrpc: "2.0", + id: 1, + result: { structuredContent: { status: "waiting_for_interaction", executionId } }, + }); + transport.onmessage!({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "resume", arguments: { executionId, action: "accept" } }, + }); + const terminal = { + structuredContent: { + status: "completed", + toolCalls: [{ path: "sample.org.test.read", status: "ok" }], + }, + }; + await expect(transport.send({ jsonrpc: "2.0", id: 2, result: terminal })).rejects.toBe(failure); + transport.onmessage!({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "resume", arguments: { executionId, action: "accept" } }, + }); + await transport.send({ jsonrpc: "2.0", id: 3, result: terminal }); + transport.onclose!(); + expect(events).toHaveLength(1); + expect(events[0]).toMatchObject({ + mcpTool: "execute", + targetTool: "tools.sample.org.test.read", + status: "error", + }); + expect(JSON.stringify(events)).not.toMatch(/transport-secret|exec_/); + }); + + it("records one execute event per distinct connected tool, or one without a target", async () => { + const events: ToolUsageEvent[] = []; + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: async () => {}, + }; + observeToolUsageTransport(transport, memberHash, (event) => events.push(event)); + await transport.start(); + const request = (id: number): JSONRPCMessage => ({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name: "execute", arguments: { code: "code-secret" } }, + }); + transport.onmessage!(request(1)); + transport.onmessage!(request(2)); + transport.onmessage!(request(3)); + await transport.send({ + jsonrpc: "2.0", + id: 1, + result: { + content: [{ type: "text", text: "result-secret" }], + structuredContent: { + status: "completed", + result: "result-secret", + toolPaths: [ + "sample.org.test.read", + "tools.other.user.mine.write", + "sample.org.test.read", + "bearer secret@example.test", + 7, + ], + logs: ["log-secret"], + }, + }, + }); + await transport.send({ + jsonrpc: "2.0", + id: 2, + result: { content: [], structuredContent: { status: "completed", result: 42, logs: [] } }, + }); + await transport.send({ + jsonrpc: "2.0", + id: 3, + result: { + isError: true, + content: [{ type: "text", text: "Error: error-secret" }], + structuredContent: { status: "error", error: "error-secret", logs: [] }, + }, + }); + expect( + events.map((event) => [event.mcpTool, event.targetTool, event.integrationSlug, event.status]), + ).toEqual([ + ["execute", "tools.sample.org.test.read", "sample", "ok"], + ["execute", "tools.other.user.mine.write", "other", "ok"], + ["execute", null, null, "ok"], + ["execute", null, null, "error"], + ]); + expect(events[0]!.durationMs).toBe(events[1]!.durationMs); + expect(events[0]!.responseBytes).toBe(events[1]!.responseBytes); + expect(events[0]!.responseBytes).toBeGreaterThan(0); + expect(JSON.stringify(events)).not.toMatch(/code-secret|result-secret|log-secret|error-secret/); + }); + + it("caps execute targets and ignores unexpected shapes", () => { + const many = Array.from( + { length: USAGE_EXECUTE_MAX_TARGETS + 5 }, + (_, index) => `sample.org.test.read${index}`, + ); + expect( + usageExecuteTargets({ + jsonrpc: "2.0", + id: 1, + result: { content: [], structuredContent: { toolPaths: many } }, + }), + ).toHaveLength(USAGE_EXECUTE_MAX_TARGETS); + expect( + usageExecuteTargets({ + jsonrpc: "2.0", + id: 1, + result: { content: [], structuredContent: { toolPaths: "sample.org.test.read" } }, + }), + ).toEqual([]); + expect( + usageExecuteTargets({ jsonrpc: "2.0", id: 1, error: { code: -32603, message: "x" } }), + ).toEqual([]); + }); + + it("reduces mixed outcomes per path, validates shapes and still updates targets after the cap", () => { + const targets = usageExecuteTargets({ + jsonrpc: "2.0", + id: 1, + result: { + structuredContent: { + toolCalls: [ + null, + 7, + "secret", + {}, + { path: "sample.org.test.bad", status: "error-secret" }, + { path: "secret@example.test", status: "error" }, + { path: "sample.org.test." + "x".repeat(512), status: "ok" }, + ...Array.from({ length: 35 }, (_, index) => ({ + path: `sample.org.test.read${index}`, + status: "ok", + args: "argument-secret", + result: "result-secret", + })), + { path: "tools.sample.org.test.read0", status: "error" }, + { path: "sample.org.test.read0", status: "blocked" }, + { path: "sample.org.test.read0", status: "ok" }, + { path: "sample.org.test.read1", status: "error" }, + ], + toolPaths: ["sample.org.test.read0"], + }, + }, + }); + expect(targets).toHaveLength(32); + expect(targets[0]).toEqual({ + targetTool: "tools.sample.org.test.read0", + integrationSlug: "sample", + status: "blocked", + }); + expect(targets[1]!.status).toBe("error"); + expect(targets[31]!.targetTool).toBe("tools.sample.org.test.read31"); + expect(JSON.stringify(targets)).not.toMatch(/secret|args|result/); + for (const toolCalls of [ + null, + 3, + "secret", + {}, + [null, { path: "sample.org.test.read", status: "constructor" }], + ]) { + expect( + usageExecuteTargets({ + jsonrpc: "2.0", + id: 1, + result: { structuredContent: { toolCalls } }, + }), + ).toEqual([]); + expect( + usageExecuteTargets({ + jsonrpc: "2.0", + id: 1, + result: { + structuredContent: { + toolCalls, + toolPaths: ["sample.org.test.read"], + }, + }, + }), + ).toEqual([{ targetTool: "tools.sample.org.test.read", integrationSlug: "sample" }]); + } + }); + + it("records per-target failures on completed scripts and execution errors on attributed failed scripts", async () => { + const events: ToolUsageEvent[] = []; + const transport: Transport = { + start: async () => {}, + close: async () => {}, + send: async () => {}, + }; + observeToolUsageTransport(transport, memberHash, (event) => events.push(event)); + await transport.start(); + const calls = [ + { path: "sample.org.test.read", status: "ok", args: "argument-secret" }, + { path: "other.user.mine.write", status: "error", error: "error-secret" }, + { path: "sample.org.test.hidden", status: "blocked", result: "result-secret" }, + { path: "sample.org.test.read", status: "error" }, + ]; + for (const id of [1, 2]) { + transport.onmessage!({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name: "execute", arguments: { code: "code-secret" } }, + }); + } + await transport.send({ + jsonrpc: "2.0", + id: 1, + result: { + content: [], + structuredContent: { + status: "completed", + toolCalls: calls, + toolPaths: ["sample.org.test.read"], + logs: ["log-secret"], + result: "result-secret", + }, + }, + }); + await transport.send({ + jsonrpc: "2.0", + id: 2, + result: { + isError: true, + content: [], + structuredContent: { + status: "error", + toolCalls: calls, + toolPaths: ["sample.org.test.read"], + error: "script-secret", + }, + }, + }); + expect(events.map(({ targetTool, status }) => [targetTool, status])).toEqual([ + ["tools.sample.org.test.read", "error"], + ["tools.other.user.mine.write", "error"], + ["tools.sample.org.test.hidden", "blocked"], + ["tools.sample.org.test.read", "error"], + ["tools.other.user.mine.write", "error"], + ["tools.sample.org.test.hidden", "error"], + ]); + expect(events[0]!.durationMs).toBe(events[2]!.durationMs); + expect(events[0]!.responseBytes).toBe(events[2]!.responseBytes); + expect(JSON.stringify(events)).not.toMatch(/secret|args|logs|result/); + }); + + it("does not retain malformed target values", () => { + expect(usageTarget("bearer secret@example.test")).toEqual({ + targetTool: null, + integrationSlug: null, + }); + expect(usageTarget({ secret: "secret" })).toEqual({ targetTool: null, integrationSlug: null }); + expect(usageTarget("tools.sample.org.test." + "x".repeat(512))).toEqual({ + targetTool: null, + integrationSlug: null, + }); + expect(usageTarget("sample.org.test.read")).toEqual({ + targetTool: "tools.sample.org.test.read", + integrationSlug: "sample", + }); + }); +}); diff --git a/apps/host-selfhost/src/mcp/tool-usage.ts b/apps/host-selfhost/src/mcp/tool-usage.ts new file mode 100644 index 0000000000..4071409c2c --- /dev/null +++ b/apps/host-selfhost/src/mcp/tool-usage.ts @@ -0,0 +1,320 @@ +import { Buffer } from "node:buffer"; +import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import type { + JSONRPCMessage, + MessageExtraInfo, + RequestId, +} from "@modelcontextprotocol/sdk/types.js"; +import type { Transport } from "@modelcontextprotocol/sdk/shared/transport.js"; +import { TOOL_USAGE_MAX_PENDING, type ToolUsageEvent } from "./tool-usage-store"; + +type Attempt = Omit & { + readonly started: number; +}; +const trackedTools = new Set(["search", "invoke", "integrations", "skills", "execute"]); +// Engine-issued opaque IDs only. Never retain arbitrary resume arguments. +const executionId = (value: unknown): string | undefined => + typeof value === "string" && /^exec_[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/i.test(value) + ? value + : undefined; + +const unavailable = "Tool not found or blocked by policy. Search for an available tool."; + +/** Distinct connected tools counted per execution; the rest of a longer list is not retained. */ +export const USAGE_EXECUTE_MAX_TARGETS = 32; + +type UsageTarget = { readonly targetTool: string | null; readonly integrationSlug: string | null }; +const noTarget: UsageTarget = { targetTool: null, integrationSlug: null }; + +/** Addresses contain identifiers only; malformed values are never retained. */ +export const usageTarget = (value: unknown): UsageTarget => { + if (typeof value !== "string" || value.length > 512) return noTarget; + const canonical = value.startsWith("tools.") ? value : `tools.${value}`; + const match = + /^tools\.([a-zA-Z0-9_-]{1,64})\.(?:org|user)\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_.-]+$/.exec(canonical); + return match && canonical.length <= 512 + ? { targetTool: canonical, integrationSlug: match[1]! } + : noTarget; +}; + +const trafficClass = (extra?: MessageExtraInfo): ToolUsageEvent["trafficClass"] => { + const value = extra?.requestInfo?.headers["x-executor-traffic-class"]; + return value === "benchmark" || value === "monitor" ? value : "agent"; +}; + +type UsageExecuteTarget = UsageTarget & { readonly status?: ToolUsageEvent["status"] }; +const outcomeSeverity = { ok: 0, error: 1, blocked: 2 } as const; + +/** + * Connected tools a sandbox execution attempted, read only from structured + * identifiers and enum outcomes; never code, arguments, logs or results. + * Older engines provide successful `toolPaths` without per-target outcomes. + */ +export const usageExecuteTargets = (message: JSONRPCMessage): UsageExecuteTarget[] => { + if (!("result" in message)) return []; + const structured = message.result.structuredContent; + if (typeof structured !== "object" || structured === null) return []; + const calls = "toolCalls" in structured ? structured.toolCalls : undefined; + const paths = "toolPaths" in structured ? structured.toolPaths : undefined; + const targets = new Map(); + if (Array.isArray(calls)) { + for (const call of calls) { + if ( + typeof call !== "object" || + call === null || + (call.status !== "ok" && call.status !== "error" && call.status !== "blocked") + ) + continue; + const status: "ok" | "error" | "blocked" = call.status; + const target = usageTarget(call.path); + if (target.targetTool === null) continue; + const previous = targets.get(target.targetTool); + if ( + previous + ? outcomeSeverity[status] > outcomeSeverity[previous.status ?? "ok"] + : targets.size < USAGE_EXECUTE_MAX_TARGETS + ) { + targets.set(target.targetTool, { ...target, status }); + } + } + } + if (Array.isArray(paths)) { + for (const path of paths) { + const target = usageTarget(path); + if ( + target.targetTool !== null && + !targets.has(target.targetTool) && + targets.size < USAGE_EXECUTE_MAX_TARGETS + ) + targets.set(target.targetTool, target); + } + } + return [...targets.values()]; +}; + +export const usageStatus = (message: JSONRPCMessage): ToolUsageEvent["status"] => { + if ("error" in message) return "error"; + if (!("result" in message) || message.result.isError !== true) return "ok"; + const structured = message.result.structuredContent; + if ( + typeof structured === "object" && + structured !== null && + "error" in structured && + typeof structured.error === "object" && + structured.error !== null && + "code" in structured.error && + structured.error.code === "tool_blocked" + ) + return "blocked"; + const content = message.result.content; + // This fixed denial also covers unknown/hidden tools; never expose which case. + if ( + Array.isArray(content) && + content.some( + (item: unknown) => + typeof item === "object" && + item !== null && + "text" in item && + typeof item.text === "string" && + (item.text === unavailable || item.text.startsWith("Error: Tool blocked by policy ")), + ) + ) + return "blocked"; + return "error"; +}; + +/** Observe the SDK transport through public callbacks, including validation failures. */ +export const observeToolUsageTransport = ( + transport: Transport, + memberHash: string, + record: (event: ToolUsageEvent) => void, + onDrop: () => void = () => {}, +): void => { + const attempts = new Map(); + const paused = new Map(); + const resumes = new Map(); + const dropOldest = (pending: Map) => { + if (pending.size < TOOL_USAGE_MAX_PENDING) return; + pending.delete(pending.keys().next().value!); + onDrop(); + }; + // An execute call records one event per distinct connected tool it called, + // each carrying its outcome and the whole execution's duration and response size. + const recordAttempt = ( + attempt: Attempt, + status: ToolUsageEvent["status"], + responseBytes: number, + targets: readonly UsageExecuteTarget[] = [], + ) => { + const durationMs = Math.max(0, performance.now() - attempt.started); + for (const target of targets.length > 0 ? targets : [attempt]) { + record({ + timestampMs: attempt.timestampMs, + memberHash: attempt.memberHash, + mcpTool: attempt.mcpTool, + targetTool: target.targetTool, + integrationSlug: target.integrationSlug, + trafficClass: attempt.trafficClass, + // A script or transport failure makes every attributed row an execution error. + status: + status === "error" + ? "error" + : (("status" in target ? target.status : undefined) ?? status), + durationMs, + responseBytes, + }); + } + }; + const complete = ( + id: RequestId, + status: ToolUsageEvent["status"], + bytes: number, + targets: readonly UsageExecuteTarget[] = [], + ) => { + const attempt = attempts.get(id); + if (!attempt) return; + attempts.delete(id); + recordAttempt(attempt, status, bytes, targets); + }; + const start = transport.start.bind(transport); + transport.start = async () => { + const receive = transport.onmessage; + transport.onmessage = (message, extra) => { + if ( + "method" in message && + message.method === "tools/call" && + "id" in message && + message.params && + typeof message.params.name === "string" && + (trackedTools.has(message.params.name) || message.params.name === "resume") + ) { + const args = message.params.arguments; + if (message.params.name === "resume") { + const id = + typeof args === "object" && args !== null && "executionId" in args + ? executionId(args.executionId) + : undefined; + // Only the session that observed execute can attribute its resume. + if (id && paused.has(id)) { + dropOldest(resumes); + resumes.set(message.id, id); + } + receive?.(message, extra); + return; + } + const tool = message.params.name as ToolUsageEvent["mcpTool"]; + const target = + tool === "invoke" && typeof args === "object" && args !== null && "tool" in args + ? usageTarget(args.tool) + : noTarget; + // Drop incomplete observations rather than invent a completion status. + dropOldest(attempts); + attempts.set(message.id, { + timestampMs: Date.now(), + started: performance.now(), + memberHash, + mcpTool: tool, + ...target, + trafficClass: trafficClass(extra), + }); + } + receive?.(message, extra); + }; + await start(); + }; + const send = transport.send.bind(transport); + transport.send = async (message, options) => { + if (!("id" in message) || message.id === undefined || "method" in message) + return send(message, options); + const resumeId = resumes.get(message.id); + const attempt = resumeId === undefined ? attempts.get(message.id) : paused.get(resumeId); + if (!attempt) { + resumes.delete(message.id); + return send(message, options); + } + const structured = "result" in message ? message.result.structuredContent : undefined; + const executionStatus = + typeof structured === "object" && structured !== null && "status" in structured + ? structured.status + : undefined; + const waiting = + attempt.mcpTool === "execute" && + (executionStatus === "waiting_for_interaction" || + executionStatus === "user_approval_required"); + const nextId = + waiting && + typeof structured === "object" && + structured !== null && + "executionId" in structured + ? executionId(structured.executionId) + : undefined; + // Register the pause before send: an in-memory client can submit resume + // as soon as it receives this response, before send's promise settles. + if (waiting) { + if (resumeId === undefined) attempts.delete(message.id); + else paused.delete(resumeId); + if (nextId) { + dropOldest(paused); + paused.set(nextId, attempt); + } else onDrop(); + } + let status = usageStatus(message); + const targets = attempt.mcpTool === "execute" ? usageExecuteTargets(message) : []; + let bytes = 0; + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: metrics sizing must never alter SDK results, even for non-serializable responses + try { + bytes = Buffer.byteLength(JSON.stringify(message), "utf8"); + } catch { + // The actual transport remains responsible for serializing its response. + } + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: preserve transport failures and count the attempted dispatch + try { + await send(message, options); + } catch (error) { + status = "error"; + // oxlint-disable-next-line executor/no-try-catch-or-throw -- boundary: preserve the original SDK transport rejection + throw error; + } finally { + resumes.delete(message.id); + // A concurrent or cached resume may replay the same result. The first + // response owns the transition; later responses never record it again. + const owns = + resumeId === undefined + ? attempts.get(message.id) === attempt + : paused.get(resumeId) === attempt; + if (owns && !waiting) { + if (resumeId === undefined) { + complete(message.id, status, bytes, targets); + } else if (executionStatus === "completed" || executionStatus === "error") { + paused.delete(resumeId); + recordAttempt(attempt, status, bytes, targets); + } + // Resume validation errors and missing decisions leave the execution + // paused. They are not another execution or its terminal outcome. + } + } + }; + const close = transport.onclose; + transport.onclose = () => { + for (const id of attempts.keys()) complete(id, "error", 0); + // Closing a paused session provides no terminal connected-tool outcomes. + // Report observation loss instead of inventing null-target success/error. + for (let count = 0; count < paused.size; count++) onDrop(); + paused.clear(); + resumes.clear(); + close?.(); + }; +}; + +export const observeToolUsageServer = ( + server: McpServer, + memberHash: string, + record: (event: ToolUsageEvent) => void, + onDrop: () => void = () => {}, +): void => { + const connect = server.connect.bind(server); + server.connect = (transport) => { + observeToolUsageTransport(transport, memberHash, record, onDrop); + return connect(transport); + }; +}; diff --git a/e2e/scenarios/mcp-passthrough.test.ts b/e2e/scenarios/mcp-passthrough.test.ts index 05b0ce5ae4..015259360c 100644 --- a/e2e/scenarios/mcp-passthrough.test.ts +++ b/e2e/scenarios/mcp-passthrough.test.ts @@ -4,6 +4,7 @@ import { createServer, type IncomingMessage } from "node:http"; import { expect } from "@effect/vitest"; import { Effect, Schema } from "effect"; +import { HttpClient } from "effect/unstable/http"; import { composePluginApi } from "@executor-js/api/server"; import { openApiHttpPlugin } from "@executor-js/plugin-openapi/api"; import { @@ -304,12 +305,26 @@ scenario( expect(guide.ok).toBe(true); expect(guide.text).toContain("integrations({})"); expect((yield* passthrough.call("skills", { name: "execute" })).ok).toBe(false); + const compact = decodeToolSearch( + (yield* passthrough.call("search", { + query: "notes", + integration: slug, + owner: "org", + connection: "main", + })).raw, + ).structuredContent; + // Default hits are compact: an argument summary, no inputSchema. + expect(compact.items.every((tool) => tool.inputSchema === undefined)).toBe(true); + expect( + compact.items.find((tool) => tool.id.endsWith(".createNote"))?.arguments, + ).toContain("text"); const found = decodeToolSearch( (yield* passthrough.call("search", { query: "notes", integration: slug, owner: "org", connection: "main", + detail: "full", })).raw, ).structuredContent; expect(found.items.every((tool) => tool.integration === slug)).toBe(true); @@ -342,13 +357,25 @@ scenario( ).structuredContent; expect(other.items.some((tool) => tool.integration === otherSlug)).toBe(true); + // A local port probe can GET / before invoke. Keep such a request in + // the ledger so selecting the first GET reproduces #22 every run. + yield* HttpClient.get(upstream.url).pipe(Effect.flatMap((response) => response.text)); + expect(upstream.requests).toContainEqual({ + method: "GET", + path: "/", + authorization: undefined, + body: "", + }); + // --- A read call reaches the upstream with the connection's credential. --- const listed = yield* passthrough.call("invoke", { tool: listTool, arguments: {} }); expect(listed.ok, `the read call completes: ${listed.text}`).toBe(true); expect(listed.text, "the upstream payload comes back").toContain("existing"); - const listReq = upstream.requests.find((r) => r.method === "GET"); - expect(listReq, "the GET reached the upstream").toBeDefined(); - expect(listReq?.authorization, "the connection's credential was applied").toBe( + const listRequests = upstream.requests.filter( + (r) => r.method === "GET" && r.path === "/notes", + ); + expect(listRequests, "one GET /notes reached the upstream").toHaveLength(1); + expect(listRequests[0]?.authorization, "the connection's credential was applied").toBe( `Bearer tok_${slug}`, ); @@ -362,7 +389,9 @@ scenario( ); expect(created.text, "the call did not pause").not.toContain("Execution paused"); expect(created.text, "the call did not ask for a resume").not.toContain("executionId"); - const createReq = upstream.requests.find((r) => r.method === "POST"); + const createReq = upstream.requests.find( + (r) => r.method === "POST" && r.path === "/notes", + ); expect(createReq, "the POST reached the upstream").toBeDefined(); expect(createReq?.body, "the JSON body went over the wire").toContain('"text":"hello"'); expect( diff --git a/e2e/scenarios/support/search-invoke.ts b/e2e/scenarios/support/search-invoke.ts index 0de674d34f..cbcaed5eef 100644 --- a/e2e/scenarios/support/search-invoke.ts +++ b/e2e/scenarios/support/search-invoke.ts @@ -1,7 +1,8 @@ import { Schema } from "effect"; import { ToolAnnotationsView } from "@executor-js/sdk"; -/** Parse the public search result, including schemas and account identity. */ +/** Parse the public search result: account identity plus either the compact + * `arguments` summary (default) or the full `inputSchema` (`detail: "full"`). */ export const decodeToolSearch = Schema.decodeUnknownSync( Schema.Struct({ structuredContent: Schema.Struct({ @@ -12,7 +13,9 @@ export const decodeToolSearch = Schema.decodeUnknownSync( integration: Schema.String, owner: Schema.String, connection: Schema.String, - inputSchema: Schema.Record(Schema.String, Schema.Unknown), + description: Schema.optional(Schema.String), + arguments: Schema.optional(Schema.String), + inputSchema: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)), annotations: Schema.optional(ToolAnnotationsView), }), ), diff --git a/packages/core/api/src/server/scoped-executor.ts b/packages/core/api/src/server/scoped-executor.ts index 749839be3e..aa8432af36 100644 --- a/packages/core/api/src/server/scoped-executor.ts +++ b/packages/core/api/src/server/scoped-executor.ts @@ -126,6 +126,8 @@ export interface HostConfigShape { * operator knob. */ readonly toolsSyncTtlMs?: number | null; + /** Wait budget for stale catalog refresh. Zero serves persisted rows immediately. */ + readonly toolsSyncGraceMs?: number | null; /** * Forwarded to `ExecutorConfig.waitUntil`: the host's keep-alive * for background work that outlives a request (stale tool-catalog rebuilds @@ -334,6 +336,9 @@ export const makeScopedExecutor = < fetch: hostedFetch, onIntegrationChange: config.onIntegrationChange, ...(config.toolsSyncTtlMs !== undefined ? { toolsSyncTtlMs: config.toolsSyncTtlMs } : {}), + ...(config.toolsSyncGraceMs !== undefined + ? { toolsSyncGraceMs: config.toolsSyncGraceMs } + : {}), ...(waitUntil !== undefined ? { waitUntil } : {}), onElicitation: "accept-all", ...(options?.orgWrites === undefined ? {} : { orgWrites: options.orgWrites }), diff --git a/packages/core/execution/src/attachment-delivery.test.ts b/packages/core/execution/src/attachment-delivery.test.ts new file mode 100644 index 0000000000..21f428593d --- /dev/null +++ b/packages/core/execution/src/attachment-delivery.test.ts @@ -0,0 +1,133 @@ +import { describe, expect, it } from "@effect/vitest"; +import { Effect } from "effect"; +import { makeQuickJsExecutor } from "@executor-js/runtime-quickjs"; +import { withAttachmentDelivery } from "./attachment-delivery"; + +const image = { type: "image", mimeType: "image/gif", data: "R0lGODlh" }; +const video = { + type: "resource", + resource: { uri: "https://slides.example/slides.mp4", mimeType: "video/mp4", blob: "AAAA" }, +}; + +const run = (value: unknown, code: string) => { + const delivery = withAttachmentDelivery({ invoke: () => Effect.succeed(value) }); + return makeQuickJsExecutor().execute(code, delivery.invoker).pipe(Effect.map(delivery.finish)); +}; + +describe("native attachment delivery", () => { + it.effect("removes serialized attachment bytes from returns, logs, and emitted text", () => + Effect.gen(function* () { + const media = { ...image, data: "R0lGODlhAQABAIAAAAAAAP///yH5BAEAAAAALAAAAAABAAEAAAIBRAA7" }; + const result = yield* run( + { ok: true, data: { content: [media] } }, + 'const r = await tools.slides.org.main.render({}); const text = JSON.stringify(r); console.log(text); emit({type:"text",text}); return text;', + ); + expect(JSON.stringify({ result: result.result, logs: result.logs })).not.toContain( + media.data, + ); + expect(result.output?.[0]).toEqual({ type: "content", content: media }); + expect(JSON.stringify(result.output?.[1])).not.toContain(media.data); + }), + ); + it.effect("delivers every MCP attachment without emit or a download", () => + Effect.gen(function* () { + const result = yield* run( + { ok: true, data: { content: [image, video], structuredContent: { render_id: "abc" } } }, + "return await tools.slides.org.main.render({});", + ); + expect(result.output).toEqual([ + { type: "content", content: image }, + { type: "content", content: video }, + ]); + expect(JSON.stringify(result.result)).not.toContain(image.data); + expect(JSON.stringify(result.result)).not.toContain(video.resource.blob); + expect(result.result).toMatchObject({ + ok: true, + data: { structuredContent: { render_id: "abc" } }, + }); + }), + ); + + it.effect("delivers files even when the script returns only metadata", () => + Effect.gen(function* () { + const file = { + _tag: "ToolFile", + encoding: "base64", + name: "report.pdf", + mimeType: "application/pdf", + data: "JVBERg==", + byteLength: 4, + }; + const result = yield* run( + { ok: true, data: file }, + "await tools.reports.org.main.get({}); return {done:true};", + ); + expect(result.output).toEqual([{ type: "file", file }]); + expect(result.result).toEqual({ done: true }); + }), + ); + + it.effect("preserves explicit output order without duplicate attachments", () => + Effect.gen(function* () { + const result = yield* run( + { ok: true, data: { content: [image] } }, + 'const r = await tools.slides.org.main.render({}); emit({type:"text",text:"caption"}); emit(r.data.content[0]); return r;', + ); + expect(result.output).toEqual([ + { type: "content", content: { type: "text", text: "caption" } }, + { type: "content", content: image }, + ]); + }), + ); + + it.effect("keeps attachment bytes available for tool-to-tool uploads", () => + Effect.gen(function* () { + const result = yield* run( + { ok: true, data: { content: [image] } }, + "const r = await tools.slides.org.main.render({}); return {bytesAvailable: r.data.content[0].data === 'R0lGODlh'};", + ); + expect(result.result).toEqual({ bytesAvailable: true }); + }), + ); + + it.effect("does not deliver failed tool results", () => + Effect.gen(function* () { + for (const value of [ + { ok: false, error: { content: [image] } }, + { ok: true, data: { isError: true, content: [image] } }, + ]) { + const result = yield* run(value, "return await tools.slides.org.main.render({});"); + expect(result.output ?? []).toEqual([]); + expect(JSON.stringify(result.result)).not.toContain(image.data); + } + }), + ); + + it.effect("keeps successful attachments when a later script step fails", () => + Effect.gen(function* () { + const result = yield* run( + { ok: true, data: { content: [image] } }, + "await tools.slides.org.main.render({}); throw new Error('later');", + ); + expect(result.error).toContain("later"); + expect(result.output).toEqual([{ type: "content", content: image }]); + }), + ); + + it.effect("preserves ordinary text and resource links", () => + Effect.gen(function* () { + const value = { + ok: true, + data: { + content: [ + { type: "text", text: "hello" }, + { type: "resource_link", uri: "https://example.com", name: "report" }, + ], + }, + }; + const result = yield* run(value, "return await tools.slides.org.main.render({});"); + expect(result.result).toEqual(value); + expect(result.output).toBeUndefined(); + }), + ); +}); diff --git a/packages/core/execution/src/attachment-delivery.ts b/packages/core/execution/src/attachment-delivery.ts new file mode 100644 index 0000000000..78c0e38e79 --- /dev/null +++ b/packages/core/execution/src/attachment-delivery.ts @@ -0,0 +1,135 @@ +import { Effect } from "effect"; +import { isToolFile } from "@executor-js/sdk"; +import type { ExecuteResult, SandboxToolInvoker } from "@executor-js/codemode-core"; + +type Output = NonNullable[number]; +const isRecord = (value: unknown): value is Record => + value !== null && typeof value === "object" && !Array.isArray(value); + +/** Keep attachment bytes on the host and expose metadata in the result preview. */ +export const withAttachmentDelivery = (invoker: SandboxToolInvoker) => { + const attachments: Output[] = []; + const seen = new Set(); + const bytes = new Set(); + + const redact = (text: string): string => { + for (const encoded of bytes) { + if (encoded.length >= 16) text = text.replaceAll(encoded, "[attachment bytes omitted]"); + } + return text; + }; + + const capture = (value: unknown, deliver: boolean): unknown => { + if (isToolFile(value)) { + bytes.add(value.data); + const key = `${value.mimeType}:${value.data}`; + if (deliver && !seen.has(key)) { + seen.add(key); + attachments.push({ type: "file", file: value }); + } + return { + name: value.name, + mimeType: value.mimeType, + byteLength: value.byteLength, + delivery: deliver ? "attachment" : "omitted", + }; + } + if (Array.isArray(value)) return value.map((item) => capture(item, deliver)); + if (!isRecord(value)) return typeof value === "string" ? redact(value) : value; + // Failed calls must never deliver attachments, including nested MCP errors. + deliver = deliver && value.ok !== false && value.isError !== true; + + const binary = + (value.type === "image" || value.type === "audio") && + typeof value.data === "string" && + typeof value.mimeType === "string" + ? { mimeType: value.mimeType, data: value.data } + : value.type === "resource" && + isRecord(value.resource) && + typeof value.resource.blob === "string" + ? { + mimeType: value.resource.mimeType, + data: value.resource.blob, + uri: value.resource.uri, + } + : undefined; + if (binary) { + bytes.add(binary.data); + const key = `${binary.mimeType}:${binary.data}`; + if (deliver && !seen.has(key)) { + seen.add(key); + attachments.push({ type: "content", content: value }); + } + return { + type: "text", + text: JSON.stringify({ + delivery: deliver ? "attachment" : "omitted", + mimeType: binary.mimeType, + uri: binary.uri, + }), + }; + } + return Object.fromEntries( + Object.entries(value).map(([key, item]) => [key, capture(item, deliver)]), + ); + }; + + return { + invoker: { + invoke: (input) => + invoker.invoke(input).pipe( + Effect.tap((value) => + Effect.sync(() => { + capture(value, true); + }), + ), + ), + } satisfies SandboxToolInvoker, + finish: (result: ExecuteResult): ExecuteResult => { + // Returning a file or native block directly also delivers it. Explicit emit + // remains supported; match its bytes to avoid sending the attachment twice. + const compactResult = capture(result.result, true); + const key = (output: Output): string | undefined => { + const value = output.type === "file" ? output.file : output.content; + if (isToolFile(value)) return `${value.mimeType}:${value.data}`; + if (!isRecord(value)) return undefined; + if ((value.type === "image" || value.type === "audio") && typeof value.data === "string") + return `${value.mimeType}:${value.data}`; + if ( + value.type === "resource" && + isRecord(value.resource) && + typeof value.resource.blob === "string" + ) + return `${value.resource.mimeType}:${value.resource.blob}`; + return undefined; + }; + const explicitKeys = new Set(); + const explicit = (result.output ?? []) + .map((item) => { + if ( + item.type === "content" && + isRecord(item.content) && + item.content.type === "text" && + typeof item.content.text === "string" + ) { + return { ...item, content: { ...item.content, text: redact(item.content.text) } }; + } + return item; + }) + .filter((item) => { + const identity = key(item); + if (identity === undefined) return true; + if (explicitKeys.has(identity)) return false; + explicitKeys.add(identity); + return true; + }); + const output = [...attachments.filter((item) => !explicitKeys.has(key(item)!)), ...explicit]; + return { + ...result, + result: compactResult, + ...(result.logs ? { logs: result.logs.map(redact) } : {}), + ...(output.length ? { output } : {}), + }; + }, + }; +}; diff --git a/packages/core/execution/src/description.test.ts b/packages/core/execution/src/description.test.ts index 365f115585..4a2af7c34b 100644 --- a/packages/core/execution/src/description.test.ts +++ b/packages/core/execution/src/description.test.ts @@ -7,6 +7,8 @@ import { IntegrationSlug, ProviderItemId, ProviderKey, + ToolAddress, + ToolName, createExecutor, definePlugin, type CredentialProvider, @@ -38,12 +40,19 @@ const SLACK = IntegrationSlug.make("slack"); const TEMPLATE = AuthTemplateSlug.make("apiKey"); // The execute description lists the top-level integrations the user has -// connected: one bare line per integration slug, deduped across connections, -// names only (no per-integration descriptions). +// connected and can call at least one tool of: one line per integration slug, +// deduped across connections, with the integration's capability description +// when the catalog carries a real one (legacy slug/name-only descriptions are +// suppressed). Each test plugin produces one tool per connection so its +// integration qualifies; `bareIntegrationPlugin` below produces none. +const oneTool = (name: string) => + Effect.succeed({ tools: [{ name: ToolName.make(name), description: name }] }); + const githubPlugin = definePlugin(() => ({ id: "github-plugin" as const, credentialProviders: [memoryProvider()], storage: () => ({}), + resolveTools: () => oneTool("issues.list"), extension: (ctx) => ({ seed: () => ctx.core.integrations.register({ @@ -57,6 +66,7 @@ const githubPlugin = definePlugin(() => ({ const slackPlugin = definePlugin(() => ({ id: "slack-plugin" as const, storage: () => ({}), + resolveTools: () => oneTool("chat.post"), extension: (ctx) => ({ seed: () => ctx.core.integrations.register({ @@ -68,9 +78,260 @@ const slackPlugin = definePlugin(() => ({ }), }))(); +// An integration whose connections carry no tools (the plugin produces none): +// nothing is callable under `tools.bare.…`, so the inventory omits it. +const BARE = IntegrationSlug.make("bare"); +const bareIntegrationPlugin = definePlugin(() => ({ + id: "bare-plugin" as const, + storage: () => ({}), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: BARE, + name: "Bare", + description: "Registered with connections but no tools.", + config: {}, + }), + }), +}))(); + const occurrences = (haystack: string, needle: string): number => haystack.split(needle).length - 1; describe("buildExecuteDescription", () => { + it.effect("names every permitted integration beyond 50, with prose only for the first 50", () => + Effect.gen(function* () { + const slugs = Array.from({ length: 56 }, (_, index) => + IntegrationSlug.make(`integration_${String(index).padStart(3, "0")}`), + ); + const inventoryPlugin = definePlugin(() => ({ + id: "inventory-plugin" as const, + credentialProviders: [memoryProvider()], + storage: () => ({}), + resolveTools: () => oneTool("list"), + extension: (ctx) => ({ + seed: (slug: IntegrationSlug) => + ctx.core.integrations.register({ + slug, + name: String(slug), + description: `Read records from ${slug}.`, + config: {}, + }), + }), + }))(); + const executor = yield* createExecutor( + makeTestConfig({ plugins: [inventoryPlugin] as const }), + ); + // Reverse insertion order so the assertions also check sorting. + for (const slug of [...slugs].reverse()) { + yield* executor["inventory-plugin"].seed(slug); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration: slug, + template: TEMPLATE, + value: "token", + }); + } + yield* executor.connections.create({ + owner: "user", + name: ConnectionName.make("personal"), + integration: slugs[55]!, + template: TEMPLATE, + value: "token", + }); + yield* executor.policies.create({ + owner: "org", + pattern: `${slugs[0]}.*`, + action: "block", + }); + yield* executor.policies.create({ + owner: "org", + pattern: `${slugs[55]}.*`, + action: "require_approval", + }); + + const description = yield* buildExecuteDescription(executor); + const expected = slugs.slice(1).map(String); + const callable = [ + ...new Set((yield* executor.tools.list()).map((tool) => String(tool.integration))), + ].sort(); + expect(callable).toEqual(expected); + expect(parseIntegrationInventory(description)).toEqual(expected); + const lines = description.split("\n").filter((line) => line.startsWith("- `")); + expect(lines).toHaveLength(55); + expect(lines.slice(0, 50)).toEqual( + expected.slice(0, 50).map((slug) => `- \`${slug}\` — Read records from ${slug}.`), + ); + expect(lines.slice(50)).toEqual(expected.slice(50).map((slug) => `- \`${slug}\``)); + expect(description).not.toContain("- ..."); + }), + ); + + it.effect("keeps tools that require approval from policy or plugin annotations", () => + Effect.gen(function* () { + const approvalSlug = IntegrationSlug.make("approval"); + const approvalPlugin = definePlugin(() => ({ + id: "approval-plugin" as const, + storage: () => ({}), + resolveTools: () => + Effect.succeed({ + tools: [ + { + name: ToolName.make("write"), + description: "Write records.", + annotations: { requiresApproval: true }, + }, + ], + }), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: approvalSlug, + description: "Write records.", + config: {}, + }), + }), + }))(); + const executor = yield* createExecutor( + makeTestConfig({ plugins: [githubPlugin, approvalPlugin] as const }), + ); + yield* executor["github-plugin"].seed(); + yield* executor["approval-plugin"].seed(); + for (const integration of [GITHUB, approvalSlug]) { + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration, + template: TEMPLATE, + value: "token", + }); + } + yield* executor.policies.create({ + owner: "org", + pattern: "github.*", + action: "require_approval", + }); + const tools = yield* executor.tools.list(); + expect(tools).toHaveLength(2); + expect( + tools.find((tool) => tool.integration === approvalSlug)?.annotations?.requiresApproval, + ).toBe(true); + expect( + (yield* executor.policies.resolve(ToolAddress.make("tools.github.org.main.issues.list"))) + .action, + ).toBe("require_approval"); + expect(parseIntegrationInventory(yield* buildExecuteDescription(executor))).toEqual([ + "approval", + "github", + ]); + }), + ); + + it.effect("keeps an org connection exception to an integration block", () => + Effect.gen(function* () { + const executor = yield* createExecutor(makeTestConfig({ plugins: [githubPlugin] as const })); + yield* executor["github-plugin"].seed(); + for (const name of ["allowed", "blocked"]) { + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make(name), + integration: GITHUB, + template: TEMPLATE, + value: "token", + }); + } + yield* executor.policies.create({ owner: "org", pattern: "github.*", action: "block" }); + expect(parseIntegrationInventory(yield* buildExecuteDescription(executor))).toEqual([]); + yield* executor.policies.create({ + owner: "org", + pattern: "github.org.allowed.*", + action: "approve", + }); + expect((yield* executor.tools.list()).map((tool) => String(tool.address))).toEqual([ + "tools.github.org.allowed.issues.list", + ]); + expect(parseIntegrationInventory(yield* buildExecuteDescription(executor))).toEqual([ + "github", + ]); + }), + ); + + it.effect("keeps an org block when a user approves a specific connection", () => + Effect.gen(function* () { + const executor = yield* createExecutor(makeTestConfig({ plugins: [githubPlugin] as const })); + yield* executor["github-plugin"].seed(); + yield* executor.connections.create({ + owner: "user", + name: ConnectionName.make("personal"), + integration: GITHUB, + template: TEMPLATE, + value: "token", + }); + yield* executor.policies.create({ owner: "org", pattern: "github.*", action: "block" }); + yield* executor.policies.create({ + owner: "user", + pattern: "github.user.personal.*", + action: "approve", + }); + expect( + (yield* executor.policies.resolve( + ToolAddress.make("tools.github.user.personal.issues.list"), + )).action, + ).toBe("block"); + expect(yield* executor.tools.list()).toEqual([]); + expect(yield* executor.connections.list()).toHaveLength(1); + expect(parseIntegrationInventory(yield* buildExecuteDescription(executor))).toEqual([]); + }), + ); + + it.effect("follows the host catalog scope while retaining approval-gated namespaces", () => + Effect.gen(function* () { + const config = makeTestConfig({ plugins: [githubPlugin, slackPlugin] as const }); + const executor = yield* createExecutor(config); + yield* executor["github-plugin"].seed(); + yield* executor["slack-plugin"].seed(); + for (const integration of [GITHUB, SLACK]) { + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration, + template: TEMPLATE, + value: "token", + }); + } + let visibleIntegration = "github"; + const hostScopePlugin = definePlugin(() => ({ + id: "host-scope-plugin" as const, + storage: () => ({}), + toolPolicyProvider: () => ({ + list: () => + Effect.succeed([ + { + id: "host-grant", + pattern: `${visibleIntegration}.org.main.*`, + action: "require_approval" as const, + position: "a0", + }, + ]), + }), + }))(); + const scoped = yield* createExecutor({ + ...config, + plugins: [githubPlugin, slackPlugin, hostScopePlugin] as const, + }); + expect(yield* executor.connections.list()).toHaveLength(2); + expect((yield* scoped.tools.list()).map((tool) => String(tool.integration))).toEqual([ + "github", + ]); + expect(parseIntegrationInventory(yield* buildExecuteDescription(scoped))).toEqual(["github"]); + visibleIntegration = "slack"; + expect((yield* scoped.tools.list()).map((tool) => String(tool.integration))).toEqual([ + "slack", + ]); + expect(parseIntegrationInventory(yield* buildExecuteDescription(scoped))).toEqual(["slack"]); + }), + ); + it.effect("lists the connected integrations, not the connection prefixes", () => Effect.gen(function* () { const executor = yield* createExecutor( @@ -118,7 +379,7 @@ describe("buildExecuteDescription", () => { }), ); - it.effect("lists integration names only, with no descriptions", () => + it.effect("renders capability descriptions, suppressing slug/name repeats", () => Effect.gen(function* () { const executor = yield* createExecutor( makeTestConfig({ plugins: [slackPlugin, githubPlugin] as const }), @@ -142,12 +403,10 @@ describe("buildExecuteDescription", () => { const description = yield* buildExecuteDescription(executor); - // Bare slugs, sorted, with no per-integration description riding the line - // (Slack's "Send and read workspace messages." is dropped). - expect(description).toContain("- `github`"); - expect(description).toContain("- `slack`"); - expect(description).not.toContain("Send and read workspace messages"); - expect(description).not.toContain("- `slack` —"); + // Slack's real capability description rides its line; github's legacy + // description ("GitHub", a name repeat) is suppressed. + expect(description).toContain("- `slack` — Send and read workspace messages."); + expect(description).toContain("- `github`\n"); expect(description).not.toContain("- `github` —"); }), ); @@ -179,6 +438,44 @@ describe("buildExecuteDescription", () => { }), ); + it.effect("truncates long descriptions to one scannable line", () => + Effect.gen(function* () { + const verbosePlugin = definePlugin(() => ({ + id: "verbose-plugin" as const, + credentialProviders: [memoryProvider()], + storage: () => ({}), + resolveTools: () => oneTool("say"), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: IntegrationSlug.make("verbose"), + name: "Verbose", + description: `${"word ".repeat(60).trim()} trailing\nsecond line ignored`, + config: {}, + }), + }), + }))(); + const executor = yield* createExecutor(makeTestConfig({ plugins: [verbosePlugin] as const })); + yield* executor["verbose-plugin"].seed(); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration: IntegrationSlug.make("verbose"), + template: TEMPLATE, + value: "verbose-token", + }); + + const description = yield* buildExecuteDescription(executor); + const line = description.split("\n").find((l) => l.startsWith("- `verbose`")) ?? ""; + + expect(line.startsWith("- `verbose` — word")).toBe(true); + expect(line.endsWith("…")).toBe(true); + expect(line.length).toBeLessThan(140); + expect(line).not.toContain("trailing"); + expect(line).not.toContain("second line"); + }), + ); + it.effect("omits the Available integrations section when no connections exist", () => Effect.gen(function* () { const executor = yield* createExecutor(makeTestConfig({ plugins: [] as const })); @@ -189,6 +486,104 @@ describe("buildExecuteDescription", () => { expect(description).not.toContain("## Available integrations"); }), ); + + // justcarlson/executor#29: the inventory is what the model picks a namespace + // from, so an integration with no callable tool for this caller must not be + // listed — a `tools.search` under it returns nothing and invites a retry. + it.effect("drops an integration whose tools the org policy blocks", () => + Effect.gen(function* () { + const executor = yield* createExecutor( + makeTestConfig({ plugins: [slackPlugin, githubPlugin] as const }), + ); + yield* executor["slack-plugin"].seed(); + yield* executor["github-plugin"].seed(); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration: SLACK, + template: TEMPLATE, + value: "slack-token", + }); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("prod"), + integration: GITHUB, + template: TEMPLATE, + value: "org-token", + }); + // Both are listed before the block lands. + expect(parseIntegrationInventory(yield* buildExecuteDescription(executor))).toEqual([ + "github", + "slack", + ]); + + yield* executor.policies.create({ owner: "org", pattern: "slack.*", action: "block" }); + + const description = yield* buildExecuteDescription(executor); + + expect(description).toContain("## Available integrations"); + expect(description).toContain("- `github`"); + expect(description).not.toContain("- `slack`"); + expect(description).not.toContain("workspace messages"); + // The connection itself still exists; only the inventory hides it. + expect((yield* executor.connections.list()).map((c) => String(c.integration))).toContain( + "slack", + ); + }), + ); + + it.effect("drops an integration whose connections carry no tools", () => + Effect.gen(function* () { + const executor = yield* createExecutor( + makeTestConfig({ plugins: [bareIntegrationPlugin, githubPlugin] as const }), + ); + yield* executor["bare-plugin"].seed(); + yield* executor["github-plugin"].seed(); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration: BARE, + template: TEMPLATE, + value: "bare-token", + }); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("prod"), + integration: GITHUB, + template: TEMPLATE, + value: "org-token", + }); + // The catalog really holds nothing under `bare`. + expect(yield* executor.tools.list({ integration: BARE })).toEqual([]); + + const description = yield* buildExecuteDescription(executor); + + expect(description).toContain("- `github`"); + expect(description).not.toContain("- `bare`"); + expect(description).not.toContain("no tools"); + }), + ); + + it.effect("omits the section when every connected integration is blocked", () => + Effect.gen(function* () { + const executor = yield* createExecutor(makeTestConfig({ plugins: [githubPlugin] as const })); + yield* executor["github-plugin"].seed(); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("prod"), + integration: GITHUB, + template: TEMPLATE, + value: "org-token", + }); + yield* executor.policies.create({ owner: "org", pattern: "github.*", action: "block" }); + + const description = yield* buildExecuteDescription(executor); + + expect(description).toContain("Execute TypeScript in a sandboxed runtime"); + expect(description).not.toContain("## Available integrations"); + expect(parseIntegrationInventory(description)).toEqual([]); + }), + ); }); describe("parseIntegrationInventory", () => { @@ -220,11 +615,40 @@ describe("parseIntegrationInventory", () => { }), ); + it.effect("round-trips the policy-filtered list, not the raw connections", () => + Effect.gen(function* () { + const executor = yield* createExecutor( + makeTestConfig({ plugins: [slackPlugin, githubPlugin, bareIntegrationPlugin] as const }), + ); + yield* executor["slack-plugin"].seed(); + yield* executor["github-plugin"].seed(); + yield* executor["bare-plugin"].seed(); + for (const integration of [SLACK, GITHUB, BARE]) { + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration, + template: TEMPLATE, + value: "token", + }); + } + yield* executor.policies.create({ owner: "org", pattern: "github.*", action: "block" }); + + const description = yield* buildExecuteDescription(executor); + + // github is blocked, bare has no tools; only slack is callable. + expect(parseIntegrationInventory(description)).toEqual(["slack"]); + expect( + [...new Set((yield* executor.connections.list()).map((c) => String(c.integration)))].sort(), + ).toEqual(["bare", "github", "slack"]); + }), + ); + it("returns nothing for a description without an inventory block", () => { expect(parseIntegrationInventory("Execute TypeScript in a sandboxed runtime.")).toEqual([]); }); - it("reads item lines only, not the overflow marker or prose", () => { + it("reads item lines only, not the overflow marker, prose, or descriptions", () => { const description = [ "Execute TypeScript in a sandboxed runtime.", "", @@ -232,7 +656,7 @@ describe("parseIntegrationInventory", () => { "", "Integrations you have connected. Their tools live under `tools..…`.", "- `github`", - "- `google_gmail`", + "- `google_gmail` — search, read, and send mail", "- ... 3 more", ].join("\n"); diff --git a/packages/core/execution/src/description.ts b/packages/core/execution/src/description.ts index dec2876fd1..343426687c 100644 --- a/packages/core/execution/src/description.ts +++ b/packages/core/execution/src/description.ts @@ -1,5 +1,5 @@ import { Effect } from "effect"; -import type { Connection, Executor } from "@executor-js/sdk/core"; +import type { Connection, Executor, Integration, Tool } from "@executor-js/sdk/core"; /** * Builds the `execute` tool description dynamically. @@ -9,8 +9,18 @@ import type { Connection, Executor } from "@executor-js/sdk/core"; * behind the `skills` tool, see ./skills.ts, to keep this always-loaded * description small) * 2. Available integrations (the live, per-session inventory): the top-level - * integration slugs the user has connected, deduped across connections, - * names only. The same block is appended to the `execute` skill content. + * integration slugs the user has connected AND can call at least one tool + * of, deduped across connections, each with its one-line capability + * description from the integration catalog (user-editable via the + * integrations API). The same block is appended to the `execute` skill + * content. + * + * Why the tool-visibility filter: `connections.list()` is not policy-scoped. + * A member whose org blocks an integration's tools, or whose host scope omits + * it, still owns its connections, so the plain connection inventory names + * namespaces whose `tools.search` returns nothing (justcarlson/executor#29). + * `tools.list` already applies the caller's effective policy, so the inventory + * keeps only the integrations with at least one tool visible there. */ /** The header that opens the live integration inventory. Exported so the host @@ -25,13 +35,47 @@ export const buildExecuteDescription = (executor: Executor): Effect.Effect | null = yield* executor.tools + .list({ includeAnnotations: false }) + .pipe( + Effect.map( + (tools: readonly Tool[]) => new Set(tools.map((tool) => String(tool.integration))), + ), + Effect.catch((error) => + Effect.logWarning("execute description tool inventory read failed", { error }).pipe( + Effect.as(null), + ), + ), + Effect.withSpan("executor.tools.list"), + ); + + const listed = + callableIntegrations === null + ? connections + : connections.filter((connection) => + callableIntegrations.has(String(connection.integration)), + ); + const description = yield* Effect.sync(() => { const lines = [ "Execute TypeScript in a sandboxed runtime.", "", 'Before writing code, call `skills({ name: "execute" })` for the workflow on how to use this tool.', ]; - const inventory = formatIntegrationInventory(connections); + const inventory = formatIntegrationInventory(listed, integrations); if (inventory.length > 0) { lines.push(""); lines.push(inventory); @@ -45,6 +89,10 @@ export const buildExecuteDescription = (executor: Executor): Effect.Effect { return address.startsWith("tools.") ? address.slice("tools.".length) : address; }; -// The live inventory block: the top-level integrations the user has connected, -// one bare line per integration slug (deduped across connections, sorted), no -// per-connection prefixes and no descriptions. Empty string when nothing is -// connected. -const INVENTORY_LIMIT = 50; +// The live inventory block: the top-level integrations the user has connected +// (pre-filtered by the caller to those with a visible tool), one line per +// integration slug (deduped across connections, sorted) with its capability +// description when the catalog carries one. No per-connection prefixes. Empty +// string when nothing is connected. +// Cap capability prose, while naming every callable integration. +const INVENTORY_DESCRIBED_INTEGRATION_LIMIT = 50; -/** One inventory line per integration: `` - `slug` ``. Owned here beside the - * formatter below so {@link parseIntegrationInventory} cannot drift from it. */ -const INVENTORY_ITEM_PATTERN = /^- `([^`]+)`$/; +/** Longest rendered capability description. The block is always-loaded prompt + * context, so one scannable line per integration is the budget. */ +const INVENTORY_DESCRIPTION_LIMIT = 120; + +/** One inventory line per integration: `` - `slug` `` or + * `` - `slug` — description ``. Owned here beside the formatter below so + * {@link parseIntegrationInventory} cannot drift from it. */ +const INVENTORY_ITEM_PATTERN = /^- `([^`]+)`(?: — .*)?$/; /** * Recover the integration slugs from a built execute description — the exact @@ -96,20 +151,53 @@ export const parseIntegrationInventory = (description: string): readonly string[ return slugs; }; -const formatIntegrationInventory = (connections: readonly Connection[]): string => { +/** One scannable line: first line of the catalog description, whitespace + * collapsed, capped at {@link INVENTORY_DESCRIPTION_LIMIT} on a word edge. */ +const formatInventoryDescription = (description: string | null | undefined): string => { + if (!description) return ""; + const flat = description.split("\n", 1)[0]!.replace(/\s+/g, " ").trim(); + if (flat.length === 0) return ""; + if (flat.length <= INVENTORY_DESCRIPTION_LIMIT) return flat; + const cut = flat.slice(0, INVENTORY_DESCRIPTION_LIMIT); + const edge = cut.lastIndexOf(" "); + return `${edge > 0 ? cut.slice(0, edge) : cut}…`; +}; + +const formatIntegrationInventory = ( + connections: readonly Connection[], + integrations: readonly Integration[], +): string => { const slugs = [...new Set(connections.map((connection) => String(connection.integration)))].sort( (a, b) => a.localeCompare(b), ); if (slugs.length === 0) return ""; - const shown = slugs.slice(0, INVENTORY_LIMIT); + const descriptions = new Map( + integrations.map((integration) => [ + String(integration.slug), + { + name: integration.name, + summary: formatInventoryDescription(integration.description), + }, + ]), + ); const lines = [ INTEGRATION_INVENTORY_HEADER, "", "Integrations you have connected. Their tools live under `tools..…`.", - ...shown.map((slug) => `- \`${slug}\``), + ...slugs.map((slug, index) => { + if (index >= INVENTORY_DESCRIBED_INTEGRATION_LIMIT) return `- \`${slug}\``; + const entry = descriptions.get(slug); + // Legacy rows store the slug or display name as the description; a + // suffix that only repeats the line's own slug adds nothing. + const summary = + entry && + entry.summary.length > 0 && + entry.summary.toLowerCase() !== slug.toLowerCase() && + entry.summary.toLowerCase() !== entry.name.toLowerCase() + ? entry.summary + : ""; + return summary ? `- \`${slug}\` — ${summary}` : `- \`${slug}\``; + }), ]; - if (slugs.length > shown.length) { - lines.push(`- ... ${slugs.length - shown.length} more`); - } return lines.join("\n"); }; diff --git a/packages/core/execution/src/engine.test.ts b/packages/core/execution/src/engine.test.ts index 9cd3609072..12ac2fff79 100644 --- a/packages/core/execution/src/engine.test.ts +++ b/packages/core/execution/src/engine.test.ts @@ -44,6 +44,56 @@ const emptyPlugin = definePlugin(() => ({ const makeExecutor = () => createExecutor(makeTestConfig({ plugins: [emptyPlugin()] as const })); +describe("automatic attachment delivery through the execution engine", () => { + for (const mode of ["inline", "pausable", "operator"] as const) { + it.effect(`delivers native content in ${mode} mode without emit`, () => + Effect.gen(function* () { + const image = { type: "image", mimeType: "image/gif", data: "R0lGODlh" }; + const plugin = definePlugin(() => ({ + id: "attachment-test" as const, + storage: () => ({}), + staticIntegrations: () => [ + { + id: "delivery.files", + kind: "in-memory" as const, + name: "Files", + tools: [ + tool({ + name: "render", + description: "Render an attachment.", + inputSchema: Schema.toStandardSchemaV1( + Schema.toStandardJSONSchemaV1(Schema.Struct({})), + ), + execute: () => Effect.succeed({ content: [image] }), + }), + ], + }, + ], + })); + const executor = yield* createExecutor(makeTestConfig({ plugins: [plugin()] as const })); + const engine = createExecutionEngine({ executor, codeExecutor: makeQuickJsExecutor() }); + yield* Effect.addFinalizer(() => + engine.shutdown.pipe(Effect.andThen(executor.close()), Effect.ignore), + ); + const code = "return await tools.delivery.files.render({});"; + const result = + mode === "inline" + ? yield* engine.execute(code, { + onElicitation: () => Effect.succeed({ action: "accept" }), + }) + : yield* engine.executeWithPause(code, { autoApprove: mode === "operator" }); + const completed = + "status" in result && result.status === "completed" ? result.result : result; + expect(completed).toMatchObject({ output: [{ type: "content", content: image }] }); + expect(JSON.stringify(completed)).toContain("attachment"); + expect(JSON.stringify("result" in completed ? completed.result : null)).not.toContain( + image.data, + ); + }).pipe(Effect.scoped), + ); + } +}); + describe("executeWithPause failure propagation", () => { it.effect("surfaces a fast codeExecutor failure as an Exit.Failure", () => Effect.gen(function* () { @@ -350,20 +400,79 @@ describe("formatExecuteResult output identity", () => { status: "completed", result: { issues: [] }, toolName: "linear.org.work.issues.list", + toolPaths: ["linear.org.work.issues.list"], logs: [], }); }); - it("omits a tool name when distinct connected tools were used", () => { + it("omits a tool name but lists distinct paths in first-call order when several tools were used", () => { const result = { result: { issues: [], projects: [] }, logs: [], - toolPaths: ["linear.org.work.issues.list", "linear.org.work.projects.list"], + toolPaths: [ + "linear.org.work.projects.list", + "linear.org.work.issues.list", + "linear.org.work.projects.list", + ], } as ExecuteResult & { readonly toolPaths: readonly string[] }; const formatted = formatExecuteResult(result); expect(formatted.structured).not.toHaveProperty("toolName"); + expect(formatted.structured["toolPaths"]).toEqual([ + "linear.org.work.projects.list", + "linear.org.work.issues.list", + ]); + }); + + it("omits tool paths when no connected tool was called", () => { + const formatted = formatExecuteResult({ result: 1, logs: [], toolPaths: [] }); + + expect(formatted.structured).not.toHaveProperty("toolPaths"); + expect(formatted.structured).not.toHaveProperty("toolName"); + }); + + it("retains successful paths and outcome identifiers in script-error envelopes", () => { + const formatted = formatExecuteResult({ + result: null, + error: "script-secret", + logs: ["log-secret"], + toolPaths: ["sample.org.main.read"], + toolCalls: [{ path: "sample.org.main.read", status: "ok" }], + }); + expect(formatted.isError).toBe(true); + expect(formatted.structured).toMatchObject({ + status: "error", + toolPaths: ["sample.org.main.read"], + toolCalls: [{ path: "sample.org.main.read", status: "ok" }], + }); + expect(formatted.structured).not.toHaveProperty("toolName"); + }); + + it("caps outcome identifiers and reduces repeats after the cap without retaining extra fields", () => { + const calls = Array.from({ length: 35 }, (_, index) => ({ + path: `sample.org.main.read${index}`, + status: "ok" as const, + args: "argument-secret", + result: "result-secret", + error: "error-secret", + })); + const formatted = formatExecuteResult({ + result: null, + toolCalls: [ + ...calls, + { path: "sample.org.main.read0", status: "error" }, + { path: "tools.sample.org.main.read0", status: "blocked" }, + { path: "sample.org.main.read0", status: "ok" }, + { path: "bearer secret@example.test", status: "error" }, + { path: "sample.org.main." + "x".repeat(512), status: "error" }, + ], + }); + expect(formatted.structured["toolCalls"]).toEqual([ + { path: "sample.org.main.read0", status: "blocked" }, + ...calls.slice(1, 32).map(({ path, status }) => ({ path, status })), + ]); + expect(JSON.stringify(formatted.structured)).not.toMatch(/secret|args|error/); }); it("truncates a long preview with the exact suffix and untouched structured value", () => { diff --git a/packages/core/execution/src/engine.ts b/packages/core/execution/src/engine.ts index dad73e2025..bce4a07020 100644 --- a/packages/core/execution/src/engine.ts +++ b/packages/core/execution/src/engine.ts @@ -27,6 +27,7 @@ import { } from "./tool-invoker"; import { ExecutionToolError } from "./errors"; import { buildExecuteDescription } from "./description"; +import { withAttachmentDelivery } from "./attachment-delivery"; // --------------------------------------------------------------------------- // Types @@ -146,9 +147,38 @@ const truncate = (value: string, max: number): string => ? `${value.slice(0, max)}\n... [truncated ${value.length - max} chars]` : value; -const soleConnectedToolName = (toolPaths: readonly string[] | undefined): string | undefined => { - const names = [...new Set(toolPaths ?? [])]; - return names.length === 1 ? names[0] : undefined; +/** Distinct connected tool paths in first-call order; never the call trace. */ +const distinctConnectedToolPaths = (toolPaths: readonly string[] | undefined): string[] => [ + ...new Set(toolPaths ?? []), +]; + +const soleConnectedToolName = (toolPaths: readonly string[]): string | undefined => + toolPaths.length === 1 ? toolPaths[0] : undefined; + +/** Keep the first 32 targets; repeated outcomes reduce as blocked > error > ok. */ +const connectedToolOutcomes = () => { + const calls = new Map[number]>(); + const severity = { ok: 0, error: 1, blocked: 2 } as const; + const record = (path: unknown, status: unknown) => { + if ( + typeof path !== "string" || + path.length > 512 || + (status !== "ok" && status !== "error" && status !== "blocked") + ) + return; + const canonical = path.startsWith("tools.") ? path : `tools.${path}`; + if ( + canonical.length > 512 || + !/^tools\.[a-zA-Z0-9_-]{1,64}\.(?:org|user)\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_.-]+$/.test(canonical) + ) + return; + const normalized = canonical.slice("tools.".length); + const previous = calls.get(normalized); + if (previous ? severity[status] > severity[previous.status] : calls.size < 32) { + calls.set(normalized, { path: normalized, status }); + } + }; + return { record, values: () => [...calls.values()] }; }; export const formatExecuteResult = ( @@ -175,6 +205,18 @@ export const formatExecuteResult = ( const emittedNote = emitted > 0 ? `${emitted} item${emitted === 1 ? "" : "s"} emitted to the user` : null; const emittedField = emitted > 0 ? { emitted } : {}; + const toolPaths = distinctConnectedToolPaths(result.toolPaths); + const outcomes = connectedToolOutcomes(); + if (Array.isArray(result.toolCalls)) { + for (const call of result.toolCalls) { + if (typeof call === "object" && call !== null) outcomes.record(call.path, call.status); + } + } + const toolCalls = outcomes.values(); + const toolFields = { + ...(toolPaths.length > 0 ? { toolPaths } : {}), + ...(toolCalls.length > 0 ? { toolCalls } : {}), + }; if (result.error) { const parts = [`Error: ${result.error}`, ...(logText ? [`\nLogs:\n${logText}`] : [])]; @@ -183,6 +225,7 @@ export const formatExecuteResult = ( structured: { status: "error", error: result.error, + ...toolFields, ...emittedField, logs: result.logs ?? [], }, @@ -196,13 +239,14 @@ export const formatExecuteResult = ( ? `(no return value; ${emittedNote})` : "(no result)"; const parts = [resultPart, ...(logText ? [`\nLogs:\n${logText}`] : [])]; - const toolName = soleConnectedToolName(result.toolPaths); + const toolName = soleConnectedToolName(toolPaths); return { text: parts.join("\n"), structured: { status: "completed", result: result.result ?? null, ...(toolName ? { toolName } : {}), + ...toolFields, ...emittedField, logs: result.logs ?? [], }, @@ -344,8 +388,13 @@ const makeFullInvoker = ( invokeOptions: InvokeOptions, toolDiscoveryProvider: ToolDiscoveryProvider, onConnectedToolCall?: (path: string) => void, + onConnectedToolOutcome?: (path: string, status: "ok" | "error" | "blocked") => void, ): SandboxToolInvoker => { - const base = makeExecutorToolInvoker(executor, { invokeOptions, onConnectedToolCall }); + const base = makeExecutorToolInvoker(executor, { + invokeOptions, + onConnectedToolCall, + onConnectedToolOutcome, + }); return { invoke: ({ path, args }) => { if (path === "search") { @@ -721,15 +770,23 @@ export const createExecutionEngine = toolPaths.push(path), + toolCalls.record, ); + const delivery = withAttachmentDelivery(invoker); fiber = yield* Effect.forkDetach( - codeExecutor.execute(code, invoker).pipe( - Effect.map((result) => (toolPaths.length === 0 ? result : { ...result, toolPaths })), + codeExecutor.execute(code, delivery.invoker).pipe( + Effect.map(delivery.finish), + Effect.map((result) => ({ + ...result, + ...(toolPaths.length > 0 ? { toolPaths } : {}), + ...(toolCalls.values().length > 0 ? { toolCalls: toolCalls.values() } : {}), + })), Effect.withSpan("executor.code.exec"), ), ); @@ -858,6 +915,7 @@ export const createExecutionEngine = toolPaths.push(path), + toolCalls.record, ); - const result = yield* codeExecutor.execute(code, invoker).pipe( - Effect.map((result) => (toolPaths.length === 0 ? result : { ...result, toolPaths })), + const delivery = withAttachmentDelivery(invoker); + const result = yield* codeExecutor.execute(code, delivery.invoker).pipe( + Effect.map(delivery.finish), + Effect.map((result) => ({ + ...result, + ...(toolPaths.length > 0 ? { toolPaths } : {}), + ...(toolCalls.values().length > 0 ? { toolCalls: toolCalls.values() } : {}), + })), Effect.withSpan("executor.code.exec"), ); yield* annotateExecuteOutcome(result); diff --git a/packages/core/execution/src/index.ts b/packages/core/execution/src/index.ts index 23175a01b3..18ffedcc81 100644 --- a/packages/core/execution/src/index.ts +++ b/packages/core/execution/src/index.ts @@ -25,6 +25,12 @@ export { skillCatalogFor, type Skill, } from "./skills"; +export { + integrationAliasText, + resolveIntegrationAlias, + type IntegrationCatalogEntry, + type IntegrationResolution, +} from "./integration-aliases"; export { PROVIDED_GLOBAL_NAMES } from "./provided-globals"; export { ExecutionToolError } from "./errors"; export { diff --git a/packages/core/execution/src/integration-aliases.ts b/packages/core/execution/src/integration-aliases.ts new file mode 100644 index 0000000000..eb9478f3d8 --- /dev/null +++ b/packages/core/execution/src/integration-aliases.ts @@ -0,0 +1,82 @@ +/** + * Integration alias resolution shared by every discovery surface: the + * passthrough `search`/`integrations` MCP tools and the sandbox + * `tools.search({ namespace })` call resolve the name a person would use + * (`gmail`, `Google Calendar`) to the catalog slug the same way. + */ + +/** What `integrations` and `search` resolve aliases against. */ +export type IntegrationCatalogEntry = { + readonly slug: string; + readonly name: string; + readonly description?: string; +}; + +/** + * How an `integration` argument resolved: `exact` is the slug as given, + * `alias` is one unambiguous match, `ambiguous` lists every slug that fits, + * `none` matched nothing. The caller decides what to do with each. + */ +export type IntegrationResolution = + | { readonly kind: "exact" | "alias"; readonly slug: string } + | { readonly kind: "ambiguous"; readonly slugs: readonly string[] } + | { readonly kind: "none" }; + +const aliasTokens = (value: string): string[] => + value + .replace(/([a-z0-9])([A-Z])/g, "$1 $2") + .toLowerCase() + .split(/[^a-z0-9]+/) + .filter(Boolean); + +const squash = (value: string): string => aliasTokens(value).join(""); + +/** + * Resolve a human alias to catalog slugs: `gmail` → `google_gmail`, + * `Google Calendar` → `google_calendar`. Exact slug first; then a slug or + * name that is the same once separators and case are dropped; then every + * alias word appearing in the slug or name; last, every word appearing in + * the description. The first tier with any match decides. + */ +export const resolveIntegrationAlias = ( + input: string, + catalog: readonly IntegrationCatalogEntry[], +): IntegrationResolution => { + const trimmed = input.trim(); + if (catalog.some((entry) => entry.slug === trimmed)) return { kind: "exact", slug: trimmed }; + const words = aliasTokens(trimmed); + const squashed = squash(trimmed); + if (words.length === 0) return { kind: "none" }; + const covers = (haystack: readonly string[]) => words.every((word) => haystack.includes(word)); + const tiers: ReadonlyArray<(entry: IntegrationCatalogEntry) => boolean> = [ + (entry) => squash(entry.slug) === squashed || squash(entry.name) === squashed, + (entry) => covers([...aliasTokens(entry.slug), ...aliasTokens(entry.name)]), + (entry) => covers(aliasTokens(entry.description ?? "")), + ]; + for (const matches of tiers) { + const slugs = [...new Set(catalog.filter(matches).map((entry) => entry.slug))].sort(); + if (slugs.length === 1) return { kind: "alias", slug: slugs[0]! }; + if (slugs.length > 1) return { kind: "ambiguous", slugs }; + } + return { kind: "none" }; +}; + +/** + * Display-name text per slug for ranking, limited to words the slug does not + * already have: `google_gmail` / "Gmail" adds nothing; `scrape_api` / + * "ScrapeCreators" adds "creators scrapecreators" (the split words and the + * one-word spelling a person types). + */ +export const integrationAliasText = ( + catalog: readonly IntegrationCatalogEntry[], +): ReadonlyMap => { + const aliases = new Map(); + for (const entry of catalog) { + const slugTokens = new Set(aliasTokens(entry.slug)); + const nameTokens = aliasTokens(entry.name); + const candidates = nameTokens.length > 1 ? [...nameTokens, nameTokens.join("")] : nameTokens; + const extra = [...new Set(candidates)].filter((token) => !slugTokens.has(token)); + if (extra.length > 0) aliases.set(entry.slug, extra.join(" ")); + } + return aliases; +}; diff --git a/packages/core/execution/src/skills.test.ts b/packages/core/execution/src/skills.test.ts index 49a1229cef..d24b3a2524 100644 --- a/packages/core/execution/src/skills.test.ts +++ b/packages/core/execution/src/skills.test.ts @@ -16,6 +16,19 @@ describe("skills registry", () => { expect(EXECUTE_SKILL.body).toContain("read `result.data.connections`"); }); + // The workflow steers to the fastest path: a narrowed, small search, then + // the call. Describing a tool is the exception, not a fixed step. + it("opens the execute workflow with a namespaced search of limit 3", () => { + const workflow = EXECUTE_SKILL.body.slice( + EXECUTE_SKILL.body.indexOf("## Workflow"), + EXECUTE_SKILL.body.indexOf("## Rules"), + ); + expect(workflow).toContain('namespace: "", limit: 3'); + expect(workflow).not.toContain("limit: 12"); + expect(workflow).toContain("Only when the argument shape is unknown"); + expect(workflow.indexOf("tools.search(")).toBeLessThan(workflow.indexOf("tools[path](input)")); + }); + it("finds a skill by exact name and misses unknown names", () => { expect(findSkill("execute")).toBe(EXECUTE_SKILL); expect(findSkill("Execute")).toBeUndefined(); diff --git a/packages/core/execution/src/skills.ts b/packages/core/execution/src/skills.ts index 9bf8f43ef0..0cde1c3209 100644 --- a/packages/core/execution/src/skills.ts +++ b/packages/core/execution/src/skills.ts @@ -32,27 +32,26 @@ const EXECUTE_SKILL_BODY = [ "", "## Workflow", "", - '1. `const { items: matches } = await tools.search({ query: "", limit: 12 });`', + '1. `const { items: matches } = await tools.search({ query: "", namespace: "", limit: 3 });` — omit `namespace` when the integration is unknown.', '2. `const path = matches[0]?.path; if (!path) return "No matching tools found.";`', - "3. `const details = await tools.describe.tool({ path });`", - "4. Use `details.inputTypeScript` / `details.outputTypeScript` and `details.typeScriptDefinitions` for compact shapes.", - "5. For live saved-connection inventory, call `tools.executor.coreTools.connections.list({})`; after checking `result.ok`, read `result.data.connections`.", - "6. Call the tool: `const result = await tools.(input);`", + "3. Only when the argument shape is unknown: `const details = await tools.describe.tool({ path });` and use `details.inputTypeScript` / `details.outputTypeScript` and `details.typeScriptDefinitions` for compact shapes.", + "4. For live saved-connection inventory, call `tools.executor.coreTools.connections.list({})`; after checking `result.ok`, read `result.data.connections`.", + "5. Call the tool: `const result = await tools[path](input);`", "", "## Rules", "", "- `tools.search()` returns paginated, ranked matches: `{ items, total, hasMore, nextOffset }`. Best-first. Use short intent phrases like `github issues`, `repo details`, or `create calendar event`.", - '- When you already know the namespace, narrow with `tools.search({ namespace: "github", query: "issues" })`.', + '- When you already know the integration, narrow with `tools.search({ namespace: "github", query: "issues", limit: 3 })`. `namespace` accepts the slug or an unambiguous alias (`gmail` resolves to `google_gmail`); an alias that fits several integrations is used as a prefix instead.', "- `tools.executor.coreTools.connections.list({})` returns saved connections with `{ address, integration, owner, name, ... }`. The `address` field includes the leading `tools.` root.", "- Tool calls return a value union: `{ ok: true, data }` for success or `{ ok: false, error: { code, message, status?, details?, retryable? } }` for expected tool/domain failures. Branch on `result.ok`.", "- `data` is the upstream payload itself. HTTP-backed tools (OpenAPI) also set `http: { status, headers }` beside `data` — read `result.http?.headers` for pagination (Link) or rate-limit headers.", "- Use `emit(value)` to append user-visible output. Plain values become MCP text content. MCP content blocks are forwarded as-is. `ToolFile` values are rendered by MIME. Emitting and returning compose: emitted items come first in the tool result, the returned value follows, and the envelope reports an `emitted` count so you can confirm the items landed.", - '- File-returning tools may return `ToolFile` values: `{ _tag: "ToolFile", name?, mimeType, encoding: "base64", data, byteLength }`. A "file-returning tool" includes APIs that return file bytes inside a JSON field, such as Base64-encoded `content`; wrap those payloads in a `ToolFile` and `emit()` them. Emit any attachment with `emit(result.data)`.', + '- File-returning tools may return `ToolFile` values: `{ _tag: "ToolFile", name?, mimeType, encoding: "base64", data, byteLength }`. A "file-returning tool" includes APIs that return file bytes inside a JSON field, such as Base64-encoded `content`; wrap those payloads in a `ToolFile` and `emit()` them. Attachments are delivered automatically; no download or `emit()` is needed.', "- Never decode or transcode bytes yourself — the sandbox has no `Buffer`, `atob`, `btoa`, `TextDecoder`, or `TextEncoder`. For base64-encoded bytes in a JSON field, wrap the payload in a `ToolFile` with its `mimeType` and `emit()` it; to forward them to an upload tool, pass the `ToolFile`'s base64 `data` as that tool's `bodyBase64` (both sides speak base64, so nothing is decoded).", '- To emit MCP-native content directly, pass an MCP content block to `emit(...)`, such as `{ type: "image", data, mimeType }`, `{ type: "audio", data, mimeType }`, `{ type: "text", text }`, `{ type: "resource", resource }`, or `{ type: "resource_link", uri, name, ... }`.', "- `emit(ToolFile)` is MIME-based: `image/*` becomes MCP image content, `audio/*` becomes MCP audio content, text-like files become decoded text, and other binary files become embedded MCP resources.", - "- `return` is only for ordinary structured data. Returning a `ToolFile`, a `ToolResult`, an MCP content block, or a bare base64 string does not emit content to the MCP client.", - "- Some providers, including Gmail, return attachment bytes as a `ToolFile` with no public URL to hand off — the bytes themselves are the payload, so `emit(result.data)` to display it, or pass its base64 `data` as another tool's `bodyBase64` to forward it.", + "- Successful tool results containing `ToolFile` values or native MCP image, audio, or embedded binary resources are delivered automatically. Returned previews contain compact metadata instead of file bytes. Explicit `emit()` remains supported and attachments are deduplicated. A bare base64 string is ordinary data.", + "- Some providers, including Gmail, return attachment bytes as a `ToolFile` with no public URL to hand off — the bytes themselves are the payload. Executor delivers the attachment automatically; pass its base64 `data` as another tool's `bodyBase64` to forward it.", "- If `tools.search()` returns `hasMore: true` and you didn't find what you need, fetch the next page: `tools.search({ query, offset: nextOffset, limit })`.", "- Always use the full address when calling tools: `tools....(args)`. The `path` returned by `tools.search()` / `tools.describe.tool()` is already the exact path under `tools` — call `tools[path]` rather than guessing segments.", "- The `tools` object is a lazy proxy — enumerating it (`Object.keys(tools)`, spread, `for...in`) throws. Use `tools.search()` or `tools.executor.coreTools.connections.list({})` instead.", diff --git a/packages/core/execution/src/tool-invoker.test.ts b/packages/core/execution/src/tool-invoker.test.ts index a2aa222d17..7a70346b64 100644 --- a/packages/core/execution/src/tool-invoker.test.ts +++ b/packages/core/execution/src/tool-invoker.test.ts @@ -29,7 +29,7 @@ import { } from "@executor-js/sdk/testing"; import { makeQuickJsExecutor } from "@executor-js/runtime-quickjs"; import type { CodeExecutor, ExecuteResult } from "@executor-js/codemode-core"; -import { createExecutionEngine } from "./engine"; +import { createExecutionEngine, formatExecuteResult } from "./engine"; import { ExecutionToolError } from "./errors"; import { describeTool, @@ -171,6 +171,8 @@ const withPrivateAnnotations = (annotations: ToolAnnotations) => ({ const makeTestPlugin = (config: { readonly pluginId: string; readonly integration: string; + /** Catalog display name; omitted, the name falls back to the slug. */ + readonly displayName?: string; readonly tools: readonly TestToolSpec[]; }) => { const slug = IntegrationSlug.make(config.integration); @@ -206,6 +208,7 @@ const makeTestPlugin = (config: { seed: () => ctx.core.integrations.register({ slug, + ...(config.displayName === undefined ? {} : { name: config.displayName }), description: config.integration, config: {}, }), @@ -511,6 +514,110 @@ describe("tool discovery", () => { }), ); + it.effect("ranks an integration's tools when the query names it by display name", () => + Effect.gen(function* () { + const executor = yield* makeSearchExecutor(); + + // "People" is nowhere in the crm slug, names, or descriptions. + const without = yield* searchTools(executor, "people contact", 5); + expect(without.items.map((match) => match.path)).toEqual([]); + + const aliases = new Map([["crm", "People Desk"]]); + const withAlias = yield* searchTools(executor, "people contact", 5, { + integrationAliases: aliases, + }); + expect(withAlias.items.map((match) => match.integration)).toEqual(["crm", "crm"]); + expect(withAlias.items.map((match) => match.path)).toEqual([ + "crm.org.main.createContact", + "crm.org.main.listContacts", + ]); + + // Alias text only ever adds signal; a query that already matched keeps its order. + const before = yield* searchTools(executor, "github issues", 5); + const after = yield* searchTools(executor, "github issues", 5, { + integrationAliases: new Map([["github", "GitHub"]]), + }); + expect(after.items.map((match) => match.path)).toEqual( + before.items.map((match) => match.path), + ); + }), + ); + + it.effect("resolves a namespace alias through the integration catalog", () => + Effect.gen(function* () { + const gmail = makeTestPlugin({ + pluginId: "gmail-alias-test", + integration: "google_gmail", + displayName: "Gmail", + tools: [ + { + name: "listMessages", + description: "List messages", + inputJsonSchema: EmptyInputJson, + validator: EmptyValidator, + handler: () => Effect.succeed([]), + }, + { + name: "sendMessage", + description: "Send a message", + inputJsonSchema: EmptyInputJson, + validator: EmptyValidator, + handler: () => Effect.succeed([]), + }, + ], + }); + const calendar = makeTestPlugin({ + pluginId: "calendar-alias-test", + integration: "google_calendar", + displayName: "Google Calendar", + tools: [ + { + name: "insertEvent", + description: "Create an event", + inputJsonSchema: EmptyInputJson, + validator: EmptyValidator, + handler: () => Effect.succeed({}), + }, + ], + }); + const executor = yield* makeExecutorWith([gmail, calendar] as const); + yield* provision(executor as never, [ + { pluginId: "gmail-alias-test", integration: "google_gmail" }, + { pluginId: "calendar-alias-test", integration: "google_calendar" }, + ]); + + // The production failure: "gmail" is not a token prefix of + // "google_gmail", so the namespace used to match nothing. + const ranked = yield* searchTools(executor, "list messages", 5, { namespace: "gmail" }); + expect(ranked.items.map((item) => item.path)).toEqual(["google_gmail.org.main.listMessages"]); + + // A display name resolves too, and enumeration scopes by the resolved slug. + const enumerated = yield* searchTools(executor, "", 10, { namespace: "Google Calendar" }); + expect(enumerated.items.map((item) => item.path)).toEqual([ + "google_calendar.org.main.insertEvent", + ]); + expect(enumerated.total).toBe(1); + + // An ambiguous alias keeps today's semantics: exact-slug enumeration + // finds no integration named "google"; ranked search still prefix-matches. + const ambiguous = yield* searchTools(executor, "", 10, { namespace: "google" }); + expect(ambiguous.total).toBe(0); + const prefix = yield* searchTools(executor, "event", 10, { namespace: "google" }); + expect(prefix.items.map((item) => item.integration)).toEqual(["google_calendar"]); + + // Display-name words rank an integration's tools without a namespace. + const byName = yield* searchTools(executor, "gmail send", 5); + expect(byName.items[0]?.path).toBe("google_gmail.org.main.sendMessage"); + + // A discovery object without the catalog (the passthrough path, which + // resolves aliases itself) is unchanged: nothing resolves. + const bare = yield* searchTools({ tools: executor.tools }, "list messages", 5, { + namespace: "gmail", + }); + expect(bare.items).toEqual([]); + }), + ); + it.effect("returns no matches for empty queries instead of listing arbitrary tools", () => Effect.gen(function* () { const executor = yield* makeSearchExecutor(); @@ -2030,3 +2137,147 @@ describe("pause/resume with multiple elicitations", () => { }), ); }); + +describe("connected tool outcome metadata", () => { + it.effect( + "reports real successes, tool failures and policy denials separately from successful paths", + () => + Effect.gen(function* () { + const executor = yield* makeExecutorWith([crmPlugin, errorPlugin] as const); + yield* provision(executor as never, [ + { pluginId: "crm-test", integration: "crm" }, + { pluginId: "error-test", integration: "records" }, + ]); + yield* Effect.addFinalizer(() => executor.close().pipe(Effect.ignore)); + const paths: string[] = []; + const calls: { path: string; status: string }[] = []; + const invoker = makeExecutorToolInvoker(executor, { + invokeOptions: { onElicitation: acceptAll }, + onConnectedToolCall: (path) => paths.push(path), + onConnectedToolOutcome: (path, status) => calls.push({ path, status }), + }); + yield* invoker.invoke({ + path: "crm.org.main.listContacts", + args: { secret: "argument-secret" }, + }); + yield* invoker.invoke({ path: "records.org.main.queryRows", args: {} }); + yield* executor.policies.create({ + owner: "org", + pattern: "crm.org.main.listContacts", + action: "block", + }); + const blocked = yield* invoker.invoke({ path: "crm.org.main.listContacts", args: {} }); + expect(blocked).toMatchObject({ ok: false, error: { code: "tool_blocked" } }); + expect(paths).toEqual(["crm.org.main.listContacts"]); + expect(calls).toEqual([ + { path: "crm.org.main.listContacts", status: "ok" }, + { path: "records.org.main.queryRows", status: "error" }, + { path: "crm.org.main.listContacts", status: "blocked" }, + ]); + yield* invoker.invoke({ path: "bad.org.main.secret@example.test", args: {} }); + expect(calls).toHaveLength(3); + expect(JSON.stringify(calls)).not.toMatch(/secret|DisplayName|args/); + }).pipe(Effect.scoped), + ); + + it.effect("reports a real approval decline as blocked even when invocation fails", () => + Effect.gen(function* () { + const executor = yield* makeExecutorWith([crmPlugin] as const); + yield* provision(executor as never, [{ pluginId: "crm-test", integration: "crm" }]); + yield* Effect.addFinalizer(() => executor.close().pipe(Effect.ignore)); + const calls: { path: string; status: string }[] = []; + const invoker = makeExecutorToolInvoker(executor, { + invokeOptions: { + onElicitation: () => Effect.succeed(ElicitationResponse.make({ action: "decline" })), + }, + onConnectedToolOutcome: (path, status) => calls.push({ path, status }), + }); + yield* Effect.exit( + invoker.invoke({ + path: "crm.org.main.createContact", + args: { email: "argument-secret@example.test" }, + }), + ); + expect(calls).toEqual([{ path: "crm.org.main.createContact", status: "blocked" }]); + }).pipe(Effect.scoped), + ); + + it.effect("reports an invocation defect as an error without passing the cause to telemetry", () => + Effect.gen(function* () { + const executor = yield* makeExecutorWith([unmarkedErrorPlugin] as const); + yield* provision(executor as never, [ + { pluginId: "unmarked-error-test", integration: "unmarked" }, + ]); + yield* Effect.addFinalizer(() => executor.close().pipe(Effect.ignore)); + const calls: { path: string; status: string }[] = []; + const invoker = makeExecutorToolInvoker(executor, { + invokeOptions: { onElicitation: acceptAll }, + onConnectedToolOutcome: (path, status) => calls.push({ path, status }), + }); + yield* Effect.exit(invoker.invoke({ path: "unmarked.org.main.explode", args: {} })); + expect(calls).toEqual([{ path: "unmarked.org.main.explode", status: "error" }]); + }).pipe(Effect.scoped), + ); + + for (const mode of ["inline", "pausable", "operator"] as const) { + it.effect( + `retains real tool outcomes on completed and script-error responses in ${mode} mode`, + () => + Effect.gen(function* () { + const executor = yield* makeExecutorWith([crmPlugin, errorPlugin] as const); + yield* provision(executor as never, [ + { pluginId: "crm-test", integration: "crm" }, + { pluginId: "error-test", integration: "records" }, + ]); + yield* executor.policies.create({ + owner: "org", + pattern: "crm.org.main.createContact", + action: "block", + }); + const engine = createExecutionEngine({ executor, codeExecutor }); + yield* Effect.addFinalizer(() => + engine.shutdown.pipe(Effect.andThen(executor.close()), Effect.ignore), + ); + const run = (code: string) => + mode === "inline" + ? engine.execute(code, { onElicitation: acceptAll }) + : engine.executeWithPause(code, { autoApprove: mode === "operator" }).pipe( + Effect.map((outcome) => { + return outcome.status === "completed" + ? outcome.result + : { result: null, error: "Unexpected pause" }; + }), + ); + const completed = yield* run( + [ + "await tools.crm.org.main.listContacts({});", + "await tools.records.org.main.queryRows({});", + "await tools.crm.org.main.createContact({ email: 'argument-secret@example.test' });", + "await tools.records.org.main.queryRows({});", + "return 42;", + ].join("\n"), + ); + expect(completed.error).toBeUndefined(); + expect(completed.toolPaths).toEqual(["crm.org.main.listContacts"]); + expect(formatExecuteResult(completed).structured).toMatchObject({ + status: "completed", + toolName: "crm.org.main.listContacts", + toolCalls: [ + { path: "crm.org.main.listContacts", status: "ok" }, + { path: "records.org.main.queryRows", status: "error" }, + { path: "crm.org.main.createContact", status: "blocked" }, + ], + }); + const failed = yield* run( + "await tools.crm.org.main.listContacts({}); throw new Error('script-secret');", + ); + expect(failed.error).toContain("script-secret"); + expect(formatExecuteResult(failed).structured).toMatchObject({ + status: "error", + toolPaths: ["crm.org.main.listContacts"], + toolCalls: [{ path: "crm.org.main.listContacts", status: "ok" }], + }); + }).pipe(Effect.scoped), + ); + } +}); diff --git a/packages/core/execution/src/tool-invoker.ts b/packages/core/execution/src/tool-invoker.ts index a0c59a2cb8..5cc4a42dc5 100644 --- a/packages/core/execution/src/tool-invoker.ts +++ b/packages/core/execution/src/tool-invoker.ts @@ -19,6 +19,7 @@ import { } from "@executor-js/sdk/core"; import type { SandboxToolInvoker } from "@executor-js/codemode-core"; import { ExecutionToolError } from "./errors"; +import { integrationAliasText, resolveIntegrationAlias } from "./integration-aliases"; const OPAQUE_DEFECT_MESSAGE = "Internal tool error"; const TOOL_DESCRIBE_SUGGESTION_LIMIT = 5; @@ -316,6 +317,7 @@ export const makeExecutorToolInvoker = ( options: { readonly invokeOptions: InvokeOptions; readonly onConnectedToolCall?: (path: string) => void; + readonly onConnectedToolOutcome?: (path: string, status: "ok" | "error" | "blocked") => void; }, ): SandboxToolInvoker => ({ invoke: Effect.fn("mcp.tool.dispatch")(function* ({ path, args }) { @@ -325,6 +327,17 @@ export const makeExecutorToolInvoker = ( }); const address = pathToAddress(path); + // Telemetry accepts identifiers only, even when the sandbox supplies a malformed path. + const telemetryPath = + String(address).length <= 512 && + /^tools\.[a-zA-Z0-9_-]{1,64}\.(?:org|user)\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_.-]+$/.test( + String(address), + ) + ? addressToPath(String(address)) + : undefined; + const reportOutcome = (status: "ok" | "error" | "blocked") => { + if (telemetryPath) options.onConnectedToolOutcome?.(telemetryPath, status); + }; const result = yield* executor.execute(address, args, options.invokeOptions).pipe( Effect.catchTag("CredentialResolutionError", (err) => Effect.succeed( @@ -343,6 +356,7 @@ export const makeExecutorToolInvoker = ( return Effect.succeed(ToolResult.fail(expected)); } if (isElicitationDeclinedError(err)) { + reportOutcome("blocked"); return Effect.fail( new ExecutionToolError({ message: `Tool "${addressToPath(String(err.address))}" requires approval but the request was ${err.action === "cancel" ? "cancelled" : "declined"} by the user.`, @@ -356,6 +370,7 @@ export const makeExecutorToolInvoker = ( // can't leak through Error.message into the sandbox. The full // cause is logged with the same correlation id so operators can // still trace the failure. + reportOutcome("error"); const correlationId = newCorrelationId(); return Effect.logError("tool dispatch failed", cause).pipe( Effect.annotateLogs({ @@ -381,6 +396,13 @@ export const makeExecutorToolInvoker = ( // Expected failures resolve through the success channel, so without the // outcome annotation the dispatch span reads as healthy even when the // caller hit an upstream error or auth wall. + reportOutcome( + !isToolResult(result) || result.ok + ? "ok" + : result.error.code === "tool_blocked" + ? "blocked" + : "error", + ); yield* annotateToolResultOutcome(result); const connectedToolPath = parseToolAddress(String(address)) ? addressToPath(String(address)) @@ -601,7 +623,11 @@ const matchesNamespace = (tool: SearchableTool, namespace?: string): boolean => return isPrefixMatch(integrationTokens) || isPrefixMatch(pathTokens); }; -const scoreToolMatch = (tool: SearchableTool, query: string): ToolDiscoveryResult | null => { +const scoreToolMatch = ( + tool: SearchableTool, + query: string, + integrationAliases?: ReadonlyMap, +): ToolDiscoveryResult | null => { const normalizedQuery = normalizeSearchText(query); const queryTokens = tokenizeSearchText(query); @@ -610,7 +636,12 @@ const scoreToolMatch = (tool: SearchableTool, query: string): ToolDiscoveryResul } const path = prepareField(tool.path); - const integration = prepareField(tool.integration); + // The integration field carries the slug plus any alias text the caller + // knows for it (its display name), so a query word that names the + // integration the way a person would ("gmail", "Google Calendar") scores + // and counts toward coverage the same as the slug's own tokens. + const alias = integrationAliases?.get(tool.integration); + const integration = prepareField(alias ? `${tool.integration} ${alias}` : tool.integration); const name = prepareField(tool.name); const description = prepareField(tool.description); @@ -672,10 +703,23 @@ const scoreToolMatch = (tool: SearchableTool, query: string): ToolDiscoveryResul /** What `tools.search()` calls inside the sandbox. */ export const searchTools = Effect.fn("executor.tools.search")(function* ( - executor: { readonly tools: Pick }, + executor: { + readonly tools: Pick; + /** The integration catalog, when the caller has one. With it, `namespace` + * accepts an unambiguous alias (`gmail` → `google_gmail`) and query words + * that name an integration by display name rank its tools. Passthrough + * resolves both before calling and passes a discovery object without it. */ + readonly integrations?: Pick; + }, query: string, limit = 12, - options?: { readonly namespace?: string; readonly offset?: number }, + options?: { + readonly namespace?: string; + readonly offset?: number; + /** Extra searchable text per integration slug (its display name), so + * query words naming the integration rank its tools. */ + readonly integrationAliases?: ReadonlyMap; + }, ) { const offset = options?.offset ?? 0; yield* Effect.annotateCurrentSpan({ @@ -701,6 +745,34 @@ export const searchTools = Effect.fn("executor.tools.search")(function* ( } satisfies PagedResult; } + // The namespace an agent typed, resolved against the catalog when one is + // available: an unambiguous alias becomes the slug for both enumeration and + // ranked search. An exact slug, an ambiguous alias (`google` naming two + // integrations) and an unknown name keep today's token-prefix behaviour, so + // no existing result changes. + let namespace = hasNamespace ? options!.namespace!.trim() : undefined; + let integrationAliases = options?.integrationAliases; + if (executor.integrations !== undefined) { + const catalog = (yield* executor.integrations.list().pipe( + Effect.mapError( + (cause) => + new ExecutionToolError({ + message: "Failed to list integrations for search", + cause, + }), + ), + )).map((integration) => ({ + slug: String(integration.slug), + name: integration.name, + description: integration.description, + })); + if (namespace !== undefined) { + const resolved = resolveIntegrationAlias(namespace, catalog); + if (resolved.kind === "alias") namespace = resolved.slug; + } + integrationAliases ??= integrationAliasText(catalog); + } + const all = yield* executor.tools.list({ includeAnnotations: false }).pipe( Effect.mapError( (cause) => @@ -722,7 +794,7 @@ export const searchTools = Effect.fn("executor.tools.search")(function* ( // `executor.integrations.list`'s per-integration toolCount. const ranked: readonly ToolDiscoveryResult[] = emptyQuery ? searchable - .filter((tool) => tool.integration === options?.namespace?.trim()) + .filter((tool) => tool.integration === namespace) .sort((left, right) => left.path.localeCompare(right.path)) .map((tool) => ({ path: tool.path, @@ -732,8 +804,8 @@ export const searchTools = Effect.fn("executor.tools.search")(function* ( ...(tool.description !== undefined ? { description: tool.description } : {}), })) : searchable - .filter((tool: SearchableTool) => matchesNamespace(tool, options?.namespace)) - .map((tool: SearchableTool) => scoreToolMatch(tool, query)) + .filter((tool: SearchableTool) => matchesNamespace(tool, namespace)) + .map((tool: SearchableTool) => scoreToolMatch(tool, query, integrationAliases)) .filter(Predicate.isNotNull) .sort((left, right) => right.score - left.score || left.path.localeCompare(right.path)); diff --git a/packages/core/sdk/src/catalog-refresh.test.ts b/packages/core/sdk/src/catalog-refresh.test.ts new file mode 100644 index 0000000000..1b385f1ce2 --- /dev/null +++ b/packages/core/sdk/src/catalog-refresh.test.ts @@ -0,0 +1,572 @@ +import { describe, expect, it } from "@effect/vitest"; +import { Deferred, Effect, Fiber } from "effect"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { createExecutor } from "./executor"; +import { ConnectionName, IntegrationSlug, NO_AUTH_TEMPLATE, ToolName } from "./ids"; +import { definePlugin, type ResolveToolsResult } from "./plugin"; +import { makeTestConfig, memoryCredentialsPlugin } from "./testing"; + +const integration = IntegrationSlug.make("refresh-fixture"); +const ref = { owner: "org" as const, integration, name: ConnectionName.make("main") }; +const catalog: ResolveToolsResult = { + tools: [ + { + name: ToolName.make("list"), + description: "List items", + inputSchema: { type: "object", properties: { id: { $ref: "#/$defs/Id" } } }, + outputSchema: { type: "array", items: { type: "string" } }, + annotations: { requiresApproval: false }, + }, + { name: ToolName.make("alpha"), description: "Another tool" }, + ], + definitions: { Id: { type: "string" } }, +}; + +describe("file-backed catalog refresh", () => { + it.effect("keeps an unchanged refresh stamp after an unrelated transaction rolls back", () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + const held = yield* Deferred.make(); + const release = yield* Deferred.make(); + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + remoteToolCatalog: true, + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => Effect.succeed(catalog), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + rollback: () => + ctx.transaction( + Effect.gen(function* () { + yield* ctx.connections.update(ref, { description: "Rolled back description" }); + yield* Deferred.succeed(held, undefined); + yield* Deferred.await(release); + return yield* Effect.fail("rollback" as const); + }), + ), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor(config); + yield* executor["refresh-fixture"].seed(); + yield* executor.connections.create({ ...ref, template: NO_AUTH_TEMPLATE, inputs: {} }); + const before = yield* executor.connections.get(ref); + const tools = yield* Effect.promise(() => config.db.findMany("tool", {})); + const definitions = yield* Effect.promise(() => config.db.findMany("definition", {})); + yield* Effect.promise(() => + config.db.updateMany("connection", { + set: { tools_synced_at: null }, + }), + ); + const writer = yield* Effect.forkChild(executor["refresh-fixture"].rollback()); + yield* Deferred.await(held); + const refresh = yield* Effect.forkChild(executor.connections.refresh(ref)); + yield* Effect.promise(() => new Promise((resolve) => setTimeout(resolve, 50))); + const finishedEarly = refresh.pollUnsafe() !== undefined; + yield* Deferred.succeed(release, undefined); + expect(yield* Fiber.join(writer).pipe(Effect.flip)).toBe("rollback"); + yield* Fiber.join(refresh); + expect(finishedEarly).toBe(false); + expect( + (yield* Effect.promise(() => config.db.findFirst("connection", {})))?.tools_synced_at, + ).not.toBeNull(); + expect((yield* executor.connections.get(ref))?.description).toBe(before?.description); + expect(yield* Effect.promise(() => config.db.findMany("tool", {}))).toEqual(tools); + expect(yield* Effect.promise(() => config.db.findMany("definition", {}))).toEqual( + definitions, + ); + }), + ), + ); + + for (const [mode, failure] of [ + ["explicit", "incomplete"], + ["background", "incomplete"], + ["background", "empty"], + ] as const) { + it.effect(`keeps ${mode} ${failure} refresh stamps after an unrelated rollback`, () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + const held = yield* Deferred.make(); + const release = yield* Deferred.make(); + const completions: Promise[] = []; + let failing = false; + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + remoteToolCatalog: true, + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => + Effect.sync(() => + failing + ? { + tools: [], + incomplete: failure === "incomplete", + incompleteReason: "temporary outage", + } + : catalog, + ), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + rollback: () => + ctx.transaction( + Effect.gen(function* () { + yield* ctx.connections.update(ref, { description: "Rolled back description" }); + yield* Deferred.succeed(held, undefined); + yield* Deferred.await(release); + return yield* Effect.fail("rollback" as const); + }), + ), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor({ + ...config, + toolsSyncGraceMs: 0, + waitUntil: (promise) => completions.push(promise), + }); + yield* executor["refresh-fixture"].seed(); + yield* executor.connections.create({ ...ref, template: NO_AUTH_TEMPLATE, inputs: {} }); + yield* Effect.promise(() => Promise.all(completions)); + completions.length = 0; + const before = yield* executor.connections.get(ref); + const tools = yield* Effect.promise(() => config.db.findMany("tool", {})); + const definitions = yield* Effect.promise(() => config.db.findMany("definition", {})); + yield* Effect.promise(() => + config.db.updateMany("connection", { + set: { tools_synced_at: null }, + }), + ); + const writer = yield* Effect.forkChild(executor["refresh-fixture"].rollback()); + yield* Deferred.await(held); + failing = true; + const finish = + mode === "explicit" + ? executor.connections.refresh(ref).pipe(Effect.asVoid) + : executor.tools + .list() + .pipe(Effect.andThen(Effect.promise(() => Promise.all(completions)))); + const refresh = yield* Effect.forkChild(finish); + yield* Effect.promise(() => new Promise((resolve) => setTimeout(resolve, 50))); + const finishedEarly = refresh.pollUnsafe() !== undefined; + yield* Deferred.succeed(release, undefined); + expect(yield* Fiber.join(writer).pipe(Effect.flip)).toBe("rollback"); + yield* Fiber.join(refresh); + expect(finishedEarly).toBe(false); + const row = yield* Effect.promise(() => config.db.findFirst("connection", {})); + expect(row?.tools_synced_at).not.toBeNull(); + expect(row?.last_health).not.toBeNull(); + expect((yield* executor.connections.get(ref))?.description).toBe(before?.description); + expect(yield* Effect.promise(() => config.db.findMany("tool", {}))).toEqual(tools); + expect(yield* Effect.promise(() => config.db.findMany("definition", {}))).toEqual( + definitions, + ); + }), + ), + ); + } + + it.effect("prioritizes cached reads and releases discovery if a reader is interrupted", () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + const reading = yield* Deferred.make(); + const releaseRead = yield* Deferred.make(); + const completions: Promise[] = []; + let holdRead = false; + let discoveries = 0; + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + remoteToolCatalog: true, + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => + Effect.sync(() => { + discoveries += 1; + return catalog; + }), + toolPolicyProvider: () => ({ + list: () => + Effect.gen(function* () { + if (holdRead) { + yield* Deferred.succeed(reading, undefined); + yield* Deferred.await(releaseRead); + } + return [{ id: "allow", pattern: "*", action: "approve" as const, position: "a0" }]; + }), + }), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor({ + ...config, + toolsSyncGraceMs: 0, + waitUntil: (promise) => completions.push(promise), + }); + yield* executor["refresh-fixture"].seed(); + yield* executor.connections.create({ ...ref, template: NO_AUTH_TEMPLATE, inputs: {} }); + const before = yield* executor.tools.list(); + yield* Effect.promise(() => Promise.all(completions)); + yield* Effect.promise(() => + config.db.updateMany("connection", { set: { tools_synced_at: null } }), + ); + discoveries = 0; + holdRead = true; + const reader = yield* Effect.forkChild(executor.tools.list()); + yield* Deferred.await(reading); + yield* Effect.promise(() => new Promise((resolve) => setTimeout(resolve, 0))); + expect(discoveries).toBe(0); + yield* Fiber.interrupt(reader); + holdRead = false; + yield* Effect.promise(() => Promise.all(completions)); + expect(discoveries).toBe(1); + expect(yield* executor.tools.list()).toEqual(before); + }), + ), + ); + + it.effect("keeps identical tool and definition rows while reads overlap discovery", () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + const started = yield* Deferred.make(); + const release = yield* Deferred.make(); + let refreshing = false; + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + remoteToolCatalog: true, + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => + Effect.gen(function* () { + if (refreshing) { + yield* Deferred.succeed(started, undefined); + yield* Deferred.await(release); + } + return catalog; + }), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor({ ...config, toolsSyncGraceMs: 0 }); + yield* executor["refresh-fixture"].seed(); + yield* executor.connections.create({ ...ref, template: NO_AUTH_TEMPLATE, inputs: {} }); + const beforeTools = yield* Effect.promise(() => config.db.findMany("tool", {})); + const beforeDefinitions = yield* Effect.promise(() => config.db.findMany("definition", {})); + const before = yield* executor.tools.list(); + refreshing = true; + const refresh = yield* Effect.forkChild(executor.connections.refresh(ref)); + yield* Deferred.await(started); + const during = yield* executor.tools.list(); + expect(during).toEqual(before); + yield* Deferred.succeed(release, undefined); + yield* Fiber.join(refresh); + expect(yield* executor.tools.list()).toEqual(before); + expect(yield* Effect.promise(() => config.db.findMany("tool", {}))).toEqual(beforeTools); + expect(yield* Effect.promise(() => config.db.findMany("definition", {}))).toEqual( + beforeDefinitions, + ); + }), + ), + ); + + it.effect("serves another request while a wave of failed refreshes is still running", () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + const started = yield* Deferred.make(); + const completions: Promise[] = []; + let refreshing = false; + let discoveries = 0; + const count = 30; + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + remoteToolCatalog: true, + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => + Effect.gen(function* () { + if (!refreshing) return catalog; + discoveries += 1; + yield* Deferred.succeed(started, undefined); + return { tools: [], incomplete: true, incompleteReason: "fixture offline" }; + }), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor({ + ...config, + toolsSyncGraceMs: 0, + waitUntil: (promise) => completions.push(promise), + }); + yield* executor["refresh-fixture"].seed(); + for (let index = 0; index < count; index += 1) { + yield* executor.connections.create({ + ...ref, + name: ConnectionName.make(`connection-${index}`), + template: NO_AUTH_TEMPLATE, + inputs: {}, + }); + } + const before = yield* executor.tools.list(); + yield* Effect.promise(() => Promise.all(completions)); + yield* Effect.promise(() => + config.db.updateMany("connection", { set: { tools_synced_at: null } }), + ); + refreshing = true; + const read = yield* Effect.forkChild(executor.tools.list()); + yield* Deferred.await(started); + // Schedule a new incoming request through the event loop, as HTTP does. + yield* Effect.promise(() => new Promise((resolve) => setTimeout(resolve, 0))); + const observed = discoveries; + const during = yield* executor.tools.list(); + expect(yield* Fiber.join(read)).toEqual(before); + expect(observed).toBeLessThan(count); + expect(during).toEqual(before); + yield* Effect.promise(() => Promise.all(completions)); + expect(discoveries).toBe(count); + expect(yield* executor.tools.list()).toEqual(before); + }), + ), + ); + + it.effect("an explicit waiter on an incomplete background sync gets the retained schemas", () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + const started = yield* Deferred.make(); + const release = yield* Deferred.make(); + const completions: Promise[] = []; + let refreshing = false; + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + remoteToolCatalog: true, + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => + Effect.gen(function* () { + if (!refreshing) return catalog; + yield* Deferred.succeed(started, undefined); + yield* Deferred.await(release); + return { tools: [], incomplete: true, incompleteReason: "fixture offline" }; + }), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor({ + ...config, + toolsSyncGraceMs: 0, + waitUntil: (promise) => completions.push(promise), + }); + yield* executor["refresh-fixture"].seed(); + yield* executor.connections.create({ ...ref, template: NO_AUTH_TEMPLATE, inputs: {} }); + yield* Effect.promise(() => + config.db.updateMany("connection", { set: { tools_synced_at: null } }), + ); + refreshing = true; + const read = yield* Effect.forkChild(executor.tools.list()); + yield* Deferred.await(started); + const explicit = yield* Effect.forkChild(executor.connections.refresh(ref)); + const during = yield* Fiber.join(read); + expect(during).toHaveLength(catalog.tools.length); + yield* Deferred.succeed(release, undefined); + const result = yield* Fiber.join(explicit); + const retained = result.find((tool) => String(tool.name) === "list"); + expect(retained?.inputSchema).toEqual(catalog.tools[0]?.inputSchema); + expect(retained?.outputSchema).toEqual(catalog.tools[0]?.outputSchema); + expect(retained?.description).toBe("List items"); + yield* Effect.promise(() => Promise.all(completions)); + }), + ), + ); + + it.effect("replaces changed descriptions, schemas, annotations, definitions and tool sets", () => + Effect.scoped( + Effect.gen(function* () { + const dir = yield* Effect.acquireRelease( + Effect.promise(() => mkdtemp(join(tmpdir(), "executor-catalog-"))), + (path) => Effect.promise(() => rm(path, { recursive: true, force: true })), + ); + let listing = catalog; + const fixture = definePlugin(() => ({ + id: "refresh-fixture" as const, + storage: () => ({}), + describeAuthMethods: () => [ + { id: "none", label: "No authentication", kind: "none", template: "none" }, + ], + resolveTools: () => Effect.sync(() => listing), + invokeTool: () => Effect.succeed(null), + extension: (ctx) => ({ + seed: () => + ctx.core.integrations.register({ + slug: integration, + description: "Refresh fixture", + config: {}, + }), + }), + }))(); + const config = makeTestConfig({ + dataDir: dir, + plugins: [memoryCredentialsPlugin(), fixture] as const, + }); + yield* Effect.addFinalizer(() => Effect.promise(() => config.testDb.close())); + const executor = yield* createExecutor(config); + yield* executor["refresh-fixture"].seed(); + yield* executor.connections.create({ ...ref, template: NO_AUTH_TEMPLATE, inputs: {} }); + const variants: readonly ResolveToolsResult[] = [ + { ...catalog, tools: [{ ...catalog.tools[0]!, description: "Changed description" }] }, + { ...catalog, tools: [{ ...catalog.tools[0]!, inputSchema: { type: "string" } }] }, + { ...catalog, tools: [{ ...catalog.tools[0]!, outputSchema: { type: "number" } }] }, + { + ...catalog, + tools: [{ ...catalog.tools[0]!, annotations: { requiresApproval: true } }], + }, + { ...catalog, definitions: { Id: { type: "number" }, Extra: { type: "boolean" } } }, + { ...catalog, definitions: {} }, + { + tools: [...catalog.tools, { name: ToolName.make("added"), description: "Added tool" }], + }, + { tools: [{ name: ToolName.make("replacement"), description: "Replacement tool" }] }, + { tools: [] }, + ]; + for (const variant of variants) { + listing = variant; + const refreshed = yield* executor.connections.refresh(ref); + expect(refreshed.map((tool) => String(tool.name))).toEqual( + variant.tools.map((tool) => String(tool.name)), + ); + const tools = yield* Effect.promise(() => config.db.findMany("tool", {})); + expect( + tools + .map((row) => ({ + name: row.name, + description: row.description, + input_schema: row.input_schema, + output_schema: row.output_schema, + annotations: row.annotations, + })) + .sort((a, b) => String(a.name).localeCompare(String(b.name))), + ).toEqual( + variant.tools + .map((tool) => ({ + name: tool.name, + description: tool.description ?? "", + input_schema: tool.inputSchema ?? null, + output_schema: tool.outputSchema ?? null, + annotations: tool.annotations ?? null, + })) + .sort((a, b) => String(a.name).localeCompare(String(b.name))), + ); + const definitions = yield* Effect.promise(() => config.db.findMany("definition", {})); + expect(Object.fromEntries(definitions.map((row) => [row.name, row.schema]))).toEqual( + variant.definitions ?? {}, + ); + } + }), + ), + ); +}); diff --git a/packages/core/sdk/src/connections.test.ts b/packages/core/sdk/src/connections.test.ts index 51c7f8d110..4ce5ff4ecc 100644 --- a/packages/core/sdk/src/connections.test.ts +++ b/packages/core/sdk/src/connections.test.ts @@ -2353,7 +2353,10 @@ describe("tool catalog sync safety", () => { } return { tools: [ - { name: ToolName.make(`deploy_${String(connection.name)}`), description: "d" }, + { + name: ToolName.make(`deploy_${String(connection.name)}`), + description: latched ? "updated" : "d", + }, ], }; }), @@ -2609,6 +2612,13 @@ const makeHealthHarness = (options?: { return (context: unknown) => interceptHealthWrites((value as (context: unknown) => FumaDb).call(target, context)); } + if (prop === "transaction") { + return (run: (tx: FumaDb) => Promise) => + (value as (run: (tx: FumaDb) => Promise) => Promise).call( + target, + (tx) => run(interceptHealthWrites(tx)), + ); + } if (prop === "updateMany") { return async ( table: string, diff --git a/packages/core/sdk/src/executor.test.ts b/packages/core/sdk/src/executor.test.ts index fb2835a529..1076a9ddcb 100644 --- a/packages/core/sdk/src/executor.test.ts +++ b/packages/core/sdk/src/executor.test.ts @@ -775,6 +775,48 @@ describe("createExecutor", () => { }), ); + it.effect("tools.schemas serves the same views and definitions as tools.schema", () => + Effect.gen(function* () { + const executor = yield* makeTestExecutor({ + plugins: [demoPlugin] as const, + coreTools: { webBaseUrl: "http://localhost:3000" }, + }); + yield* executor.demo.seed(); + yield* executor.connections.create({ + owner: "org", + name: CONN, + integration: INTEG, + template: TEMPLATE, + from: { + provider: ProviderKey.make("memory"), + id: ProviderItemId.make("v"), + }, + }); + // A catalog tool with definitions, a schemaless one, a static core + // tool, and a missing one, in one batch. + const addresses = [ + addr("inspect"), + addr("run"), + ToolAddress.make("executor.coreTools.integrations.list"), + addr("absent"), + ]; + const batched = yield* executor.tools.schemas(addresses); + const single = yield* Effect.forEach(addresses, (address) => executor.tools.schema(address)); + expect(batched).toEqual(single); + expect(Object.keys(batched[0]?.schemaDefinitions ?? {}).sort()).toEqual([ + "Cat", + "Collar", + "Dog", + "Owner", + "Pet", + ]); + expect(batched[0]?.inputTypeScript).toContain("pet"); + expect(batched[1]?.name).toBe("run"); + expect(batched[2]?.name).toBe("coreTools.integrations.list"); + expect(batched[3]).toBeNull(); + }), + ); + it.effect("execute dispatches a connection-produced tool to the owning plugin", () => Effect.gen(function* () { const executor = yield* makeTestExecutor({ diff --git a/packages/core/sdk/src/executor.ts b/packages/core/sdk/src/executor.ts index cd89ab9075..786fe5758a 100644 --- a/packages/core/sdk/src/executor.ts +++ b/packages/core/sdk/src/executor.ts @@ -157,8 +157,10 @@ import { isValidPattern, matchPattern, positionForNewPattern, + prepareToolPolicies, resolveEffectivePolicy, rowToToolPolicy, + type PreparedToolPolicies, type CreateToolPolicyInput, type EffectivePolicy, type RemoveToolPolicyInput, @@ -469,7 +471,18 @@ export type Executor = { readonly tools: { readonly list: (filter?: ToolListFilter) => Effect.Effect; - readonly schema: (address: ToolAddress) => Effect.Effect; + readonly schema: ( + address: ToolAddress, + options?: ToolSchemaOptions, + ) => Effect.Effect; + /** The `schema` view for several addresses in one read: one policy + * snapshot for the set, one catalog read and one definitions read per + * connection. Answers in input order, `null` where `schema` answers + * `null`. */ + readonly schemas: ( + addresses: readonly ToolAddress[], + options?: ToolSchemaOptions, + ) => Effect.Effect; }; readonly providers: { @@ -853,6 +866,12 @@ export interface ExecutorConfig[], + proposed: readonly Record[], + columns: readonly string[], +): boolean => { + if (stored.length !== proposed.length) return false; + const byName = new Map(proposed.map((row) => [row["name"], row])); + return stored.every((row) => { + const candidate = byName.get(row["name"]); + return ( + candidate !== undefined && + JSON.stringify(columns.map((column) => row[column])) === + JSON.stringify(columns.map((column) => candidate[column])) + ); + }); +}; + // Projects a tool's annotations onto the schema view. Plugins persist extra // keys alongside the declared contract (the mcp plugin stores its upstream tool // name and `_meta` there so they survive to invokeTool), so the three declared @@ -3519,12 +3558,25 @@ export const createExecutor = (effect: Effect.Effect) => catalogPersistLock.withPermits(1)(transaction(effect)); + let catalogReaders = 0; + let catalogReadersDone: Deferred.Deferred | null = null; + const yieldBackgroundCatalog = Effect.sleep(0).pipe( + Effect.andThen( + Effect.suspend(() => + catalogReadersDone === null ? Effect.void : Deferred.await(catalogReadersDone), + ), + ), + ); + const produceConnectionToolsUnshared = ( integrationRow: IntegrationRow, ref: ConnectionRef, mode: () => "explicit" | "background", - ): Effect.Effect => + ): Effect.Effect => Effect.gen(function* () { + // A detached fiber still shares the request's event loop. Give pending + // HTTP I/O a turn between background listings, including fast failures. + if (mode() === "background") yield* yieldBackgroundCatalog; const runtime = runtimes.get(integrationRow.plugin_id); const keys = yield* Effect.try({ try: () => ownedKeys(ref.owner), @@ -3586,30 +3638,32 @@ export const createExecutor = - findConnectionRow(ref).pipe( - Effect.flatMap((fresh) => - fresh === null - ? Effect.void - : core - .updateMany("connection", { - where: (b: AnyCb) => - b.and( - connectionWhere(b), - b("updated_at", "=", fresh.updated_at), - fresh.tools_synced_at == null - ? b.isNull("tools_synced_at") - : b("tools_synced_at", "=", fresh.tools_synced_at), - ), - set: - oauthReauthRequiredFromProviderState(fresh.provider_state) !== null - ? { tools_synced_at: Date.now() } - : { - tools_synced_at: Date.now(), - last_health: health ?? toolSyncHealth(reason), - updated_at: new Date(), - }, - }) - .pipe(Effect.asVoid), + transaction( + findConnectionRow(ref).pipe( + Effect.flatMap((fresh) => + fresh === null + ? Effect.void + : core + .updateMany("connection", { + where: (b: AnyCb) => + b.and( + connectionWhere(b), + b("updated_at", "=", fresh.updated_at), + fresh.tools_synced_at == null + ? b.isNull("tools_synced_at") + : b("tools_synced_at", "=", fresh.tools_synced_at), + ), + set: + oauthReauthRequiredFromProviderState(fresh.provider_state) !== null + ? { tools_synced_at: Date.now() } + : { + tools_synced_at: Date.now(), + last_health: health ?? toolSyncHealth(reason), + updated_at: new Date(), + }, + }) + .pipe(Effect.asVoid), + ), ), ); @@ -3668,6 +3722,8 @@ export const createExecutor = rowToTool(row as ConnectionToolRow)); } @@ -3690,8 +3750,8 @@ export const createExecutor = 0) { + const keptCount = yield* core.count("tool", { where }); + if (keptCount > 0) { const reason = "background tool sync produced an authoritative empty catalog for a connection with existing tools"; yield* stampSyncedWithHealth(reason); @@ -3699,8 +3759,10 @@ export const createExecutor = rowToTool(row as ConnectionToolRow)); } } @@ -3734,16 +3796,51 @@ export const createExecutor = + catalogRowsMatch(storedTools, toolRows, toolColumns) && + catalogRowsMatch(storedDefinitions, definitionRows, definitionColumns), + catch: (cause) => storageFailureFromUnknown("comparing tool catalog", cause), + }); + if (mode() === "background") yield* yieldBackgroundCatalog; + if (unchanged) { + // Even one UPDATE must use the transaction queue. Otherwise it + // can join an unrelated SQLite transaction and roll back with it. + yield* transaction(stampSynced(existingRow)); + } else { + yield* transaction( + Effect.gen(function* () { + yield* core.deleteMany("tool", { where }); + yield* core.deleteMany("definition", { where }); + yield* core.createMany("tool", toolRows); + yield* core.createMany("definition", definitionRows); + yield* stampSynced(existingRow); + }), + ); + } }), ); + if (mode() === "background") return null; return result.tools.map((tool: ToolDef) => rowToTool( { @@ -3768,7 +3865,7 @@ export const createExecutor = ; + readonly deferred: Deferred.Deferred; mode: "explicit" | "background"; } const toolProductionInFlight = new Map(); @@ -3779,14 +3876,33 @@ export const createExecutor = => Effect.suspend(() => { const key = `${ref.owner}:${String(ref.integration)}:${String(ref.name)}`; + const awaitProduction = (entry: ToolProductionInFlight) => + Deferred.await(entry.deferred).pipe( + Effect.flatMap((tools) => + tools !== null + ? Effect.succeed(tools) + : requestedMode === "background" + ? Effect.succeed([] as readonly Tool[]) + : core + .findMany("tool", { + where: (b: AnyCb) => + b.and( + byOwner(ref.owner)(b), + b("integration", "=", String(ref.integration)), + b("connection", "=", String(ref.name)), + ), + }) + .pipe(Effect.map((rows) => rows.map((row) => rowToTool(row)))), + ), + ); const existing = toolProductionInFlight.get(key); if (existing) { if (requestedMode === "explicit") existing.mode = "explicit"; - return Deferred.await(existing.deferred); + return awaitProduction(existing); } const entry: ToolProductionInFlight = { - deferred: Deferred.makeUnsafe(), + deferred: Deferred.makeUnsafe(), mode: requestedMode, }; toolProductionInFlight.set(key, entry); @@ -3795,7 +3911,7 @@ export const createExecutor = Deferred.done(entry.deferred, exit)), Effect.ensuring(Effect.sync(() => void toolProductionInFlight.delete(key))), ); - return Effect.forkDetach(run).pipe(Effect.andThen(Deferred.await(entry.deferred))); + return Effect.forkDetach(run).pipe(Effect.andThen(awaitProduction(entry))); }); // ------------------------------------------------------------------ @@ -5391,7 +5507,12 @@ export const createExecutor = b.id ? 1 : 0; }; + // `sortedRules` must already be in `compareProviderPolicyRule` order; the + // rule set sorts once per snapshot instead of once per tool. const resolveProviderPolicyFromRules = ( toolId: string, - rules: readonly ToolPolicyProviderRule[], + sortedRules: readonly ToolPolicyProviderRule[], ): EffectivePolicy => { - for (const rule of [...rules].sort(compareProviderPolicyRule)) { + for (const rule of sortedRules) { if (!matchPattern(rule.pattern, toolId)) continue; return { action: rule.action, @@ -5436,7 +5559,13 @@ export const createExecutor = => + // `forToolIds` narrows the global rule read to rows that can match those + // tools: an exact pattern (no `*` segment) matches only when it equals a + // tool id, so every other exact row is irrelevant to their resolution. The + // remaining rows keep their relative order, so each answer is unchanged. + const listActivePolicyRuleSet = ( + forToolIds?: string | readonly string[], + ): Effect.Effect => activeToolPolicyProvider ? // Batched per-operation resolver: fetch all policy + connection state // once, then resolve every tool in this operation against that @@ -5458,32 +5587,61 @@ export const createExecutor = ({ kind: "provider" as const, provider: activeToolPolicyProvider!, - rules, + rules: [...rules].sort(compareProviderPolicyRule), })), ) : core - .findMany("tool_policy", {}) - .pipe(Effect.map((rows) => ({ kind: "global" as const, rows }))); + .findMany( + "tool_policy", + forToolIds === undefined + ? {} + : { + where: (b: AnyCb) => + b.or( + typeof forToolIds === "string" + ? b("pattern", "=", forToolIds) + : forToolIds.length === 0 + ? false + : b("pattern", "in", [...forToolIds]), + b("pattern", "contains", "*"), + ), + }, + ) + .pipe( + Effect.map((rows) => ({ + kind: "global" as const, + rows, + prepared: prepareToolPolicies(rows, ownerRankForRow), + })), + ); - const resolvePolicyFromRuleSet = ( + // Synchronous resolution for rule sets that need no further I/O. Returns + // `undefined` only for a provider that resolves per tool. + const resolvePolicyFromRuleSetSync = ( toolId: string, ruleSet: ActivePolicyRuleSet, defaultRequiresApproval?: boolean, - ): Effect.Effect => + ): EffectivePolicy | undefined => ruleSet.kind === "prepared" - ? Effect.succeed(ruleSet.resolve({ toolId, defaultRequiresApproval })) + ? ruleSet.resolve({ toolId, defaultRequiresApproval }) : ruleSet.kind === "provider" ? ruleSet.provider.resolve - ? ruleSet.provider.resolve({ toolId, defaultRequiresApproval }) - : Effect.succeed(resolveProviderPolicyFromRules(toolId, ruleSet.rules ?? [])) - : Effect.succeed( - resolveEffectivePolicy( - toolId, - ruleSet.rows, - ownerRankForRow, - defaultRequiresApproval, - ), - ); + ? undefined + : resolveProviderPolicyFromRules(toolId, ruleSet.rules ?? []) + : ruleSet.prepared.resolveEffective(toolId, defaultRequiresApproval); + + const resolvePolicyFromRuleSet = ( + toolId: string, + ruleSet: ActivePolicyRuleSet, + defaultRequiresApproval?: boolean, + ): Effect.Effect => { + if (ruleSet.kind === "provider" && ruleSet.provider.resolve) { + return ruleSet.provider.resolve({ toolId, defaultRequiresApproval }); + } + return Effect.succeed( + resolvePolicyFromRuleSetSync(toolId, ruleSet, defaultRequiresApproval)!, + ); + }; // ------------------------------------------------------------------ // Tools (read surface) @@ -5628,7 +5786,11 @@ export const createExecutor = | null = null; const awaitStaleSyncWithinGrace = (graceMs: number) => Effect.gen(function* () { - const fiber = yield* Effect.forkDetach( - syncStaleConnectionTools.pipe( - Effect.catch((error) => - Effect.logWarning("executor stale tool sync scan failed", { - error: describeSyncFailure(error), + const fiber = yield* Effect.suspend(() => { + // Overlapping reads share the scan, including discoveries queued + // behind a slow server. Per-connection single-flight alone cannot + // deduplicate work that has not reached the front of that queue yet. + if (staleSyncFiber !== null) return Effect.succeed(staleSyncFiber); + return Effect.forkDetach( + syncStaleConnectionTools.pipe( + Effect.catch((error) => + Effect.logWarning("executor stale tool sync scan failed", { + error: describeSyncFailure(error), + }), + ), + Effect.ensuring( + Effect.sync(() => { + staleSyncFiber = null; + }), + ), + ), + ).pipe( + Effect.tap((running) => + Effect.sync(() => { + staleSyncFiber = running; }), ), - ), - ); + ); + }); // On hosts that cancel request-scoped I/O once the response settles // (Cloudflare Workers), hand the host the rebuilds' completion so the // catalog still converges after the read stops waiting. @@ -5669,123 +5849,159 @@ export const createExecutor = => - Effect.gen(function* () { - if (toolsSyncGraceMs === null) { - yield* syncStaleConnectionTools; - } else { - yield* awaitStaleSyncWithinGrace(toolsSyncGraceMs); - } - // Projected: the list surface is metadata (address, description, - // annotations) — loading every tool's input/output schema JSON made - // an unbounded list scale with schema bytes, not tool count. - const rows = yield* core.findMany("tool", { - where: (b: AnyCb) => - b.and( - filter?.integration === undefined - ? true - : b("integration", "=", String(filter.integration)), - filter?.owner === undefined ? true : b("owner", "=", filter.owner), - filter?.connection === undefined - ? true - : b("connection", "=", String(filter.connection)), - ), - select: TOOL_INVOCATION_COLUMNS, - }); - const includeBlocked = filter?.includeBlocked ?? false; - const policyRules = yield* listActivePolicyRuleSet(); - // Only tools whose integration is still in the catalog. A tool row - // whose integration was removed is an orphan (a removal that could - // not reach this subject's rows): listing it invites an invoke that - // cannot resolve its config and a reconnect that cannot mint. - const catalogSlugs = yield* listCatalogSlugs(); - const tools: Tool[] = []; - for (const row of rows) { - if (!catalogSlugs.has(String(row.integration))) continue; - const tool = rowToTool(row); - if (!matchesToolFilter(tool, filter)) continue; - if (!includeBlocked) { - const effective = yield* resolvePolicyFromRuleSet( - normalizedPolicyId(tool), - policyRules, - tool.annotations?.requiresApproval, - ); - if (effective.action === "block") continue; - } - tools.push(tool); - } - for (const entry of staticTools.values()) { - const tool = staticToolToTool(entry); - if (!matchesToolFilter(tool, filter)) continue; - if (!includeBlocked) { - const effective = yield* resolvePolicyFromRuleSet( - normalizedPolicyId(tool), - policyRules, - tool.annotations?.requiresApproval, - ); - if (effective.action === "block") continue; + Effect.acquireUseRelease( + Effect.sync(() => { + // Zero-grace reads never wait for discovery. Keep its next CPU/DB + // phase parked until all current catalog readers have finished. + if (toolsSyncGraceMs === 0 && catalogReaders++ === 0) { + catalogReadersDone = Deferred.makeUnsafe(); } - tools.push(tool); - } - return tools; - }); + }), + () => + Effect.gen(function* () { + if (toolsSyncGraceMs === null) { + yield* syncStaleConnectionTools; + } else { + yield* awaitStaleSyncWithinGrace(toolsSyncGraceMs); + } + // Projected: the list surface is metadata (address, description, + // annotations) — loading every tool's input/output schema JSON made + // an unbounded list scale with schema bytes, not tool count. + const rows = yield* core.findMany("tool", { + where: (b: AnyCb) => + b.and( + filter?.integration === undefined + ? true + : b("integration", "=", String(filter.integration)), + filter?.owner === undefined ? true : b("owner", "=", filter.owner), + filter?.connection === undefined + ? true + : b("connection", "=", String(filter.connection)), + ), + select: TOOL_INVOCATION_COLUMNS, + }); + const includeBlocked = filter?.includeBlocked ?? false; + const policyRules = yield* listActivePolicyRuleSet(); + // Only tools whose integration is still in the catalog. A tool row + // whose integration was removed is an orphan (a removal that could + // not reach this subject's rows): listing it invites an invoke that + // cannot resolve its config and a reconnect that cannot mint. + const catalogSlugs = yield* listCatalogSlugs(); + const tools: Tool[] = []; + for (const row of rows) { + if (!catalogSlugs.has(String(row.integration))) continue; + const tool = rowToTool(row); + if (!matchesToolFilter(tool, filter)) continue; + if (!includeBlocked) { + const toolId = normalizedPolicyId(tool); + const requiresApproval = tool.annotations?.requiresApproval; + // Resolve in-line when the rule set needs no I/O; a per-tool + // Effect here dominated large catalog listings. + const effective = + resolvePolicyFromRuleSetSync(toolId, policyRules, requiresApproval) ?? + (yield* resolvePolicyFromRuleSet(toolId, policyRules, requiresApproval)); + if (effective.action === "block") continue; + } + tools.push(tool); + } + for (const entry of staticTools.values()) { + const tool = staticToolToTool(entry); + if (!matchesToolFilter(tool, filter)) continue; + if (!includeBlocked) { + const toolId = normalizedPolicyId(tool); + const requiresApproval = tool.annotations?.requiresApproval; + // Resolve in-line when the rule set needs no I/O; a per-tool + // Effect here dominated large catalog listings. + const effective = + resolvePolicyFromRuleSetSync(toolId, policyRules, requiresApproval) ?? + (yield* resolvePolicyFromRuleSet(toolId, policyRules, requiresApproval)); + if (effective.action === "block") continue; + } + tools.push(tool); + } + return tools; + }), + () => + Effect.suspend(() => { + if (toolsSyncGraceMs !== 0 || --catalogReaders !== 0) return Effect.void; + const done = catalogReadersDone!; + catalogReadersDone = null; + return Deferred.succeed(done, undefined); + }), + ); - const toolSchema = ( + // ------------------------------------------------------------------ + // Schema views. `toolSchema` and `toolSchemas` share these builders, so a + // batched read serves the same view per address as a single read. Policy + // visibility is the caller's: a builder runs only for a tool that already + // resolved as not blocked. + // ------------------------------------------------------------------ + + const staticToolSchemaView = ( address: ToolAddress, - ): Effect.Effect => + tool: Tool, + includeTypeScript: boolean, + ): Effect.Effect => Effect.gen(function* () { - const policyRules = yield* listActivePolicyRuleSet(); - const staticEntry = staticTools.get(String(address)); - if (staticEntry) { - const tool = staticToolToTool(staticEntry); - const effective = yield* resolvePolicyFromRuleSet( - normalizedPolicyId(tool), - policyRules, - tool.annotations?.requiresApproval, - ); - if (effective.action === "block") return null; - const preview = yield* Effect.tryPromise({ - try: () => - buildToolTypeScriptPreview({ - inputSchema: tool.inputSchema, - outputSchema: tool.outputSchema, - defs: new Map(), - }), - catch: (cause) => - storageFailureFromUnknown("Failed to build static tool TypeScript preview", cause), - }).pipe(Effect.option); - return ToolSchemaView.make({ - address, - name: tool.name, - description: tool.description, - inputSchema: tool.inputSchema, - outputSchema: tool.outputSchema, - inputTypeScript: Option.getOrUndefined(preview)?.inputTypeScript, - outputTypeScript: Option.getOrUndefined(preview)?.outputTypeScript, - typeScriptDefinitions: Option.getOrUndefined(preview)?.typeScriptDefinitions, - annotations: toolAnnotationsView(tool.annotations), - }); - } + const preview = includeTypeScript + ? yield* Effect.tryPromise({ + try: () => + buildToolTypeScriptPreview({ + inputSchema: tool.inputSchema, + outputSchema: tool.outputSchema, + defs: new Map(), + }), + catch: (cause) => + storageFailureFromUnknown("Failed to build static tool TypeScript preview", cause), + }).pipe(Effect.option) + : Option.none(); + return ToolSchemaView.make({ + address, + name: tool.name, + description: tool.description, + inputSchema: tool.inputSchema, + outputSchema: tool.outputSchema, + inputTypeScript: Option.getOrUndefined(preview)?.inputTypeScript, + outputTypeScript: Option.getOrUndefined(preview)?.outputTypeScript, + typeScriptDefinitions: Option.getOrUndefined(preview)?.typeScriptDefinitions, + annotations: toolAnnotationsView(tool.annotations), + }); + }); - const parsed = parseToolAddress(String(address)); - if (!parsed) return null; - const row = yield* core.findFirst("tool", { + type ConnectionIdentity = Pick; + + // Every definition a connection persisted, decoded once. A view only + // attaches the subgraph its schemas reference. + const connectionDefinitions = ( + identity: ConnectionIdentity, + ): Effect.Effect, StorageFailure> => + core + .findMany("definition", { where: (b: AnyCb) => b.and( - byOwner(parsed.owner)(b), - b("integration", "=", String(parsed.integration)), - b("connection", "=", String(parsed.connection)), - b("name", "=", String(parsed.tool)), + byOwner(identity.owner)(b), + b("integration", "=", String(identity.integration)), + b("connection", "=", String(identity.connection)), ), - }); - if (!row) return null; - const tool = rowToTool(row); - const effective = yield* resolvePolicyFromRuleSet( - normalizedPolicyId(tool), - policyRules, - tool.annotations?.requiresApproval, + }) + .pipe( + Effect.map((definitionRows) => { + const defs = new Map(); + for (const def of definitionRows) defs.set(def.name, decodeJsonColumn(def.schema)); + return defs; + }), ); - if (effective.action === "block") return null; + const catalogToolSchemaView = (input: { + readonly address: ToolAddress; + readonly parsed: ParsedToolAddress; + readonly row: ToolRow; + readonly tool: Tool; + readonly includeTypeScript: boolean; + readonly definitions: Effect.Effect, StorageFailure>; + }): Effect.Effect => + Effect.gen(function* () { + const { address, parsed, row, tool, includeTypeScript } = input; const runtime = runtimes.get(row.plugin_id); const projected = runtime?.plugin.projectToolSchema ? yield* runtime.plugin @@ -5825,33 +6041,27 @@ export const createExecutor = - b.and( - byOwner(parsed.owner)(b), - b("integration", "=", String(parsed.integration)), - b("connection", "=", String(parsed.connection)), - ), - }); - const defs = new Map(); - for (const def of definitionRows) defs.set(def.name, decodeJsonColumn(def.schema)); - + const defs = yield* input.definitions; const referenced = collectReferencedDefinitions([inputSchema, effectiveOutputSchema], defs); // Compile against the referenced subgraph only. The compiler walks // every definition it is handed (its parser scans the whole `$defs` // map per node), so passing the connection's full component set made a // single describe of a large spec cost seconds of CPU on the shared // session isolate. Unreferenced definitions never appear in the output. - const preview = yield* Effect.tryPromise({ - try: () => - buildToolTypeScriptPreview({ - inputSchema, - outputSchema: effectiveOutputSchema, - defs: new Map(Object.entries(referenced)), - }), - catch: (cause) => - storageFailureFromUnknown("Failed to build tool TypeScript preview", cause), - }).pipe(Effect.option); + // Callers that only need JSON Schema (MCP passthrough search/invoke) + // skip the TypeScript compile, the dominant cost of a schema read. + const preview = includeTypeScript + ? yield* Effect.tryPromise({ + try: () => + buildToolTypeScriptPreview({ + inputSchema, + outputSchema: effectiveOutputSchema, + defs: new Map(Object.entries(referenced)), + }), + catch: (cause) => + storageFailureFromUnknown("Failed to build tool TypeScript preview", cause), + }).pipe(Effect.option) + : Option.none(); const view = preview; return ToolSchemaView.make({ @@ -5877,6 +6087,191 @@ export const createExecutor = => + Effect.gen(function* () { + const includeTypeScript = options?.typeScript ?? true; + const staticEntry = staticTools.get(String(address)); + if (staticEntry) { + const tool = staticToolToTool(staticEntry); + const policyRules = yield* listActivePolicyRuleSet(normalizedPolicyId(tool)); + const effective = yield* resolvePolicyFromRuleSet( + normalizedPolicyId(tool), + policyRules, + tool.annotations?.requiresApproval, + ); + if (effective.action === "block") return null; + return yield* staticToolSchemaView(address, tool, includeTypeScript); + } + + const parsed = parseToolAddress(String(address)); + if (!parsed) return null; + // Same id `normalizedPolicyId` derives from the row this lookup returns. + const policyRules = yield* listActivePolicyRuleSet( + `${parsed.integration}.${parsed.owner}.${parsed.connection}.${parsed.tool}`, + ); + const row = yield* core.findFirst("tool", { + where: (b: AnyCb) => + b.and( + byOwner(parsed.owner)(b), + b("integration", "=", String(parsed.integration)), + b("connection", "=", String(parsed.connection)), + b("name", "=", String(parsed.tool)), + ), + }); + if (!row) return null; + const tool = rowToTool(row); + const effective = yield* resolvePolicyFromRuleSet( + normalizedPolicyId(tool), + policyRules, + tool.annotations?.requiresApproval, + ); + if (effective.action === "block") return null; + return yield* catalogToolSchemaView({ + address, + parsed, + row, + tool, + includeTypeScript, + definitions: connectionDefinitions(parsed), + }); + }); + + // Several addresses in one read. Passthrough search used to call + // `toolSchema` once per hit, and each call read the policy rows, the tool + // row and the connection's whole definition set again. Here the policy + // snapshot is read once for the set, tool rows once per connection, and + // definitions once per connection with a visible hit. Each address still + // resolves against the same rows `toolSchema` would read for it, so the + // views are the same; revocation still lands between calls. + const toolSchemas = ( + addresses: readonly ToolAddress[], + options?: ToolSchemaOptions, + ): Effect.Effect => + Effect.gen(function* () { + if (addresses.length === 0) return []; + const includeTypeScript = options?.typeScript ?? true; + + type Pending = + | { readonly kind: "static"; readonly tool: Tool } + | { readonly kind: "catalog"; readonly parsed: ParsedToolAddress; readonly group: string } + | null; + const groups = new Map }>(); + const pending: Pending[] = addresses.map((address) => { + const staticEntry = staticTools.get(String(address)); + if (staticEntry) return { kind: "static", tool: staticToolToTool(staticEntry) }; + const parsed = parseToolAddress(String(address)); + if (!parsed) return null; + const group = `${parsed.owner}\u0000${parsed.integration}\u0000${parsed.connection}`; + const entry = groups.get(group) ?? { identity: parsed, names: new Set() }; + entry.names.add(String(parsed.tool)); + groups.set(group, entry); + return { kind: "catalog", parsed, group }; + }); + + const policyIds = pending.flatMap((entry) => + entry === null + ? [] + : entry.kind === "static" + ? [normalizedPolicyId(entry.tool)] + : [ + `${entry.parsed.integration}.${entry.parsed.owner}.${entry.parsed.connection}.${entry.parsed.tool}`, + ], + ); + const policyRules = yield* listActivePolicyRuleSet(policyIds); + + const rowsByGroup = new Map>(); + for (const [group, { identity, names }] of groups) { + const rows = yield* core.findMany("tool", { + where: (b: AnyCb) => + b.and( + byOwner(identity.owner)(b), + b("integration", "=", String(identity.integration)), + b("connection", "=", String(identity.connection)), + b("name", "in", [...names]), + ), + }); + rowsByGroup.set(group, new Map(rows.map((row) => [row.name, row]))); + } + + // Resolve visibility first, so definitions load only for a + // connection that serves at least one visible hit. + type Visible = + | { readonly kind: "static"; readonly address: ToolAddress; readonly tool: Tool } + | { + readonly kind: "catalog"; + readonly address: ToolAddress; + readonly parsed: ParsedToolAddress; + readonly group: string; + readonly row: ToolRow; + readonly tool: Tool; + } + | null; + const visible: Visible[] = []; + for (const [index, entry] of pending.entries()) { + const address = addresses[index]!; + if (entry === null) { + visible.push(null); + continue; + } + if (entry.kind === "static") { + const effective = yield* resolvePolicyFromRuleSet( + normalizedPolicyId(entry.tool), + policyRules, + entry.tool.annotations?.requiresApproval, + ); + visible.push( + effective.action === "block" ? null : { kind: "static", address, tool: entry.tool }, + ); + continue; + } + const row = rowsByGroup.get(entry.group)?.get(String(entry.parsed.tool)); + if (!row) { + visible.push(null); + continue; + } + const tool = rowToTool(row); + const effective = yield* resolvePolicyFromRuleSet( + normalizedPolicyId(tool), + policyRules, + tool.annotations?.requiresApproval, + ); + visible.push( + effective.action === "block" + ? null + : { kind: "catalog", address, parsed: entry.parsed, group: entry.group, row, tool }, + ); + } + + const definitionsByGroup = new Map>(); + for (const entry of visible) { + if (entry === null || entry.kind !== "catalog" || definitionsByGroup.has(entry.group)) { + continue; + } + definitionsByGroup.set(entry.group, yield* connectionDefinitions(entry.parsed)); + } + + return yield* Effect.forEach( + visible, + (entry) => + entry === null + ? Effect.succeed(null) + : entry.kind === "static" + ? staticToolSchemaView(entry.address, entry.tool, includeTypeScript) + : catalogToolSchemaView({ + address: entry.address, + parsed: entry.parsed, + row: entry.row, + tool: entry.tool, + includeTypeScript, + definitions: Effect.succeed(definitionsByGroup.get(entry.group)!), + }), + { concurrency: 4 }, + ); + }); + // ------------------------------------------------------------------ // Providers // ------------------------------------------------------------------ @@ -7201,6 +7596,7 @@ export const createExecutor = + ({ + id, + owner, + subject: owner === "org" ? "" : "u", + pattern, + action, + position, + created_at: new Date(0), + updated_at: new Date(0), + }) as ToolPolicyRow; + +const flatRank = () => 0; +const ownerRank = (row: Pick) => (row.owner === "user" ? 0 : 1); + +const expectSame = ( + toolId: string, + rows: readonly ToolPolicyRow[], + rank: (row: Pick) => number, +) => { + const prepared = prepareToolPolicies(rows, rank); + expect(prepared.resolve(toolId)).toEqual(resolveToolPolicy(toolId, rows, rank)); + for (const defaultRequiresApproval of [undefined, false, true]) { + expect(prepared.resolveEffective(toolId, defaultRequiresApproval)).toEqual( + resolveEffectivePolicy(toolId, rows, rank, defaultRequiresApproval), + ); + } +}; + +describe("prepareToolPolicies", () => { + it("returns undefined and the plugin default when there are no rules", () => { + const prepared = prepareToolPolicies([], ownerRank); + expect(prepared.resolve("a.org.default.b")).toBeUndefined(); + expect(prepared.resolveEffective("a.org.default.b", true)).toEqual({ + action: "require_approval", + source: "plugin-default", + }); + expect(prepared.resolveEffective("a.org.default.b")).toEqual({ + action: "approve", + source: "plugin-default", + }); + }); + + it("takes the first matching rule by position, exact or wildcard", () => { + const exactFirst = [ + ROW("a", "vercel.org.default.create", "approve", "a0"), + ROW("b", "vercel.*", "block", "a1"), + ]; + const wildcardFirst = [ + ROW("b", "vercel.*", "block", "a0"), + ROW("a", "vercel.org.default.create", "approve", "a1"), + ]; + expect(prepareToolPolicies(exactFirst, flatRank).resolve("vercel.org.default.create")).toEqual({ + action: "approve", + pattern: "vercel.org.default.create", + policyId: "a", + }); + expect( + prepareToolPolicies(wildcardFirst, flatRank).resolve("vercel.org.default.create"), + ).toEqual({ action: "block", pattern: "vercel.*", policyId: "b" }); + expectSame("vercel.org.default.create", exactFirst, flatRank); + expectSame("vercel.org.default.create", wildcardFirst, flatRank); + }); + + it("lets an org block override a user approve, and a user block override an org approve", () => { + const orgBlock = [ + ROW("user-allow", "lidarr.org.default.list", "approve", "a0", "user"), + ROW("org-block", "lidarr.*", "block", "a0", "org"), + ]; + const userBlock = [ + ROW("org-allow", "lidarr.org.default.list", "approve", "a0", "org"), + ROW("user-block", "*", "block", "a5", "user"), + ]; + expect( + prepareToolPolicies(orgBlock, ownerRank).resolve("lidarr.org.default.list")?.policyId, + ).toBe("org-block"); + expect( + prepareToolPolicies(userBlock, ownerRank).resolve("lidarr.org.default.list")?.policyId, + ).toBe("user-block"); + expectSame("lidarr.org.default.list", orgBlock, ownerRank); + expectSame("lidarr.org.default.list", userBlock, ownerRank); + }); + + it("keeps the earlier owner's match when both owners match with the same action", () => { + const rows = [ + ROW("org-allow", "plex.*", "approve", "a0", "org"), + ROW("user-allow", "plex.org.default.search", "approve", "a9", "user"), + ]; + // user ranks first, so its match is found first and kept on a tie. + expect(prepareToolPolicies(rows, ownerRank).resolve("plex.org.default.search")?.policyId).toBe( + "user-allow", + ); + expectSame("plex.org.default.search", rows, ownerRank); + }); + + it("uses the earliest of duplicate exact patterns and breaks position ties by id", () => { + const rows = [ + ROW("z", "a.org.default.t", "block", "a0"), + ROW("m", "a.org.default.t", "approve", "a0"), + ROW("b", "a.org.default.t", "require_approval", "a1"), + ]; + expect(prepareToolPolicies(rows, flatRank).resolve("a.org.default.t")?.policyId).toBe("m"); + expectSame("a.org.default.t", rows, flatRank); + }); + + it("matches the default block plus specific grants used in production", () => { + const rows = [ + ROW("g1", "google_gmail.org.default.users.messages.list", "approve", "a0"), + ROW("g2", "cloudflare-dns.*.*.dns.getZone", "approve", "a1"), + ROW("lid", "lidarr.*", "block", "a2"), + ROW("all", "*", "block", "a3"), + ]; + for (const toolId of [ + "google_gmail.org.default.users.messages.list", + "google_gmail.org.default.users.messages.send", + "cloudflare-dns.org.default.dns.getZone", + "cloudflare-dns.org.default.dns.deleteZone", + "lidarr.org.default.artist.list", + "plex.org.default.library", + ]) { + expectSame(toolId, rows, ownerRank); + } + const prepared = prepareToolPolicies(rows, ownerRank); + expect(prepared.resolve("cloudflare-dns.org.default.dns.getZone")?.action).toBe("approve"); + expect(prepared.resolve("lidarr.org.default.artist.list")?.action).toBe("block"); + expect(prepared.resolve("plex.org.default.library")?.policyId).toBe("all"); + }); + + it("matches subtree, mid-segment, universal and legacy pattern shapes", () => { + const patterns = [ + "*", + "a.*", + "a.b.*", + "a.*.c", + "a.*.*.d", + "a.b", + "a.b.c", + "*.b", + "*.*", + "a.b*", + "a", + "", + "a..b", + ]; + const tools = ["a", "a.b", "a.b.c", "a.x.c", "a.x.y.d", "z.b", "a.b.c.d", "a.b*", "", "a..b"]; + patterns.forEach((pattern, i) => { + const rows = [ROW(`p${i}`, pattern, "block", "a0")]; + for (const toolId of tools) expectSame(toolId, rows, flatRank); + }); + }); + + it("agrees with resolveToolPolicy on seeded random rule sets", () => { + // mulberry32 — deterministic so a failure reproduces exactly. + let seed = 0x5eed1234; + const random = () => { + seed = (seed + 0x6d2b79f5) | 0; + let t = Math.imul(seed ^ (seed >>> 15), 1 | seed); + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; + const pick = (items: readonly T[]): T => items[Math.floor(random() * items.length)]!; + const integrations = ["gmail", "plex", "lidarr", "dns"]; + const owners = ["org", "user"]; + const connections = ["default", "justin"]; + const names = ["list", "get", "send", "delete", "users.messages.list"]; + const actions: Action[] = ["approve", "require_approval", "block"]; + const positions = ["a0", "a1", "a2", "a3", "Zz", "a0V"]; + + const randomTool = () => + `${pick(integrations)}.${pick(owners)}.${pick(connections)}.${pick(names)}`; + const randomPattern = (): string => { + const segments = randomTool().split("."); + const shapes = [ + () => "*", + () => `${segments[0]}.*`, + () => `${segments.slice(0, 3).join(".")}.*`, + () => `${segments[0]}.*.*.${segments.slice(3).join(".")}`, + () => `${segments[0]}.${segments[1]}.*.${segments.slice(3).join(".")}`, + () => `*.${segments.slice(1).join(".")}`, + () => segments.join("."), + () => segments.join("."), + ]; + return pick(shapes)(); + }; + + for (let round = 0; round < 300; round++) { + const count = Math.floor(random() * 40); + const rows: ToolPolicyRow[] = []; + for (let i = 0; i < count; i++) { + rows.push( + ROW( + `id${Math.floor(random() * 1000)}-${i}`, + randomPattern(), + pick(actions), + pick(positions), + pick(owners) as "org" | "user", + ), + ); + } + const toolIds = Array.from({ length: 25 }, randomTool); + for (const rank of [ownerRank, flatRank]) { + const prepared = prepareToolPolicies(rows, rank); + for (const toolId of toolIds) { + expect(prepared.resolve(toolId)).toEqual(resolveToolPolicy(toolId, rows, rank)); + expect(prepared.resolveEffective(toolId, round % 2 === 0)).toEqual( + resolveEffectivePolicy(toolId, rows, rank, round % 2 === 0), + ); + } + } + } + }); +}); diff --git a/packages/core/sdk/src/policies.test.ts b/packages/core/sdk/src/policies.test.ts index 18c5d29a72..58a6f130f3 100644 --- a/packages/core/sdk/src/policies.test.ts +++ b/packages/core/sdk/src/policies.test.ts @@ -635,6 +635,167 @@ describe("blocked tools", () => { ); }); +describe("tools.schemas", () => { + /** Count core reads per table, so a batch proves how many policy, tool and + * definition reads it spends. */ + const countingDb = (db: FumaDb, reads: Record): FumaDb => { + const wrap = (inner: FumaDb): FumaDb => + new Proxy(inner, { + get(target, property) { + if (property === "withContext") { + return (context: unknown) => + wrap((target.withContext as (value: unknown) => FumaDb)(context)); + } + if (property === "findMany" || property === "findFirst") { + return (...args: Parameters) => { + const [table] = args; + reads[table] = (reads[table] ?? 0) + 1; + return (target[property] as (...inner: typeof args) => unknown)(...args); + }; + } + return Reflect.get(target, property); + }, + }); + return wrap(db); + }; + + const setupCounting = () => + Effect.gen(function* () { + const reads: Record = {}; + const config = makeTestConfig({ plugins: [policyTestPlugin()] as const }); + const executor = yield* createExecutor({ ...config, db: countingDb(config.db, reads) }); + yield* Effect.addFinalizer(() => + executor + .close() + .pipe(Effect.andThen(Effect.promise(() => config.testDb.close())), Effect.ignore), + ); + yield* executor.ptest.seed(); + for (const integration of [VERCEL, GITHUB]) { + yield* executor.connections.create({ + owner: "org", + name: CONN, + integration, + template: TEMPLATE, + from: { provider: ProviderKey.make("memory"), id: ProviderItemId.make("v") }, + }); + } + const reset = () => { + for (const key of Object.keys(reads)) delete reads[key]; + }; + return { executor, reads, reset }; + }); + + it.effect("answers per address exactly as schema does, in input order", () => + Effect.gen(function* () { + const executor = yield* setupExecutor(); + yield* executor.policies.create({ + owner: "org", + pattern: "vercel.*.*.delete", + action: "block", + }); + yield* executor.policies.create({ owner: "org", pattern: "github.*", action: "approve" }); + const addresses = [ + addr(VERCEL, "deploy"), + addr(VERCEL, "delete"), + addr(GITHUB, "list"), + addr(VERCEL, "missing"), + ToolAddress.make("not-a-tool-address"), + addr(VERCEL, "deploy"), + ]; + const batched = yield* executor.tools.schemas(addresses); + const single = yield* Effect.forEach(addresses, (address) => executor.tools.schema(address)); + expect(batched).toEqual(single); + expect(batched.map((view) => view?.name ?? null)).toEqual([ + "deploy", + null, + "list", + null, + null, + "deploy", + ]); + expect(batched[0]?.annotations).toEqual(single[0]?.annotations); + expect(batched[2]?.inputTypeScript).toEqual(single[2]?.inputTypeScript); + + const compact = yield* executor.tools.schemas(addresses, { typeScript: false }); + expect(compact).toEqual( + yield* Effect.forEach(addresses, (address) => + executor.tools.schema(address, { typeScript: false }), + ), + ); + expect(compact[0]?.inputTypeScript).toBeUndefined(); + expect(yield* executor.tools.schemas([])).toEqual([]); + }), + ); + + it.effect("sees a policy change between calls", () => + Effect.gen(function* () { + const executor = yield* setupExecutor(); + const addresses = [addr(VERCEL, "deploy"), addr(VERCEL, "delete")]; + const before = yield* executor.tools.schemas(addresses); + expect(before.map((view) => view?.name)).toEqual(["deploy", "delete"]); + + const rule = yield* executor.policies.create({ + owner: "org", + pattern: "vercel.*", + action: "block", + }); + const blocked = yield* executor.tools.schemas(addresses); + expect(blocked).toEqual([null, null]); + expect(blocked).toEqual( + yield* Effect.forEach(addresses, (address) => executor.tools.schema(address)), + ); + + yield* executor.policies.remove({ id: String(rule.id), owner: "org" }); + const restored = yield* executor.tools.schemas(addresses); + expect(restored.map((view) => view?.name)).toEqual(["deploy", "delete"]); + }), + ); + + it.effect("keeps a user block over an org approve for the same tool", () => + Effect.gen(function* () { + const executor = yield* setupExecutor(); + yield* executor.policies.create({ owner: "org", pattern: "vercel.*", action: "approve" }); + yield* executor.policies.create({ + owner: "user", + pattern: "vercel.*.*.deploy", + action: "block", + }); + const addresses = [addr(VERCEL, "deploy"), addr(VERCEL, "delete")]; + const batched = yield* executor.tools.schemas(addresses); + expect(batched.map((view) => view?.name ?? null)).toEqual([null, "delete"]); + expect(batched).toEqual( + yield* Effect.forEach(addresses, (address) => executor.tools.schema(address)), + ); + }), + ); + + it.effect("reads policy once per batch, and tools and definitions once per connection", () => + Effect.gen(function* () { + const { executor, reads, reset } = yield* setupCounting(); + const addresses = [ + addr(VERCEL, "deploy"), + addr(VERCEL, "delete"), + addr(GITHUB, "list"), + addr(VERCEL, "missing"), + ]; + + reset(); + yield* Effect.forEach(addresses, (address) => + executor.tools.schema(address, { typeScript: false }), + ); + expect(reads.tool_policy).toBe(4); + expect(reads.tool).toBe(4); + expect(reads.definition).toBe(3); + + reset(); + yield* executor.tools.schemas(addresses, { typeScript: false }); + expect(reads.tool_policy).toBe(1); + expect(reads.tool).toBe(2); + expect(reads.definition).toBe(2); + }).pipe(Effect.scoped), + ); +}); + describe("active tool-policy provider", () => { const staticPlugin = definePlugin(() => ({ id: "toolkit-fixture" as const, @@ -700,6 +861,11 @@ describe("active tool-policy provider", () => { ToolAddress.make("toolkit-fixture.ctl.hidden"), ); expect(hiddenSchema).toBeNull(); + const batched = yield* executor.tools.schemas([ + ToolAddress.make("toolkit-fixture.ctl.hidden"), + ToolAddress.make("toolkit-fixture.ctl.allowed"), + ]); + expect(batched).toEqual([null, allowedSchema]); const allowed = yield* executor.execute(ToolAddress.make("toolkit-fixture.ctl.allowed"), {}); expect(allowed).toBe("allowed"); @@ -790,3 +956,59 @@ describe("approve / require_approval interaction with annotations", () => { }), ); }); + +describe("tools.schema policy read", () => { + it.effect("narrowed rule reads keep list and schema visibility identical", () => + Effect.gen(function* () { + const executor = yield* setupExecutor(); + // Exact rules for other tools never match; wildcard and exact rules for + // the tool itself decide. Mix both owners' worth of shapes. + yield* executor.policies.create({ owner: "org", pattern: "*", action: "block" }); + yield* executor.policies.create({ + owner: "org", + pattern: `vercel.org.${CONN}.deploy`, + action: "approve", + }); + yield* executor.policies.create({ owner: "org", pattern: "github.*", action: "approve" }); + yield* executor.policies.create({ + owner: "org", + pattern: `github.org.${CONN}.list`, + action: "block", + }); + for (const name of ["a", "b", "c"]) { + yield* executor.policies.create({ + owner: "org", + pattern: `vercel.org.${CONN}.${name}`, + action: "approve", + }); + } + + const visible = new Set((yield* executor.tools.list()).map((tool) => String(tool.address))); + for (const address of [addr(VERCEL, "delete"), addr(GITHUB, "list")]) { + expect(yield* executor.tools.schema(address)).toBeNull(); + expect(yield* executor.tools.schema(address, { typeScript: false })).toBeNull(); + } + for (const address of [addr(VERCEL, "deploy")]) { + const full = yield* executor.tools.schema(address); + const lean = yield* executor.tools.schema(address, { typeScript: false }); + expect(full !== null).toBe(visible.has(String(address))); + expect(lean !== null).toBe(visible.has(String(address))); + expect(full).not.toBeNull(); + expect(lean).not.toBeNull(); + { + expect(lean!.inputTypeScript).toBeUndefined(); + expect(lean!.outputTypeScript).toBeUndefined(); + expect(lean!.typeScriptDefinitions).toBeUndefined(); + const { + inputTypeScript: _i, + outputTypeScript: _o, + typeScriptDefinitions: _d, + ...rest + } = full!; + expect({ ...lean }).toEqual(rest); + } + } + expect([...visible].sort()).toEqual([String(addr(VERCEL, "deploy"))]); + }), + ); +}); diff --git a/packages/core/sdk/src/policies.ts b/packages/core/sdk/src/policies.ts index 8620d9c6d3..f8fa420373 100644 --- a/packages/core/sdk/src/policies.ts +++ b/packages/core/sdk/src/policies.ts @@ -254,6 +254,147 @@ export const resolveEffectivePolicy = ( return match ? liftUser(match) : liftPlugin(defaultRequiresApproval); }; +// --------------------------------------------------------------------------- +// Prepared resolution — the same answer as `resolveToolPolicy`, built once per +// rule snapshot and reused for every tool in an operation. `resolveToolPolicy` +// re-sorts the whole rule list and re-splits every pattern for each tool, so a +// catalog listing paid O(tools × rules·log rules) string work. Here the rules +// are sorted once, patterns are split once, exact patterns (no `*` segment) +// become a map lookup, and wildcard patterns are bucketed by their literal +// first segment. Each owner still contributes its first matching rule by the +// global sort order, and owners are combined in the order their first match +// was found, so ties between owners keep the same winner. +// --------------------------------------------------------------------------- + +interface PreparedRule { + /** Index in the global (ownerRank, position, id) sort order. */ + readonly index: number; + readonly match: PolicyMatch; + /** Split pattern; only set for wildcard rules. */ + readonly segments: readonly string[]; +} + +interface PreparedOwnerRules { + /** Exact patterns → the owner's earliest rule with that pattern. */ + readonly exact: Map; + /** Wildcard patterns keyed by their literal first segment, in sort order. */ + readonly wildcardByHead: Map; + /** Wildcard patterns whose first segment is `*` (includes universal `*`). */ + readonly wildcardAnyHead: PreparedRule[]; +} + +/** `matchPattern` over pre-split segments. Kept in lockstep with it. */ +const matchSegments = ( + patternSegments: readonly string[], + toolSegments: readonly string[], +): boolean => { + for (let i = 0; i < patternSegments.length; i++) { + const seg = patternSegments[i]!; + if (seg === "*") { + if (i === patternSegments.length - 1) return toolSegments.length >= i; + if (i >= toolSegments.length) return false; + continue; + } + if (i >= toolSegments.length || toolSegments[i] !== seg) return false; + } + return patternSegments.length === toolSegments.length; +}; + +/** First rule in `rules` (sort order) that matches and sorts before `best`. */ +const firstWildcardMatch = ( + rules: readonly PreparedRule[] | undefined, + toolSegments: readonly string[], + best: PreparedRule | undefined, +): PreparedRule | undefined => { + if (!rules) return best; + for (const rule of rules) { + if (best && rule.index > best.index) return best; + if (matchSegments(rule.segments, toolSegments)) return rule; + } + return best; +}; + +export interface PreparedToolPolicies { + /** Same result as `resolveToolPolicy(toolId, policies, ownerRank)`. */ + readonly resolve: (toolId: string) => PolicyMatch | undefined; + /** Same result as `resolveEffectivePolicy(toolId, policies, ownerRank, d)`. */ + readonly resolveEffective: (toolId: string, defaultRequiresApproval?: boolean) => EffectivePolicy; +} + +export const prepareToolPolicies = ( + policies: readonly ToolPolicyRow[], + ownerRank: (row: Pick) => number, +): PreparedToolPolicies => { + const sorted = [...policies].sort((a, b) => { + const sa = ownerRank(a); + const sb = ownerRank(b); + if (sa !== sb) return sa - sb; + return comparePolicyRow(a, b); + }); + const owners = new Map(); + sorted.forEach((row, index) => { + let ownerRules = owners.get(row.owner); + if (!ownerRules) { + ownerRules = { exact: new Map(), wildcardByHead: new Map(), wildcardAnyHead: [] }; + owners.set(row.owner, ownerRules); + } + const match: PolicyMatch = { + action: row.action as ToolPolicyAction, + pattern: row.pattern, + policyId: row.id, + }; + const segments = row.pattern.split("."); + // `matchPattern` treats only a whole `*` segment as a wildcard; any other + // pattern matches exactly when it equals the tool id. + if (row.pattern !== "*" && !segments.includes("*")) { + if (!ownerRules.exact.has(row.pattern)) { + ownerRules.exact.set(row.pattern, { index, match, segments: [] }); + } + return; + } + const rule: PreparedRule = { index, match, segments: row.pattern === "*" ? ["*"] : segments }; + const head = rule.segments[0]!; + if (head === "*") { + ownerRules.wildcardAnyHead.push(rule); + } else { + const bucket = ownerRules.wildcardByHead.get(head); + if (bucket) bucket.push(rule); + else ownerRules.wildcardByHead.set(head, [rule]); + } + }); + const ownerList = [...owners.values()]; + + const resolve = (toolId: string): PolicyMatch | undefined => { + if (ownerList.length === 0) return undefined; + const toolSegments = toolId.split("."); + const found: PreparedRule[] = []; + for (const ownerRules of ownerList) { + let best = ownerRules.exact.get(toolId); + best = firstWildcardMatch( + ownerRules.wildcardByHead.get(toolSegments[0]!), + toolSegments, + best, + ); + best = firstWildcardMatch(ownerRules.wildcardAnyHead, toolSegments, best); + if (best) found.push(best); + } + // `resolveToolPolicy` combines owners in the order their first match + // appears in the global sort; equal restriction keeps the earlier one. + found.sort((a, b) => a.index - b.index); + let selected: PolicyMatch | undefined; + for (const rule of found) selected = moreRestrictive(selected, rule.match); + return selected; + }; + + return { + resolve, + resolveEffective: (toolId, defaultRequiresApproval) => { + const match = resolve(toolId); + return match ? liftUser(match) : liftPlugin(defaultRequiresApproval); + }, + }; +}; + export const effectivePolicyFromSorted = ( toolId: string, sortedPolicies: readonly (Pick & diff --git a/packages/hosts/mcp/src/codemode-instructions.ts b/packages/hosts/mcp/src/codemode-instructions.ts new file mode 100644 index 0000000000..4fa670be2b --- /dev/null +++ b/packages/hosts/mcp/src/codemode-instructions.ts @@ -0,0 +1,15 @@ +/** + * The server `instructions` for codemode (sandbox) sessions: the shortest + * path from a request to a call, mirroring `passthroughInstructions`. Every + * byte is loaded into each client session, so the long guide stays behind + * `skills({ name: "execute" })`. + */ +export const codemodeInstructions = (): string => + [ + "Executor runs tools of connected integrations inside `execute` (a TypeScript sandbox). Fastest path, one call:", + '1. `const { items } = await tools.search({ query: "", namespace: "", limit: 3 });`', + '2. `if (!items[0]) return "no matching tool"; return await tools[items[0].path]({ ...arguments });`', + "`namespace` accepts the slug or an unambiguous alias (`gmail` resolves to `google_gmail`); the `execute` description lists the connected integrations. Omit `namespace` when the integration is unknown.", + "Call `tools.describe.tool({ path })` only when the argument shape is unknown. Results are `{ ok: true, data }` or `{ ok: false, error }`.", + 'Read `skills({ name: "execute" })` once per session for files, `emit`, pagination, and resume.', + ].join("\n"); diff --git a/packages/hosts/mcp/src/envelope.ts b/packages/hosts/mcp/src/envelope.ts index 8e0b5f2252..1aaf8ce3f9 100644 --- a/packages/hosts/mcp/src/envelope.ts +++ b/packages/hosts/mcp/src/envelope.ts @@ -359,7 +359,11 @@ const mcpDispatch = (resource: McpResource) => // request. Non-authenticated outcomes render directly. Session teardown is // only safe after the store can validate the authenticated principal and MCP // resource; an auth-level Forbidden may not carry either. - const outcome = yield* auth.authenticate(request); + const outcome = yield* auth.authenticate(request).pipe( + Effect.withSpan("mcp.envelope.authenticate", { + attributes: { "http.method": request.method }, + }), + ); if (!Predicate.isTagged(outcome, "Authenticated")) { return fromWebResponse(renderAuthError(auth, request, outcome)); } diff --git a/packages/hosts/mcp/src/in-memory-session-store.test.ts b/packages/hosts/mcp/src/in-memory-session-store.test.ts index bdc51db85d..8ef053bd37 100644 --- a/packages/hosts/mcp/src/in-memory-session-store.test.ts +++ b/packages/hosts/mcp/src/in-memory-session-store.test.ts @@ -613,6 +613,119 @@ interface JsonRpcErrorBody { readonly error: { readonly code: number; readonly message: string }; } +/** Open a session on a serving store with the given client capabilities. */ +const initializeWith = async ( + sessions: ReturnType["sessions"], + capabilities: Record, +): Promise => { + const response = await dispatchPost(sessions, { + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2025-06-18", + capabilities, + clientInfo: { name: "stream-test", version: "1.0.0" }, + }, + }); + expect(response.status).toBe(200); + const sessionId = response.headers.get("mcp-session-id") ?? ""; + expect(sessionId).not.toBe(""); + return sessionId; +}; + +const dispatchGet = ( + sessions: ReturnType["sessions"], + sessionId: string, + principal: Principal = TEST_PRINCIPAL, +) => + Effect.runPromise( + sessions.store.dispatch({ + request: new Request("https://executor.test/mcp", { + method: "GET", + headers: { accept: "text/event-stream", "mcp-session-id": sessionId }, + }), + principal, + resource: defaultMcpResource, + sessionId, + method: "GET", + }), + ); + +// --------------------------------------------------------------------------- +// The standalone GET stream, offered only where it can carry something. +// --------------------------------------------------------------------------- + +describe("standalone GET stream through the in-memory session store", () => { + it("answers 405 for a session whose client can receive no server-initiated message", async () => { + const { sessions } = makeServingStore(); + const sessionId = await initializeWith(sessions, {}); + + const result = await dispatchGet(sessions, sessionId); + expect(result).toBeInstanceOf(Response); + const response = result as Response; + expect(response.status).toBe(405); + expect(response.headers.get("allow")).toBe("POST, DELETE"); + const body = (await response.json()) as JsonRpcErrorBody; + expect(body.error.code).toBe(-32000); + // The session itself is untouched: a POST on it still works. + const listed = await dispatchPostTo(sessions, sessionId, { + jsonrpc: "2.0", + id: 2, + method: "tools/list", + }); + expect(listed.status).toBe(200); + await sessions.close(); + }); + + it("still opens the stream for a client that declared elicitation", async () => { + const { sessions } = makeServingStore(); + const sessionId = await initializeWith(sessions, { elicitation: { form: {} } }); + + const result = await dispatchGet(sessions, sessionId); + expect(result).toBeInstanceOf(Response); + const response = result as Response; + expect(response.status).toBe(200); + expect(response.headers.get("content-type")).toContain("text/event-stream"); + await response.body?.cancel(); + await sessions.close(); + }); + + it("keeps ownership ahead of the stream policy", async () => { + const { sessions } = makeServingStore(); + const sessionId = await initializeWith(sessions, {}); + const other: Principal = { ...TEST_PRINCIPAL, accountId: "acct_other" }; + expect(await dispatchGet(sessions, sessionId, other)).toBe("forbidden"); + await sessions.close(); + }); +}); + +const dispatchPostTo = ( + sessions: ReturnType["sessions"], + sessionId: string, + body: unknown, +): Promise => + Effect.runPromise( + sessions.store + .dispatch({ + request: new Request("https://executor.test/mcp", { + method: "POST", + headers: { ...MCP_POST_HEADERS, "mcp-session-id": sessionId }, + body: JSON.stringify(body), + }), + principal: TEST_PRINCIPAL, + resource: defaultMcpResource, + sessionId, + method: "POST", + }) + .pipe( + Effect.map((result) => { + expect(result).toBeInstanceOf(Response); + return result as Response; + }), + ), + ); + describe("pre-initialize dispatch through the in-memory session store", () => { it("answers a valid unknown pre-session method with -32601 on a 200", async () => { const { sessions, buildCount } = makeServingStore(); diff --git a/packages/hosts/mcp/src/in-memory-session-store.ts b/packages/hosts/mcp/src/in-memory-session-store.ts index 952f9bd83f..86d6de6eee 100644 --- a/packages/hosts/mcp/src/in-memory-session-store.ts +++ b/packages/hosts/mcp/src/in-memory-session-store.ts @@ -20,6 +20,7 @@ import { type InProcessBrowserApprovalStore, } from "./browser-approval-store"; import { jsonRpcErrorBody, preInitializeMethodNotFound } from "./envelope"; +import { serverInitiatedMessagesPossible, standaloneStreamNotOffered } from "./standalone-stream"; import { McpSessionStore, MCP_ORG_WRITE_ACCESS_HEADER, @@ -396,6 +397,15 @@ export const makeInMemoryMcpSessionStore = ( if (!sessionOwnerMatches(owner, principal, resource)) return Effect.succeed("forbidden"); owners.set(sessionId, { principal, resource }); touch(sessionId); + // A standalone stream this client can never hear anything on is refused + // with 405 (see ./standalone-stream) rather than held open for nothing. + // Checked after ownership so a foreign bearer still gets its 403. + if ( + request.method === "GET" && + !serverInitiatedMessagesPossible(servers.get(sessionId)?.server.getClientCapabilities()) + ) { + return Effect.succeed(standaloneStreamNotOffered()); + } // Claim before the await, release in the finalizer — `runHandleRequest` // already recovers every failure to a 500, but `ensuring` also covers an // interrupt, so the counter cannot be left permanently raised (which would diff --git a/packages/hosts/mcp/src/namespace-search-tools.test.ts b/packages/hosts/mcp/src/namespace-search-tools.test.ts index 428fb091f9..1d6d7d2b7c 100644 --- a/packages/hosts/mcp/src/namespace-search-tools.test.ts +++ b/packages/hosts/mcp/src/namespace-search-tools.test.ts @@ -4,7 +4,19 @@ import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; import type * as Cause from "effect/Cause"; -import type { ExecutionEngine } from "@executor-js/execution"; +import { + buildExecuteDescription, + parseIntegrationInventory, + type ExecutionEngine, +} from "@executor-js/execution"; +import { + AuthTemplateSlug, + ConnectionName, + IntegrationSlug, + ToolName, + definePlugin, +} from "@executor-js/sdk"; +import { makeTestExecutor, memoryCredentialsPlugin } from "@executor-js/sdk/testing"; import { readSearchToolsEnabled } from "./browser-approval"; import { createExecutorMcpServer, type ExecutorMcpServerConfig } from "./tool-server"; @@ -41,9 +53,8 @@ const makeRecordingEngine = (): { }; }; -// The inventory block exactly as `buildExecuteDescription` renders it, -// including an overflow marker and a slug that cannot form a legal MCP tool -// name (which registration must skip, not fail on). +// A legacy inventory block, including an overflow marker and an invalid MCP +// tool slug, both of which registration must skip. const DESCRIPTION_WITH_INVENTORY = [ "Execute TypeScript in a sandboxed runtime.", "", @@ -83,7 +94,101 @@ const toolNames = async (client: Client): Promise => // Registration // --------------------------------------------------------------------------- +describe("MCP host — codemode instructions", () => { + // Every byte here is paid on every turn of every sandbox session: a short + // search-then-call recipe, with the long guide behind `skills`. + it("serves short instructions that name the search-then-call path in order", async () => { + const { engine } = makeRecordingEngine(); + await withClient({ engine, description: DESCRIPTION_WITH_INVENTORY }, async (client) => { + const instructions = client.getInstructions() ?? ""; + expect(instructions.length).toBeGreaterThan(0); + expect(instructions.length).toBeLessThan(800); + const order = [ + "tools.search({", + "limit: 3", + "tools[items[0].path](", + 'skills({ name: "execute" })', + ].map((marker) => instructions.indexOf(marker)); + expect(order.every((index) => index >= 0)).toBe(true); + expect(order[0]).toBeLessThan(order[2]!); + expect(order[2]).toBeLessThan(order[3]!); + expect(instructions).toContain("`gmail` resolves to `google_gmail`"); + }); + }); +}); + describe("MCP host — per-integration search tools", () => { + it("derives all search tools from a built inventory with more than 50 permitted integrations", async () => { + const slugs = Array.from({ length: 56 }, (_, index) => + IntegrationSlug.make(`integration_${String(index).padStart(3, "0")}`), + ); + const inventoryPlugin = definePlugin(() => ({ + id: "inventory-plugin" as const, + storage: () => ({}), + resolveTools: () => + Effect.succeed({ tools: [{ name: ToolName.make("list"), description: "Read records." }] }), + extension: (ctx) => ({ + seed: (slug: IntegrationSlug) => + ctx.core.integrations.register({ + slug, + name: String(slug), + description: `Read records from ${slug}.`, + config: {}, + }), + }), + }))(); + const description = await Effect.runPromise( + Effect.gen(function* () { + const executor = yield* makeTestExecutor({ + plugins: [memoryCredentialsPlugin(), inventoryPlugin] as const, + }); + for (const slug of [...slugs].reverse()) { + yield* executor["inventory-plugin"].seed(slug); + yield* executor.connections.create({ + owner: "org", + name: ConnectionName.make("main"), + integration: slug, + template: AuthTemplateSlug.make("apiKey"), + value: "token", + }); + } + yield* executor.policies.create({ + owner: "org", + pattern: `${slugs[0]}.*`, + action: "block", + }); + yield* executor.policies.create({ + owner: "org", + pattern: `${slugs[55]}.*`, + action: "require_approval", + }); + expect( + (yield* executor.tools.list()).map((tool) => String(tool.integration)).sort(), + ).toEqual(slugs.slice(1).map(String)); + return yield* buildExecuteDescription(executor); + }).pipe(Effect.scoped), + ); + const expected = slugs.slice(1).map(String); + expect(parseIntegrationInventory(description)).toEqual(expected); + expect(description).toContain(`- \`${slugs[50]}\` — Read records from ${slugs[50]}.`); + expect(description.split("\n")).toContain(`- \`${slugs[51]}\``); + expect(description).not.toContain("- ..."); + + const { engine, executed } = makeRecordingEngine(); + await withClient({ engine, description, searchToolsEnabled: true }, async (client) => { + const names = (await toolNames(client)).filter((name) => name.startsWith("search_")); + expect(names.sort()).toEqual(expected.map((slug) => `search_${slug}`)); + const result = await client.callTool({ + name: `search_${slugs[55]}`, + arguments: { query: "records" }, + }); + expect(result.isError ?? false).toBe(false); + expect(executed).toEqual([ + `return tools.search(${JSON.stringify({ query: "records", namespace: String(slugs[55]) })})`, + ]); + }); + }); + it("registers none by default: the option is opt-in", async () => { const { engine } = makeRecordingEngine(); await withClient({ engine, description: DESCRIPTION_WITH_INVENTORY }, async (client) => { @@ -130,7 +235,7 @@ describe("MCP host — per-integration search tools", () => { }); it("keeps each serialized definition small — the whole point is cheap context", async () => { - // A session serves one of these per connected integration (up to 50), so + // A session serves one of these per callable integration, so // definition bytes multiply. 300 serialized chars/tool keeps the full // surface around ~2k tokens; the original shipped shape was 551 chars/tool // (~5k tokens for 30 integrations), which defeated the feature's purpose. diff --git a/packages/hosts/mcp/src/passthrough-tools.test.ts b/packages/hosts/mcp/src/passthrough-tools.test.ts index 491a623740..83e75e6d0a 100644 --- a/packages/hosts/mcp/src/passthrough-tools.test.ts +++ b/packages/hosts/mcp/src/passthrough-tools.test.ts @@ -15,7 +15,15 @@ import { } from "@executor-js/sdk"; import { readToolMode } from "./browser-approval"; -import { passthroughCallCode } from "./passthrough-tools"; +import { + COMPACT_ARGUMENT_LIMIT, + COMPACT_DESCRIPTION_LIMIT, + compactArguments, + compactDescription, + integrationAliasText, + passthroughCallCode, + resolveIntegrationAlias, +} from "./passthrough-tools"; import { createExecutorMcpServer, McpPassthroughUnavailableError, @@ -60,31 +68,41 @@ const toolPort = ( catalog: readonly Tool[], schemaReads: string[] = [], lists: string[] = [], -): McpToolsPort => ({ - list: (filter) => - Effect.sync(() => { - lists.push("list"); - return catalog.filter( - (tool) => - (filter?.integration === undefined || tool.integration === filter.integration) && - (filter?.owner === undefined || tool.owner === filter.owner) && - (filter?.connection === undefined || tool.connection === filter.connection), - ); - }), - schema: (address) => - Effect.sync(() => { - schemaReads.push(String(address)); - const tool = catalog.find((item) => item.address === address); - if (!tool) return null; - return { - address, - name: tool.name, - description: tool.description, - inputSchema: tool.inputSchema, - annotations: tool.annotations, - } satisfies ToolSchemaView; - }), -}); + batches: string[][] = [], +): McpToolsPort => { + const port: McpToolsPort = { + list: (filter) => + Effect.sync(() => { + lists.push("list"); + return catalog.filter( + (tool) => + (filter?.integration === undefined || tool.integration === filter.integration) && + (filter?.owner === undefined || tool.owner === filter.owner) && + (filter?.connection === undefined || tool.connection === filter.connection), + ); + }), + schema: (address) => + Effect.sync(() => { + schemaReads.push(String(address)); + const tool = catalog.find((item) => item.address === address); + if (!tool) return null; + return { + address, + name: tool.name, + description: tool.description, + inputSchema: tool.inputSchema, + annotations: tool.annotations, + } satisfies ToolSchemaView; + }), + // The batched read answers per address exactly as `schema` would. + schemas: (addresses, options) => + Effect.suspend(() => { + batches.push(addresses.map(String)); + return Effect.forEach(addresses, (address) => port.schema(address, options)); + }), + }; + return port; +}; /** A stub engine that records every executed code string and answers with a * fixed value, so a test can prove a passthrough call became the expected @@ -198,6 +216,122 @@ describe("passthrough catalog", () => { // Wire flags // --------------------------------------------------------------------------- +describe("integration aliases", () => { + const catalog = [ + { slug: "google_gmail", name: "Gmail", description: "Send and read mail" }, + { slug: "google_calendar", name: "Google Calendar", description: "Events and calendars" }, + { slug: "cloudflare-dns", name: "Cloudflare DNS", description: "Zones and records" }, + { slug: "scrape_api", name: "ScrapeCreators", description: "Social media scraping" }, + { slug: "plex", name: "Plex", description: "Media server" }, + ]; + + it("resolves exact slugs, squashed names, word subsets, then descriptions", () => { + expect(resolveIntegrationAlias("plex", catalog)).toEqual({ kind: "exact", slug: "plex" }); + expect(resolveIntegrationAlias("gmail", catalog)).toEqual({ + kind: "alias", + slug: "google_gmail", + }); + expect(resolveIntegrationAlias("Google Calendar", catalog)).toEqual({ + kind: "alias", + slug: "google_calendar", + }); + expect(resolveIntegrationAlias("googlegmail", catalog)).toEqual({ + kind: "alias", + slug: "google_gmail", + }); + expect(resolveIntegrationAlias("cloudflare_dns", catalog)).toEqual({ + kind: "alias", + slug: "cloudflare-dns", + }); + expect(resolveIntegrationAlias("scrapecreators", catalog)).toEqual({ + kind: "alias", + slug: "scrape_api", + }); + expect(resolveIntegrationAlias("mail", catalog)).toEqual({ + kind: "alias", + slug: "google_gmail", + }); + expect(resolveIntegrationAlias("google", catalog)).toEqual({ + kind: "ambiguous", + slugs: ["google_calendar", "google_gmail"], + }); + expect(resolveIntegrationAlias("notion", catalog)).toEqual({ kind: "none" }); + expect(resolveIntegrationAlias(" ", catalog)).toEqual({ kind: "none" }); + }); + + it("adds display-name words the slug lacks as ranking text", () => { + expect([...integrationAliasText(catalog)]).toEqual([ + ["google_calendar", "googlecalendar"], + ["cloudflare-dns", "cloudflaredns"], + ["scrape_api", "creators scrapecreators"], + ]); + }); +}); + +describe("compact search hits", () => { + it("keeps the first sentence of a description within the limit", () => { + expect(compactDescription(undefined)).toBe(""); + expect(compactDescription(" \nList DNS records. Supports paging.\nMore.")).toBe( + "List DNS records.", + ); + expect(compactDescription("No. Period means something else here, keep going")).toBe( + "No. Period means something else here, keep going", + ); + const long = `Create ${"x".repeat(400)} end`; + const compact = compactDescription(long); + expect(compact.length).toBe(COMPACT_DESCRIPTION_LIMIT); + expect(compact.endsWith("…")).toBe(true); + }); + + it("summarizes arguments with types, required first, enums, arrays, refs and a cap", () => { + expect(compactArguments(undefined)).toBe("none"); + expect(compactArguments({ type: "object" })).toBe("none"); + expect(compactArguments({ oneOf: [{ type: "string" }, { type: "object" }] })).toBe( + "see full schema", + ); + expect( + compactArguments( + { + type: "object", + properties: { + optional: { type: ["string", "null"] }, + zone: { $ref: "#/$defs/Zone" }, + kind: { enum: ["A", "AAAA", "CNAME", "MX", "TXT", "NS", "SRV"] }, + tags: { type: "array", items: { $ref: "#/$defs/Tag" } }, + either: { anyOf: [{ type: "string" }, { type: "number" }] }, + nested: { properties: { a: { type: "string" } } }, + body: { + type: "object", + properties: { text: { type: "string" }, title: { type: "string" } }, + required: ["text"], + }, + fixed: { const: 7 }, + }, + required: ["zone", "kind"], + }, + { Zone: { type: "string" }, Tag: { type: "object", properties: {} } }, + ), + ).toBe( + "zone (string, required); kind (A|AAAA|CNAME|MX|TXT|NS|…, required); optional (string); tags (object[]); either (string|number); nested (object{a}); body (object{text*, title}); fixed (7)", + ); + const wideBody = Object.fromEntries( + Array.from({ length: 9 }, (_, i) => [`k${i}`, { type: "string" }]), + ); + expect( + compactArguments({ + type: "object", + properties: { body: { type: "object", properties: wideBody } }, + }), + ).toBe("body (object{k0, k1, k2, k3, k4, k5, +3})"); + const wide = Object.fromEntries( + Array.from({ length: COMPACT_ARGUMENT_LIMIT + 3 }, (_, i) => [`p${i}`, { type: "string" }]), + ); + const summary = compactArguments({ type: "object", properties: wide }); + expect(summary.split("; ")).toHaveLength(COMPACT_ARGUMENT_LIMIT + 1); + expect(summary.endsWith("; +3 more")).toBe(true); + }); +}); + describe("readToolMode", () => { const request = (query: string) => new Request(`https://example.test/mcp${query}`); @@ -337,6 +471,40 @@ describe("passthrough mode server", () => { ); }); + it("reads the execute description only when a guide asks for it", async () => { + const { engine } = makeRecordingEngine(); + let reads = 0; + const counting: ExecutionEngine = { + ...engine, + getDescription: Effect.sync(() => { + reads += 1; + return "test executor\n\n## Available integrations\n- github: issues"; + }), + }; + await withClient( + { engine: counting, mode: "passthrough", tools: toolPort(CATALOG) }, + async (client) => { + // Session setup, the tool list, and a search never touch the inventory. + expect(reads).toBe(0); + await client.listTools(); + await client.callTool({ name: "search", arguments: { query: "issues" } }); + expect(reads).toBe(0); + // The guides read it once per session, then reuse it. + await client.callTool({ name: "skills", arguments: { name: "search-invoke" } }); + await client.callTool({ name: "skills", arguments: {} }); + expect(reads).toBe(1); + }, + ); + // Codemode registers the `execute` tool with the description, so it is read at build. + reads = 0; + await withClient({ engine: counting }, async (client) => { + expect(reads).toBe(1); + const guide = await client.callTool({ name: "skills", arguments: { name: "execute" } }); + expect(JSON.stringify(guide.content)).toContain("## Available integrations"); + expect(reads).toBe(1); + }); + }); + it("serves only the search/invoke guide as text", async () => { const { engine } = makeRecordingEngine(); await withClient({ engine, mode: "passthrough", tools: toolPort(CATALOG) }, async (client) => { @@ -392,7 +560,18 @@ describe("passthrough mode server", () => { }), ]; await withClient( - { engine, mode: "passthrough", tools: toolPort(catalog, schemaReads) }, + { + engine, + mode: "passthrough", + tools: toolPort(catalog, schemaReads), + integrations: { + list: () => + Effect.succeed([ + { slug: IntegrationSlug.make("github"), name: "GitHub", description: "" }, + { slug: IntegrationSlug.make("github_other"), name: "Other", description: "" }, + ]), + }, + }, async (client) => { const args = { query: "issues", @@ -428,6 +607,130 @@ describe("passthrough mode server", () => { ); }); + it("accepts integration aliases in search and integrations, and ranks by display name", async () => { + const { engine } = makeRecordingEngine(); + const lists: string[] = []; + const catalog = [ + projection({ integration: "google_gmail", name: "users.messages.send" }), + projection({ integration: "google_gmail", name: "users.messages.list" }), + projection({ integration: "google_calendar", name: "events.list" }), + projection({ + integration: "scrape_api", + name: "profiles.get", + description: "Fetch a profile", + }), + ]; + const account = (integration: string) => ({ + integration, + owner: "org" as const, + name: "main", + identityLabel: null, + description: null, + lastHealth: null, + }); + await withClient( + { + engine, + mode: "passthrough", + tools: toolPort(catalog, [], lists), + connections: { + list: () => + Effect.succeed([ + account("google_gmail"), + account("google_calendar"), + account("scrape_api"), + ]), + }, + integrations: { + list: () => + Effect.succeed([ + { slug: IntegrationSlug.make("google_gmail"), name: "Gmail", description: "Mail" }, + { + slug: IntegrationSlug.make("google_calendar"), + name: "Google Calendar", + description: "Events", + }, + { + slug: IntegrationSlug.make("scrape_api"), + name: "ScrapeCreators", + description: "Scraping", + }, + ]), + }, + }, + async (client) => { + // An alias narrows the list read to the resolved slug and reports it. + const alias = await client.callTool({ + name: "search", + arguments: { query: "send", integration: "gmail", limit: 3 }, + }); + expect(alias.structuredContent).toMatchObject({ + integration: "google_gmail", + total: 1, + items: [{ id: "tools.google_gmail.org.main.users.messages.send" }], + }); + // An ambiguous alias searches every candidate and lists them. + const ambiguous = await client.callTool({ + name: "search", + arguments: { query: "list", integration: "google" }, + }); + expect(ambiguous.structuredContent).toMatchObject({ + integrations: ["google_calendar", "google_gmail"], + total: 2, + }); + expect( + decodeSearchItems(ambiguous.structuredContent) + .items.map((item) => item.id) + .sort(), + ).toEqual([ + "tools.google_calendar.org.main.events.list", + "tools.google_gmail.org.main.users.messages.list", + ]); + // An unknown value stays an exact filter (no hits) and explains itself. + const unknown = await client.callTool({ + name: "search", + arguments: { query: "send", integration: "notion" }, + }); + expect(unknown.structuredContent).toMatchObject({ + items: [], + total: 0, + hint: expect.stringContaining('No integration matches "notion"'), + }); + // Query words naming an integration by display name rank its tools. + const byName = await client.callTool({ + name: "search", + arguments: { query: "scrapecreators profile", limit: 3 }, + }); + expect(decodeSearchItems(byName.structuredContent).items.map((item) => item.id)).toEqual([ + "tools.scrape_api.org.main.profiles.get", + ]); + // integrations accepts the same aliases. + const accounts = await client.callTool({ + name: "integrations", + arguments: { integration: "gmail" }, + }); + expect(accounts.structuredContent).toMatchObject({ + integration: "google_gmail", + total: 1, + items: [{ integration: "google_gmail" }], + }); + const both = await client.callTool({ + name: "integrations", + arguments: { integration: "google" }, + }); + expect(both.structuredContent).toMatchObject({ + integrations: ["google_calendar", "google_gmail"], + total: 2, + }); + const none = await client.callTool({ + name: "integrations", + arguments: { integration: "notion" }, + }); + expect(none.structuredContent).toMatchObject({ total: 0, hint: expect.any(String) }); + }, + ); + }); + it("serves four discovery and call tools even for 10000 tools", async () => { const { engine, executed } = makeRecordingEngine(); const schemaReads: string[] = []; @@ -454,7 +757,18 @@ describe("passthrough mode server", () => { "search", "skills", ]); - expect(JSON.stringify(listed).length).toBeLessThan(4000); + // Every byte here is paid on every turn of every client; the long + // guide stays behind `skills`. + expect(JSON.stringify(listed).length).toBeLessThan(3600); + const instructions = client.getInstructions() ?? ""; + expect(instructions.length).toBeLessThan(1000); + // The instructions name the three-call path in order, with an example. + const order = ["integrations({})", "search({", "invoke({", "Example:"].map((marker) => + instructions.indexOf(marker), + ); + expect(order.every((index) => index >= 0)).toBe(true); + expect(order[0]).toBeLessThan(order[1]!); + expect(order[1]).toBeLessThan(order[2]!); expect(lists).toEqual([]); expect(schemaReads).toEqual([]); const result = await client.callTool({ @@ -507,6 +821,58 @@ describe("passthrough mode server", () => { ); }); + it("reads a search page's schemas in one batch and drops hits the batch hides", async () => { + const { engine } = makeRecordingEngine(); + const schemaReads: string[] = []; + const lists: string[] = []; + const batches: string[][] = []; + const catalog = Array.from({ length: 6 }, (_, i) => + projection({ + integration: "bench", + name: `record${i}`, + description: `benchmark record ${i}`, + }), + ); + const port = toolPort(catalog, schemaReads, lists, batches); + const hidden = new Set(); + const tools: McpToolsPort = { + ...port, + // A policy change between listing and the schema read hides a hit. + schemas: (addresses, options) => + port + .schemas(addresses, options) + .pipe( + Effect.map((views) => + views.map((view, index) => (hidden.has(String(addresses[index])) ? null : view)), + ), + ), + }; + await withClient({ engine, mode: "passthrough", tools }, async (client) => { + const page = await client.callTool({ + name: "search", + arguments: { query: "benchmark record", limit: 4 }, + }); + const ids = decodeSearchItems(page.structuredContent).items.map((item) => item.id); + expect(ids).toHaveLength(4); + // One batch per search call, in result order; no per-hit reads outside it. + expect(batches).toEqual([ids]); + expect(schemaReads).toEqual(ids); + + hidden.add(ids[1]!); + const again = await client.callTool({ + name: "search", + arguments: { query: "benchmark record", limit: 4 }, + }); + expect(decodeSearchItems(again.structuredContent).items.map((item) => item.id)).toEqual([ + ids[0], + ids[2], + ids[3], + ]); + expect(again.structuredContent).toMatchObject({ total: 6, hasMore: true, nextOffset: 4 }); + expect(batches).toEqual([ids, ids]); + }); + }); + it("uses current schemas and excludes static configuration tools", async () => { const recording = makeRecordingEngine(); const dynamic = projection({ @@ -516,7 +882,11 @@ describe("passthrough mode server", () => { }); const catalog = [dynamic, projection({ integration: "settings", name: "erase", static: true })]; let current: ToolSchemaView = { address: dynamic.address, inputSchema: dynamic.inputSchema }; - const tools: McpToolsPort = { ...toolPort(catalog), schema: () => Effect.succeed(current) }; + const tools: McpToolsPort = { + ...toolPort(catalog), + schema: () => Effect.succeed(current), + schemas: (addresses) => Effect.succeed(addresses.map(() => current)), + }; await withClient({ engine: recording.engine, mode: "passthrough", tools }, async (client) => { const hidden = await client.callTool({ name: "search", arguments: { query: "settings" } }); expect(decodeSearchItems(hidden.structuredContent).items).toEqual([]); @@ -539,15 +909,75 @@ describe("passthrough mode server", () => { arguments: { tool: String(dynamic.address), arguments: {} }, }); expect(invalid.isError).toBe(true); + // The validation error carries the current schema so the model can + // correct the call without another search. + const [invalidText] = invalid.content as ReadonlyArray<{ text: string }>; + expect(invalidText?.text).toContain("Invalid arguments"); + expect(invalidText?.text).toContain('"required":["title"]'); expect(recording.executed).toEqual([]); const refreshed = await client.callTool({ name: "search", arguments: { query: "notes" } }); expect(refreshed.structuredContent).toMatchObject({ + items: [{ arguments: "title (string, required)" }], + }); + const full = await client.callTool({ + name: "search", + arguments: { query: "notes", detail: "full" }, + }); + expect(full.structuredContent).toMatchObject({ items: [{ inputSchema: { required: ["title"] } }], }); }); }); - it("returns schemas and account details from search and marks invoke destructive", async () => { + it("returns compact hits by default: identity, one-line description, argument summary", async () => { + const { engine } = makeRecordingEngine(); + const catalog = [ + projection({ + integration: "github", + name: "issues.create", + description: + "Create an issue in a repository. Requires write access.\n\nLong second paragraph with details the model does not need to pick a tool.", + inputSchema: { + type: "object", + properties: { + title: { type: "string" }, + body: { $ref: "#/$defs/Body" }, + labels: { type: "array", items: { type: "string" } }, + state: { type: "string", enum: ["open", "closed"] }, + }, + required: ["title"], + $defs: { Body: { type: "string" } }, + }, + }), + ]; + await withClient({ engine, mode: "passthrough", tools: toolPort(catalog) }, async (client) => { + const result = await client.callTool({ + name: "search", + arguments: { query: "github issues create", limit: 1 }, + }); + expect(result.structuredContent).toEqual({ + items: [ + { + id: "tools.github.org.main.issues.create", + name: "issues.create", + integration: "github", + owner: "org", + connection: "main", + description: "Create an issue in a repository.", + arguments: + "title (string, required); body (string); labels (string[]); state (open|closed)", + }, + ], + total: 1, + hasMore: false, + nextOffset: null, + }); + expect(JSON.stringify(result.structuredContent)).not.toContain("inputSchema"); + expect(JSON.stringify(result.structuredContent)).not.toContain("second paragraph"); + }); + }); + + it("returns full schemas and account details on request and marks invoke destructive", async () => { const { engine } = makeRecordingEngine(); await withClient({ engine, mode: "passthrough", tools: toolPort(CATALOG) }, async (client) => { const listed = await client.listTools(); @@ -561,7 +991,7 @@ describe("passthrough mode server", () => { }); const result = await client.callTool({ name: "search", - arguments: { query: "github issues create", limit: 1 }, + arguments: { query: "github issues create", limit: 1, detail: "full" }, }); expect(result.structuredContent).toMatchObject({ items: [ diff --git a/packages/hosts/mcp/src/passthrough-tools.ts b/packages/hosts/mcp/src/passthrough-tools.ts index 5ad32f1f6e..50434cddf6 100644 --- a/packages/hosts/mcp/src/passthrough-tools.ts +++ b/packages/hosts/mcp/src/passthrough-tools.ts @@ -23,13 +23,21 @@ export const passthroughCallCode = (address: string, args: unknown): string => { return `return await tools[${JSON.stringify(bare)}](${JSON.stringify(args ?? {})});`; }; -/** Describe the fixed search/invoke surface without listing the underlying catalog. */ +/** + * The server `instructions`: the shortest path from a request to a call, + * with one worked example. Everything longer lives in the search-invoke + * guide, which costs context only when a model asks for it. + */ export const passthroughInstructions = (): string => - 'Use integrations to see connected accounts and skills({ name: "search-invoke" }) for the workflow. ' + - "Find connected integration tools with search, then call invoke with the returned tool ID and JSON arguments. " + - "Search returns input schemas and account details. Use its nextOffset to get more matches. " + - "Invoke can change external state; your client handles approval for each call. Workspace block policies remain enforced. " + - "No general execute or resume tools are exposed. When artifacts are enabled, use skills to read their guides."; + [ + "Executor runs tools of connected integrations. Fastest path, three calls:", + "1. integrations({}) once, to learn the integration slugs and accounts (skip when you already know the slug).", + '2. search({ query: "", integration: "", limit: 3 }). Hits are compact: id, one-line description, and an argument summary.', + "3. invoke({ tool: , arguments: { ... } }).", + 'Example: search({ query: "create issue", integration: "github", limit: 3 }) then invoke({ tool: "tools.github.org.main.issues.create", arguments: { title: "Bug" } }).', + 'Use detail: "full" on search only when the argument summary is not enough; an invoke rejected for invalid arguments also returns the schema.', + 'Invoke can change external state; your client handles approval and workspace blocks stay enforced. skills({ name: "search-invoke" }) has the long guide.', + ].join("\n"); /** On-demand guidance for the JSON tool surface; no sandbox or artifact instructions. */ export const SEARCH_INVOKE_SKILL: Skill = { @@ -39,11 +47,11 @@ export const SEARCH_INVOKE_SKILL: Skill = { "# Search and invoke", "", "1. Call `integrations({})` to see connected integrations and accounts. Each item includes an integration description, account label, and last recorded health. A null health verdict means the account has not been checked; a saved connection does not guarantee a working credential.", - '2. Call `search({ query: "create issue", integration: "github", owner: "org", connection: "main" })`. Use the exact integration, owner, and connection returned by integrations to select an account. Omit filters to search across accounts visible to you.', - "3. Read the matching tool's `inputSchema`. Call `invoke({ tool: , arguments: })`. Do not guess tool IDs or arguments.", + '2. Call `search({ query: "create issue", integration: "github", owner: "org", connection: "main" })`. Use the integration, owner, and connection returned by integrations to select an account. `integration` also accepts an unambiguous alias (`gmail` resolves to `google_gmail`); the result reports the resolved slug as `integration`, lists candidates as `integrations` when the alias is ambiguous, or adds a `hint` when nothing matches. Omit filters to search across accounts visible to you.', + '3. Read the hit\'s `arguments` summary (`name (type, required); other (type)`). Call `invoke({ tool: , arguments: })`. Do not guess tool IDs or arguments. When the summary is not enough, repeat the search with `detail: "full"` and `limit: 1` to get the complete `inputSchema`; an invoke rejected for invalid arguments also returns the schema.', "", "## Pagination", - "Both integrations and search return `{ items, total, hasMore, nextOffset }`. If hasMore is true, repeat the call with the same filters and `offset: nextOffset`. Search also needs the same query. Search returns at most 20 tools per page, with schemas only for those matches.", + "Both integrations and search return `{ items, total, hasMore, nextOffset }`. If hasMore is true, repeat the call with the same filters and `offset: nextOffset`. Search also needs the same query. Search returns at most 20 tools per page; prefer `limit: 3` with an `integration` filter.", "Tool-search pagination is separate from an upstream API's pagination. Follow the invoked tool's schema and response for cursor or page arguments when retrieving more records.", "", "## Results and approval", @@ -52,3 +60,147 @@ export const SEARCH_INVOKE_SKILL: Skill = { "This mode accepts JSON tool arguments. It does not expose general execute or resume tools. When artifacts are enabled, use create-artifact, edit-artifact, list-artifacts, and show-artifact; read the create-artifact and artifact-style guides through skills first. The skills tool serves only this server's guides, not files or skills from your harness or project.", ].join("\n"), }; + +// --------------------------------------------------------------------------- +// Compact search results +// --------------------------------------------------------------------------- + +/** Characters kept from a tool description in a compact search hit. */ +export const COMPACT_DESCRIPTION_LIMIT = 200; +/** Properties named in a compact argument summary before "+N more". */ +export const COMPACT_ARGUMENT_LIMIT = 12; +/** Enum values spelled out in a compact argument summary. */ +const COMPACT_ENUM_LIMIT = 6; +/** Keys named for a nested object argument. */ +const COMPACT_NESTED_KEY_LIMIT = 6; + +const isRecord = (value: unknown): value is Record => + typeof value === "object" && value !== null && !Array.isArray(value); + +/** + * The first sentence or line of a description, capped. Search hits are read + * once to pick a tool; the full text stays behind `detail: "full"`. + */ +export const compactDescription = (description: string | undefined): string => { + if (description === undefined) return ""; + const firstLine = description.split(/\r?\n/).find((line) => line.trim().length > 0) ?? ""; + const collapsed = firstLine.replace(/\s+/g, " ").trim(); + const sentence = collapsed.match(/^.*?[.!?](?=\s|$)/)?.[0] ?? collapsed; + // A very short first "sentence" ("No." / "v2.") is usually not one. + const kept = sentence.length >= 12 || sentence.length === collapsed.length ? sentence : collapsed; + return kept.length <= COMPACT_DESCRIPTION_LIMIT + ? kept + : `${kept.slice(0, COMPACT_DESCRIPTION_LIMIT - 1).trimEnd()}…`; +}; + +const resolveRef = ( + node: Record, + defs: ReadonlyMap, +): Record => { + const ref = node.$ref; + if (typeof ref !== "string") return node; + const name = ref.replace(/^#\/(\$defs|definitions)\//, ""); + const target = defs.get(name); + return isRecord(target) ? target : node; +}; + +/** A short type label for one JSON Schema node: `string`, `A|AAAA`, `string[]`, `object`. */ +const typeLabel = (value: unknown, defs: ReadonlyMap, depth = 0): string => { + if (!isRecord(value)) return "any"; + const node = resolveRef(value, defs); + if (Array.isArray(node.enum)) { + const values = node.enum.map((item) => String(item)); + return values.length <= COMPACT_ENUM_LIMIT + ? values.join("|") + : `${values.slice(0, COMPACT_ENUM_LIMIT).join("|")}|…`; + } + if (node.const !== undefined) return JSON.stringify(node.const); + const variants = [node.oneOf, node.anyOf].find(Array.isArray); + if (variants && depth < 1) { + const labels = [ + ...new Set(variants.map((variant: unknown) => typeLabel(variant, defs, depth + 1))), + ]; + return labels.length <= 3 ? labels.join("|") : "any"; + } + const type = Array.isArray(node.type) + ? node.type.filter((item) => item !== "null").join("|") + : typeof node.type === "string" + ? node.type + : undefined; + if (type === "array") { + const items = depth < 1 ? typeLabel(node.items, defs, depth + 1) : "any"; + return items.includes("|") ? `(${items})[]` : `${items}[]`; + } + if (type === "object" || (type === undefined && isRecord(node.properties))) + return objectLabel(node, depth); + return type ?? "any"; +}; + +/** + * `object{text*, title}`: the first-level keys of a nested object, required + * ones starred. OpenAPI tools wrap their inputs (`body`, `query`, `path`), so + * without this a summary would say only `body (object, required)`. + */ +const objectLabel = (node: Record, depth: number): string => { + if (depth >= 1 || !isRecord(node.properties)) return "object"; + const required = new Set(Array.isArray(node.required) ? node.required.map(String) : []); + const names = Object.keys(node.properties); + if (names.length === 0) return "object"; + const shown = names + .slice(0, COMPACT_NESTED_KEY_LIMIT) + .map((name) => (required.has(name) ? `${name}*` : name)); + const rest = names.length - shown.length; + return `object{${shown.join(", ")}${rest > 0 ? `, +${rest}` : ""}}`; +}; + +/** + * One line naming a tool's arguments: `name (type, required); other (type)`. + * Enough to call most tools without the full schema; `detail: "full"` has the + * rest. Schemas that are not a plain object come back as `see full schema`. + */ +export const compactArguments = ( + inputSchema: unknown, + definitions?: Record, +): string => { + if (!isRecord(inputSchema)) return "none"; + // Definitions may ride inside the schema (`$defs`) or beside it (the view's + // `schemaDefinitions`); both resolve the same `#/$defs/` pointers. + const defs = new Map([ + ...Object.entries(definitions ?? {}), + ...Object.entries(isRecord(inputSchema.definitions) ? inputSchema.definitions : {}), + ...Object.entries(isRecord(inputSchema.$defs) ? inputSchema.$defs : {}), + ]); + const schema = resolveRef(inputSchema, defs); + const properties = schema.properties; + if (!isRecord(properties)) { + if (Array.isArray(schema.oneOf) || Array.isArray(schema.anyOf) || Array.isArray(schema.allOf)) + return "see full schema"; + return "none"; + } + const required = new Set(Array.isArray(schema.required) ? schema.required.map(String) : []); + const names = Object.keys(properties); + if (names.length === 0) return "none"; + const ordered = [ + ...names.filter((name) => required.has(name)), + ...names.filter((name) => !required.has(name)), + ]; + const shown = ordered.slice(0, COMPACT_ARGUMENT_LIMIT).map((name) => { + const label = typeLabel(properties[name], defs); + return required.has(name) ? `${name} (${label}, required)` : `${name} (${label})`; + }); + const rest = ordered.length - shown.length; + return rest > 0 ? `${shown.join("; ")}; +${rest} more` : shown.join("; "); +}; + +// --------------------------------------------------------------------------- +// Integration aliases live in @executor-js/execution so the sandbox +// `tools.search({ namespace })` resolves them the same way; re-exported here +// so the host surface is unchanged. +// --------------------------------------------------------------------------- + +export { + integrationAliasText, + resolveIntegrationAlias, + type IntegrationCatalogEntry, + type IntegrationResolution, +} from "@executor-js/execution"; diff --git a/packages/hosts/mcp/src/standalone-stream.test.ts b/packages/hosts/mcp/src/standalone-stream.test.ts new file mode 100644 index 0000000000..9bff35c9ad --- /dev/null +++ b/packages/hosts/mcp/src/standalone-stream.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from "@effect/vitest"; + +import { serverInitiatedMessagesPossible, standaloneStreamNotOffered } from "./standalone-stream"; + +describe("serverInitiatedMessagesPossible", () => { + it("keeps the stream available before capabilities are negotiated", () => { + expect(serverInitiatedMessagesPossible(undefined)).toBe(true); + }); + + it("refuses the stream for a client that declared no capabilities", () => { + expect(serverInitiatedMessagesPossible({})).toBe(false); + }); + + it("offers the stream to clients the server may send requests to", () => { + expect(serverInitiatedMessagesPossible({ elicitation: {} })).toBe(true); + expect(serverInitiatedMessagesPossible({ elicitation: { form: {} } })).toBe(true); + expect(serverInitiatedMessagesPossible({ sampling: {} })).toBe(true); + expect(serverInitiatedMessagesPossible({ roots: { listChanged: true } })).toBe(true); + }); + + it("offers the stream to an MCP Apps host, whose tool list can change", () => { + expect( + serverInitiatedMessagesPossible({ + extensions: { "io.modelcontextprotocol/ui": { mimeTypes: ["text/html;profile=mcp-app"] } }, + } as never), + ).toBe(true); + }); + + it("ignores capabilities that admit no server-initiated message", () => { + expect(serverInitiatedMessagesPossible({ experimental: { anything: {} } })).toBe(false); + expect( + serverInitiatedMessagesPossible({ + extensions: { "io.modelcontextprotocol/ui": { mimeTypes: ["text/plain"] } }, + } as never), + ).toBe(false); + }); +}); + +describe("standaloneStreamNotOffered", () => { + it("is a 405 that names the methods the endpoint still serves", async () => { + const response = standaloneStreamNotOffered(); + expect(response.status).toBe(405); + expect(response.headers.get("allow")).toBe("POST, DELETE"); + expect(response.headers.get("content-type")).toBe("application/json"); + const body = (await response.json()) as { error: { code: number }; id: null }; + expect(body.error.code).toBe(-32000); + expect(body.id).toBeNull(); + }); +}); diff --git a/packages/hosts/mcp/src/standalone-stream.ts b/packages/hosts/mcp/src/standalone-stream.ts new file mode 100644 index 0000000000..2079044bb8 --- /dev/null +++ b/packages/hosts/mcp/src/standalone-stream.ts @@ -0,0 +1,50 @@ +import type { ClientCapabilities } from "@modelcontextprotocol/sdk/types.js"; +import { getUiCapability, RESOURCE_MIME_TYPE } from "@modelcontextprotocol/ext-apps/server"; + +// --------------------------------------------------------------------------- +// Whether a session's standalone `GET /mcp` stream can ever carry anything. +// +// The standalone SSE stream exists for server-initiated messages: requests the +// server makes of the client (elicitation, sampling, roots) and notifications +// such as a tool-list change when the MCP Apps capability toggles a tool. A +// client that negotiated none of those capabilities can never receive any of +// them, so the stream it opens stays silent for the life of the session. The +// streamable-HTTP spec lets a server answer that GET with 405 instead of +// offering a stream, and the client SDKs read 405 as "no stream here" and +// move on. Answering 405 saves such clients a request and the server an open +// stream per session — which matters for a gateway that opens a fresh session +// for every tool call. +// +// Capabilities arrive with `initialize`. Before that, or for any client that +// declared one of the capabilities above, the stream is offered as before. +// --------------------------------------------------------------------------- + +type ClientCapabilitiesWithExtensions = ClientCapabilities & { + readonly extensions?: Record; +}; + +export const serverInitiatedMessagesPossible = ( + capabilities: ClientCapabilities | undefined, +): boolean => { + if (!capabilities) return true; + if (capabilities.elicitation || capabilities.sampling || capabilities.roots) return true; + const ui = getUiCapability(capabilities as ClientCapabilitiesWithExtensions); + return Boolean(ui?.mimeTypes?.includes(RESOURCE_MIME_TYPE)); +}; + +/** The 405 a session answers its standalone GET with when no server-initiated + * message can ever reach this client. An inner response: the envelope adds + * CORS. `allow` names what the endpoint still serves. */ +export const standaloneStreamNotOffered = (): Response => + new Response( + JSON.stringify({ + jsonrpc: "2.0", + error: { + code: -32000, + message: + "Standalone SSE stream not offered: the negotiated client capabilities admit no server-initiated messages", + }, + id: null, + }), + { status: 405, headers: { "content-type": "application/json", allow: "POST, DELETE" } }, + ); diff --git a/packages/hosts/mcp/src/tool-server.ts b/packages/hosts/mcp/src/tool-server.ts index f7c1a349f7..3ee7a5c8b5 100644 --- a/packages/hosts/mcp/src/tool-server.ts +++ b/packages/hosts/mcp/src/tool-server.ts @@ -87,10 +87,16 @@ import { } from "./artifact-bindings"; import { MCP_ORG_WRITE_ACCESS_HEADER } from "./seams"; import { + compactArguments, + compactDescription, + integrationAliasText, + type IntegrationResolution, passthroughCallCode, passthroughInstructions, + resolveIntegrationAlias, SEARCH_INVOKE_SKILL, } from "./passthrough-tools"; +import { codemodeInstructions } from "./codemode-instructions"; import type { McpToolMode } from "./browser-approval"; // --------------------------------------------------------------------------- @@ -321,8 +327,9 @@ export type McpIntegrationsPort = { >; }; -/** The same list and schema APIs used by codemode discovery. */ -export type McpToolsPort = Pick; +/** The same list and schema APIs used by codemode discovery, plus the batched + * schema read a search page uses. */ +export type McpToolsPort = Pick; /** A passthrough session was requested but the host gave the factory no * catalog to serve. A configuration defect, not a runtime condition. */ @@ -1267,6 +1274,32 @@ const passthroughInputSchema = (view: ToolSchemaView): unknown => new Map(Object.entries(view.schemaDefinitions ?? {})), ); +/** + * The slugs an `integration` argument selects, and the envelope fields that + * tell the model how the alias resolved. An exact slug adds nothing; an alias + * reports the slug it became; an ambiguous alias reports every candidate and + * selects all of them; an unknown value stays an exact filter and adds a hint. + */ +const selectedIntegrations = ( + resolution: IntegrationResolution | undefined, + input: string | undefined, +): { readonly slugs: ReadonlySet | null; readonly envelope: Record } => { + if (resolution === undefined || input === undefined) return { slugs: null, envelope: {} }; + if (resolution.kind === "exact") return { slugs: new Set([resolution.slug]), envelope: {} }; + if (resolution.kind === "alias") + return { slugs: new Set([resolution.slug]), envelope: { integration: resolution.slug } }; + if (resolution.kind === "ambiguous") + return { slugs: new Set(resolution.slugs), envelope: { integrations: resolution.slugs } }; + // Keep the value as an exact filter (it may be a slug the metadata catalog + // does not list), and say what to do if it was a typo. + return { + slugs: new Set([input]), + envelope: { + hint: `No integration matches "${input}". Call integrations to list the connected slugs.`, + }, + }; +}; + /** Register discovery over the existing APIs, with no catalog work at connection time. */ const registerPassthroughTools = ( server: McpServer, @@ -1300,9 +1333,14 @@ const registerPassthroughTools = ( "integrations", { description: - "List connected integrations and accounts visible to you. Returns integration descriptions, account labels, exact search filters, and last recorded health (null means unchecked). One item per account; use nextOffset for more. Does not load tool schemas or check credentials.", + "List connected integrations and accounts: slug, name, owner, connection, last health. Call once to learn the exact slug for search; skip it when you already know the slug.", inputSchema: { - integration: z.string().trim().min(1).optional().describe("Exact integration slug."), + integration: z + .string() + .trim() + .min(1) + .optional() + .describe("Slug or alias (gmail resolves to google_gmail)."), owner: z.enum(["org", "user"]).optional(), limit: z.number().int().min(1).max(50).default(20), offset: z.number().int().min(0).default(0), @@ -1317,12 +1355,25 @@ const registerPassthroughTools = ( integrations.list(), ]); const metadata = new Map(catalog.map((item) => [String(item.slug), item])); + const selected = selectedIntegrations( + integration === undefined + ? undefined + : resolveIntegrationAlias( + integration, + catalog.map((item) => ({ + slug: String(item.slug), + name: item.name, + description: item.description, + })), + ), + integration, + ); const visible = accounts .flatMap((account) => { const item = metadata.get(account.integration); if ( !item || - (integration !== undefined && account.integration !== integration) || + (selected.slugs !== null && !selected.slugs.has(account.integration)) || (owner !== undefined && account.owner !== owner) ) return []; @@ -1358,6 +1409,7 @@ const registerPassthroughTools = ( total: visible.length, hasMore, nextOffset: hasMore ? offset + items.length : null, + ...selected.envelope, }; return { content: [{ type: "text" as const, text: JSON.stringify(result) }], @@ -1371,79 +1423,133 @@ const registerPassthroughTools = ( "search", { description: - "Search connected integration tools by action, integration, or account. Returns matching tool IDs, account details, and full JSON input schemas. Pass the returned ID and arguments to invoke. Use integrations to discover accounts, then pass exact integration, owner, and connection filters. Use nextOffset to page through matches.", + "Find tools by action words. Fastest: pass integration (slug or alias) and limit 3. Each hit has id, one-line description, and arguments (name (type, required)); pass id and arguments to invoke. detail: full adds the complete inputSchema (large) when the summary is not enough. Page with offset: nextOffset.", inputSchema: { query: z .string() .trim() .min(1) .max(500) - .describe("Keywords describing the tool or task, such as github create issue."), + .describe("Action words, such as create issue or list dns records."), integration: z .string() .trim() .min(1) .optional() - .describe("Exact integration slug from integrations."), + .describe( + "Slug or alias (gmail resolves to google_gmail); the result reports the slug.", + ), owner: z.enum(["org", "user"]).optional(), connection: z .string() .trim() .min(1) .optional() - .describe( - "Exact account name from integrations; pair with integration and owner to select one account.", - ), + .describe("Account name from integrations; pair with integration and owner."), limit: z.number().int().min(1).max(20).default(10), offset: z.number().int().min(0).default(0), + detail: z + .enum(["compact", "full"]) + .default("compact") + .describe("full adds the complete inputSchema per hit; use with limit 1-3."), }, annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false }, }, - ({ query, integration, owner, connection, limit, offset }, extra) => + ({ query, integration, owner, connection, limit, offset, detail }, extra) => boundary( Effect.gen(function* () { + // Catalog metadata is cheap and lets `integration` accept the + // name a person would use ("gmail" → google_gmail) and lets + // query words that name an integration rank its tools. + const catalog = (yield* integrations.list()).map((item) => ({ + slug: String(item.slug), + name: item.name, + description: item.description, + })); + const selected = selectedIntegrations( + integration === undefined + ? undefined + : resolveIntegrationAlias(integration, catalog), + integration, + ); + // One slug narrows the list read itself (the fast path); several + // candidates read the whole visible catalog once and keep theirs. + const exactSlug = + selected.slugs !== null && selected.slugs.size === 1 + ? [...selected.slugs][0] + : undefined; const discovery = { tools: { list: (filter?: Parameters[0]) => tools .list({ ...filter, - ...(integration === undefined + ...(exactSlug === undefined ? {} - : { integration: IntegrationSlug.make(integration) }), + : { integration: IntegrationSlug.make(exactSlug) }), ...(owner === undefined ? {} : { owner }), ...(connection === undefined ? {} : { connection: ConnectionName.make(connection) }), }) - .pipe(Effect.map((items) => items.filter((tool) => tool.static !== true))), + .pipe( + Effect.map((items) => + items.filter( + (tool) => + tool.static !== true && + (selected.slugs === null || selected.slugs.has(tool.integration)), + ), + ), + ), }, }; - const page = yield* searchTools(discovery, query, limit, { offset }); - const candidates = yield* Effect.forEach( - page.items, - (match) => - Effect.gen(function* () { - const address = ToolAddress.make(`tools.${match.path}`); - const identity = parseToolAddress(String(address)); - if (!identity) return null; - const schema = yield* tools.schema(address); - // Visibility can change between listing and schema lookup. - if (!schema) return null; - return { - id: String(address), - name: match.name, - integration: identity.integration, - owner: identity.owner, - connection: identity.connection, + const page = yield* searchTools(discovery, query, limit, { + offset, + integrationAliases: integrationAliasText(catalog), + }); + const addresses = page.items.map((match) => ToolAddress.make(`tools.${match.path}`)); + // One read for the page: one policy snapshot, one catalog and + // one definitions read per connection. Search returns JSON + // Schema only; skip the TypeScript preview. + const schemas = yield* tools.schemas(addresses, { typeScript: false }); + const candidates = page.items.map((match, index) => { + const address = addresses[index]!; + const identity = parseToolAddress(String(address)); + if (!identity) return null; + const schema = schemas[index] ?? null; + // Visibility can change between listing and schema lookup. + if (!schema) return null; + const identityFields = { + id: String(address), + name: match.name, + integration: identity.integration, + owner: identity.owner, + connection: identity.connection, + }; + const annotations = schema.annotations ? { annotations: schema.annotations } : {}; + // Compact hits carry what a model needs to choose a tool + // and usually to call it; the full schema is one more + // search away (`detail: "full"`) or comes back with an + // invoke validation error. + return detail === "full" + ? { + ...identityFields, description: match.description, inputSchema: passthroughInputSchema(schema), - ...(schema.annotations ? { annotations: schema.annotations } : {}), + ...annotations, + } + : { + ...identityFields, + description: compactDescription(match.description), + arguments: compactArguments(schema.inputSchema, schema.schemaDefinitions), + ...annotations, }; - }), - { concurrency: 4 }, - ); - const result = { ...page, items: candidates.filter(Predicate.isNotNull) }; + }); + const result = { + ...page, + items: candidates.filter(Predicate.isNotNull), + ...selected.envelope, + }; return { content: [{ type: "text" as const, text: JSON.stringify(result) }], structuredContent: result, @@ -1456,12 +1562,12 @@ const registerPassthroughTools = ( "invoke", { description: - "Call one connected integration tool using the exact ID and JSON input schema returned by search. May read or change external state. Your client handles approval for this call; workspace blocks remain enforced.", + "Call one tool by the exact id from search with a JSON arguments object. Returns the tool's result; an invalid-arguments error includes the expected inputSchema, so correct and retry. May change external state; your client handles approval and workspace blocks stay enforced.", inputSchema: { - tool: z.string().min(1).describe("Exact tool ID returned by search."), + tool: z.string().min(1).describe("Exact id from a search hit."), arguments: z .record(z.string(), z.unknown()) - .describe("Tool arguments matching the inputSchema returned by search."), + .describe("JSON object matching the hit's arguments summary or inputSchema."), }, annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: true }, }, @@ -1490,7 +1596,7 @@ const registerPassthroughTools = ( }); if (!visible.some((tool) => tool.static !== true && tool.address === address)) return unavailable; - const schema = yield* tools.schema(address); + const schema = yield* tools.schema(address, { typeScript: false }); if (!schema) return unavailable; // The SDK validator checks this dynamic JSON schema at the MCP boundary. const validate = validator.getValidator( @@ -1503,7 +1609,9 @@ const registerPassthroughTools = ( content: [ { type: "text" as const, - text: `Invalid arguments for tool ${id}: ${checked.errorMessage ?? "invalid"}`, + // The schema rides along so the model can correct the + // call without another search. + text: `Invalid arguments for tool ${id}: ${checked.errorMessage ?? "invalid"}. Expected inputSchema: ${JSON.stringify(passthroughInputSchema(schema))}`, }, ], }; @@ -1524,12 +1632,17 @@ export const createExecutorMcpServer = ( ): Effect.Effect => Effect.gen(function* () { const engine = "engine" in config ? config.engine : createExecutionEngine(config); - const description = - config.description ?? - (yield* engine.getDescription.pipe(Effect.withSpan("mcp.host.get_description"))); - // The same live integration inventory the description carries, re-used by - // the `skills` tool so the `execute` guide lists what is connected too. - const executeInventory = extractInventory(description); + // The `execute` description reads the live connection and integration + // inventory, two catalog reads per session. Codemode needs it at + // construction: it is the `execute` tool's description and the source of + // the per-integration search tools. Passthrough registers neither, so it + // reads the description only on demand (the `execute` guide from `skills`), + // once per session — a session that never asks never pays for it. + const loadDescription = yield* Effect.cached( + config.description !== undefined + ? Effect.succeed(config.description) + : engine.getDescription.pipe(Effect.withSpan("mcp.host.get_description")), + ); // Artifacts are on unless this connection opted out (`?artifacts=false`). // One flag decides the whole surface: the tools, the shell resource, and // the skills catalog below. @@ -1559,6 +1672,10 @@ export const createExecutorMcpServer = ( "passthrough mode requires tool list/schema, connection list, and integration list APIs", }); } + const description = passthrough ? "" : yield* loadDescription; + // The same live integration inventory the description carries, re-used by + // the `skills` tool so the `execute` guide lists what is connected too. + const executeInventory = loadDescription.pipe(Effect.map(extractInventory)); // Captured at construction time. SDK callbacks fire later (often // deferred past the outer Effect's await), so we use the runtime to @@ -1659,11 +1776,7 @@ export const createExecutorMcpServer = ( // per host. capabilities: { resources: {}, tools: {} }, jsonSchemaValidator: new CfWorkerJsonSchemaValidator(), - ...(passthrough - ? { - instructions: passthroughInstructions(), - } - : {}), + instructions: passthrough ? passthroughInstructions() : codemodeInstructions(), }, ), ).pipe(Effect.withSpan("mcp.host.create_server")); @@ -2074,7 +2187,7 @@ export const createExecutorMcpServer = ( { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false }, description: passthrough - ? 'Documentation for this server only, not harness or project skills. Call with no name to list guides, or skills({ name: "search-invoke" }) for account discovery, tool search, invocation, and pagination.' + ? 'Guides for this server only, not harness or project skills. skills({ name: "search-invoke" }) explains discovery, aliases, pagination, and approval; no name lists the guides.' : [ "Documentation for THIS server's own tools. Not a general skill reader: it serves a short, fixed set of how-to docs about using `execute` and artifacts here, and it cannot reach your harness's skills, a SKILL.md on disk, or any user- or project-authored skill. The argument is a name from its own catalog, never a path or an outside skill's id.", "These docs hold the long-form guidance that would otherwise bloat another tool's always-loaded description.", @@ -2086,12 +2199,19 @@ export const createExecutorMcpServer = ( .string() .optional() .describe( - `A doc from this server's own catalog, e.g. "${passthrough ? "search-invoke" : "execute"}". Omit to list the catalog.`, + passthrough + ? 'Guide name, e.g. "search-invoke". Omit to list guides.' + : `A doc from this server's own catalog, e.g. "execute". Omit to list the catalog.`, ), }, }, ({ name }, extra) => - runToolEffect(Effect.succeed(skillsResult(name, executeInventory, skillCatalog)), extra), + runToolEffect( + executeInventory.pipe( + Effect.map((inventory) => skillsResult(name, inventory, skillCatalog)), + ), + extra, + ), ), ).pipe( Effect.withSpan("mcp.host.register_tool", { diff --git a/packages/kernel/core/src/types.ts b/packages/kernel/core/src/types.ts index d6dee10ad4..5002ecb621 100644 --- a/packages/kernel/core/src/types.ts +++ b/packages/kernel/core/src/types.ts @@ -48,6 +48,11 @@ export type ExecuteResult = { logs?: string[]; /** Successful connected-tool paths observed during this execution. */ toolPaths?: readonly string[]; + /** Distinct attempted connected tools and bounded outcomes; no call payloads. */ + toolCalls?: readonly { + readonly path: string; + readonly status: "ok" | "error" | "blocked"; + }[]; }; /** diff --git a/packages/plugins/encrypted-secrets/src/index.ts b/packages/plugins/encrypted-secrets/src/index.ts index feae781e8a..99e1b6ca5d 100644 --- a/packages/plugins/encrypted-secrets/src/index.ts +++ b/packages/plugins/encrypted-secrets/src/index.ts @@ -42,6 +42,19 @@ const PAYLOAD_VERSION = "v1"; /** Derive a 32-byte AES key from an arbitrary-length master key string. */ const deriveKey = (master: string): Buffer => scryptSync(master, KEY_SALT, 32); +// Plugin factories run for each session. Only key derivation is reusable: +// providers below still bind fresh storage and ownership for each executor. +// Keep one entry so rotating a host key replaces the cache without retaining +// every key a long-lived process has used. Existing providers keep their key. +let cachedKey: { readonly master: string; readonly derived: Buffer } | undefined; + +const pluginKey = (master: string): Buffer => { + if (cachedKey?.master === master) return cachedKey.derived; + const derived = deriveKey(master); + cachedKey = { master, derived }; + return derived; +}; + const encryptSecret = (key: Buffer, plaintext: string): Effect.Effect => Effect.try({ try: () => { @@ -138,7 +151,7 @@ export const encryptedSecretsPlugin = definePlugin((options?: EncryptedSecretsPl // oxlint-disable-next-line executor/no-try-catch-or-throw, executor/no-error-constructor -- boundary: a secret store with no master key is unsafe; fail loud at construction throw new Error("encryptedSecretsPlugin requires a non-empty `key`"); } - const derivedKey = deriveKey(master); + const derivedKey = pluginKey(master); return { id: "encryptedSecrets" as const, storage: () => ({}), diff --git a/packages/plugins/encrypted-secrets/src/key-cache.test.ts b/packages/plugins/encrypted-secrets/src/key-cache.test.ts new file mode 100644 index 0000000000..2cd0982d99 --- /dev/null +++ b/packages/plugins/encrypted-secrets/src/key-cache.test.ts @@ -0,0 +1,99 @@ +import { Effect, Exit } from "effect"; +import { afterEach, describe, expect, test } from "@effect/vitest"; +// oxlint-disable-next-line executor/no-vitest-import -- boundary: Vitest hoists vi.mock only with a direct vitest import +import { vi } from "vitest"; +import { scryptSync } from "node:crypto"; +import { ProviderItemId, type CredentialProvider, type PluginCtx } from "@executor-js/sdk"; + +import { decryptSecret, deriveKey, encryptSecret, encryptedSecretsPlugin } from "./index"; + +vi.mock("node:crypto", async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, scryptSync: vi.fn(actual.scryptSync) }; +}); + +afterEach(() => vi.mocked(scryptSync).mockClear()); + +const providerFor = (key: string, ctx: PluginCtx): CredentialProvider => { + const plugin = encryptedSecretsPlugin({ key }); + const providers = plugin.credentialProviders as ( + ctx: PluginCtx, + ) => readonly CredentialProvider[]; + return providers(ctx)[0]!; +}; + +describe("session key derivation", () => { + test("derives once for repeated plugin builds and invalidates on every key change", () => { + encryptedSecretsPlugin({ key: "cache-a" }); + encryptedSecretsPlugin({ key: "cache-a" }); + expect(scryptSync).toHaveBeenCalledTimes(1); + encryptedSecretsPlugin({ key: "cache-b" }); + encryptedSecretsPlugin({ key: "cache-b" }); + expect(scryptSync).toHaveBeenCalledTimes(2); + encryptedSecretsPlugin({ key: "cache-a" }); + expect(scryptSync).toHaveBeenCalledTimes(3); + expect(() => encryptedSecretsPlugin({ key: "" })).toThrow(/non-empty/); + expect(scryptSync).toHaveBeenCalledTimes(3); + }); + + test("cached keys keep provider storage and ownership separate", async () => { + const getA = vi.fn(() => Effect.succeed(null)); + const getB = vi.fn(() => Effect.succeed(null)); + const putA = vi.fn(() => Effect.succeed(undefined)); + const putB = vi.fn(() => Effect.succeed(undefined)); + // oxlint-disable-next-line executor/no-double-cast -- boundary: only get and put are used by this provider test + const ctxA = { + owner: { tenant: "a", subject: "alice" }, + pluginStorage: { get: getA, put: putA }, + } as unknown as PluginCtx; + // oxlint-disable-next-line executor/no-double-cast -- boundary: only get and put are used by this provider test + const ctxB = { + owner: { tenant: "b", subject: "bob" }, + pluginStorage: { get: getB, put: putB }, + } as unknown as PluginCtx; + const a = providerFor("isolation-key", ctxA); + const b = providerFor("isolation-key", ctxB); + expect(scryptSync).toHaveBeenCalledTimes(1); + expect(a).not.toBe(b); + const id = ProviderItemId.make("test"); + await Effect.runPromise(a.set!(id, "alice-value")); + await Effect.runPromise(b.set!(id, "bob-value")); + expect(putA).toHaveBeenCalledWith(expect.objectContaining({ owner: "user", key: "test" })); + expect(putB).toHaveBeenCalledWith(expect.objectContaining({ owner: "user", key: "test" })); + await Effect.runPromise(a.get(id)); + expect(getA).toHaveBeenCalledTimes(1); + expect(getB).not.toHaveBeenCalled(); + await Effect.runPromise(b.get(id)); + expect(getB).toHaveBeenCalledTimes(1); + }); + + test("rotation preserves old providers and uses the new key for new providers", async () => { + let payload = ""; + // oxlint-disable-next-line executor/no-double-cast -- boundary: this test uses only provider storage get and put + const ctx = { + owner: { tenant: "rotation", subject: "alice" }, + pluginStorage: { + get: () => Effect.succeed({ data: payload }), + put: ({ data }: { data: string }) => + Effect.sync(() => { + payload = data; + }), + }, + } as unknown as PluginCtx; + const old = providerFor("rotation-old", ctx); + const next = providerFor("rotation-new", ctx); + const id = ProviderItemId.make("test"); + await Effect.runPromise(old.set!(id, "old-value")); + expect(await Effect.runPromise(old.get(id))).toBe("old-value"); + expect(Exit.isFailure(await Effect.runPromiseExit(next.get(id)))).toBe(true); + await Effect.runPromise(next.set!(id, "new-value")); + expect(await Effect.runPromise(next.get(id))).toBe("new-value"); + expect(Exit.isFailure(await Effect.runPromiseExit(old.get(id)))).toBe(true); + expect(await Effect.runPromise(decryptSecret(deriveKey("rotation-new"), payload))).toBe( + "new-value", + ); + const legacy = await Effect.runPromise(encryptSecret(deriveKey("rotation-old"), "legacy")); + payload = legacy; + expect(await Effect.runPromise(old.get(id))).toBe("legacy"); + }); +}); diff --git a/packages/plugins/mcp/src/sdk/catalog-sync.test.ts b/packages/plugins/mcp/src/sdk/catalog-sync.test.ts index f3328876b3..7963353082 100644 --- a/packages/plugins/mcp/src/sdk/catalog-sync.test.ts +++ b/packages/plugins/mcp/src/sdk/catalog-sync.test.ts @@ -469,6 +469,29 @@ const serveLatchedMutableServer = () => }); describe("MCP stale-refresh grace budget", () => { + it.live("zero grace returns cached tools while a listing is parked, then converges", () => + Effect.gen(function* () { + const fixture = yield* serveLatchedMutableServer(); + const executor = yield* makeCatalogTestExecutor(fixture.url, { + toolsSyncTtlMs: 0, + toolsSyncGraceMs: 0, + }); + const before = toolNames(yield* executor.tools.list()); + expect(before).toContain("alpha"); + yield* fixture.rename; + yield* fixture.arm; + const cached = yield* executor.tools.list().pipe(Effect.timeoutOption("500 millis")); + expect(Option.isSome(cached)).toBe(true); + expect(toolNames(Option.getOrThrow(cached))).toEqual(before); + yield* fixture.release; + const converged = yield* Effect.gen(function* () { + while (!toolNames(yield* executor.tools.list()).includes("beta")) { + yield* Effect.sleep("10 millis"); + } + }).pipe(Effect.timeoutOption("10 seconds")); + expect(Option.isSome(converged)).toBe(true); + }), + ); // `it.live` (real clock): the grace timeout must actually fire while a real // HTTP listing stays parked. it.live("a read outlasting the grace serves the stored catalog, then converges", () =>