From 217f28a08681ca7afa19b18544df376a8e2e1758 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Wed, 30 Sep 2026 01:34:57 -0700 Subject: [PATCH 1/4] fix(sandbox): withhold workbench certification after an unprovenanced file mount A persistent chat workbench stays "clean" only while every input it received was classified secret-free, and the scratch-file read hands its bytes to the model on that basis. File mounts resolved from platform file objects were counted only when a provenance source existed, so a mount whose key has no canonical metadata record (or no principal to bind one) left the machine certified. Both the `files` parameter and a mount marker in context variables reach this resolver from model-supplied Function parameters. The resolver now reports how many mounts had no provenance source. When a workbench session receives any, the session request carries `unprovenancedInputs` and the code boundary records the machine as unknown. Workflow runs have no session and keep their existing absence policy. --- .../sim/lib/execution/remote-sandbox/index.ts | 8 +- .../session-input-certification.test.ts | 112 ++++++++++++++++++ .../sim/lib/execution/remote-sandbox/types.ts | 5 + .../execute-request.test.ts | 27 +++++ .../lib/function-execution/execute-request.ts | 6 +- .../function-execution/sandbox-mounts.test.ts | 36 ++++++ .../lib/function-execution/sandbox-mounts.ts | 26 ++-- 7 files changed, 209 insertions(+), 11 deletions(-) create mode 100644 apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts diff --git a/apps/sim/lib/execution/remote-sandbox/index.ts b/apps/sim/lib/execution/remote-sandbox/index.ts index e1bd2fb1638..f3ea137224a 100644 --- a/apps/sim/lib/execution/remote-sandbox/index.ts +++ b/apps/sim/lib/execution/remote-sandbox/index.ts @@ -891,7 +891,9 @@ async function executeInSandboxWithinBudget( await recordSessionFileInput( req.session.key, { providerId: created.providerId, sandboxId }, - sandboxSessionInputsSafe() && !Object.keys(selected?.envs ?? {}).length + sandboxSessionInputsSafe() && + !req.session.unprovenancedInputs && + !Object.keys(selected?.envs ?? {}).length ) await provisionWithinBudget(sandbox, selected, signal) await writeSandboxInputs(sandbox, req.sandboxFiles, { @@ -1078,7 +1080,9 @@ async function executeShellInSandboxWithinBudget( await recordSessionFileInput( req.session.key, { providerId: created.providerId, sandboxId }, - sandboxSessionInputsSafe() && !Object.keys(selected?.envs ?? {}).length + sandboxSessionInputsSafe() && + !req.session.unprovenancedInputs && + !Object.keys(selected?.envs ?? {}).length ) await provisionWithinBudget(sandbox, selected, signal) await writeSandboxInputs(sandbox, req.sandboxFiles, { diff --git a/apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts b/apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts new file mode 100644 index 00000000000..afa89b4de6b --- /dev/null +++ b/apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts @@ -0,0 +1,112 @@ +/** + * The persistent workbench's input history is what lets a scratch file reach the model. These + * run the real code boundary against the real history recorder, so a mount the caller could not + * classify has to leave the machine uncertified. + */ +import { redisConfigMockFns } from '@sim/testing/mocks/redis-config.mock' +import { + remoteSandboxProviderMock, + remoteSandboxProviderMockFns, +} from '@sim/testing/mocks/remote-sandbox-provider.mock' +import { generateShortId } from '@sim/utils/id' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { CodeLanguage } from '@/lib/execution/languages' +import type { SandboxHandle, SandboxProvider } from '@/lib/execution/remote-sandbox/types' + +const { mockFindSessionSandbox } = vi.hoisted(() => ({ mockFindSessionSandbox: vi.fn() })) + +vi.mock('@/lib/execution/remote-sandbox/provider', () => remoteSandboxProviderMock) +vi.mock('@/lib/execution/remote-sandbox/resolve', () => ({ + resolveWorkspaceSandbox: vi.fn().mockResolvedValue(null), + provisionRuntimeDependencies: vi.fn(), + repairMissingSandboxImage: vi.fn().mockResolvedValue(null), + RUNTIME_INSTALL_TIMEOUT_MS: 60_000, +})) +vi.mock('@/lib/core/execution-limits/metrics', () => ({ + recordSandboxTeardownFailure: vi.fn(), + recordSandboxProviderLimit: vi.fn(), +})) + +import { + executeInSandbox, + executeShellInSandbox, + SIM_RESULT_PREFIX, +} from '@/lib/execution/remote-sandbox' +import { observeSandboxSessionInputs } from '@/lib/execution/remote-sandbox/execution-observer' +import { + initializeSessionFileProvenance, + isSessionFileProvenanceClean, +} from '@/lib/execution/remote-sandbox/session-file-provenance' + +/** Same one-way semantics as the Lua history script. */ +const records = new Map() +redisConfigMockFns.mockGetRedisClient.mockImplementation(() => ({ + set: async (key: string, value: string) => { + if (!records.has(key)) records.set(key, value) + return 'OK' + }, + get: async (key: string) => records.get(key) ?? null, + eval: async (_script: string, _count: number, key: string, input: string) => { + records.set(key, records.get(key) === 'clean' && input === 'clean' ? 'clean' : 'unknown') + return records.get(key) + }, +})) +remoteSandboxProviderMockFns.mockResolveProvider.mockImplementation( + (): SandboxProvider => ({ + id: 'e2b', + dependencyStrategy: 'prebuilt', + resolveLifetimeMs: (ms: number) => ms, + create: vi.fn(), + findSessionSandbox: mockFindSessionSandbox, + }) +) + +function machine(sandboxId: string): SandboxHandle { + return { + sandboxId, + runCode: async () => ({ text: `${SIM_RESULT_PREFIX}{"ok":true}`, stdout: '', stderr: '' }), + runCommand: async () => ({ stdout: '', stderr: '', exitCode: 0 }), + extendLifetime: async () => {}, + getFileSize: async () => 0, + readFile: async () => '', + readFileWithLimit: async () => ({ content: '', byteLength: 0 }), + writeFile: async () => {}, + removeFile: async () => {}, + listFiles: async () => [], + kill: async () => {}, + } +} + +beforeEach(() => { + records.clear() +}) + +describe('workbench certification at the code boundary', () => { + it.each([ + ['code', false], + ['code', true], + ['shell', false], + ['shell', true], + ] as const)('%s with unprovenanced mounts %s', async (kind, unprovenanced) => { + const sandboxId = `machine-${generateShortId(8)}` + const key = `chat-${generateShortId(8)}` + const identity = { providerId: 'e2b', sandboxId } as const + mockFindSessionSandbox.mockResolvedValue(machine(sandboxId)) + await initializeSessionFileProvenance(key, identity) + const session = { key, ...(unprovenanced ? { unprovenancedInputs: true } : {}) } + const request = { + code: 'print(1)', + language: CodeLanguage.Python, + timeoutMs: 30_000, + session, + } + await observeSandboxSessionInputs( + () => true, + () => + kind === 'code' + ? executeInSandbox(request) + : executeShellInSandbox({ ...request, envs: {} }) + ) + expect(await isSessionFileProvenanceClean(key, identity)).toBe(!unprovenanced) + }) +}) diff --git a/apps/sim/lib/execution/remote-sandbox/types.ts b/apps/sim/lib/execution/remote-sandbox/types.ts index 5ff085e4526..473a4aa3732 100644 --- a/apps/sim/lib/execution/remote-sandbox/types.ts +++ b/apps/sim/lib/execution/remote-sandbox/types.ts @@ -91,6 +91,11 @@ export interface SandboxSessionRequest { cli?: { path: string; content: string; runtime?: { path: string; content: string } } /** Extra environment variables present on every execution in the session. */ envs?: Record + /** + * This execution mounts bytes whose secret provenance is unknown, so the machine's input + * history must not stay certified clean even when the caller's own inputs are. + */ + unprovenancedInputs?: boolean } export interface SandboxShellExecutionRequest { diff --git a/apps/sim/lib/function-execution/execute-request.test.ts b/apps/sim/lib/function-execution/execute-request.test.ts index 474c5688f7f..9c192e3759a 100644 --- a/apps/sim/lib/function-execution/execute-request.test.ts +++ b/apps/sim/lib/function-execution/execute-request.test.ts @@ -153,6 +153,7 @@ vi.mock('@/lib/function-execution/sandbox-mounts', () => ({ }) => ({ contributingFiles: mockMountContributors(), renderedContributingFiles: mockRenderedMountContributors(), + unprovenancedMountCount: mockMountContributors() ? 0 : planned.length, sandboxFiles: planned.map(({ mountPath }) => ({ type: 'url' as const, path: mountPath, @@ -2560,6 +2561,32 @@ describe('Function execution request', () => { expect(mockWriteWorkspaceFileByPath).not.toHaveBeenCalled() }) + it.each([ + ['a mount with no provenance source', true], + ['no mounts', false], + ] as const)('withholds workbench certification for %s', async (_label, mounted) => { + envFlagsMock.isMothershipSandboxEnabled = true + hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({ + success: true, + userId: 'user-123', + authType: 'internal_jwt', + sandboxProfile: 'mothership', + }) + const response = await POST( + createMockRequest('POST', { + code: 'x', + language: 'python', + workspaceId: 'workspace-1', + sandboxSessionKey: 'chat-session', + ...(mounted ? { contextVariables: { doc: MOUNT_REF } } : {}), + }) + ) + expect(response.status).toBe(200) + const session = mockExecuteInSandbox.mock.calls.at(-1)?.[0].session + expect(session.key).toBe('chat-session') + expect(session.unprovenancedInputs === true).toBe(mounted) + }) + it('gives overlapping calls in one persistent workbench distinct automatic export directories', async () => { envFlagsMock.isMothershipSandboxEnabled = true hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({ diff --git a/apps/sim/lib/function-execution/execute-request.ts b/apps/sim/lib/function-execution/execute-request.ts index dd5b2f8e752..cdc1e7a7067 100644 --- a/apps/sim/lib/function-execution/execute-request.ts +++ b/apps/sim/lib/function-execution/execute-request.ts @@ -2385,7 +2385,7 @@ export async function executeFunctionRequest( // would leave `{{OTHER_SECRET}}` resolving, which is a hole, not a scope. const envVars = scopeEnvironmentVariables(rawEnvVars, secretScope, mountedSecrets) const admittedChatOwner = activeSandboxChatOwner() - const mothershipSession = + const admittedSession = usesMothershipSandbox && !selectedSandboxId && sandboxSessionKey && @@ -2716,6 +2716,10 @@ export async function executeFunctionRequest( ) } const { sandboxFiles: userFileMounts, manifest: mountManifest } = resolvedMounts + const mothershipSession = + admittedSession && resolvedMounts.unprovenancedMountCount > 0 + ? { ...admittedSession, unprovenancedInputs: true } + : admittedSession const sandboxFiles = mergeSandboxFileMounts(_sandboxFiles, userFileMounts) // Every `` marker becomes the path its file was mounted at, diff --git a/apps/sim/lib/function-execution/sandbox-mounts.test.ts b/apps/sim/lib/function-execution/sandbox-mounts.test.ts index 6ad4c418829..570649e2008 100644 --- a/apps/sim/lib/function-execution/sandbox-mounts.test.ts +++ b/apps/sim/lib/function-execution/sandbox-mounts.test.ts @@ -214,6 +214,42 @@ describe('resolveUserFileMounts', () => { expect(mockDownloadServableFileFromStorage).not.toHaveBeenCalled() }) + /** + * A persistent workbench certifies its machine from these counts: a mount whose bytes have no + * provenance source can hold resolved secret plaintext nobody recorded, so it must be reported. + */ + it.each([ + ['has no metadata record', null, 1], + ['has a canonical metadata record', 'recorded', 0], + ] as const)('reports a mounted file whose key %s', async (_label, metadata, expected) => { + const file = executionFile() + mockGetFileMetadataByKey.mockResolvedValue( + metadata + ? { + id: 'canonical-file-id', + key: file.key, + context: 'execution', + workspaceId: WORKSPACE_ID, + userId: 'user-1', + contentUpdatedAt: new Date('2026-01-01T00:00:00Z'), + } + : null + ) + const result = await resolveUserFileMounts({ + planned: planUserFileMounts([file]), + context: { ...executionContext, principal: createSessionPrincipal() }, + }) + expect(result.unprovenancedMountCount).toBe(expected) + }) + + it('reports every mount as unprovenanced when no principal can bind its source', async () => { + const result = await resolveUserFileMounts({ + planned: planUserFileMounts([executionFile()]), + context: executionContext, + }) + expect(result.unprovenancedMountCount).toBe(1) + }) + it('preserves contributors introduced when an inline mount renders generated source', async () => { const contributor = { fileId: 'image-file', diff --git a/apps/sim/lib/function-execution/sandbox-mounts.ts b/apps/sim/lib/function-execution/sandbox-mounts.ts index 8ba9069dcef..feec6b0ba34 100644 --- a/apps/sim/lib/function-execution/sandbox-mounts.ts +++ b/apps/sim/lib/function-execution/sandbox-mounts.ts @@ -277,7 +277,14 @@ export async function resolveUserFileMounts(args: { manifest: SandboxMountManifestEntry[] contributingFiles?: readonly WorkspaceFileSecretProvenanceIdentity[] renderedContributingFiles?: readonly WorkspaceFileSecretProvenanceIdentity[] + /** + * Mounts whose own bytes have no provenance source (no principal to bind one, or a key with no + * canonical metadata record). Workflow runs keep their legacy absence policy; a persistent + * workbench must not certify a machine that received one. + */ + unprovenancedMountCount: number }> { + let unprovenancedMountCount = 0 const sandboxFiles: SandboxFile[] = [] const manifest: SandboxMountManifestEntry[] = [] const budget = createSandboxMountBudget() @@ -297,14 +304,16 @@ export async function resolveUserFileMounts(args: { for (const { userFile, mountPath } of args.planned) { const storageContext = resolveTrustedFileContext(userFile.key, userFile.context) await assertUserFileContentAccess(userFile, args.context) - if (args.context.principal && args.context.workspaceId) { - const source = await resolveStoredFileProvenanceSource(userFile, { - ...args.context, - principal: args.context.principal, - workspaceId: args.context.workspaceId, - }) - if (source) addContributor(source.identity) - } + const source = + args.context.principal && args.context.workspaceId + ? await resolveStoredFileProvenanceSource(userFile, { + ...args.context, + principal: args.context.principal, + workspaceId: args.context.workspaceId, + }) + : undefined + if (source) addContributor(source.identity) + else unprovenancedMountCount += 1 await pushSandboxFileMount( sandboxFiles, @@ -361,6 +370,7 @@ export async function resolveUserFileMounts(args: { return { sandboxFiles, manifest, + unprovenancedMountCount, ...(contributingFiles.size > 0 ? { contributingFiles: [...contributingFiles.values()] } : {}), ...(renderedContributingFiles.size > 0 ? { renderedContributingFiles: [...renderedContributingFiles.values()] } From 41818fbf4f8c1fcda7a3927a40ba281e09de0adc Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Wed, 30 Sep 2026 01:43:20 -0700 Subject: [PATCH 2/4] test(sandbox): certify workbench history against real Redis and close the mount bypass - Replace the certification unit test, which restated the history script in a fake Redis, with an integration suite that runs the real code boundary against the real script in a disposable Redis. - Have the route test's mocked mount resolver return a fixed count per test rather than restating the counting rule. - Strip model-supplied `_sandboxFiles` from Copilot Function calls. Only resolved inputs may populate it, and a supplied URL mount would skip their provenance. - Document that public storage contexts always count as unprovenanced mounts. --- ...session-input-certification.integration.ts | 121 ++++++++++++++++++ .../session-input-certification.test.ts | 112 ---------------- .../execute-request.test.ts | 6 +- .../lib/function-execution/sandbox-mounts.ts | 4 + .../function-execute-provenance.test.ts | 27 ++++ .../tools/handlers/function-execute.ts | 2 + 6 files changed, 159 insertions(+), 113 deletions(-) create mode 100644 apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts delete mode 100644 apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts diff --git a/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts new file mode 100644 index 00000000000..6d64a82b660 --- /dev/null +++ b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts @@ -0,0 +1,121 @@ +/** + * The persistent workbench's input history is what lets a scratch file reach the model. This runs + * the real code boundary against the real history script in a disposable Redis, so a mount the + * caller could not classify has to leave the machine uncertified. Only the sandbox provider is a + * stand-in: it hands back an existing machine that executes nothing. + * + * Set TEST_REDIS_URL to an isolated local Redis service. + */ +import { createHash } from 'node:crypto' +import { + remoteSandboxProviderMock, + remoteSandboxProviderMockFns, +} from '@sim/testing/mocks/remote-sandbox-provider.mock' +import { generateShortId } from '@sim/utils/id' +import { afterAll, describe, expect, it, vi } from 'vitest' + +const { redisUrl, inheritedRedisUrl, mockFindSessionSandbox } = await vi.hoisted(async () => { + const { readTestRedisUrl } = await import('@sim/db/testing/test-infrastructure') + const url = readTestRedisUrl() + const inheritedRedisUrl = process.env.REDIS_URL + /** The real Redis module reads this at import. */ + if (url) process.env.REDIS_URL = url + return { redisUrl: url, inheritedRedisUrl, mockFindSessionSandbox: vi.fn() } +}) + +vi.mock('@/lib/execution/remote-sandbox/provider', () => remoteSandboxProviderMock) +vi.mock('@/lib/execution/remote-sandbox/resolve', () => ({ + resolveWorkspaceSandbox: async () => null, + provisionRuntimeDependencies: async () => {}, + repairMissingSandboxImage: async () => null, + RUNTIME_INSTALL_TIMEOUT_MS: 60_000, +})) + +import { getRedisClient } from '@/lib/core/config/redis' +import { CodeLanguage } from '@/lib/execution/languages' +import { + executeInSandbox, + executeShellInSandbox, + SIM_RESULT_PREFIX, +} from '@/lib/execution/remote-sandbox' +import { observeSandboxSessionInputs } from '@/lib/execution/remote-sandbox/execution-observer' +import { + initializeSessionFileProvenance, + isSessionFileProvenanceClean, +} from '@/lib/execution/remote-sandbox/session-file-provenance' +import type { SandboxHandle, SandboxProvider } from '@/lib/execution/remote-sandbox/types' + +remoteSandboxProviderMockFns.mockResolveProvider.mockImplementation( + () => + ({ + id: 'e2b', + dependencyStrategy: 'prebuilt', + resolveLifetimeMs: (ms: number) => ms, + create: async () => { + throw new Error('The certification fixture only reuses an existing machine') + }, + findSessionSandbox: mockFindSessionSandbox, + }) satisfies SandboxProvider +) + +function machine(sandboxId: string): SandboxHandle { + return { + sandboxId, + runCode: async () => ({ text: `${SIM_RESULT_PREFIX}{"ok":true}`, stdout: '', stderr: '' }), + runCommand: async () => ({ stdout: '', stderr: '', exitCode: 0 }), + extendLifetime: async () => {}, + getFileSize: async () => 0, + readFile: async () => '', + readFileWithLimit: async () => ({ content: '', byteLength: 0 }), + writeFile: async () => {}, + removeFile: async () => {}, + listFiles: async () => [], + kill: async () => {}, + } +} + +/** Machine-history keys this suite created, so cleanup never touches another suite's state. */ +const createdKeys: string[] = [] + +afterAll(async () => { + const redis = getRedisClient() + if (redis && createdKeys.length) await redis.del(...createdKeys) + if (inheritedRedisUrl === undefined) process.env.REDIS_URL = undefined + else process.env.REDIS_URL = inheritedRedisUrl +}) + +describe.skipIf(!redisUrl)('workbench certification at the code boundary', () => { + it.each([ + ['code', false], + ['code', true], + ['shell', false], + ['shell', true], + ] as const)('%s with unprovenanced mounts %s', async (kind, unprovenanced) => { + const sandboxId = `machine-${generateShortId(12)}` + const key = `certification-${generateShortId(12)}` + const identity = { providerId: 'e2b', sandboxId } as const + createdKeys.push( + `mothership:workbench-provenance:v1:${createHash('sha256') + .update(JSON.stringify([key, identity.providerId, sandboxId])) + .digest('hex')}` + ) + mockFindSessionSandbox.mockResolvedValue(machine(sandboxId)) + await initializeSessionFileProvenance(key, identity) + expect(await isSessionFileProvenanceClean(key, identity)).toBe(true) + + const request = { + code: 'print(1)', + language: CodeLanguage.Python, + timeoutMs: 30_000, + session: { key, ...(unprovenanced ? { unprovenancedInputs: true } : {}) }, + } + await observeSandboxSessionInputs( + () => true, + () => + kind === 'code' + ? executeInSandbox(request) + : executeShellInSandbox({ ...request, envs: {} }) + ) + expect(await isSessionFileProvenanceClean(key, identity)).toBe(!unprovenanced) + }) +}) diff --git a/apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts b/apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts deleted file mode 100644 index afa89b4de6b..00000000000 --- a/apps/sim/lib/execution/remote-sandbox/session-input-certification.test.ts +++ /dev/null @@ -1,112 +0,0 @@ -/** - * The persistent workbench's input history is what lets a scratch file reach the model. These - * run the real code boundary against the real history recorder, so a mount the caller could not - * classify has to leave the machine uncertified. - */ -import { redisConfigMockFns } from '@sim/testing/mocks/redis-config.mock' -import { - remoteSandboxProviderMock, - remoteSandboxProviderMockFns, -} from '@sim/testing/mocks/remote-sandbox-provider.mock' -import { generateShortId } from '@sim/utils/id' -import { beforeEach, describe, expect, it, vi } from 'vitest' -import { CodeLanguage } from '@/lib/execution/languages' -import type { SandboxHandle, SandboxProvider } from '@/lib/execution/remote-sandbox/types' - -const { mockFindSessionSandbox } = vi.hoisted(() => ({ mockFindSessionSandbox: vi.fn() })) - -vi.mock('@/lib/execution/remote-sandbox/provider', () => remoteSandboxProviderMock) -vi.mock('@/lib/execution/remote-sandbox/resolve', () => ({ - resolveWorkspaceSandbox: vi.fn().mockResolvedValue(null), - provisionRuntimeDependencies: vi.fn(), - repairMissingSandboxImage: vi.fn().mockResolvedValue(null), - RUNTIME_INSTALL_TIMEOUT_MS: 60_000, -})) -vi.mock('@/lib/core/execution-limits/metrics', () => ({ - recordSandboxTeardownFailure: vi.fn(), - recordSandboxProviderLimit: vi.fn(), -})) - -import { - executeInSandbox, - executeShellInSandbox, - SIM_RESULT_PREFIX, -} from '@/lib/execution/remote-sandbox' -import { observeSandboxSessionInputs } from '@/lib/execution/remote-sandbox/execution-observer' -import { - initializeSessionFileProvenance, - isSessionFileProvenanceClean, -} from '@/lib/execution/remote-sandbox/session-file-provenance' - -/** Same one-way semantics as the Lua history script. */ -const records = new Map() -redisConfigMockFns.mockGetRedisClient.mockImplementation(() => ({ - set: async (key: string, value: string) => { - if (!records.has(key)) records.set(key, value) - return 'OK' - }, - get: async (key: string) => records.get(key) ?? null, - eval: async (_script: string, _count: number, key: string, input: string) => { - records.set(key, records.get(key) === 'clean' && input === 'clean' ? 'clean' : 'unknown') - return records.get(key) - }, -})) -remoteSandboxProviderMockFns.mockResolveProvider.mockImplementation( - (): SandboxProvider => ({ - id: 'e2b', - dependencyStrategy: 'prebuilt', - resolveLifetimeMs: (ms: number) => ms, - create: vi.fn(), - findSessionSandbox: mockFindSessionSandbox, - }) -) - -function machine(sandboxId: string): SandboxHandle { - return { - sandboxId, - runCode: async () => ({ text: `${SIM_RESULT_PREFIX}{"ok":true}`, stdout: '', stderr: '' }), - runCommand: async () => ({ stdout: '', stderr: '', exitCode: 0 }), - extendLifetime: async () => {}, - getFileSize: async () => 0, - readFile: async () => '', - readFileWithLimit: async () => ({ content: '', byteLength: 0 }), - writeFile: async () => {}, - removeFile: async () => {}, - listFiles: async () => [], - kill: async () => {}, - } -} - -beforeEach(() => { - records.clear() -}) - -describe('workbench certification at the code boundary', () => { - it.each([ - ['code', false], - ['code', true], - ['shell', false], - ['shell', true], - ] as const)('%s with unprovenanced mounts %s', async (kind, unprovenanced) => { - const sandboxId = `machine-${generateShortId(8)}` - const key = `chat-${generateShortId(8)}` - const identity = { providerId: 'e2b', sandboxId } as const - mockFindSessionSandbox.mockResolvedValue(machine(sandboxId)) - await initializeSessionFileProvenance(key, identity) - const session = { key, ...(unprovenanced ? { unprovenancedInputs: true } : {}) } - const request = { - code: 'print(1)', - language: CodeLanguage.Python, - timeoutMs: 30_000, - session, - } - await observeSandboxSessionInputs( - () => true, - () => - kind === 'code' - ? executeInSandbox(request) - : executeShellInSandbox({ ...request, envs: {} }) - ) - expect(await isSessionFileProvenanceClean(key, identity)).toBe(!unprovenanced) - }) -}) diff --git a/apps/sim/lib/function-execution/execute-request.test.ts b/apps/sim/lib/function-execution/execute-request.test.ts index 9c192e3759a..548785d80ba 100644 --- a/apps/sim/lib/function-execution/execute-request.test.ts +++ b/apps/sim/lib/function-execution/execute-request.test.ts @@ -47,6 +47,7 @@ const { mockWriteWorkspaceFileByPath, mockUploadExecutionFile, mockMountContributors, + mockUnprovenancedMountCount, mockRenderedMountContributors, } = vi.hoisted(() => ({ mockExecuteInIsolatedVM: vi.fn(), @@ -54,6 +55,7 @@ const { mockWriteWorkspaceFileByPath: vi.fn(), mockUploadExecutionFile: vi.fn(), mockMountContributors: vi.fn(), + mockUnprovenancedMountCount: vi.fn(), mockRenderedMountContributors: vi.fn(), })) @@ -153,7 +155,7 @@ vi.mock('@/lib/function-execution/sandbox-mounts', () => ({ }) => ({ contributingFiles: mockMountContributors(), renderedContributingFiles: mockRenderedMountContributors(), - unprovenancedMountCount: mockMountContributors() ? 0 : planned.length, + unprovenancedMountCount: mockUnprovenancedMountCount(), sandboxFiles: planned.map(({ mountPath }) => ({ type: 'url' as const, path: mountPath, @@ -260,6 +262,7 @@ describe('Function execution request', () => { beforeEach(() => { resetDbChainMock() mockMountContributors.mockReturnValue(undefined) + mockUnprovenancedMountCount.mockReturnValue(0) mockRenderedMountContributors.mockReturnValue(undefined) mockUploadExecutionFile.mockImplementation(async (context, buffer, name, type) => ({ id: 'execution-file-1', @@ -2566,6 +2569,7 @@ describe('Function execution request', () => { ['no mounts', false], ] as const)('withholds workbench certification for %s', async (_label, mounted) => { envFlagsMock.isMothershipSandboxEnabled = true + mockUnprovenancedMountCount.mockReturnValue(mounted ? 1 : 0) hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({ success: true, userId: 'user-123', diff --git a/apps/sim/lib/function-execution/sandbox-mounts.ts b/apps/sim/lib/function-execution/sandbox-mounts.ts index feec6b0ba34..6c9faf28557 100644 --- a/apps/sim/lib/function-execution/sandbox-mounts.ts +++ b/apps/sim/lib/function-execution/sandbox-mounts.ts @@ -281,6 +281,10 @@ export async function resolveUserFileMounts(args: { * Mounts whose own bytes have no provenance source (no principal to bind one, or a key with no * canonical metadata record). Workflow runs keep their legacy absence policy; a persistent * workbench must not certify a machine that received one. + * + * Storage contexts other than workspace and execution (chat, copilot, knowledge-base, logs, and + * the other public contexts) never have a source, so they always count here and taint a + * workbench. That is conservative by design. */ unprovenancedMountCount: number }> { diff --git a/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts b/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts index 72fb4eb81ef..9679e422a27 100644 --- a/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts +++ b/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts @@ -57,6 +57,33 @@ describe('Function physical-session input certification', () => { ) }) +describe('Function sandbox mounts', () => { + it('never forwards a model-supplied sandbox mount, which would bypass input provenance', async () => { + mocks.execute.mockImplementation(async (_tool, params) => ({ + success: true, + output: { sandboxFiles: params._sandboxFiles ?? [] }, + })) + const result = await executeFunctionExecute( + { + code: 'print(open("/tmp/sim/inputs/x").read())', + language: 'python', + _sandboxFiles: [{ type: 'url', path: '/tmp/sim/inputs/x', url: 'https://storage.test/x' }], + }, + { + userId: 'user', + workflowId: '', + workspaceId: 'workspace', + chatId: 'chat', + resolvedSecretTraceRegistry: new ResolvedSecretTraceRegistry([], { + userId: 'user', + workspaceId: 'workspace', + }), + } + ) + expect(result.output).toEqual({ sandboxFiles: [] }) + }) +}) + describe('Generic Secrets function execution', () => { it('mounts the authorized environment and propagates echoed secrets into model redaction', async () => { setEnv({ ENCRYPTION_KEY: 'a'.repeat(64) }) diff --git a/apps/sim/lib/mothership/tools/handlers/function-execute.ts b/apps/sim/lib/mothership/tools/handlers/function-execute.ts index e44bff771ef..aa04f0076a3 100644 --- a/apps/sim/lib/mothership/tools/handlers/function-execute.ts +++ b/apps/sim/lib/mothership/tools/handlers/function-execute.ts @@ -499,6 +499,8 @@ export async function executeFunctionExecute( 'internalSandboxProfile', // Server-derived below — a model-supplied value must never select a session. 'sandboxSessionKey', + // Server-derived from resolved inputs; a model-supplied mount would skip their provenance. + '_sandboxFiles', PRIVATE_SECRET_PROVENANCE_FIELD, ]) // One persistent session sandbox per chat: files and installed packages From f050a043261119a390b0e09ac8121e1ad021517c Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Wed, 30 Sep 2026 02:05:57 -0700 Subject: [PATCH 3/4] test(sandbox): restore only the env the certification suite changed The suite's cleanup assigned undefined to REDIS_URL when it had been unset, which stores the string "undefined" for later suites in the worker, and asked for a Redis client even when the suite was skipped. --- .../session-input-certification.integration.ts | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts index 6d64a82b660..cf3833c21d4 100644 --- a/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts +++ b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts @@ -78,9 +78,13 @@ function machine(sandboxId: string): SandboxHandle { const createdKeys: string[] = [] afterAll(async () => { - const redis = getRedisClient() - if (redis && createdKeys.length) await redis.del(...createdKeys) - if (inheritedRedisUrl === undefined) process.env.REDIS_URL = undefined + if (redisUrl && createdKeys.length) { + const redis = getRedisClient() + if (redis) await redis.del(...createdKeys) + } + // Only restore what the hoisted setup changed; assigning undefined would store the string "undefined". + if (!redisUrl) return + if (inheritedRedisUrl === undefined) Reflect.deleteProperty(process.env, 'REDIS_URL') else process.env.REDIS_URL = inheritedRedisUrl }) From cfce8b3ea2e851c601e049445f456a9e8163a997 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Wed, 30 Sep 2026 02:25:04 -0700 Subject: [PATCH 4/4] test(sandbox): close the shared Redis client after the certification suite --- .../remote-sandbox/session-input-certification.integration.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts index cf3833c21d4..45066ff3fcd 100644 --- a/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts +++ b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts @@ -31,7 +31,7 @@ vi.mock('@/lib/execution/remote-sandbox/resolve', () => ({ RUNTIME_INSTALL_TIMEOUT_MS: 60_000, })) -import { getRedisClient } from '@/lib/core/config/redis' +import { closeRedisConnection, getRedisClient } from '@/lib/core/config/redis' import { CodeLanguage } from '@/lib/execution/languages' import { executeInSandbox, @@ -82,6 +82,7 @@ afterAll(async () => { const redis = getRedisClient() if (redis) await redis.del(...createdKeys) } + if (redisUrl) await closeRedisConnection() // Only restore what the hoisted setup changed; assigning undefined would store the string "undefined". if (!redisUrl) return if (inheritedRedisUrl === undefined) Reflect.deleteProperty(process.env, 'REDIS_URL')