diff --git a/apps/sim/lib/execution/remote-sandbox/index.ts b/apps/sim/lib/execution/remote-sandbox/index.ts index e1bd2fb1638..f3ea137224a 100644 --- a/apps/sim/lib/execution/remote-sandbox/index.ts +++ b/apps/sim/lib/execution/remote-sandbox/index.ts @@ -891,7 +891,9 @@ async function executeInSandboxWithinBudget( await recordSessionFileInput( req.session.key, { providerId: created.providerId, sandboxId }, - sandboxSessionInputsSafe() && !Object.keys(selected?.envs ?? {}).length + sandboxSessionInputsSafe() && + !req.session.unprovenancedInputs && + !Object.keys(selected?.envs ?? {}).length ) await provisionWithinBudget(sandbox, selected, signal) await writeSandboxInputs(sandbox, req.sandboxFiles, { @@ -1078,7 +1080,9 @@ async function executeShellInSandboxWithinBudget( await recordSessionFileInput( req.session.key, { providerId: created.providerId, sandboxId }, - sandboxSessionInputsSafe() && !Object.keys(selected?.envs ?? {}).length + sandboxSessionInputsSafe() && + !req.session.unprovenancedInputs && + !Object.keys(selected?.envs ?? {}).length ) await provisionWithinBudget(sandbox, selected, signal) await writeSandboxInputs(sandbox, req.sandboxFiles, { diff --git a/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts new file mode 100644 index 00000000000..45066ff3fcd --- /dev/null +++ b/apps/sim/lib/execution/remote-sandbox/session-input-certification.integration.ts @@ -0,0 +1,126 @@ +/** + * The persistent workbench's input history is what lets a scratch file reach the model. This runs + * the real code boundary against the real history script in a disposable Redis, so a mount the + * caller could not classify has to leave the machine uncertified. Only the sandbox provider is a + * stand-in: it hands back an existing machine that executes nothing. + * + * Set TEST_REDIS_URL to an isolated local Redis service. + */ +import { createHash } from 'node:crypto' +import { + remoteSandboxProviderMock, + remoteSandboxProviderMockFns, +} from '@sim/testing/mocks/remote-sandbox-provider.mock' +import { generateShortId } from '@sim/utils/id' +import { afterAll, describe, expect, it, vi } from 'vitest' + +const { redisUrl, inheritedRedisUrl, mockFindSessionSandbox } = await vi.hoisted(async () => { + const { readTestRedisUrl } = await import('@sim/db/testing/test-infrastructure') + const url = readTestRedisUrl() + const inheritedRedisUrl = process.env.REDIS_URL + /** The real Redis module reads this at import. */ + if (url) process.env.REDIS_URL = url + return { redisUrl: url, inheritedRedisUrl, mockFindSessionSandbox: vi.fn() } +}) + +vi.mock('@/lib/execution/remote-sandbox/provider', () => remoteSandboxProviderMock) +vi.mock('@/lib/execution/remote-sandbox/resolve', () => ({ + resolveWorkspaceSandbox: async () => null, + provisionRuntimeDependencies: async () => {}, + repairMissingSandboxImage: async () => null, + RUNTIME_INSTALL_TIMEOUT_MS: 60_000, +})) + +import { closeRedisConnection, getRedisClient } from '@/lib/core/config/redis' +import { CodeLanguage } from '@/lib/execution/languages' +import { + executeInSandbox, + executeShellInSandbox, + SIM_RESULT_PREFIX, +} from '@/lib/execution/remote-sandbox' +import { observeSandboxSessionInputs } from '@/lib/execution/remote-sandbox/execution-observer' +import { + initializeSessionFileProvenance, + isSessionFileProvenanceClean, +} from '@/lib/execution/remote-sandbox/session-file-provenance' +import type { SandboxHandle, SandboxProvider } from '@/lib/execution/remote-sandbox/types' + +remoteSandboxProviderMockFns.mockResolveProvider.mockImplementation( + () => + ({ + id: 'e2b', + dependencyStrategy: 'prebuilt', + resolveLifetimeMs: (ms: number) => ms, + create: async () => { + throw new Error('The certification fixture only reuses an existing machine') + }, + findSessionSandbox: mockFindSessionSandbox, + }) satisfies SandboxProvider +) + +function machine(sandboxId: string): SandboxHandle { + return { + sandboxId, + runCode: async () => ({ text: `${SIM_RESULT_PREFIX}{"ok":true}`, stdout: '', stderr: '' }), + runCommand: async () => ({ stdout: '', stderr: '', exitCode: 0 }), + extendLifetime: async () => {}, + getFileSize: async () => 0, + readFile: async () => '', + readFileWithLimit: async () => ({ content: '', byteLength: 0 }), + writeFile: async () => {}, + removeFile: async () => {}, + listFiles: async () => [], + kill: async () => {}, + } +} + +/** Machine-history keys this suite created, so cleanup never touches another suite's state. */ +const createdKeys: string[] = [] + +afterAll(async () => { + if (redisUrl && createdKeys.length) { + const redis = getRedisClient() + if (redis) await redis.del(...createdKeys) + } + if (redisUrl) await closeRedisConnection() + // Only restore what the hoisted setup changed; assigning undefined would store the string "undefined". + if (!redisUrl) return + if (inheritedRedisUrl === undefined) Reflect.deleteProperty(process.env, 'REDIS_URL') + else process.env.REDIS_URL = inheritedRedisUrl +}) + +describe.skipIf(!redisUrl)('workbench certification at the code boundary', () => { + it.each([ + ['code', false], + ['code', true], + ['shell', false], + ['shell', true], + ] as const)('%s with unprovenanced mounts %s', async (kind, unprovenanced) => { + const sandboxId = `machine-${generateShortId(12)}` + const key = `certification-${generateShortId(12)}` + const identity = { providerId: 'e2b', sandboxId } as const + createdKeys.push( + `mothership:workbench-provenance:v1:${createHash('sha256') + .update(JSON.stringify([key, identity.providerId, sandboxId])) + .digest('hex')}` + ) + mockFindSessionSandbox.mockResolvedValue(machine(sandboxId)) + await initializeSessionFileProvenance(key, identity) + expect(await isSessionFileProvenanceClean(key, identity)).toBe(true) + + const request = { + code: 'print(1)', + language: CodeLanguage.Python, + timeoutMs: 30_000, + session: { key, ...(unprovenanced ? { unprovenancedInputs: true } : {}) }, + } + await observeSandboxSessionInputs( + () => true, + () => + kind === 'code' + ? executeInSandbox(request) + : executeShellInSandbox({ ...request, envs: {} }) + ) + expect(await isSessionFileProvenanceClean(key, identity)).toBe(!unprovenanced) + }) +}) diff --git a/apps/sim/lib/execution/remote-sandbox/types.ts b/apps/sim/lib/execution/remote-sandbox/types.ts index 5ff085e4526..473a4aa3732 100644 --- a/apps/sim/lib/execution/remote-sandbox/types.ts +++ b/apps/sim/lib/execution/remote-sandbox/types.ts @@ -91,6 +91,11 @@ export interface SandboxSessionRequest { cli?: { path: string; content: string; runtime?: { path: string; content: string } } /** Extra environment variables present on every execution in the session. */ envs?: Record + /** + * This execution mounts bytes whose secret provenance is unknown, so the machine's input + * history must not stay certified clean even when the caller's own inputs are. + */ + unprovenancedInputs?: boolean } export interface SandboxShellExecutionRequest { diff --git a/apps/sim/lib/function-execution/execute-request.test.ts b/apps/sim/lib/function-execution/execute-request.test.ts index 474c5688f7f..548785d80ba 100644 --- a/apps/sim/lib/function-execution/execute-request.test.ts +++ b/apps/sim/lib/function-execution/execute-request.test.ts @@ -47,6 +47,7 @@ const { mockWriteWorkspaceFileByPath, mockUploadExecutionFile, mockMountContributors, + mockUnprovenancedMountCount, mockRenderedMountContributors, } = vi.hoisted(() => ({ mockExecuteInIsolatedVM: vi.fn(), @@ -54,6 +55,7 @@ const { mockWriteWorkspaceFileByPath: vi.fn(), mockUploadExecutionFile: vi.fn(), mockMountContributors: vi.fn(), + mockUnprovenancedMountCount: vi.fn(), mockRenderedMountContributors: vi.fn(), })) @@ -153,6 +155,7 @@ vi.mock('@/lib/function-execution/sandbox-mounts', () => ({ }) => ({ contributingFiles: mockMountContributors(), renderedContributingFiles: mockRenderedMountContributors(), + unprovenancedMountCount: mockUnprovenancedMountCount(), sandboxFiles: planned.map(({ mountPath }) => ({ type: 'url' as const, path: mountPath, @@ -259,6 +262,7 @@ describe('Function execution request', () => { beforeEach(() => { resetDbChainMock() mockMountContributors.mockReturnValue(undefined) + mockUnprovenancedMountCount.mockReturnValue(0) mockRenderedMountContributors.mockReturnValue(undefined) mockUploadExecutionFile.mockImplementation(async (context, buffer, name, type) => ({ id: 'execution-file-1', @@ -2560,6 +2564,33 @@ describe('Function execution request', () => { expect(mockWriteWorkspaceFileByPath).not.toHaveBeenCalled() }) + it.each([ + ['a mount with no provenance source', true], + ['no mounts', false], + ] as const)('withholds workbench certification for %s', async (_label, mounted) => { + envFlagsMock.isMothershipSandboxEnabled = true + mockUnprovenancedMountCount.mockReturnValue(mounted ? 1 : 0) + hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({ + success: true, + userId: 'user-123', + authType: 'internal_jwt', + sandboxProfile: 'mothership', + }) + const response = await POST( + createMockRequest('POST', { + code: 'x', + language: 'python', + workspaceId: 'workspace-1', + sandboxSessionKey: 'chat-session', + ...(mounted ? { contextVariables: { doc: MOUNT_REF } } : {}), + }) + ) + expect(response.status).toBe(200) + const session = mockExecuteInSandbox.mock.calls.at(-1)?.[0].session + expect(session.key).toBe('chat-session') + expect(session.unprovenancedInputs === true).toBe(mounted) + }) + it('gives overlapping calls in one persistent workbench distinct automatic export directories', async () => { envFlagsMock.isMothershipSandboxEnabled = true hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({ diff --git a/apps/sim/lib/function-execution/execute-request.ts b/apps/sim/lib/function-execution/execute-request.ts index dd5b2f8e752..cdc1e7a7067 100644 --- a/apps/sim/lib/function-execution/execute-request.ts +++ b/apps/sim/lib/function-execution/execute-request.ts @@ -2385,7 +2385,7 @@ export async function executeFunctionRequest( // would leave `{{OTHER_SECRET}}` resolving, which is a hole, not a scope. const envVars = scopeEnvironmentVariables(rawEnvVars, secretScope, mountedSecrets) const admittedChatOwner = activeSandboxChatOwner() - const mothershipSession = + const admittedSession = usesMothershipSandbox && !selectedSandboxId && sandboxSessionKey && @@ -2716,6 +2716,10 @@ export async function executeFunctionRequest( ) } const { sandboxFiles: userFileMounts, manifest: mountManifest } = resolvedMounts + const mothershipSession = + admittedSession && resolvedMounts.unprovenancedMountCount > 0 + ? { ...admittedSession, unprovenancedInputs: true } + : admittedSession const sandboxFiles = mergeSandboxFileMounts(_sandboxFiles, userFileMounts) // Every `` marker becomes the path its file was mounted at, diff --git a/apps/sim/lib/function-execution/sandbox-mounts.test.ts b/apps/sim/lib/function-execution/sandbox-mounts.test.ts index 6ad4c418829..570649e2008 100644 --- a/apps/sim/lib/function-execution/sandbox-mounts.test.ts +++ b/apps/sim/lib/function-execution/sandbox-mounts.test.ts @@ -214,6 +214,42 @@ describe('resolveUserFileMounts', () => { expect(mockDownloadServableFileFromStorage).not.toHaveBeenCalled() }) + /** + * A persistent workbench certifies its machine from these counts: a mount whose bytes have no + * provenance source can hold resolved secret plaintext nobody recorded, so it must be reported. + */ + it.each([ + ['has no metadata record', null, 1], + ['has a canonical metadata record', 'recorded', 0], + ] as const)('reports a mounted file whose key %s', async (_label, metadata, expected) => { + const file = executionFile() + mockGetFileMetadataByKey.mockResolvedValue( + metadata + ? { + id: 'canonical-file-id', + key: file.key, + context: 'execution', + workspaceId: WORKSPACE_ID, + userId: 'user-1', + contentUpdatedAt: new Date('2026-01-01T00:00:00Z'), + } + : null + ) + const result = await resolveUserFileMounts({ + planned: planUserFileMounts([file]), + context: { ...executionContext, principal: createSessionPrincipal() }, + }) + expect(result.unprovenancedMountCount).toBe(expected) + }) + + it('reports every mount as unprovenanced when no principal can bind its source', async () => { + const result = await resolveUserFileMounts({ + planned: planUserFileMounts([executionFile()]), + context: executionContext, + }) + expect(result.unprovenancedMountCount).toBe(1) + }) + it('preserves contributors introduced when an inline mount renders generated source', async () => { const contributor = { fileId: 'image-file', diff --git a/apps/sim/lib/function-execution/sandbox-mounts.ts b/apps/sim/lib/function-execution/sandbox-mounts.ts index 8ba9069dcef..6c9faf28557 100644 --- a/apps/sim/lib/function-execution/sandbox-mounts.ts +++ b/apps/sim/lib/function-execution/sandbox-mounts.ts @@ -277,7 +277,18 @@ export async function resolveUserFileMounts(args: { manifest: SandboxMountManifestEntry[] contributingFiles?: readonly WorkspaceFileSecretProvenanceIdentity[] renderedContributingFiles?: readonly WorkspaceFileSecretProvenanceIdentity[] + /** + * Mounts whose own bytes have no provenance source (no principal to bind one, or a key with no + * canonical metadata record). Workflow runs keep their legacy absence policy; a persistent + * workbench must not certify a machine that received one. + * + * Storage contexts other than workspace and execution (chat, copilot, knowledge-base, logs, and + * the other public contexts) never have a source, so they always count here and taint a + * workbench. That is conservative by design. + */ + unprovenancedMountCount: number }> { + let unprovenancedMountCount = 0 const sandboxFiles: SandboxFile[] = [] const manifest: SandboxMountManifestEntry[] = [] const budget = createSandboxMountBudget() @@ -297,14 +308,16 @@ export async function resolveUserFileMounts(args: { for (const { userFile, mountPath } of args.planned) { const storageContext = resolveTrustedFileContext(userFile.key, userFile.context) await assertUserFileContentAccess(userFile, args.context) - if (args.context.principal && args.context.workspaceId) { - const source = await resolveStoredFileProvenanceSource(userFile, { - ...args.context, - principal: args.context.principal, - workspaceId: args.context.workspaceId, - }) - if (source) addContributor(source.identity) - } + const source = + args.context.principal && args.context.workspaceId + ? await resolveStoredFileProvenanceSource(userFile, { + ...args.context, + principal: args.context.principal, + workspaceId: args.context.workspaceId, + }) + : undefined + if (source) addContributor(source.identity) + else unprovenancedMountCount += 1 await pushSandboxFileMount( sandboxFiles, @@ -361,6 +374,7 @@ export async function resolveUserFileMounts(args: { return { sandboxFiles, manifest, + unprovenancedMountCount, ...(contributingFiles.size > 0 ? { contributingFiles: [...contributingFiles.values()] } : {}), ...(renderedContributingFiles.size > 0 ? { renderedContributingFiles: [...renderedContributingFiles.values()] } diff --git a/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts b/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts index 72fb4eb81ef..9679e422a27 100644 --- a/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts +++ b/apps/sim/lib/mothership/tools/handlers/function-execute-provenance.test.ts @@ -57,6 +57,33 @@ describe('Function physical-session input certification', () => { ) }) +describe('Function sandbox mounts', () => { + it('never forwards a model-supplied sandbox mount, which would bypass input provenance', async () => { + mocks.execute.mockImplementation(async (_tool, params) => ({ + success: true, + output: { sandboxFiles: params._sandboxFiles ?? [] }, + })) + const result = await executeFunctionExecute( + { + code: 'print(open("/tmp/sim/inputs/x").read())', + language: 'python', + _sandboxFiles: [{ type: 'url', path: '/tmp/sim/inputs/x', url: 'https://storage.test/x' }], + }, + { + userId: 'user', + workflowId: '', + workspaceId: 'workspace', + chatId: 'chat', + resolvedSecretTraceRegistry: new ResolvedSecretTraceRegistry([], { + userId: 'user', + workspaceId: 'workspace', + }), + } + ) + expect(result.output).toEqual({ sandboxFiles: [] }) + }) +}) + describe('Generic Secrets function execution', () => { it('mounts the authorized environment and propagates echoed secrets into model redaction', async () => { setEnv({ ENCRYPTION_KEY: 'a'.repeat(64) }) diff --git a/apps/sim/lib/mothership/tools/handlers/function-execute.ts b/apps/sim/lib/mothership/tools/handlers/function-execute.ts index e44bff771ef..aa04f0076a3 100644 --- a/apps/sim/lib/mothership/tools/handlers/function-execute.ts +++ b/apps/sim/lib/mothership/tools/handlers/function-execute.ts @@ -499,6 +499,8 @@ export async function executeFunctionExecute( 'internalSandboxProfile', // Server-derived below — a model-supplied value must never select a session. 'sandboxSessionKey', + // Server-derived from resolved inputs; a model-supplied mount would skip their provenance. + '_sandboxFiles', PRIVATE_SECRET_PROVENANCE_FIELD, ]) // One persistent session sandbox per chat: files and installed packages