Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 6 additions & 2 deletions apps/sim/lib/execution/remote-sandbox/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -891,7 +891,9 @@ async function executeInSandboxWithinBudget(
await recordSessionFileInput(
req.session.key,
{ providerId: created.providerId, sandboxId },
sandboxSessionInputsSafe() && !Object.keys(selected?.envs ?? {}).length
sandboxSessionInputsSafe() &&
!req.session.unprovenancedInputs &&
!Object.keys(selected?.envs ?? {}).length
)
await provisionWithinBudget(sandbox, selected, signal)
await writeSandboxInputs(sandbox, req.sandboxFiles, {
Expand Down Expand Up @@ -1078,7 +1080,9 @@ async function executeShellInSandboxWithinBudget(
await recordSessionFileInput(
req.session.key,
{ providerId: created.providerId, sandboxId },
sandboxSessionInputsSafe() && !Object.keys(selected?.envs ?? {}).length
sandboxSessionInputsSafe() &&
!req.session.unprovenancedInputs &&
!Object.keys(selected?.envs ?? {}).length
)
await provisionWithinBudget(sandbox, selected, signal)
await writeSandboxInputs(sandbox, req.sandboxFiles, {
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,126 @@
/**
* The persistent workbench's input history is what lets a scratch file reach the model. This runs
* the real code boundary against the real history script in a disposable Redis, so a mount the
* caller could not classify has to leave the machine uncertified. Only the sandbox provider is a
* stand-in: it hands back an existing machine that executes nothing.
*
* Set TEST_REDIS_URL to an isolated local Redis service.
*/
import { createHash } from 'node:crypto'
import {
remoteSandboxProviderMock,
remoteSandboxProviderMockFns,
} from '@sim/testing/mocks/remote-sandbox-provider.mock'
import { generateShortId } from '@sim/utils/id'
import { afterAll, describe, expect, it, vi } from 'vitest'

const { redisUrl, inheritedRedisUrl, mockFindSessionSandbox } = await vi.hoisted(async () => {
const { readTestRedisUrl } = await import('@sim/db/testing/test-infrastructure')
const url = readTestRedisUrl()
const inheritedRedisUrl = process.env.REDIS_URL
/** The real Redis module reads this at import. */
if (url) process.env.REDIS_URL = url
return { redisUrl: url, inheritedRedisUrl, mockFindSessionSandbox: vi.fn() }
})

vi.mock('@/lib/execution/remote-sandbox/provider', () => remoteSandboxProviderMock)
vi.mock('@/lib/execution/remote-sandbox/resolve', () => ({
resolveWorkspaceSandbox: async () => null,
provisionRuntimeDependencies: async () => {},
repairMissingSandboxImage: async () => null,
RUNTIME_INSTALL_TIMEOUT_MS: 60_000,
}))

import { closeRedisConnection, getRedisClient } from '@/lib/core/config/redis'
import { CodeLanguage } from '@/lib/execution/languages'
import {
executeInSandbox,
executeShellInSandbox,
SIM_RESULT_PREFIX,
} from '@/lib/execution/remote-sandbox'
import { observeSandboxSessionInputs } from '@/lib/execution/remote-sandbox/execution-observer'
import {
initializeSessionFileProvenance,
isSessionFileProvenanceClean,
} from '@/lib/execution/remote-sandbox/session-file-provenance'
import type { SandboxHandle, SandboxProvider } from '@/lib/execution/remote-sandbox/types'

remoteSandboxProviderMockFns.mockResolveProvider.mockImplementation(
() =>
({
id: 'e2b',
dependencyStrategy: 'prebuilt',
resolveLifetimeMs: (ms: number) => ms,
create: async () => {
throw new Error('The certification fixture only reuses an existing machine')
},
findSessionSandbox: mockFindSessionSandbox,
}) satisfies SandboxProvider
)

function machine(sandboxId: string): SandboxHandle {
return {
sandboxId,
runCode: async () => ({ text: `${SIM_RESULT_PREFIX}{"ok":true}`, stdout: '', stderr: '' }),
runCommand: async () => ({ stdout: '', stderr: '', exitCode: 0 }),
extendLifetime: async () => {},
getFileSize: async () => 0,
readFile: async () => '',
readFileWithLimit: async () => ({ content: '', byteLength: 0 }),
writeFile: async () => {},
removeFile: async () => {},
listFiles: async () => [],
kill: async () => {},
}
}

/** Machine-history keys this suite created, so cleanup never touches another suite's state. */
const createdKeys: string[] = []

afterAll(async () => {
if (redisUrl && createdKeys.length) {
const redis = getRedisClient()
Comment thread
waleedlatif1 marked this conversation as resolved.
if (redis) await redis.del(...createdKeys)
}
if (redisUrl) await closeRedisConnection()
// Only restore what the hoisted setup changed; assigning undefined would store the string "undefined".
if (!redisUrl) return
if (inheritedRedisUrl === undefined) Reflect.deleteProperty(process.env, 'REDIS_URL')
else process.env.REDIS_URL = inheritedRedisUrl
})

describe.skipIf(!redisUrl)('workbench certification at the code boundary', () => {
it.each([
['code', false],
['code', true],
['shell', false],
['shell', true],
] as const)('%s with unprovenanced mounts %s', async (kind, unprovenanced) => {
const sandboxId = `machine-${generateShortId(12)}`
const key = `certification-${generateShortId(12)}`
const identity = { providerId: 'e2b', sandboxId } as const
createdKeys.push(
`mothership:workbench-provenance:v1:${createHash('sha256')
.update(JSON.stringify([key, identity.providerId, sandboxId]))
.digest('hex')}`
)
mockFindSessionSandbox.mockResolvedValue(machine(sandboxId))
await initializeSessionFileProvenance(key, identity)
expect(await isSessionFileProvenanceClean(key, identity)).toBe(true)

const request = {
code: 'print(1)',
language: CodeLanguage.Python,
timeoutMs: 30_000,
session: { key, ...(unprovenanced ? { unprovenancedInputs: true } : {}) },
}
await observeSandboxSessionInputs(
() => true,
() =>
kind === 'code'
? executeInSandbox(request)
: executeShellInSandbox({ ...request, envs: {} })
)
expect(await isSessionFileProvenanceClean(key, identity)).toBe(!unprovenanced)
})
})
5 changes: 5 additions & 0 deletions apps/sim/lib/execution/remote-sandbox/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,11 @@ export interface SandboxSessionRequest {
cli?: { path: string; content: string; runtime?: { path: string; content: string } }
/** Extra environment variables present on every execution in the session. */
envs?: Record<string, string>
/**
* This execution mounts bytes whose secret provenance is unknown, so the machine's input
* history must not stay certified clean even when the caller's own inputs are.
*/
unprovenancedInputs?: boolean
}

export interface SandboxShellExecutionRequest {
Expand Down
31 changes: 31 additions & 0 deletions apps/sim/lib/function-execution/execute-request.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -47,13 +47,15 @@ const {
mockWriteWorkspaceFileByPath,
mockUploadExecutionFile,
mockMountContributors,
mockUnprovenancedMountCount,
mockRenderedMountContributors,
} = vi.hoisted(() => ({
mockExecuteInIsolatedVM: vi.fn(),
mockValidateWorkspaceFileWriteTarget: vi.fn(),
mockWriteWorkspaceFileByPath: vi.fn(),
mockUploadExecutionFile: vi.fn(),
mockMountContributors: vi.fn(),
mockUnprovenancedMountCount: vi.fn(),
mockRenderedMountContributors: vi.fn(),
}))

Expand Down Expand Up @@ -153,6 +155,7 @@ vi.mock('@/lib/function-execution/sandbox-mounts', () => ({
}) => ({
contributingFiles: mockMountContributors(),
renderedContributingFiles: mockRenderedMountContributors(),
unprovenancedMountCount: mockUnprovenancedMountCount(),
sandboxFiles: planned.map(({ mountPath }) => ({
type: 'url' as const,
path: mountPath,
Expand Down Expand Up @@ -259,6 +262,7 @@ describe('Function execution request', () => {
beforeEach(() => {
resetDbChainMock()
mockMountContributors.mockReturnValue(undefined)
mockUnprovenancedMountCount.mockReturnValue(0)
mockRenderedMountContributors.mockReturnValue(undefined)
mockUploadExecutionFile.mockImplementation(async (context, buffer, name, type) => ({
id: 'execution-file-1',
Expand Down Expand Up @@ -2560,6 +2564,33 @@ describe('Function execution request', () => {
expect(mockWriteWorkspaceFileByPath).not.toHaveBeenCalled()
})

it.each([
['a mount with no provenance source', true],
['no mounts', false],
] as const)('withholds workbench certification for %s', async (_label, mounted) => {
envFlagsMock.isMothershipSandboxEnabled = true
mockUnprovenancedMountCount.mockReturnValue(mounted ? 1 : 0)
hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({
success: true,
userId: 'user-123',
authType: 'internal_jwt',
sandboxProfile: 'mothership',
})
const response = await POST(
createMockRequest('POST', {
code: 'x',
language: 'python',
workspaceId: 'workspace-1',
sandboxSessionKey: 'chat-session',
...(mounted ? { contextVariables: { doc: MOUNT_REF } } : {}),
})
)
expect(response.status).toBe(200)
const session = mockExecuteInSandbox.mock.calls.at(-1)?.[0].session
expect(session.key).toBe('chat-session')
expect(session.unprovenancedInputs === true).toBe(mounted)
})

it('gives overlapping calls in one persistent workbench distinct automatic export directories', async () => {
envFlagsMock.isMothershipSandboxEnabled = true
hybridAuthMockFns.mockCheckInternalAuth.mockResolvedValue({
Expand Down
6 changes: 5 additions & 1 deletion apps/sim/lib/function-execution/execute-request.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2385,7 +2385,7 @@ export async function executeFunctionRequest(
// would leave `{{OTHER_SECRET}}` resolving, which is a hole, not a scope.
const envVars = scopeEnvironmentVariables(rawEnvVars, secretScope, mountedSecrets)
const admittedChatOwner = activeSandboxChatOwner()
const mothershipSession =
const admittedSession =
usesMothershipSandbox &&
!selectedSandboxId &&
sandboxSessionKey &&
Expand Down Expand Up @@ -2716,6 +2716,10 @@ export async function executeFunctionRequest(
)
}
const { sandboxFiles: userFileMounts, manifest: mountManifest } = resolvedMounts
const mothershipSession =
admittedSession && resolvedMounts.unprovenancedMountCount > 0
? { ...admittedSession, unprovenancedInputs: true }
: admittedSession
const sandboxFiles = mergeSandboxFileMounts(_sandboxFiles, userFileMounts)

// Every `<block.file.path>` marker becomes the path its file was mounted at,
Expand Down
36 changes: 36 additions & 0 deletions apps/sim/lib/function-execution/sandbox-mounts.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -214,6 +214,42 @@ describe('resolveUserFileMounts', () => {
expect(mockDownloadServableFileFromStorage).not.toHaveBeenCalled()
})

/**
* A persistent workbench certifies its machine from these counts: a mount whose bytes have no
* provenance source can hold resolved secret plaintext nobody recorded, so it must be reported.
*/
it.each([
['has no metadata record', null, 1],
['has a canonical metadata record', 'recorded', 0],
] as const)('reports a mounted file whose key %s', async (_label, metadata, expected) => {
const file = executionFile()
mockGetFileMetadataByKey.mockResolvedValue(
metadata
? {
id: 'canonical-file-id',
key: file.key,
context: 'execution',
workspaceId: WORKSPACE_ID,
userId: 'user-1',
contentUpdatedAt: new Date('2026-01-01T00:00:00Z'),
}
: null
)
const result = await resolveUserFileMounts({
planned: planUserFileMounts([file]),
context: { ...executionContext, principal: createSessionPrincipal() },
})
expect(result.unprovenancedMountCount).toBe(expected)
})

it('reports every mount as unprovenanced when no principal can bind its source', async () => {
const result = await resolveUserFileMounts({
planned: planUserFileMounts([executionFile()]),
context: executionContext,
})
expect(result.unprovenancedMountCount).toBe(1)
})

it('preserves contributors introduced when an inline mount renders generated source', async () => {
const contributor = {
fileId: 'image-file',
Expand Down
30 changes: 22 additions & 8 deletions apps/sim/lib/function-execution/sandbox-mounts.ts
Original file line number Diff line number Diff line change
Expand Up @@ -277,7 +277,18 @@ export async function resolveUserFileMounts(args: {
manifest: SandboxMountManifestEntry[]
contributingFiles?: readonly WorkspaceFileSecretProvenanceIdentity[]
renderedContributingFiles?: readonly WorkspaceFileSecretProvenanceIdentity[]
/**
* Mounts whose own bytes have no provenance source (no principal to bind one, or a key with no
* canonical metadata record). Workflow runs keep their legacy absence policy; a persistent
* workbench must not certify a machine that received one.
*
* Storage contexts other than workspace and execution (chat, copilot, knowledge-base, logs, and
* the other public contexts) never have a source, so they always count here and taint a
* workbench. That is conservative by design.
*/
unprovenancedMountCount: number
}> {
let unprovenancedMountCount = 0
const sandboxFiles: SandboxFile[] = []
const manifest: SandboxMountManifestEntry[] = []
const budget = createSandboxMountBudget()
Expand All @@ -297,14 +308,16 @@ export async function resolveUserFileMounts(args: {
for (const { userFile, mountPath } of args.planned) {
const storageContext = resolveTrustedFileContext(userFile.key, userFile.context)
await assertUserFileContentAccess(userFile, args.context)
if (args.context.principal && args.context.workspaceId) {
const source = await resolveStoredFileProvenanceSource(userFile, {
...args.context,
principal: args.context.principal,
workspaceId: args.context.workspaceId,
})
if (source) addContributor(source.identity)
}
const source =
args.context.principal && args.context.workspaceId
? await resolveStoredFileProvenanceSource(userFile, {
...args.context,
principal: args.context.principal,
workspaceId: args.context.workspaceId,
})
: undefined
if (source) addContributor(source.identity)
else unprovenancedMountCount += 1

await pushSandboxFileMount(
sandboxFiles,
Expand Down Expand Up @@ -361,6 +374,7 @@ export async function resolveUserFileMounts(args: {
return {
sandboxFiles,
manifest,
unprovenancedMountCount,
...(contributingFiles.size > 0 ? { contributingFiles: [...contributingFiles.values()] } : {}),
...(renderedContributingFiles.size > 0
? { renderedContributingFiles: [...renderedContributingFiles.values()] }
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -57,6 +57,33 @@ describe('Function physical-session input certification', () => {
)
})

describe('Function sandbox mounts', () => {
it('never forwards a model-supplied sandbox mount, which would bypass input provenance', async () => {
mocks.execute.mockImplementation(async (_tool, params) => ({
success: true,
output: { sandboxFiles: params._sandboxFiles ?? [] },
}))
const result = await executeFunctionExecute(
{
code: 'print(open("/tmp/sim/inputs/x").read())',
language: 'python',
_sandboxFiles: [{ type: 'url', path: '/tmp/sim/inputs/x', url: 'https://storage.test/x' }],
},
{
userId: 'user',
workflowId: '',
workspaceId: 'workspace',
chatId: 'chat',
resolvedSecretTraceRegistry: new ResolvedSecretTraceRegistry([], {
userId: 'user',
workspaceId: 'workspace',
}),
}
)
expect(result.output).toEqual({ sandboxFiles: [] })
})
})

describe('Generic Secrets function execution', () => {
it('mounts the authorized environment and propagates echoed secrets into model redaction', async () => {
setEnv({ ENCRYPTION_KEY: 'a'.repeat(64) })
Expand Down
2 changes: 2 additions & 0 deletions apps/sim/lib/mothership/tools/handlers/function-execute.ts
Original file line number Diff line number Diff line change
Expand Up @@ -499,6 +499,8 @@ export async function executeFunctionExecute(
'internalSandboxProfile',
// Server-derived below — a model-supplied value must never select a session.
'sandboxSessionKey',
// Server-derived from resolved inputs; a model-supplied mount would skip their provenance.
'_sandboxFiles',
PRIVATE_SECRET_PROVENANCE_FIELD,
])
// One persistent session sandbox per chat: files and installed packages
Expand Down
Loading