Skip to content

Commit 1266679

Browse files
fix(tests): purge deleted tests, bill their storage, and record versions as loaded (#8858)
* fix(tests): purge deleted tests, bill their storage, and record versions as loaded - Retention cleanup removes an expired test's row before its source file, which its foreign key blocked, and treats test sources as billed so their bytes are released. - A run records each workflow's deployment, and a draft's timestamp, as the executor loads it, so an edit or redeploy mid-run no longer makes the run look current. - Organization-chat workspace submenus leave tests out, like the composer. - Open test opens a new tab on web; doc fixes for hook inheritance and mockSampleOutput. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix(tests): take a draft's timestamp from before the run loads it Reading it after the load let an edit landing in between look tested; the snapshot from before the load can only err toward stale. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
1 parent 1897cbb commit 1266679

13 files changed

Lines changed: 194 additions & 90 deletions

File tree

‎apps/sim/app/workspace/[workspaceId]/home/components/mothership-view/components/resource-content/resource-content.tsx‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -582,7 +582,7 @@ interface EmbeddedTestActionsProps {
582582

583583
/** Runs the test the tab shows, and opens its page. */
584584
function EmbeddedTestActions({ workspaceId, name }: EmbeddedTestActionsProps) {
585-
const router = useRouter()
585+
const openInternalLink = useOpenInternalLink()
586586
const testsEnabled = useFeatureFlag('workflow-tests')
587587
const detail = useWorkflowTest(workspaceId, name)
588588
const run = useTestRunAction({
@@ -612,7 +612,7 @@ function EmbeddedTestActions({ workspaceId, name }: EmbeddedTestActionsProps) {
612612
<Tooltip.Trigger asChild>
613613
<TabStripAction
614614
variant='subtle'
615-
onClick={() => router.push(`/workspace/${workspaceId}/tests/${name}`)}
615+
onClick={() => openInternalLink(`/workspace/${workspaceId}/tests/${name}`)}
616616
aria-label='Open test'
617617
>
618618
<SquareArrowUpRight className={RESOURCE_TAB_ICON_CLASS} />

‎apps/sim/app/workspace/[workspaceId]/home/components/user-input/components/plus-menu-dropdown/plus-menu-dropdown.tsx‎

Lines changed: 4 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -80,11 +80,12 @@ function candidateKey({ type, item }: MentionCandidate): string {
8080
const MENTION_ONLY_RESOURCE_TYPES = new Set<MothershipResourceType>(['integration'])
8181

8282
/**
83-
* Families an organization chat's workspace submenus leave out: the mention-only
84-
* ones, plus Browser and Terminal, which belong to this desktop rather than to a
85-
* workspace and so sit once after the workspaces.
83+
* Families an organization chat's workspace submenus leave out: the composer's
84+
* exclusions, the mention-only ones, plus Browser and Terminal, which belong to
85+
* this desktop rather than to a workspace and so sit once after the workspaces.
8686
*/
8787
const WORKSPACE_SUBMENU_EXCLUDED_TYPES: readonly MothershipResourceType[] = [
88+
...COMPOSER_EXCLUDED_TYPES,
8889
...MENTION_ONLY_RESOURCE_TYPES,
8990
'browser',
9091
'terminal',
Lines changed: 75 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,75 @@
1+
/**
2+
* Retention cleanup of deleted workflow tests against the disposable TEST_DATABASE_URL database:
3+
* the foreign keys, the storage ledger, and the deletes all run for real.
4+
*/
5+
import { readTestDatabaseUrl } from '@sim/db/testing/test-infrastructure'
6+
import { generateId } from '@sim/utils/id'
7+
import postgres from 'postgres'
8+
import { afterAll, beforeAll, describe, expect, it } from 'vitest'
9+
import { runCleanupSoftDeletes } from '@/background/cleanup-soft-deletes'
10+
11+
const control = postgres(readTestDatabaseUrl(), { max: 2, onnotice: () => {} })
12+
const userId = generateId()
13+
const workspaceId = generateId()
14+
const expired = '2020-01-01T00:00:00Z'
15+
16+
async function seedTest(deletedAt: string | null, sizeBytes: number) {
17+
const fileId = generateId()
18+
const testId = generateId()
19+
await control`INSERT INTO workspace_files (id, key, user_id, workspace_id, context, original_name, content_type, size_bytes, deleted_at)
20+
VALUES (${fileId}, ${`test/${workspaceId}/${fileId}`}, ${userId}, ${workspaceId}, 'test',
21+
${`${testId}.test.js`}, 'text/javascript', ${sizeBytes}, ${deletedAt})`
22+
await control`INSERT INTO workflow_test (id, workspace_id, name, title, body_file_id, source_hash, deleted_at)
23+
VALUES (${testId}, ${workspaceId}, ${testId}, 'Fixture', ${fileId}, 'hash', ${deletedAt})`
24+
await control`INSERT INTO workflow_test_run (id, test_id, workspace_id, version, triggered_by_actor)
25+
VALUES (${generateId()}, ${testId}, ${workspaceId}, 'draft', ${control.json({ type: 'user' })})`
26+
return { fileId, testId }
27+
}
28+
29+
describe('Retention cleanup of deleted workflow tests', () => {
30+
beforeAll(async () => {
31+
await control`INSERT INTO "user" (id, name, email, email_verified, created_at, updated_at)
32+
VALUES (${userId}, 'Test cleanup fixture', ${`${userId}@example.test`}, true, now(), now())`
33+
await control`INSERT INTO user_stats (id, user_id, storage_used_bytes)
34+
VALUES (${generateId()}, ${userId}, 100)`
35+
await control`INSERT INTO workspace (id, name, owner_id, billed_account_user_id, storage_used_bytes)
36+
VALUES (${workspaceId}, 'Test cleanup fixtures', ${userId}, ${userId}, 100)`
37+
})
38+
39+
afterAll(async () => {
40+
await control`DELETE FROM workspace WHERE id = ${workspaceId}`
41+
await control`DELETE FROM "user" WHERE id = ${userId}`
42+
await control.end()
43+
})
44+
45+
it('purges an expired test with its source and releases the source bytes', async () => {
46+
const deleted = await seedTest(expired, 40)
47+
const live = await seedTest(null, 60)
48+
49+
await runCleanupSoftDeletes({
50+
label: 'tests-integration',
51+
plan: 'free',
52+
retentionHours: 24,
53+
workspaceIds: [workspaceId],
54+
})
55+
56+
const tests = await control<
57+
{ id: string }[]
58+
>`SELECT id FROM workflow_test WHERE workspace_id = ${workspaceId}`
59+
expect(tests.map(({ id }) => id)).toEqual([live.testId])
60+
const files = await control<
61+
{ id: string }[]
62+
>`SELECT id FROM workspace_files WHERE workspace_id = ${workspaceId}`
63+
expect(files.map(({ id }) => id)).toEqual([live.fileId])
64+
const runs = await control`SELECT 1 FROM workflow_test_run WHERE test_id = ${deleted.testId}`
65+
expect(runs).toHaveLength(0)
66+
const [ws] = await control<
67+
{ storage_used_bytes: number }[]
68+
>`SELECT storage_used_bytes::int FROM workspace WHERE id = ${workspaceId}`
69+
expect(ws.storage_used_bytes).toBe(60)
70+
const [stats] = await control<
71+
{ storage_used_bytes: number }[]
72+
>`SELECT storage_used_bytes::int FROM user_stats WHERE user_id = ${userId}`
73+
expect(stats.storage_used_bytes).toBe(60)
74+
})
75+
})

‎apps/sim/background/cleanup-soft-deletes.test.ts‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -170,8 +170,11 @@ describe('cleanup soft deletes', () => {
170170
expect(dbChainMockFns.transaction.mock.invocationCallOrder[0]).toBeLessThan(
171171
mockReleaseWorkspaceFileVersionsForPurgeInTx.mock.invocationCallOrder[0]
172172
)
173+
const fileDelete = dbChainMockFns.delete.mock.calls.findIndex(
174+
([table]) => table === schemaMock.workspaceFiles
175+
)
173176
expect(mockReleaseWorkspaceFileVersionsForPurgeInTx.mock.invocationCallOrder[0]).toBeLessThan(
174-
dbChainMockFns.delete.mock.invocationCallOrder[0]
177+
dbChainMockFns.delete.mock.invocationCallOrder[fileDelete]
175178
)
176179
})
177180

‎apps/sim/background/cleanup-soft-deletes.ts‎

Lines changed: 19 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,7 @@ import {
99
userTableDefinitions,
1010
workflow,
1111
workflowMcpServer,
12+
workflowTest,
1213
workspaceFile,
1314
workspaceFiles,
1415
} from '@sim/db/schema'
@@ -255,6 +256,9 @@ async function deleteExpiredLegacyWorkspaceFileRows(
255256
return result
256257
}
257258

259+
/** Contexts whose bytes count toward the workspace's billed storage: a test's source is one. */
260+
const BILLED_FILE_CONTEXTS: readonly StorageContext[] = ['workspace', 'test']
261+
258262
async function deleteExpiredUnbilledWorkspaceFileRows(
259263
rows: WorkspaceFileScope['multiContextRows'],
260264
retentionDate: Date,
@@ -263,7 +267,7 @@ async function deleteExpiredUnbilledWorkspaceFileRows(
263267
const result = { deleted: 0, failed: 0 }
264268
const rowsByContext = new Map<StorageContext, WorkspaceFileScope['multiContextRows']>()
265269
for (const row of rows) {
266-
if (row.context === 'workspace') continue
270+
if (BILLED_FILE_CONTEXTS.includes(row.context)) continue
267271
const bucket = rowsByContext.get(row.context)
268272
if (bucket) bucket.push(row)
269273
else rowsByContext.set(row.context, [row])
@@ -305,7 +309,7 @@ async function deleteExpiredBillableWorkspaceFileRows(
305309
const result = { deleted: 0, failed: 0 }
306310
const rowsByWorkspace = new Map<string, WorkspaceFileScope['multiContextRows']>()
307311
for (const row of rows) {
308-
if (row.context !== 'workspace') continue
312+
if (!BILLED_FILE_CONTEXTS.includes(row.context)) continue
309313
if (!row.workspaceId) {
310314
result.failed++
311315
logger.error(`[${label}/workspaceFiles] Billable row has no workspace attribution`, {
@@ -334,21 +338,24 @@ async function deleteExpiredBillableWorkspaceFileRows(
334338
for (const batch of chunkArray(workspaceRows, DEFAULT_DELETE_CHUNK_SIZE)) {
335339
try {
336340
const deletedCount = await db.transaction(async (tx) => {
337-
await releaseWorkspaceFileVersionsForPurgeInTx(
338-
tx,
339-
batch.map(({ id }) => id),
340-
retentionDate
341-
)
341+
const fileIds = batch.map(({ id }) => id)
342+
await tx
343+
.delete(workflowTest)
344+
.where(
345+
and(
346+
inArray(workflowTest.bodyFileId, fileIds),
347+
isNotNull(workflowTest.deletedAt),
348+
lt(workflowTest.deletedAt, retentionDate)
349+
)
350+
)
351+
await releaseWorkspaceFileVersionsForPurgeInTx(tx, fileIds, retentionDate)
342352
const deletedRows = await tx
343353
.delete(workspaceFiles)
344354
.where(
345355
and(
346-
inArray(
347-
workspaceFiles.id,
348-
batch.map(({ id }) => id)
349-
),
356+
inArray(workspaceFiles.id, fileIds),
350357
eq(workspaceFiles.workspaceId, workspaceId),
351-
eq(workspaceFiles.context, 'workspace'),
358+
inArray(workspaceFiles.context, BILLED_FILE_CONTEXTS),
352359
isNotNull(workspaceFiles.deletedAt),
353360
lt(workspaceFiles.deletedAt, retentionDate)
354361
)

‎apps/sim/executor/execution/executor.ts‎

Lines changed: 10 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -89,6 +89,7 @@ export class DAGExecutor {
8989
async execute(workflowId: string, triggerBlockId?: string): Promise<ExecutionResult> {
9090
await this.contextExtensions.testHooks?.enterWorkflow({
9191
workflowId,
92+
deploymentVersionId: this.loadedDeploymentVersionId(workflowId),
9293
blocks: this.workflow.blocks.map((block) => ({
9394
id: block.id,
9495
name: block.metadata?.name ?? block.id,
@@ -284,6 +285,15 @@ export class DAGExecutor {
284285
return result
285286
}
286287

288+
/** The version this execution's state was loaded from, as its delegation authority records it. */
289+
private loadedDeploymentVersionId(workflowId: string): string | null {
290+
const current = this.contextExtensions.executorDelegationOrigin?.currentWorkflow
291+
if (current?.workflowId !== workflowId) {
292+
throw new Error(`Test run of workflow ${workflowId} has no loaded version`)
293+
}
294+
return current.mode === 'deployment' ? current.deploymentVersionId : null
295+
}
296+
287297
private restoreSavedIncomingEdges(dag: DAG, savedIncomingEdges?: Record<string, string[]>): void {
288298
if (!savedIncomingEdges) return
289299

‎apps/sim/executor/execution/test-hooks.test.ts‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -79,6 +79,10 @@ function run(
7979
workspaceId: 'ws',
8080
executionId: 'exec-1',
8181
principal: PRINCIPAL,
82+
executorDelegationOrigin: {
83+
workflowId: 'wf',
84+
currentWorkflow: { workflowId: 'wf', mode: 'draft' },
85+
},
8286
testHooks: channel.hooks,
8387
...(abortSignal ? { abortSignal } : {}),
8488
},

‎apps/sim/executor/execution/types.ts‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -288,12 +288,15 @@ export interface TestWorkflowBlock {
288288
* and spies against those blocks by name. A mocked block awaits `resolveMock` in place of its
289289
* handler; everything after the handler (normalization, redaction, logging, edges) runs as
290290
* usual. A mocked tool call awaits `resolveToolMock` in place of the tool. Child workflow
291-
* executions inherit the hooks.
291+
* executions inherit the hooks, except a custom block's source workflow, which a test mocks as a
292+
* whole, and a workflow an Agent calls as a tool, which runs without them.
292293
*/
293294
export interface ExecutionTestHooks {
294295
/** `resolvedSecretTraceRegistry` redacts what this run sends back to the test. */
295296
enterWorkflow(workflow: {
296297
workflowId: string
298+
/** The deployment this execution loaded; null when it loaded the draft. */
299+
deploymentVersionId: string | null
297300
blocks: TestWorkflowBlock[]
298301
resolvedSecretTraceRegistry: ResolvedSecretTraceRegistry | undefined
299302
}): Promise<void>

‎apps/sim/lib/api/contracts/mothership-management-tools.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -24,7 +24,7 @@ export const managementToolContracts = [
2424
route: 'sim',
2525
scope: 'all',
2626
description:
27-
'Create and run the selected workspace\u2019s workflow tests. create takes a name, a one-line title for the concern, and an optional description, and returns tests/<name>.test.js; write the cases into that file with the file tools. A test file is plain vitest: import { describe, it, expect, vi } from "vitest" and { runWorkflow, mockBlock, mockTool, spyOnBlock } from "sim:test", one top-level describe, an it per case. runWorkflow(name, input) runs a workflow and returns { output } (pass { trigger: \"Trigger block name\" } as a third argument when it has several triggers); mockBlock(blockName) returns a vi.fn whose value replaces that block\u2019s output and records its inputs; mockTool(toolId) or mockTool(agentBlockName, toolId) answers an Agent\u2019s calls to that tool the same way while the model still runs, naming built-in tools by id (slack_message), MCP tools by server as mockTool({ mcp: "Server name", tool: "tool_name" }), and custom tools by title as mockTool({ customTool: "Title" }); .mockSampleOutput({ ...overrides }) on either mock returns a placeholder output shaped like the real one, with your fields merged in; await expect(value).toMatchRubric(rubric) asks a model judge for pass or fail. Every write is checked and refused if the file does not load. run takes a version (draft while editing, deployed before shipping), waits, and returns each file\u2019s failures with line numbers and messages. list and get report status; update changes title or description.',
27+
'Create and run the selected workspace\u2019s workflow tests. create takes a name, a one-line title for the concern, and an optional description, and returns tests/<name>.test.js; write the cases into that file with the file tools. A test file is plain vitest: import { describe, it, expect, vi } from "vitest" and { runWorkflow, mockBlock, mockTool, spyOnBlock } from "sim:test", one top-level describe, an it per case. runWorkflow(name, input) runs a workflow and returns { output } (pass { trigger: \"Trigger block name\" } as a third argument when it has several triggers); mockBlock(blockName) returns a vi.fn whose value replaces that block\u2019s output and records its inputs; mockTool(toolId) or mockTool(agentBlockName, toolId) answers an Agent\u2019s calls to that tool the same way while the model still runs, naming built-in tools by id (slack_message), MCP tools by server as mockTool({ mcp: "Server name", tool: "tool_name" }), and custom tools by title as mockTool({ customTool: "Title" }); .mockSampleOutput({ ...overrides }) on either mock makes it answer with a placeholder output shaped like the real one, with your fields merged in; await expect(value).toMatchRubric(rubric) asks a model judge for pass or fail. Every write is checked and refused if the file does not load. run takes a version (draft while editing, deployed before shipping), waits, and returns each file\u2019s failures with line numbers and messages. list and get report status; update changes title or description.',
2828
inputSchema: mothershipTestsInputSchema,
2929
},
3030
{

‎apps/sim/lib/workflow-tests/application/run-tests.ts‎

Lines changed: 2 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -55,7 +55,7 @@ async function executeTestRun(params: {
5555
const buffer = await fetchWorkspaceFileBuffer(file, { maxBytes: MAX_TEST_SOURCE_BYTES })
5656
const source = buffer.toString('utf-8')
5757
sourceHash = testSourceHash(source)
58-
const { report, enteredWorkflowIds } = await runWorkflowTestFile({
58+
const { report, entered } = await runWorkflowTestFile({
5959
principal,
6060
workspaceId: test.workspaceId,
6161
source,
@@ -69,9 +69,8 @@ async function executeTestRun(params: {
6969
)
7070
const ranAgainst = await readRanAgainst({
7171
executionIds,
72-
enteredWorkflowIds,
72+
entered,
7373
workspaceId: test.workspaceId,
74-
version,
7574
})
7675
await completeWorkflowTestRun(runId, report, sourceHash, ranAgainst)
7776
} catch (error) {

0 commit comments

Comments
 (0)