From e20a3a4630db7a48ef6b752a192a80330aaa1dc9 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 18:22:08 -0700 Subject: [PATCH 1/4] chore(trigger): tighten task limits, fix queues that never serialized, schedule missing self-hosted crons --- .../platform/self-hosting/background-jobs.mdx | 3 +++ apps/sim/background/cleanup-dispatch.ts | 1 + apps/sim/background/cleanup-file-versions.ts | 8 +++++- .../background/cleanup-stale-executions.ts | 13 ++++++---- apps/sim/background/cleanup-table-row-ttl.ts | 1 + apps/sim/background/fork-content-copy.ts | 8 ++++++ apps/sim/background/knowledge-processing.ts | 7 ++++-- apps/sim/background/lifecycle-email.ts | 2 ++ .../background/mothership-inbox-execution.ts | 1 + .../background/quickbooks-webhook-ingress.ts | 1 + apps/sim/background/run-data-drain.ts | 12 +++++++++ apps/sim/background/table-run-dispatcher.ts | 8 +++++- apps/sim/lib/table/dispatcher.ts | 5 ++-- apps/sim/trigger.config.ts | 2 +- docker/crontab | 6 +++++ helm/sim/values.yaml | 25 +++++++++++++++++++ 16 files changed, 90 insertions(+), 13 deletions(-) diff --git a/apps/docs/content/docs/platform/self-hosting/background-jobs.mdx b/apps/docs/content/docs/platform/self-hosting/background-jobs.mdx index b34de9913ce..d2807cdb3ec 100644 --- a/apps/docs/content/docs/platform/self-hosting/background-jobs.mdx +++ b/apps/docs/content/docs/platform/self-hosting/background-jobs.mdx @@ -48,11 +48,13 @@ Point cron at an **internal** address where possible (the in-cluster Service, or | Outbox processing | `/api/webhooks/outbox/process` | `*/1 * * * *` | Transactional-outbox retries for billing, membership, enterprise issuance, and workflow-deployment side effects | | Workspace file search dispatch | `/api/cron/workspace-file-search-dispatch` | `*/1 * * * *` | Dispatches indexing work for workspace file search | | Knowledge projection | `/api/cron/knowledge-projection` | `*/1 * * * *` | Brings knowledge base search up to date with document, permission, and chunk changes | +| Table row-change fold | `/api/cron/fold-table-row-changes` | `*/1 * * * *` | Folds the table row-change log into each table's stored row count and version | | Connector sync | `/api/knowledge/connectors/sync` | `*/5 * * * *` | Knowledge base connector syncs | | Connector member sync | `/api/knowledge/connectors/member-sync` | `*/5 * * * *` | Per-member content/access sync for indexed connectors; live Search sources are skipped | | Connector directory sync | `/api/knowledge/connectors/directory-sync` | `*/5 * * * *` | Refreshes stored groups for indexed administrator-mode connectors; live Search sources are skipped | | Workspace events poll | `/api/workspace-events/poll` | `*/15 * * * *` | Workspace event triggers | | Table row TTL cleanup | `/api/cron/cleanup-table-row-ttl` | `*/15 * * * *` | Deletes table rows whose TTL column has expired | +| Stale execution cleanup | `/api/cron/cleanup-stale-executions` | `*/30 * * * *` | Fails runs whose worker died without finishing — workflow executions, background jobs, table jobs and table runs — so they stop showing as running | | OAuth token cleanup | `/api/cron/cleanup-oauth-tokens` | `0 * * * *` | Deletes access and refresh tokens after the retention tail | | SCIM reconciliation | `/api/cron/scim-reconcile` | `17 * * * *` | Re-applies directory group mappings | | Data drains | `/api/cron/run-data-drains` | `0 * * * *` | Enterprise data drains | @@ -61,6 +63,7 @@ Point cron at an **internal** address where possible (the in-cluster Service, or | Reconcile billing seats | `/api/cron/reconcile-billing-seats` | `0 * * * *` | Billing only — safe to disable when self-hosted | | Reconcile inbox entitlement | `/api/cron/reconcile-inbox-entitlement` | `0 3 * * *` | Inbox access reconciliation | | Cleanup sandbox images | `/api/cron/cleanup-sandbox-images` | `30 4 * * *` | Reclaims sandbox images | +| Idempotency key cleanup | `/api/webhooks/cleanup/idempotency` | `0 5 * * *` | Deletes deduplication records older than seven days | **Subscription renewal** covers Microsoft Teams chat triggers, whose Microsoft Graph subscriptions are hard-capped at about three days. Without it, Teams triggers work for a couple of days and then quietly stop. Gmail, Outlook, Drive, Calendar, and Sheets triggers are **polled** instead — they depend on the per-minute poll jobs above, not on this one. diff --git a/apps/sim/background/cleanup-dispatch.ts b/apps/sim/background/cleanup-dispatch.ts index d0e1f94f5d2..827b0abfbf0 100644 --- a/apps/sim/background/cleanup-dispatch.ts +++ b/apps/sim/background/cleanup-dispatch.ts @@ -15,6 +15,7 @@ export const CLEANUP_DISPATCH_MAX_ATTEMPTS = 3 */ export const cleanupDispatchTask = task({ id: 'cleanup-dispatch', + maxDuration: 900, queue: { concurrencyLimit: 1 }, retry: { maxAttempts: CLEANUP_DISPATCH_MAX_ATTEMPTS }, run: ({ jobType }: CleanupDispatchPayload) => dispatchCleanupJobs(jobType), diff --git a/apps/sim/background/cleanup-file-versions.ts b/apps/sim/background/cleanup-file-versions.ts index 0564806c50d..4771e8b3312 100644 --- a/apps/sim/background/cleanup-file-versions.ts +++ b/apps/sim/background/cleanup-file-versions.ts @@ -188,7 +188,13 @@ export async function runCleanupFileVersions(payload: CleanupJobPayload): Promis export const cleanupFileVersionsTask = task({ id: 'cleanup-file-versions', - machine: 'large-1x', + /** + * Sized from production telemetry: peak RSS 219 MB across ~15,600 runs, so + * `small-2x` (1 GB) holds over four times the worst case. Each pass works in + * bounded batches of version rows, so the peak does not grow with backlog. + */ + machine: 'small-2x', + maxDuration: 600, queue: retentionCleanupQueue, retry: { maxAttempts: 1 }, run: runCleanupFileVersions, diff --git a/apps/sim/background/cleanup-stale-executions.ts b/apps/sim/background/cleanup-stale-executions.ts index 5fce17f1720..9bc26f1e216 100644 --- a/apps/sim/background/cleanup-stale-executions.ts +++ b/apps/sim/background/cleanup-stale-executions.ts @@ -49,17 +49,19 @@ const GENERIC_STALE_PROCESSING_ERROR = `Job terminated: stuck in processing for const EXECUTION_DEADLINE_ERROR = getTimeoutErrorMessage() /** * Table jobs run as detached workers with progress heartbeats, independently of workflow timeout - * policy. Preserve their historical 90-minute task window plus five-minute cleanup grace. + * policy. A running job is treated as dead once its last progress write (`updatedAt`) is older + * than this. It measures the gap since the last heartbeat, not run length, so it does not track + * any task's `maxDuration`; it only has to exceed the longest stretch a live worker goes without + * writing progress. */ const TABLE_JOB_STALE_THRESHOLD_MINUTES = 95 /** Terminal table-jobs older than this are pruned; only the latest job per table is ever read. */ const TABLE_JOB_RETENTION_HOURS = 24 /** * A table run dispatch whose holder has not made progress for this long is - * treated as dead. Same shape and window as the table-job threshold above: the - * 90-minute Trigger.dev task ceiling (`maxDuration` in `trigger.config.ts`) plus - * five minutes of cleanup grace, measured from the dispatcher's own per-window - * heartbeat rather than from when the run was requested. + * treated as dead. Same shape and window as the table-job threshold above, + * measured from the dispatcher's own per-window heartbeat rather than from when + * the run was requested, so it is independent of the task's `maxDuration`. */ const TABLE_DISPATCH_STALE_THRESHOLD_MINUTES = 95 /** Per-run ceiling on reaped dispatches, so one tick cannot fan out unbounded SSE. */ @@ -795,6 +797,7 @@ export async function runCleanupStaleExecutions() { export const cleanupStaleExecutionsTask = task({ id: 'cleanup-stale-executions', + maxDuration: 1800, queue: { concurrencyLimit: 1 }, run: () => runCleanupStaleExecutions(), }) diff --git a/apps/sim/background/cleanup-table-row-ttl.ts b/apps/sim/background/cleanup-table-row-ttl.ts index 9a40bbafdaf..b43f6d5e4a9 100644 --- a/apps/sim/background/cleanup-table-row-ttl.ts +++ b/apps/sim/background/cleanup-table-row-ttl.ts @@ -335,6 +335,7 @@ export async function runCleanupTableRowTtl( export const cleanupTableRowTtlTask = task({ id: 'cleanup-table-row-ttl', + maxDuration: 600, queue: { concurrencyLimit: 1 }, run: () => runCleanupTableRowTtl(), }) diff --git a/apps/sim/background/fork-content-copy.ts b/apps/sim/background/fork-content-copy.ts index 8b721cecc62..6e2db728f2a 100644 --- a/apps/sim/background/fork-content-copy.ts +++ b/apps/sim/background/fork-content-copy.ts @@ -20,6 +20,14 @@ import { export const forkContentCopyTask = task({ id: 'fork-content-copy', machine: 'large-2x', + /** + * Inside `STALE_ACTIVE_MS` (30 minutes, in the background-work store), after + * which the outbox cron marks the fork's status row failed. A run outliving + * that would keep copying under a status that already says it failed; the + * remaining five minutes absorb queue wait, since the row starts its clock at + * fork time. + */ + maxDuration: 1500, retry: { maxAttempts: 1 }, queue: { name: 'fork-content-copy', diff --git a/apps/sim/background/knowledge-processing.ts b/apps/sim/background/knowledge-processing.ts index f31b04ba234..f080f061941 100644 --- a/apps/sim/background/knowledge-processing.ts +++ b/apps/sim/background/knowledge-processing.ts @@ -306,13 +306,16 @@ export async function runDocumentProcessing( */ export const interactiveProcessingQueue = queue({ name: INTERACTIVE_PROCESSING_QUEUE_NAME, - concurrencyLimit: envNumber(env.KB_CONFIG_CONCURRENCY_LIMIT, 20), + concurrencyLimit: envNumber(env.KB_CONFIG_CONCURRENCY_LIMIT, 20, { min: 1, integer: true }), }) /** Referenced by no dispatch site: named per trigger, declared here so the deploy registers it. */ export const backfillProcessingQueue = queue({ name: BACKFILL_PROCESSING_QUEUE_NAME, - concurrencyLimit: envNumber(env.KB_CONFIG_BACKFILL_CONCURRENCY_LIMIT, 20), + concurrencyLimit: envNumber(env.KB_CONFIG_BACKFILL_CONCURRENCY_LIMIT, 20, { + min: 1, + integer: true, + }), }) export const processDocument = task({ diff --git a/apps/sim/background/lifecycle-email.ts b/apps/sim/background/lifecycle-email.ts index 7179ff398d5..b2ffdfbc5ad 100644 --- a/apps/sim/background/lifecycle-email.ts +++ b/apps/sim/background/lifecycle-email.ts @@ -59,6 +59,8 @@ async function sendLifecycleEmail({ userId, type }: LifecycleEmailParams): Promi export const lifecycleEmailTask = task({ id: LIFECYCLE_EMAIL_TASK_ID, + maxDuration: 120, + queue: { concurrencyLimit: 10 }, retry: { maxAttempts: 2 }, run: async (params: LifecycleEmailParams) => { await sendLifecycleEmail(params) diff --git a/apps/sim/background/mothership-inbox-execution.ts b/apps/sim/background/mothership-inbox-execution.ts index 82a706189e7..9a1ceddc757 100644 --- a/apps/sim/background/mothership-inbox-execution.ts +++ b/apps/sim/background/mothership-inbox-execution.ts @@ -11,6 +11,7 @@ export interface MothershipInboxExecutionParams { export const mothershipInboxExecution = task({ id: 'mothership-inbox-execution', machine: { preset: 'medium-1x' }, + queue: { concurrencyLimit: 10 }, retry: { maxAttempts: 2, minTimeoutInMs: 5000, diff --git a/apps/sim/background/quickbooks-webhook-ingress.ts b/apps/sim/background/quickbooks-webhook-ingress.ts index 7e7436c49ba..c1629fdac6c 100644 --- a/apps/sim/background/quickbooks-webhook-ingress.ts +++ b/apps/sim/background/quickbooks-webhook-ingress.ts @@ -143,6 +143,7 @@ export async function enqueueQuickBooksWebhookIngress( export const quickBooksWebhookIngressTask = task({ id: 'quickbooks-webhook-ingress', machine: 'small-1x', + maxDuration: 300, retry: { maxAttempts: QUICKBOOKS_WEBHOOK_INGRESS_MAX_ATTEMPTS, factor: 2, diff --git a/apps/sim/background/run-data-drain.ts b/apps/sim/background/run-data-drain.ts index 21d462055df..9dac3ef1ac0 100644 --- a/apps/sim/background/run-data-drain.ts +++ b/apps/sim/background/run-data-drain.ts @@ -9,6 +9,18 @@ interface RunDataDrainPayload { export const runDataDrainTask = task({ id: 'run-data-drain', + /** + * The drain cursor commits only after the last chunk is delivered, so a run + * cut off by its ceiling restarts from the old cursor and re-exports the same + * rows. A shorter cap would loop a large drain forever. + */ + maxDuration: 5400, + /** + * Enqueue sites key runs by `data-drain:`; this limit is what makes that + * key serialize one drain's runs instead of granting each key the + * environment's full concurrency. + */ + queue: { concurrencyLimit: 1 }, run: async ({ drainId, trigger }: RunDataDrainPayload, { signal }) => runDrain(drainId, trigger, { signal }), }) diff --git a/apps/sim/background/table-run-dispatcher.ts b/apps/sim/background/table-run-dispatcher.ts index 7e1b41e29f7..cba9c177157 100644 --- a/apps/sim/background/table-run-dispatcher.ts +++ b/apps/sim/background/table-run-dispatcher.ts @@ -29,9 +29,15 @@ export const tableRunDispatcherTask = task({ * 0.03 for p90, so the larger preset is bought for its RAM. */ machine: 'small-2x', + maxDuration: 3600, + /** + * The trigger site keys each run by its `dispatchId`, so the limit applies + * per dispatch: 1 serializes a duplicate run of the same dispatch without + * throttling distinct dispatches against each other. + */ queue: { name: 'table-run-dispatcher', - concurrencyLimit: 8, + concurrencyLimit: 1, }, run: async (payload: TableRunDispatcherPayload) => { const { dispatchId, concurrency } = payload diff --git a/apps/sim/lib/table/dispatcher.ts b/apps/sim/lib/table/dispatcher.ts index 8e87416ceda..2d812ff8b94 100644 --- a/apps/sim/lib/table/dispatcher.ts +++ b/apps/sim/lib/table/dispatcher.ts @@ -52,9 +52,8 @@ const STALE_DISPATCH_EVENT_CONCURRENCY = 10 * How long past its stale threshold a dispatch may be spared by cell activity * before it is reclaimed anyway. Bounds the one masking case the probe cannot * resolve — two table-wide dispatches sharing a group — which continuous - * activity would otherwise hide forever. Far beyond any single window: the - * Trigger.dev run ceiling is ninety minutes, and a live dispatch heartbeats - * between windows no matter what its cells are doing. + * activity would otherwise hide forever. Far beyond any single window: a live + * dispatch heartbeats between windows no matter what its cells are doing. */ const DISPATCH_ABSOLUTE_STALE_MS = 24 * 60 * 60 * 1000 diff --git a/apps/sim/trigger.config.ts b/apps/sim/trigger.config.ts index 5c2ae2d6cff..ead1fa8ef25 100644 --- a/apps/sim/trigger.config.ts +++ b/apps/sim/trigger.config.ts @@ -87,7 +87,7 @@ const grafanaTelemetry = grafanaFullyConfigured export default defineConfig({ project: env.TRIGGER_PROJECT_ID!, runtime: 'node-24', - logLevel: 'log', + logLevel: 'info', maxDuration: 5400, retries: { enabledInDev: false, diff --git a/docker/crontab b/docker/crontab index 94cfe25fabc..91f8c6e07bf 100644 --- a/docker/crontab +++ b/docker/crontab @@ -56,6 +56,9 @@ SHELL=/bin/sh # Deletes table rows whose TTL column has expired */15 * * * * curl -fsS -m 60 -o /dev/null -H "Authorization: Bearer $CRON_SECRET" "$SIM_URL/api/cron/cleanup-table-row-ttl" +# Fails runs whose worker died without a terminal state and prunes their history +*/30 * * * * curl -fsS -m 60 -o /dev/null -H "Authorization: Bearer $CRON_SECRET" "$SIM_URL/api/cron/cleanup-stale-executions" + # Deletes OAuth token rows after the retention tail 0 * * * * curl -fsS -m 120 -o /dev/null -H "Authorization: Bearer $CRON_SECRET" "$SIM_URL/api/cron/cleanup-oauth-tokens" @@ -68,6 +71,9 @@ SHELL=/bin/sh # Reclaims sandbox images 30 4 * * * curl -fsS -m 300 -o /dev/null -H "Authorization: Bearer $CRON_SECRET" "$SIM_URL/api/cron/cleanup-sandbox-images" +# Deletes idempotency records past the seven-day dedupe window +0 5 * * * curl -fsS -m 300 -o /dev/null -H "Authorization: Bearer $CRON_SECRET" "$SIM_URL/api/webhooks/cleanup/idempotency" + # Inbox entitlement reconciliation 0 3 * * * curl -fsS -m 120 -o /dev/null -H "Authorization: Bearer $CRON_SECRET" "$SIM_URL/api/cron/reconcile-inbox-entitlement" diff --git a/helm/sim/values.yaml b/helm/sim/values.yaml index 403c06a78fe..d8e46e8503c 100644 --- a/helm/sim/values.yaml +++ b/helm/sim/values.yaml @@ -1593,6 +1593,31 @@ cronjobs: successfulJobsHistoryLimit: 3 failedJobsHistoryLimit: 1 + # Fails runs whose worker died without writing a terminal state (workflow + # executions, async jobs, table jobs and dispatches) and prunes old terminal + # bookkeeping rows. + cleanupStaleExecutions: + enabled: true + name: cleanup-stale-executions + schedule: "*/30 * * * *" + path: "/api/cron/cleanup-stale-executions" + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: 3 + failedJobsHistoryLimit: 1 + + # Deletes idempotency records older than seven days, the longest dedupe + # window any caller uses. Runs inline in the request for up to five minutes. + idempotencyCleanup: + enabled: true + name: idempotency-cleanup + schedule: "0 5 * * *" + path: "/api/webhooks/cleanup/idempotency" + requestTimeoutSeconds: 300 + activeDeadlineSeconds: 360 + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: 3 + failedJobsHistoryLimit: 1 + # Global CronJob settings image: repository: curlimages/curl From 8a3bdd9b9b6850efd386c6c389455253bd8c5b58 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 18:36:32 -0700 Subject: [PATCH 2/4] refactor(trigger): derive the fork ceiling from STALE_ACTIVE_MS and trim sizing comments --- apps/sim/background/cleanup-file-versions.ts | 5 ++--- apps/sim/background/cleanup-stale-executions.ts | 9 +++------ apps/sim/background/fork-content-copy.ts | 11 +++++------ .../ee/workspace-forking/lib/background-work/store.ts | 2 +- 4 files changed, 11 insertions(+), 16 deletions(-) diff --git a/apps/sim/background/cleanup-file-versions.ts b/apps/sim/background/cleanup-file-versions.ts index 4771e8b3312..e98d907bc1a 100644 --- a/apps/sim/background/cleanup-file-versions.ts +++ b/apps/sim/background/cleanup-file-versions.ts @@ -189,9 +189,8 @@ export async function runCleanupFileVersions(payload: CleanupJobPayload): Promis export const cleanupFileVersionsTask = task({ id: 'cleanup-file-versions', /** - * Sized from production telemetry: peak RSS 219 MB across ~15,600 runs, so - * `small-2x` (1 GB) holds over four times the worst case. Each pass works in - * bounded batches of version rows, so the peak does not grow with backlog. + * Each pass works in bounded batches of version rows, so peak memory stays well under + * `small-2x` (1 GB) and does not grow with backlog. */ machine: 'small-2x', maxDuration: 600, diff --git a/apps/sim/background/cleanup-stale-executions.ts b/apps/sim/background/cleanup-stale-executions.ts index 9bc26f1e216..5fb8189ffdb 100644 --- a/apps/sim/background/cleanup-stale-executions.ts +++ b/apps/sim/background/cleanup-stale-executions.ts @@ -50,18 +50,15 @@ const EXECUTION_DEADLINE_ERROR = getTimeoutErrorMessage() /** * Table jobs run as detached workers with progress heartbeats, independently of workflow timeout * policy. A running job is treated as dead once its last progress write (`updatedAt`) is older - * than this. It measures the gap since the last heartbeat, not run length, so it does not track - * any task's `maxDuration`; it only has to exceed the longest stretch a live worker goes without - * writing progress. + * than this, so it must exceed the longest gap between a live worker's heartbeats. */ const TABLE_JOB_STALE_THRESHOLD_MINUTES = 95 /** Terminal table-jobs older than this are pruned; only the latest job per table is ever read. */ const TABLE_JOB_RETENTION_HOURS = 24 /** * A table run dispatch whose holder has not made progress for this long is - * treated as dead. Same shape and window as the table-job threshold above, - * measured from the dispatcher's own per-window heartbeat rather than from when - * the run was requested, so it is independent of the task's `maxDuration`. + * treated as dead. Same rule as the table-job threshold above, measured from the + * dispatcher's own per-window heartbeat rather than from when the run was requested. */ const TABLE_DISPATCH_STALE_THRESHOLD_MINUTES = 95 /** Per-run ceiling on reaped dispatches, so one tick cannot fan out unbounded SSE. */ diff --git a/apps/sim/background/fork-content-copy.ts b/apps/sim/background/fork-content-copy.ts index 6e2db728f2a..0f5c9aae72d 100644 --- a/apps/sim/background/fork-content-copy.ts +++ b/apps/sim/background/fork-content-copy.ts @@ -1,4 +1,5 @@ import { task } from '@trigger.dev/sdk' +import { STALE_ACTIVE_MS } from '@/ee/workspace-forking/lib/background-work/store' import { type ForkContentCopyPayload, runForkContentCopy, @@ -21,13 +22,11 @@ export const forkContentCopyTask = task({ id: 'fork-content-copy', machine: 'large-2x', /** - * Inside `STALE_ACTIVE_MS` (30 minutes, in the background-work store), after - * which the outbox cron marks the fork's status row failed. A run outliving - * that would keep copying under a status that already says it failed; the - * remaining five minutes absorb queue wait, since the row starts its clock at - * fork time. + * Ends inside {@link STALE_ACTIVE_MS}, after which the outbox cron marks the fork's status + * row failed, so a run never keeps copying under a status that already says it failed. The + * five-minute margin absorbs queue wait, since the row's clock starts at fork time. */ - maxDuration: 1500, + maxDuration: STALE_ACTIVE_MS / 1000 - 5 * 60, retry: { maxAttempts: 1 }, queue: { name: 'fork-content-copy', diff --git a/apps/sim/ee/workspace-forking/lib/background-work/store.ts b/apps/sim/ee/workspace-forking/lib/background-work/store.ts index bc9eb612f2d..a465d97bba2 100644 --- a/apps/sim/ee/workspace-forking/lib/background-work/store.ts +++ b/apps/sim/ee/workspace-forking/lib/background-work/store.ts @@ -53,7 +53,7 @@ const BACKGROUND_WORK_PAGE_SIZE_MAX = 100 * sweeps these via {@link reapStaleBackgroundWork}, marking them `failed` so the UI * surfaces the failure instead of spinning forever. */ -const STALE_ACTIVE_MS = 30 * 60 * 1000 +export const STALE_ACTIVE_MS = 30 * 60 * 1000 /** Terminal rows older than this are pruned by the cron so the audit trail stays bounded. */ const RETENTION_MS = 30 * 24 * 60 * 60 * 1000 From a500cf2ccc9495fdd1bfb90453860f1201a51145 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 18:50:10 -0700 Subject: [PATCH 3/4] chore(trigger): drop caps whose work grows with user data or that lose cleanup on kill --- apps/sim/background/cleanup-dispatch.ts | 1 - apps/sim/background/cleanup-stale-executions.ts | 1 - apps/sim/background/fork-content-copy.ts | 7 ------- apps/sim/background/knowledge-processing.ts | 7 ++----- apps/sim/background/mothership-inbox-execution.ts | 1 - apps/sim/background/table-run-dispatcher.ts | 1 - apps/sim/ee/workspace-forking/lib/background-work/store.ts | 2 +- 7 files changed, 3 insertions(+), 17 deletions(-) diff --git a/apps/sim/background/cleanup-dispatch.ts b/apps/sim/background/cleanup-dispatch.ts index 827b0abfbf0..d0e1f94f5d2 100644 --- a/apps/sim/background/cleanup-dispatch.ts +++ b/apps/sim/background/cleanup-dispatch.ts @@ -15,7 +15,6 @@ export const CLEANUP_DISPATCH_MAX_ATTEMPTS = 3 */ export const cleanupDispatchTask = task({ id: 'cleanup-dispatch', - maxDuration: 900, queue: { concurrencyLimit: 1 }, retry: { maxAttempts: CLEANUP_DISPATCH_MAX_ATTEMPTS }, run: ({ jobType }: CleanupDispatchPayload) => dispatchCleanupJobs(jobType), diff --git a/apps/sim/background/cleanup-stale-executions.ts b/apps/sim/background/cleanup-stale-executions.ts index 5fb8189ffdb..198ca2eab97 100644 --- a/apps/sim/background/cleanup-stale-executions.ts +++ b/apps/sim/background/cleanup-stale-executions.ts @@ -794,7 +794,6 @@ export async function runCleanupStaleExecutions() { export const cleanupStaleExecutionsTask = task({ id: 'cleanup-stale-executions', - maxDuration: 1800, queue: { concurrencyLimit: 1 }, run: () => runCleanupStaleExecutions(), }) diff --git a/apps/sim/background/fork-content-copy.ts b/apps/sim/background/fork-content-copy.ts index 0f5c9aae72d..8b721cecc62 100644 --- a/apps/sim/background/fork-content-copy.ts +++ b/apps/sim/background/fork-content-copy.ts @@ -1,5 +1,4 @@ import { task } from '@trigger.dev/sdk' -import { STALE_ACTIVE_MS } from '@/ee/workspace-forking/lib/background-work/store' import { type ForkContentCopyPayload, runForkContentCopy, @@ -21,12 +20,6 @@ import { export const forkContentCopyTask = task({ id: 'fork-content-copy', machine: 'large-2x', - /** - * Ends inside {@link STALE_ACTIVE_MS}, after which the outbox cron marks the fork's status - * row failed, so a run never keeps copying under a status that already says it failed. The - * five-minute margin absorbs queue wait, since the row's clock starts at fork time. - */ - maxDuration: STALE_ACTIVE_MS / 1000 - 5 * 60, retry: { maxAttempts: 1 }, queue: { name: 'fork-content-copy', diff --git a/apps/sim/background/knowledge-processing.ts b/apps/sim/background/knowledge-processing.ts index f080f061941..f31b04ba234 100644 --- a/apps/sim/background/knowledge-processing.ts +++ b/apps/sim/background/knowledge-processing.ts @@ -306,16 +306,13 @@ export async function runDocumentProcessing( */ export const interactiveProcessingQueue = queue({ name: INTERACTIVE_PROCESSING_QUEUE_NAME, - concurrencyLimit: envNumber(env.KB_CONFIG_CONCURRENCY_LIMIT, 20, { min: 1, integer: true }), + concurrencyLimit: envNumber(env.KB_CONFIG_CONCURRENCY_LIMIT, 20), }) /** Referenced by no dispatch site: named per trigger, declared here so the deploy registers it. */ export const backfillProcessingQueue = queue({ name: BACKFILL_PROCESSING_QUEUE_NAME, - concurrencyLimit: envNumber(env.KB_CONFIG_BACKFILL_CONCURRENCY_LIMIT, 20, { - min: 1, - integer: true, - }), + concurrencyLimit: envNumber(env.KB_CONFIG_BACKFILL_CONCURRENCY_LIMIT, 20), }) export const processDocument = task({ diff --git a/apps/sim/background/mothership-inbox-execution.ts b/apps/sim/background/mothership-inbox-execution.ts index 9a1ceddc757..82a706189e7 100644 --- a/apps/sim/background/mothership-inbox-execution.ts +++ b/apps/sim/background/mothership-inbox-execution.ts @@ -11,7 +11,6 @@ export interface MothershipInboxExecutionParams { export const mothershipInboxExecution = task({ id: 'mothership-inbox-execution', machine: { preset: 'medium-1x' }, - queue: { concurrencyLimit: 10 }, retry: { maxAttempts: 2, minTimeoutInMs: 5000, diff --git a/apps/sim/background/table-run-dispatcher.ts b/apps/sim/background/table-run-dispatcher.ts index cba9c177157..bb8593a6973 100644 --- a/apps/sim/background/table-run-dispatcher.ts +++ b/apps/sim/background/table-run-dispatcher.ts @@ -29,7 +29,6 @@ export const tableRunDispatcherTask = task({ * 0.03 for p90, so the larger preset is bought for its RAM. */ machine: 'small-2x', - maxDuration: 3600, /** * The trigger site keys each run by its `dispatchId`, so the limit applies * per dispatch: 1 serializes a duplicate run of the same dispatch without diff --git a/apps/sim/ee/workspace-forking/lib/background-work/store.ts b/apps/sim/ee/workspace-forking/lib/background-work/store.ts index a465d97bba2..bc9eb612f2d 100644 --- a/apps/sim/ee/workspace-forking/lib/background-work/store.ts +++ b/apps/sim/ee/workspace-forking/lib/background-work/store.ts @@ -53,7 +53,7 @@ const BACKGROUND_WORK_PAGE_SIZE_MAX = 100 * sweeps these via {@link reapStaleBackgroundWork}, marking them `failed` so the UI * surfaces the failure instead of spinning forever. */ -export const STALE_ACTIVE_MS = 30 * 60 * 1000 +const STALE_ACTIVE_MS = 30 * 60 * 1000 /** Terminal rows older than this are pruned by the cron so the audit trail stays bounded. */ const RETENTION_MS = 30 * 24 * 60 * 60 * 1000 From 7c357664b67b6ed368abf09851c0046da7d9f8e6 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Tue, 6 Oct 2026 18:58:26 -0700 Subject: [PATCH 4/4] chore(trigger): keep unbounded-fan-out tasks on the global ceiling; bump chart for new cronjobs --- apps/sim/background/cleanup-file-versions.ts | 7 +------ apps/sim/background/lifecycle-email.ts | 1 - apps/sim/background/quickbooks-webhook-ingress.ts | 1 - helm/sim/Chart.yaml | 2 +- 4 files changed, 2 insertions(+), 9 deletions(-) diff --git a/apps/sim/background/cleanup-file-versions.ts b/apps/sim/background/cleanup-file-versions.ts index e98d907bc1a..0564806c50d 100644 --- a/apps/sim/background/cleanup-file-versions.ts +++ b/apps/sim/background/cleanup-file-versions.ts @@ -188,12 +188,7 @@ export async function runCleanupFileVersions(payload: CleanupJobPayload): Promis export const cleanupFileVersionsTask = task({ id: 'cleanup-file-versions', - /** - * Each pass works in bounded batches of version rows, so peak memory stays well under - * `small-2x` (1 GB) and does not grow with backlog. - */ - machine: 'small-2x', - maxDuration: 600, + machine: 'large-1x', queue: retentionCleanupQueue, retry: { maxAttempts: 1 }, run: runCleanupFileVersions, diff --git a/apps/sim/background/lifecycle-email.ts b/apps/sim/background/lifecycle-email.ts index b2ffdfbc5ad..a0546b590f4 100644 --- a/apps/sim/background/lifecycle-email.ts +++ b/apps/sim/background/lifecycle-email.ts @@ -59,7 +59,6 @@ async function sendLifecycleEmail({ userId, type }: LifecycleEmailParams): Promi export const lifecycleEmailTask = task({ id: LIFECYCLE_EMAIL_TASK_ID, - maxDuration: 120, queue: { concurrencyLimit: 10 }, retry: { maxAttempts: 2 }, run: async (params: LifecycleEmailParams) => { diff --git a/apps/sim/background/quickbooks-webhook-ingress.ts b/apps/sim/background/quickbooks-webhook-ingress.ts index c1629fdac6c..7e7436c49ba 100644 --- a/apps/sim/background/quickbooks-webhook-ingress.ts +++ b/apps/sim/background/quickbooks-webhook-ingress.ts @@ -143,7 +143,6 @@ export async function enqueueQuickBooksWebhookIngress( export const quickBooksWebhookIngressTask = task({ id: 'quickbooks-webhook-ingress', machine: 'small-1x', - maxDuration: 300, retry: { maxAttempts: QUICKBOOKS_WEBHOOK_INGRESS_MAX_ATTEMPTS, factor: 2, diff --git a/helm/sim/Chart.yaml b/helm/sim/Chart.yaml index 0ad8c459861..1b0be5399dd 100644 --- a/helm/sim/Chart.yaml +++ b/helm/sim/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: sim description: A Helm chart for Sim - the open-source AI workspace where teams build, deploy, and manage AI agents type: application -version: 1.11.6 +version: 1.11.7 appVersion: "v0.8.26" kubeVersion: ">=1.25.0-0" home: https://sim.ai