diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 96feb39e..40bce045 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -26,7 +26,7 @@ Thank you for your interest in contributing. Please read the [Code of Conduct](C git config core.hooksPath .githooks ``` -See [docs/development.md](docs/development.md) for daemon bootstrap, Login Items, and database paths. +See [docs/Design.md](docs/Design.md) for architecture notes. Local SQLite lives under the Derrick app group; reset with `./scripts/reset-local-state.sh`. ## Build from the command line @@ -48,7 +48,7 @@ Do not commit provisioning profiles (`.mobileprovision`, `.p12`, `.pem`). - Run `./scripts/verify-no-secrets.sh --staged` before committing. - Run `./scripts/build.sh test` when you change build-affecting code. - Keep changes focused; match existing Swift style and module boundaries. -- Update README or ADRs when behavior or architecture changes. +- Update README or `docs/Design.md` when behavior or architecture changes. ## License diff --git a/check_browser_deps.py b/check_browser_deps.py deleted file mode 100644 index bb9af009..00000000 --- a/check_browser_deps.py +++ /dev/null @@ -1,6 +0,0 @@ -import subprocess -try: - result = subprocess.run(["apt-cache", "search", "chromium"], capture_output=True, text=True) - print(result.stdout[:500]) -except Exception as e: - print(str(e)) diff --git a/check_playwright.py b/check_playwright.py deleted file mode 100644 index 5c9a7b84..00000000 --- a/check_playwright.py +++ /dev/null @@ -1,9 +0,0 @@ -import subprocess -try: - # Try to check if chromium is installable in the container - # I will try to update apt cache first - this might take time/fail due to read-only container - print("Checking if chromium is available via apt-cache...") - result = subprocess.run(["apt-cache", "search", "chromium"], capture_output=True, text=True) - print(result.stdout[:500]) -except Exception as e: - print(str(e)) diff --git a/docs/Design.md b/docs/Design.md index 5a9171d9..3d097e29 100644 --- a/docs/Design.md +++ b/docs/Design.md @@ -8,6 +8,21 @@ Derrick does not provide many tools to agents. Instead Derrick provides a small ## Structure This application is Protocol first. All major features must have a Protocol and internally use GoF Design Patterns. No exceptions. The Protocols can be found in the Structure spm. -## Plugins -- Plugins are using the Agent Plugin standard -- Plugins and `script_exec` run in the unified Go worker Docker image (`derrick-worker:go-v1`) +## Guardrail +Guardrail is Derrick's control plane in Structure (`Sources/Guardrail`). + +Flow: **Policy evaluates rules → adapters apply `GuardrailDecision` → chokepoints only call those two.** + +Naming (no exceptions): + +- `Guardrail*` — control-plane types +- `*Evaluating` — rule interpreters (`Request` → `GuardrailDecision`) +- `*Applying` — decision adapters (decision → effect) +- `StoreBacked*Evaluating` — SQLite-backed interpreters in `packages/PolicyRuntime` + +- Workflow starts: `StoreBackedWorkflowStartEvaluating` (`workflow_start`) → `WorkflowStartGuardrailApplying` in `WorkflowRuntimeEngine`. +- MCP tools/effectors: `StoreBackedToolInvocationEvaluating` (`tool_invocation`) → `ToolInvocationGuardrailApplying` in the chat pipeline and MCPService. +- Content: `StoreBackedAssistantContentEvaluating` → `AssistantContentGuardrailApplying`. +- HITL: `GuardrailHITLPresenting` (shared); adapters take a presenter, chokepoints do not switch on decisions. +- `WorkflowKind.pluginFactoryEdit` is denied by a Policy rule until editability ships. +- Plugins propose work; they do not authorize control outcomes. diff --git a/fetch_ms.py b/fetch_ms.py deleted file mode 100644 index 95156d5f..00000000 --- a/fetch_ms.py +++ /dev/null @@ -1,16 +0,0 @@ -import requests, json -from bs4 import BeautifulSoup -headers = { - "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0.0.0 Safari/537.36", - "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8", - "Accept-Language": "en-US,en;q=0.9", - "Referer": "https://www.google.com/" -} -try: - r = requests.get("https://microsoft.com", timeout=20, headers=headers) - r.raise_for_status() - soup = BeautifulSoup(r.text, "lxml") - out = {"url": r.url, "status_code": r.status_code, "title": soup.title.string.strip() if soup.title and soup.title.string else ""} - print(json.dumps(out)) -except Exception as e: - print(str(e)) diff --git a/master-todo.md b/master-todo.md index affeecf7..8564906d 100644 --- a/master-todo.md +++ b/master-todo.md @@ -118,8 +118,7 @@ In-use lease TTL (default 7 minutes) stays as the anti-hoard cap. No idle TTL (n ### Docs -- [ ] Sweep `docs/`: drop or mark stale ADRs, fix Python-vs-Swift guest, file extractor vs native reads, container recreate-on-handoff, messaging roadmap items that already shipped. -- [ ] Files to revisit: `docs/adr-swift-script-runtime.md`, `docs/development.md`, `docs/messaging-design.md` remaining table, `docs/opensource-plan.md`, `readme.md` if it still implies one-shot containers only. +- [x] Dropped stale ADRs / plan docs; `docs/Design.md` is the remaining architecture note. README/CONTRIBUTING no longer link to deleted files. ### Startup: crawler image build blocks the app diff --git a/packages/DerrickBackend/Package.swift b/packages/DerrickBackend/Package.swift index e38d7ba7..6b7fea31 100644 --- a/packages/DerrickBackend/Package.swift +++ b/packages/DerrickBackend/Package.swift @@ -15,6 +15,7 @@ let package = Package( .package(path: "../DBRepository"), .package(path: "../DockerRunnerXPC"), .package(path: "../Plugin"), + .package(path: "../PolicyRuntime"), ], targets: [ .target( @@ -24,6 +25,7 @@ let package = Package( "DBRepository", "DockerRunnerXPC", "Plugin", + "PolicyRuntime", ], path: "Sources/DerrickBackend", swiftSettings: [ @@ -32,7 +34,7 @@ let package = Package( ), .testTarget( name: "DerrickBackendTests", - dependencies: ["DerrickBackend", "DBRepository", "Plugin", "Structure"], + dependencies: ["DerrickBackend", "DBRepository", "Plugin", "Structure", "PolicyRuntime"], path: "Tests/DerrickBackendTests", swiftSettings: [ .enableUpcomingFeature("ApproachableConcurrency") diff --git a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift index c622c88c..3334a510 100644 --- a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift +++ b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift @@ -1,5 +1,6 @@ import DBRepository import Foundation +import PolicyRuntime import Structure /// Durable workflow coordinator (Process Manager) running inside derrickd. @@ -16,6 +17,25 @@ public actor WorkflowRuntimeEngine { repositoryProvider: @escaping @Sendable () async throws -> DBRepository ) async throws -> WorkflowHandleDTO { let repo = try await repositoryProvider() + try await DefaultGuardrailPolicySeeds.seedWorkflowStartRulesIfNeeded( + store: repo, + applicationName: DerrickAppSupport.defaultApplicationName + ) + let decision = try await StoreBackedWorkflowStartEvaluating( + store: repo, + applicationName: DerrickAppSupport.defaultApplicationName + ).evaluate(request) + let hitl = ClosureGuardrailHITLPresenting { [self] presentation in + let approved = await awaitWorkflowStartHITL( + request: request, + repository: repo, + hitl: presentation.hitl + ) + return approved + ? .approved(editedPayloadJSON: nil, actor: nil) + : .cancelled(actor: nil) + } + try await WorkflowStartGuardrailApplying(hitl: hitl).apply(decision, for: request) let idempotencyKey = WorkflowRuntimeIdempotency.key( sessionID: request.sessionID, kind: request.kind, @@ -336,4 +356,57 @@ public actor WorkflowRuntimeEngine { message: message ) } + + private static let workflowHITLPollNanoseconds: UInt64 = 1_000_000_000 + private static let workflowHITLTimeoutNanoseconds: UInt64 = 15 * 60 * 1_000_000_000 + + /// Present HITL for a workflow_start confirm decision; returns true when approved. + private func awaitWorkflowStartHITL( + request: WorkflowStartRequest, + repository: DBRepository, + hitl: GuardrailHITLRequest + ) async -> Bool { + let approvalID = UUID().uuidString + let requiredJSON = (try? JSONEncoder().encode(hitl.requiredFields)) + .flatMap { String(data: $0, encoding: .utf8) } ?? "[]" + let isJob: Bool = { + if case .job = request.principal { return true } + return false + }() + let row = PendingHITLApprovalRow( + id: approvalID, + turnID: request.turnID ?? request.sessionID, + sessionID: request.sessionID, + toolName: "workflow_start:\(request.kind.rawValue)", + argumentsJSON: request.inputJSON, + requiredFieldsJSON: requiredJSON, + isJobContext: isJob + ) + do { + try await repository.insertPendingHITLApproval(row) + } catch { + fputs("[workflow] HITL persist failed: \(error.localizedDescription)\n", stderr) + return false + } + DerrickHITLNotificationSignal.postPoll() + + let deadline = Date().addingTimeInterval( + Double(Self.workflowHITLTimeoutNanoseconds) / 1_000_000_000 + ) + while Date() < deadline { + if Task.isCancelled { return false } + if let decision = try? await repository.fetchPendingHITLApproval(id: approvalID), + decision.status != .pending { + return decision.status == .approved + } + try? await Task.sleep(nanoseconds: Self.workflowHITLPollNanoseconds) + } + try? await repository.resolveHITLApproval( + id: approvalID, + status: .timeout, + editedArgumentsJSON: nil, + actor: "system-timeout" + ) + return false + } } diff --git a/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift b/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift index 85382f6b..c0b8338f 100644 --- a/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift +++ b/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift @@ -123,66 +123,70 @@ import Testing ) let repository = DBRepository(configuration: configuration) _ = try await repository.createEmptyDatabaseIfNeeded(username: "app-user", password: "app-secret") + try await DefaultGuardrailPolicySeeds.seedWorkflowStartRulesIfNeeded( + store: repository, + applicationName: "ui" + ) let previousMCP = InProcessServiceBridges.mcpCallTool defer { InProcessServiceBridges.mcpCallTool = previousMCP } - InProcessServiceBridges.mcpCallTool = { request in - switch request.toolName { - case "web.crawl": - let pages: String - if request.argumentsJSON.contains("agent-plugins.org") { - pages = #"{"pages":[{"url":"https://agent-plugins.org/specification","title":"Spec","text":"plugin.json and skills/SKILL.md are required."}]}"# - } else { - pages = #"{"pages":[{"url":"https://api.slack.com/docs","title":"Slack","text":"auth"}]}"# + // Hold the first run in `running` until after the second start (dedupe only matches running). + actor ReleaseGate { + private var released = false + private var waiters: [CheckedContinuation] = [] + + func wait() async { + if released { return } + await withCheckedContinuation { (c: CheckedContinuation) in + waiters.append(c) } - let outcome = try ToolExecutionOutcome.completed( - output: ToolExecutionOutcome.Output(format: .json, value: pages) - ).encodedJSON() - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: true, - isError: false, - text: outcome - ) - case "plugin_factory_build": - let receipt = """ - {"plugin_id":"slack-connector","version":"1.0.0","content_hash":"abc","review_summary":"ok","secrets":[]} - """ - let outcome = try ToolExecutionOutcome.completed( - output: ToolExecutionOutcome.Output(format: .json, value: receipt) - ).encodedJSON() - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: true, - isError: false, - text: outcome - ) - default: - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: false, - isError: true, - text: "", - message: "unexpected tool \(request.toolName)" - ) } + + func release() { + released = true + for waiter in waiters { + waiter.resume() + } + waiters.removeAll() + } + } + let gate = ReleaseGate() + + InProcessServiceBridges.mcpCallTool = { request in + await gate.wait() + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: "held for dedupe test" + ) } + let inputJSON = try PluginFactoryCreateInput.makeConnector( + vendor: .slack, + scope: .fullSync, + userDescription: "Post alerts." + ).encodedJSON() let request = WorkflowStartRequest( kind: .pluginFactoryCreate, - sessionID: "session-1", + sessionID: "session-dedupe", agentID: "ui", - inputJSON: "slack connector", - principal: .agent(sessionID: "session-1", agentID: "ui") + inputJSON: inputJSON, + principal: .agent(sessionID: "session-dedupe", agentID: "ui") ) let provider: @Sendable () async throws -> DBRepository = { repository } let first = try await WorkflowRuntimeEngine.shared.startWorkflow(request, repositoryProvider: provider) + // Give the run task a turn to reach the gated MCP call while still running. + try await Task.sleep(nanoseconds: 50_000_000) let second = try await WorkflowRuntimeEngine.shared.startWorkflow(request, repositoryProvider: provider) #expect(first.deduplicated == false) #expect(second.deduplicated == true) #expect(first.workflowID == second.workflowID) + await gate.release() + var status = WorkflowRunStatus.running for _ in 0..<50 { try await Task.sleep(nanoseconds: 100_000_000) diff --git a/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift b/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift deleted file mode 100644 index 8f766309..00000000 --- a/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift +++ /dev/null @@ -1,53 +0,0 @@ -import Foundation -import Structure - -public struct DefaultPolicyInterceptor: PolicyInterceptor { - private let policy: PolicyEvaluator? - - public init(policy: PolicyEvaluator? = nil) { - self.policy = policy - } - - public func interceptAssistantChunk(_ event: AssistantChunkEvent) async throws -> AssistantContentInterceptResult { - guard let policy else { return .allowed(event.content) } - - let outcome = try await policy.evaluateAssistantChunk(event) - switch outcome { - case .allow: - return .allowed(event.content) - case .deny(let reason): - return .denied(reason: reason) - case .redact(let pattern, let replacement): - let redacted = event.content.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - return .allowed(redacted) - case .confirm: - // Streaming chunks: never modal mid-token. Completion path enforces confirm. - return .allowed(event.content) - } - } - - public func interceptAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> AssistantContentInterceptResult { - guard let policy else { return .allowed(event.fullCompletion) } - - let outcome = try await policy.evaluateAssistantCompletion(event) - switch outcome { - case .allow: - return .allowed(event.fullCompletion) - case .deny(let reason): - return .denied(reason: reason) - case .redact(let pattern, let replacement): - let redacted = event.fullCompletion.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - return .allowed(redacted) - case .confirm(let requiredFields): - return .confirm(content: event.fullCompletion, requiredFields: requiredFields) - } - } -} diff --git a/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift b/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift deleted file mode 100644 index 2043b5a7..00000000 --- a/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift +++ /dev/null @@ -1,87 +0,0 @@ -import Foundation -import Structure - -public struct DefaultToolRequestInterceptor: ToolRequestInterceptor { - private let policy: ToolGovernancePolicy? - - public init(policy: ToolGovernancePolicy? = nil) { - self.policy = policy - } - - public func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInterceptionDecision { - guard let policy else { - return .allow(event) - } - - let outcome = try await policy.evaluateToolInvocation(event) - switch outcome { - case .allow: - return .allow(event) - case .deny(let reason): - return .deny(reason: reason) - case .confirm(let requiredFields): - return .confirm(event, requiredFields: requiredFields) - case .redact(let key, let pattern, let replacement): - guard let parsedArgs = try? JSONSerialization.jsonObject(with: event.argumentsJSON.data(using: .utf8) ?? Data(), options: []) as? [String: Any] else { - return .allow(event) - } - - var mutableArgs = parsedArgs - if let stringValue = mutableArgs[key] as? String { - mutableArgs[key] = stringValue.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - } - - guard let redactedJSON = try? JSONSerialization.data(withJSONObject: mutableArgs, options: [.sortedKeys]), - let redactedString = String(data: redactedJSON, encoding: .utf8) else { - return .allow(event) - } - - return .allow( - ToolInvocationEvent( - sessionID: event.sessionID, - toolName: event.toolName, - argumentsJSON: redactedString, - timestamp: event.timestamp - ) - ) - } - } - - public func interceptToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInvocationEvent? { - switch try await evaluateToolInvocation(event) { - case .allow(let processedEvent): - return processedEvent - case .deny: - return nil - case .confirm(let processedEvent, _): - return processedEvent - } - } - - public func interceptAndRun( - _ event: ToolInvocationEvent, - confirm: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent, [String]) async throws -> ToolInvocationConfirmation, - proceed: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent) async throws -> R - ) async throws -> R { - switch try await evaluateToolInvocation(event) { - case .allow(let allowedEvent): - return try await proceed(allowedEvent) - case .deny(let reason): - throw ToolInvocationInterceptionError.denied(reason: reason) - case .confirm(let confirmEvent, let requiredFields): - switch try await confirm(confirmEvent, requiredFields) { - case .approved(let approvedEvent): - return try await proceed(approvedEvent) - case .cancelled(let actor): - let suffix = actor.map { " by \($0)" } ?? "" - throw ToolInvocationInterceptionError.cancelled( - reason: "User cancelled the approval request\(suffix)" - ) - } - } - } -} diff --git a/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift b/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift deleted file mode 100644 index 903058fb..00000000 --- a/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift +++ /dev/null @@ -1,234 +0,0 @@ -import XCTest -import Structure -@testable import MemorySystem - -final class PolicyInterceptionTests: XCTestCase { - func test_assistantChunkEvent_creation() { - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Hello, " - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.sessionID, "session-1") - XCTAssertEqual(event.chunkIndex, 0) - XCTAssertEqual(event.content, "Hello, ") - } - - func test_assistantCompletionEvent_creation() { - let event = AssistantCompletionEvent( - sessionID: "session-1", - fullCompletion: "Hello, world!", - chunkCount: 2 - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.fullCompletion, "Hello, world!") - XCTAssertEqual(event.chunkCount, 2) - } - - func test_toolInvocationEvent_creation() { - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/etc/config.txt"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.toolName, "read_file") - XCTAssertTrue(event.argumentsJSON.contains("path")) - } - - func test_toolResultEvent_creation() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: #"{"content": "config data"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.toolName, "read_file") - XCTAssertNil(event.error) - } - - func test_statusUpdateEvent_creation() { - let event = StatusUpdateEvent( - sessionID: "session-1", - message: "Processing..." - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.message, "Processing...") - } - - func test_policyInterceptionEvent_timestamp() { - let now = Date() - let chunkEvent = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test", - timestamp: now - ) - let event = PolicyInterceptionEvent.assistantChunk(chunkEvent) - - XCTAssertEqual(event.timestamp, now) - } - - func test_defaultPolicyInterceptor_allowsContent() async throws { - let interceptor = DefaultPolicyInterceptor() - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Safe content" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("Safe content")) - } - - func test_defaultPolicyInterceptor_withoutPolicy_passesThrough() async throws { - let interceptor = DefaultPolicyInterceptor(policy: nil) - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Any content" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("Any content")) - } - - func test_events_are_hashable() { - let event1 = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test" - ) - let event2 = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test" - ) - - XCTAssertNotEqual(event1, event2) - } - - func test_events_are_sendable() async { - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test" - ) - - let task = Task { - return event.content - } - - let result = await task.value - XCTAssertEqual(result, "test") - } - - func test_toolResultEvent_withError() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: "{}", - error: "File not found" - ) - - XCTAssertEqual(event.error, "File not found") - } -} - -struct MockResponseContentPolicy: PolicyEvaluator { - var shouldAllow = true - var shouldDeny = false - var denyReason = "Policy rejected" - - func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> PolicyDecisionOutcome { - if shouldDeny { - return .deny(reason: denyReason) - } - return shouldAllow ? .allow : .confirm(requiredFields: []) - } - - func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome { - if shouldDeny { - return .deny(reason: denyReason) - } - return shouldAllow ? .allow : .confirm(requiredFields: []) - } -} - -final class PolicyInterceptorTests: XCTestCase { - func test_interceptor_allows_with_policy() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = true - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Hello" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("Hello")) - } - - func test_interceptor_denies_with_policy() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = false - policy.shouldDeny = true - policy.denyReason = "Policy violation" - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Unsafe content" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .denied(reason: "Policy violation")) - } - - func test_interceptor_redacts_content() async throws { - let interceptor = DefaultPolicyInterceptor() - let event = AssistantCompletionEvent( - sessionID: "session-1", - fullCompletion: "Email is test@example.com here", - chunkCount: 1 - ) - - let result = try await interceptor.interceptAssistantCompletion(event) - XCTAssertEqual(result, .allowed("Email is test@example.com here")) - } - - func test_interceptor_completion_confirm_does_not_silently_allow() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = false - policy.shouldDeny = false - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantCompletionEvent( - sessionID: "session-1", - fullCompletion: "Contact me at a@b.co", - chunkCount: 1 - ) - let result = try await interceptor.interceptAssistantCompletion(event) - XCTAssertEqual(result, .confirm(content: "Contact me at a@b.co", requiredFields: [])) - } - - func test_interceptor_chunk_confirm_still_soft_allows() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = false - policy.shouldDeny = false - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantChunkEvent(sessionID: "s", chunkIndex: 0, content: "partial") - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("partial")) - } -} diff --git a/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift b/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift deleted file mode 100644 index eec6f240..00000000 --- a/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift +++ /dev/null @@ -1,409 +0,0 @@ -import XCTest -import Structure -@testable import MemorySystem - -final class ToolRequestInterceptionTests: XCTestCase { - func test_toolInvocationEvent_creation() { - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/etc/config.txt"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.sessionID, "session-1") - XCTAssertEqual(event.toolName, "read_file") - XCTAssertTrue(event.argumentsJSON.contains("path")) - } - - func test_toolResultEvent_creation() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: #"{"content": "file content"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.toolName, "read_file") - XCTAssertNil(event.error) - } - - func test_toolResultEvent_with_error() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: "{}", - error: "File not found" - ) - - XCTAssertEqual(event.error, "File not found") - } - - func test_toolInvocationEvent_timestamp() { - let now = Date() - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}", - timestamp: now - ) - - XCTAssertEqual(event.timestamp, now) - } - - func test_toolResultEvent_timestamp() { - let now = Date() - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: "{}", - timestamp: now - ) - - XCTAssertEqual(event.timestamp, now) - } - - func test_toolInvocationEvent_sendable() async { - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - - let task = Task { - return event.toolName - } - - let result = await task.value - XCTAssertEqual(result, "read_file") - } - - func test_toolRequestEvent_hashable() { - let event1 = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - let event2 = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - - XCTAssertNotEqual(event1, event2) - } -} - -struct MockToolGovernancePolicy: ToolGovernancePolicy { - var shouldAllow = true - var shouldDeny = false - var denyReason = "Tool not allowed" - - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - if shouldDeny { - return .deny(reason: denyReason) - } - return shouldAllow ? .allow : .confirm(requiredFields: ["approval"]) - } -} - -final class ToolRequestInterceptorTests: XCTestCase { - func test_defaultInterceptor_allows_tool() async throws { - var policy = MockToolGovernancePolicy() - policy.shouldAllow = true - - let interceptor = DefaultToolRequestInterceptor(policy: policy) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/etc/config.txt"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNotNil(result) - XCTAssertEqual(result?.toolName, "read_file") - } - - func test_defaultInterceptor_denies_tool() async throws { - var policy = MockToolGovernancePolicy() - policy.shouldAllow = false - policy.shouldDeny = true - policy.denyReason = "Tool is restricted" - - let interceptor = DefaultToolRequestInterceptor(policy: policy) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "shell_exec", - argumentsJSON: #"{"cmd": "rm -rf /"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNil(result) - } - - func test_defaultInterceptor_without_policy() async throws { - let interceptor = DefaultToolRequestInterceptor(policy: nil) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/home/user/file.txt"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertEqual(result?.toolName, "read_file") - XCTAssertEqual(result?.argumentsJSON, event.argumentsJSON) - } - - func test_defaultInterceptor_redacts_arguments() async throws { - struct RedactingPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - return .redact(argumentKey: "password", pattern: ".+", replacement: "[REDACTED]") - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: RedactingPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "authenticate", - argumentsJSON: #"{"username": "admin", "password": "secret123"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNotNil(result) - XCTAssertTrue(result!.argumentsJSON.contains("[REDACTED]")) - XCTAssertFalse(result!.argumentsJSON.contains("secret123")) - } - - func test_interceptionEvent_in_enum() { - let invocationEvent = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - let event = PolicyInterceptionEvent.toolInvocation(invocationEvent) - - XCTAssertEqual(event.timestamp, invocationEvent.timestamp) - } - - func test_toolResult_interceptionEvent() { - let resultEvent = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: #"{"content": "data"}"# - ) - let event = PolicyInterceptionEvent.toolResult(resultEvent) - - XCTAssertEqual(event.timestamp, resultEvent.timestamp) - } - - func test_confirm_outcome_preserves_tool_name() async throws { - struct ConfirmingPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - return .confirm(requiredFields: ["user_consent"]) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmingPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{"path": "/tmp/file.txt"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNotNil(result) - XCTAssertEqual(result?.toolName, "delete_file") - } - - func test_evaluateToolInvocation_returns_confirm_with_requiredFields() async throws { - struct ConfirmingPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirm(requiredFields: ["user_consent", "ticket_id"]) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmingPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{"path": "/tmp/file.txt"}"# - ) - - let decision = try await interceptor.evaluateToolInvocation(event) - switch decision { - case .confirm(let confirmedEvent, let requiredFields): - XCTAssertEqual(confirmedEvent.toolName, "delete_file") - XCTAssertEqual(requiredFields, ["user_consent", "ticket_id"]) - default: - XCTFail("Expected confirm decision") - } - } - - // MARK: - interceptAndRun (item 4: confirm-before-proceed) - - private final class CallProbe: @unchecked Sendable { - var confirmCalled = false - var proceedCalled = false - var proceedTool: String? - var seenFields: [String] = [] - } - - func test_interceptAndRun_allow_calls_proceed_only() async throws { - struct AllowPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .allow - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: AllowPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{}"# - ) - - let probe = CallProbe() - let result = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in - probe.confirmCalled = true - return .cancelled(actor: nil) - }, - proceed: { gated in - probe.proceedTool = gated.toolName - return "ok" - } - ) - - XCTAssertEqual(result, "ok") - XCTAssertFalse(probe.confirmCalled) - XCTAssertEqual(probe.proceedTool, "read_file") - } - - func test_interceptAndRun_deny_throws_without_proceed() async throws { - struct DenyPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .deny(reason: "blocked by policy") - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: DenyPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "shell_exec", - argumentsJSON: #"{}"# - ) - - let probe = CallProbe() - do { - _ = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in .cancelled(actor: nil) }, - proceed: { _ -> String in - probe.proceedCalled = true - return "should not run" - } - ) - XCTFail("Expected deny error") - } catch ToolInvocationInterceptionError.denied(let reason) { - XCTAssertEqual(reason, "blocked by policy") - } - XCTAssertFalse(probe.proceedCalled) - } - - func test_interceptAndRun_confirm_approved_then_proceed() async throws { - struct ConfirmPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirm(requiredFields: ["user_approval"]) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{"path":"/tmp/a"}"# - ) - - let probe = CallProbe() - let result = try await interceptor.interceptAndRun( - event, - confirm: { confirmEvent, fields in - probe.seenFields = fields - return .approved( - ToolInvocationEvent( - sessionID: confirmEvent.sessionID, - toolName: confirmEvent.toolName, - argumentsJSON: #"{"path":"/tmp/edited"}"#, - timestamp: confirmEvent.timestamp - ) - ) - }, - proceed: { gated in - XCTAssertEqual(gated.argumentsJSON, #"{"path":"/tmp/edited"}"#) - return gated.argumentsJSON - } - ) - - XCTAssertEqual(probe.seenFields, ["user_approval"]) - XCTAssertEqual(result, #"{"path":"/tmp/edited"}"#) - } - - func test_interceptAndRun_confirm_cancelled_throws_without_proceed() async throws { - struct ConfirmPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirm(requiredFields: ["user_approval"]) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{}"# - ) - - let probe = CallProbe() - do { - _ = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in .cancelled(actor: "tester") }, - proceed: { _ -> String in - probe.proceedCalled = true - return "no" - } - ) - XCTFail("Expected cancelled error") - } catch ToolInvocationInterceptionError.cancelled(let reason) { - XCTAssertTrue(reason.contains("cancelled")) - XCTAssertTrue(reason.contains("tester")) - } - XCTAssertFalse(probe.proceedCalled) - } - - func test_interceptAndRun_redact_then_proceed_with_redacted_event() async throws { - struct RedactPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .redact(argumentKey: "token", pattern: ".+", replacement: "[REDACTED]") - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: RedactPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "auth", - argumentsJSON: #"{"token":"secret"}"# - ) - - let result = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in .cancelled(actor: nil) }, - proceed: { gated in - gated.argumentsJSON - } - ) - - XCTAssertTrue(result.contains("[REDACTED]")) - XCTAssertFalse(result.contains("secret")) - } -} diff --git a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedGuardrailEvaluating.swift similarity index 89% rename from packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift rename to packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedGuardrailEvaluating.swift index 33846c0b..d4d11bd7 100644 --- a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift +++ b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedGuardrailEvaluating.swift @@ -1,9 +1,8 @@ import Foundation -import MemorySystem import Structure /// Store-backed tool governance: loads rules from `PolicyStore`, matches, returns first enabled hit. -public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { +public struct StoreBackedToolInvocationEvaluating: GuardrailEvaluating { private let store: any PolicyStore private let applicationName: String @@ -12,19 +11,19 @@ public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { self.applicationName = applicationName } - public func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - let rules = try await loadRules(scopes: ["tool_invocation", "tool_call"]) + public func evaluate(_ request: ToolInvocationEvent) async throws -> GuardrailDecision { + let rules = try await loadRules(scopes: GuardrailPolicyScope.toolInvocationScopes.map(\.rawValue)) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) } - let argumentsObject = parseJSONObject(from: event.argumentsJSON) + let argumentsObject = parseJSONObject(from: request.argumentsJSON) for rule in rules { guard rule.enabled else { continue } guard let matcher = try? decode(ToolMatcher.self, from: rule.matcherJSON) else { continue } - guard matcher.matches(event: event, arguments: argumentsObject) else { + guard matcher.matches(event: request, arguments: argumentsObject) else { continue } guard let outcome = try? decode(OutcomeRule.self, from: rule.outcomeJSON) else { @@ -57,7 +56,7 @@ public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { } /// Store-backed assistant content policy for chunks and full completions. -public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { +public struct StoreBackedAssistantContentEvaluating: Sendable { private let store: any PolicyStore private let applicationName: String @@ -66,8 +65,8 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { self.applicationName = applicationName } - public func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> PolicyDecisionOutcome { - let rules = try await loadRules(scopes: ["assistant_chunk"]) + public func evaluate(_ request: AssistantChunkEvent) async throws -> GuardrailDecision { + let rules = try await loadRules(scopes: [GuardrailPolicyScope.assistantChunk.rawValue]) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) } @@ -77,7 +76,7 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { guard let matcher = try? decode(ContentMatcher.self, from: rule.matcherJSON) else { continue } - guard matcher.matches(content: event.content) else { + guard matcher.matches(content: request.content) else { continue } guard let outcome = try? decode(OutcomeRule.self, from: rule.outcomeJSON) else { @@ -89,19 +88,19 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { return .deny(reason: Self.noMatchingRuleReason) } - public func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome { - let rules = try await loadRules(scopes: ["assistant_completion_content", "assistant_completion"]) + public func evaluate(_ request: AssistantCompletionEvent) async throws -> GuardrailDecision { + let rules = try await loadRules(scopes: GuardrailPolicyScope.assistantCompletionScopes.map(\.rawValue)) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) } - let detectedPatterns = detectSensitivePatterns(in: event.fullCompletion) + let detectedPatterns = detectSensitivePatterns(in: request.fullCompletion) for rule in rules { guard rule.enabled else { continue } guard let matcher = try? decode(CompletionMatcher.self, from: rule.matcherJSON) else { continue } - guard matcher.matches(content: event.fullCompletion, detectedPatterns: detectedPatterns) else { + guard matcher.matches(content: request.fullCompletion, detectedPatterns: detectedPatterns) else { continue } guard let outcome = try? decode(OutcomeRule.self, from: rule.outcomeJSON) else { @@ -415,30 +414,38 @@ private struct OutcomeRule: Decodable { case replacement } - var toolOutcome: ToolGovernanceOutcome { + var toolOutcome: GuardrailDecision { switch action.lowercased() { case "deny": return .deny(reason: reason ?? "Tool invocation denied by policy.") case "confirm": - return .confirm(requiredFields: requiredFields ?? ["user_approval"]) + return .confirmHITL( + GuardrailHITLRequest(requiredFields: requiredFields ?? ["user_approval"]) + ) case "allow": return .allow case "redact": guard let argumentKey, let pattern else { return .deny(reason: "Invalid redact outcome for tool rule (missing argument_key/pattern).") } - return .redact(argumentKey: argumentKey, pattern: pattern, replacement: replacement ?? "[REDACTED]") + return .redactArgument( + argumentKey: argumentKey, + pattern: pattern, + replacement: replacement ?? "[REDACTED]" + ) default: return .deny(reason: "Unknown tool policy action '\(action)'; denying by default.") } } - func contentOutcome(fallbackPattern: String?) -> PolicyDecisionOutcome { + func contentOutcome(fallbackPattern: String?) -> GuardrailDecision { switch action.lowercased() { case "deny": return .deny(reason: reason ?? "Assistant content denied by policy.") case "confirm": - return .confirm(requiredFields: requiredFields ?? ["review_confirmation"]) + return .confirmHITL( + GuardrailHITLRequest(requiredFields: requiredFields ?? ["review_confirmation"]) + ) case "allow": return .allow case "redact": @@ -446,13 +453,14 @@ private struct OutcomeRule: Decodable { guard let patternToUse else { return .deny(reason: "Invalid redact outcome for content rule (missing pattern).") } - return .redact(pattern: patternToUse, replacement: replacement ?? "[REDACTED]") + return .redactContent(pattern: patternToUse, replacement: replacement ?? "[REDACTED]") default: return .deny(reason: "Unknown content policy action '\(action)'; denying by default.") } } } + private func parseJSONObject(from json: String) -> [String: Any] { guard let data = json.data(using: .utf8), let object = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { diff --git a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowStartEvaluating.swift b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowStartEvaluating.swift new file mode 100644 index 00000000..747e5aaa --- /dev/null +++ b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowStartEvaluating.swift @@ -0,0 +1,118 @@ +import Foundation +import Structure + +/// Store-backed `workflow_start` interpreter. Implements `GuardrailEvaluating`. +public struct StoreBackedWorkflowStartEvaluating: GuardrailEvaluating { + private let store: any PolicyStore + private let applicationName: String + + public init(store: any PolicyStore, applicationName: String) { + self.store = store + self.applicationName = applicationName + } + + public func evaluate(_ request: WorkflowStartRequest) async throws -> GuardrailDecision { + let rules = try await loadRules() + guard !rules.isEmpty else { + return .deny(reason: Self.noRulesConfiguredReason) + } + + for rule in rules { + guard rule.enabled else { continue } + guard let matcher = try? decode(WorkflowMatcher.self, from: rule.matcherJSON) else { + continue + } + guard matcher.matches(kind: request.kind) else { + continue + } + guard let outcome = try? decode(WorkflowOutcomeRule.self, from: rule.outcomeJSON) else { + continue + } + return outcome.decision + } + + return .deny(reason: Self.noMatchingRuleReason) + } + + public static let noRulesConfiguredReason = + "No workflow_start rules are configured; denying by default." + + public static let noMatchingRuleReason = + "No workflow_start rule matched this kind; denying by default." + + private func loadRules() async throws -> [PolicyRule] { + try await store.loadRules(applicationName: applicationName, scope: "workflow_start") + .filter(\.enabled) + .sorted { lhs, rhs in + if lhs.priority != rhs.priority { return lhs.priority > rhs.priority } + return lhs.createdAt > rhs.createdAt + } + } + + private func decode(_ type: T.Type, from json: String) throws -> T { + guard let data = json.data(using: .utf8) else { + throw DecodingError.dataCorrupted( + .init(codingPath: [], debugDescription: "Invalid UTF-8 in policy JSON.") + ) + } + return try JSONDecoder().decode(T.self, from: data) + } +} + +private struct WorkflowMatcher: Decodable { + let workflowKind: String? + let workflowKindAny: [String]? + + enum CodingKeys: String, CodingKey { + case workflowKind = "workflow_kind" + case workflowKindAny = "workflow_kind_any" + } + + func matches(kind: WorkflowKind) -> Bool { + if let workflowKind, kind.rawValue != workflowKind { + return false + } + if let workflowKindAny { + guard workflowKindAny.contains(kind.rawValue) else { return false } + } + return true + } +} + +private struct WorkflowOutcomeRule: Decodable { + let action: String + let reason: String? + let requiredFields: [String]? + let title: String? + let message: String? + + enum CodingKeys: String, CodingKey { + case action + case reason + case requiredFields = "required_fields" + case title + case message + } + + var decision: GuardrailDecision { + switch action.lowercased() { + case "allow": + return .allow + case "deny": + return .deny(reason: reason ?? "Workflow start denied by policy.") + case "confirm": + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: requiredFields ?? ["user_approval"], + title: title, + message: message + ) + ) + case "require_workflow": + // Already starting a workflow; treat as deny. + return .deny(reason: reason ?? "require_workflow is not valid for workflow_start.") + default: + return .deny(reason: "Unknown workflow_start action '\(action)'; denying by default.") + } + } +} diff --git a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedGuardrailEvaluatingTests.swift similarity index 82% rename from packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift rename to packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedGuardrailEvaluatingTests.swift index 2fbc2869..dbe0353b 100644 --- a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift +++ b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedGuardrailEvaluatingTests.swift @@ -3,7 +3,7 @@ import MemorySystem @testable import PolicyRuntime import Structure -final class StoreBackedPolicyEvaluatorsTests: XCTestCase { +final class StoreBackedGuardrailEvaluatingTests: XCTestCase { func test_toolPolicy_loadsRelevantScopeAndDeniesMatch() async throws { let store = MockPolicyStore(rulesByScope: [ "tool_invocation": [ @@ -16,13 +16,13 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") let event = ToolInvocationEvent( sessionID: "s1", toolName: "delete_file", argumentsJSON: #"{"path":"/tmp/a"}"# ) - let outcome = try await policy.evaluateToolInvocation(event) + let outcome = try await policy.evaluate(event) XCTAssertEqual(outcome, .deny(reason: "blocked")) } @@ -39,8 +39,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent( sessionID: "factory-1", toolName: "tool_search", @@ -73,8 +73,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) XCTAssertEqual(outcome, .allow) @@ -82,11 +82,11 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { func test_toolPolicy_deniesWhenNoRulesConfigured() async throws { let store = MockPolicyStore(rulesByScope: [:]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) - XCTAssertEqual(outcome, .deny(reason: StoreBackedToolGovernancePolicy.noRulesConfiguredReason)) + XCTAssertEqual(outcome, .deny(reason: StoreBackedToolInvocationEvaluating.noRulesConfiguredReason)) } func test_toolPolicy_deniesWhenNoRuleMatches() async throws { @@ -102,11 +102,11 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) - XCTAssertEqual(outcome, .deny(reason: StoreBackedToolGovernancePolicy.noMatchingRuleReason)) + XCTAssertEqual(outcome, .deny(reason: StoreBackedToolInvocationEvaluating.noMatchingRuleReason)) } func test_toolPolicy_prefersHigherPriorityAcrossScopes() async throws { @@ -132,8 +132,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "file_write", argumentsJSON: "{}") ) XCTAssertEqual(outcome, .deny(reason: "high priority")) @@ -151,15 +151,15 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedCompletionContentPolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateAssistantCompletion( + let policy = StoreBackedAssistantContentEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "Please email me at hi@example.com", chunkCount: 1 ) ) - XCTAssertEqual(outcome, .confirm(requiredFields: ["review"])) + XCTAssertEqual(outcome, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) } func test_completionPolicy_skipsDisabledRule() async throws { @@ -185,8 +185,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedCompletionContentPolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateAssistantCompletion( + let policy = StoreBackedAssistantContentEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "hi@example.com", @@ -218,18 +218,18 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let withCode = try await policy.evaluateToolInvocation( + let withCode = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", argumentsJSON: #"{"code":"print(1)"}"# ) ) - XCTAssertEqual(withCode, .confirm(requiredFields: ["review"])) + XCTAssertEqual(withCode, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) - let withoutCode = try await policy.evaluateToolInvocation( + let withoutCode = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", @@ -238,7 +238,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(withoutCode, .allow) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "read_file", @@ -268,19 +268,19 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let fileWrite = try await policy.evaluateToolInvocation( + let fileWrite = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "file_write", argumentsJSON: "{}") ) XCTAssertEqual(fileWrite, .deny(reason: "mutating tool")) - let deleteFile = try await policy.evaluateToolInvocation( + let deleteFile = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "delete_file", argumentsJSON: "{}") ) XCTAssertEqual(deleteFile, .deny(reason: "mutating tool")) - let readFile = try await policy.evaluateToolInvocation( + let readFile = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "read_file", argumentsJSON: "{}") ) XCTAssertEqual(readFile, .allow) @@ -306,14 +306,14 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let sessionTool = try await policy.evaluateToolInvocation( + let sessionTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "session_memory_search", argumentsJSON: "{}") ) XCTAssertEqual(sessionTool, .allow) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) XCTAssertEqual(otherTool, .deny(reason: "only session memory allowed")) @@ -346,18 +346,18 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let offlineScript = try await policy.evaluateToolInvocation( + let offlineScript = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", argumentsJSON: #"{"code":"x"}"# ) ) - XCTAssertEqual(offlineScript, .confirm(requiredFields: ["offline_ok"])) + XCTAssertEqual(offlineScript, .confirmHITL(GuardrailHITLRequest(requiredFields: ["offline_ok"]))) - let networkedScript = try await policy.evaluateToolInvocation( + let networkedScript = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", @@ -366,7 +366,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(networkedScript, .allow) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "read_file", argumentsJSON: "{}") ) XCTAssertEqual(otherTool, .allow) @@ -392,13 +392,13 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let sessionTool = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let sessionTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "session_memory_search", argumentsJSON: "{}") ) - XCTAssertEqual(sessionTool, .confirm(requiredFields: ["ok"])) + XCTAssertEqual(sessionTool, .confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"]))) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) XCTAssertEqual(otherTool, .allow) @@ -424,17 +424,17 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedCompletionContentPolicy(store: store, applicationName: "ui") - let ssn = try await policy.evaluateAssistantCompletion( + let policy = StoreBackedAssistantContentEvaluating(store: store, applicationName: "ui") + let ssn = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "SSN 123-45-6789", chunkCount: 1 ) ) - XCTAssertEqual(ssn, .confirm(requiredFields: ["review"])) + XCTAssertEqual(ssn, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) - let plain = try await policy.evaluateAssistantCompletion( + let plain = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "hello world", diff --git a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowStartEvaluatingTests.swift b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowStartEvaluatingTests.swift new file mode 100644 index 00000000..d38c1c0f --- /dev/null +++ b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowStartEvaluatingTests.swift @@ -0,0 +1,90 @@ +import XCTest +import Structure +@testable import PolicyRuntime + +final class StoreBackedWorkflowStartEvaluatingTests: XCTestCase { + func test_allowsSeededCreateAndDeniesEdit() async throws { + let store = MockWorkflowPolicyStore() + for rule in DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: "ui") { + try await store.saveRule(rule) + } + let policy = StoreBackedWorkflowStartEvaluating(store: store, applicationName: "ui") + + let create = try await policy.evaluate( + WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + ) + XCTAssertEqual(create, .allow) + + let edit = try await policy.evaluate( + WorkflowStartRequest( + kind: .pluginFactoryEdit, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + ) + guard case .deny = edit else { + return XCTFail("edit placeholder must be denied by Policy rule") + } + } + + func test_confirmOutcomeReturnsHITL() async throws { + let store = MockWorkflowPolicyStore(rulesByScope: [ + "workflow_start": [ + PolicyRule( + applicationName: "ui", + name: "confirm-create", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"plugin_factory_create"}"#, + outcomeJSON: #"{"action":"confirm","required_fields":["review"],"title":"Approve create"}"# + ) + ] + ]) + let policy = StoreBackedWorkflowStartEvaluating(store: store, applicationName: "ui") + let decision = try await policy.evaluate( + WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + ) + guard case .confirmHITL(let hitl) = decision else { + return XCTFail("expected confirmHITL") + } + XCTAssertEqual(hitl.requiredFields, ["review"]) + XCTAssertEqual(hitl.title, "Approve create") + } +} + +private final class MockWorkflowPolicyStore: PolicyStore, @unchecked Sendable { + private var rulesByScope: [String: [PolicyRule]] + + init(rulesByScope: [String: [PolicyRule]] = [:]) { + self.rulesByScope = rulesByScope + } + + func loadRules(applicationName: String, scope: String) async throws -> [PolicyRule] { + rulesByScope[scope] ?? [] + } + + func saveRule(_ rule: PolicyRule) async throws { + var rules = rulesByScope[rule.scope] ?? [] + rules.removeAll { $0.name == rule.name } + rules.append(rule) + rulesByScope[rule.scope] = rules + } + + func saveApproval(_ approval: PolicyApproval) async throws {} + func loadApprovals(sessionID: String, limit: Int) async throws -> [PolicyApproval] { [] } + func logAuditEntry(_ entry: PolicyAuditLogEntry) async throws {} + func auditLog(sessionID: String, limit: Int, page: Int) async throws -> [PolicyAuditLogEntry] { [] } +} diff --git a/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift b/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift index bccbeecd..e1cb2b73 100644 --- a/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift +++ b/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift @@ -10,6 +10,7 @@ public enum AppLayerFeature: String, CaseIterable, Sendable { case scriptReview case mcpToolCatalog case conversationMCPBridge + case guardrail } public struct AppLayerFeatureContract: Sendable, Hashable { @@ -78,6 +79,23 @@ public enum AppLayerFeatureCatalog { persistenceMayBeSQLite: false, uiMustNotImportMCPServer: false ), + AppLayerFeatureContract( + feature: .guardrail, + structureTypeNames: [ + "Guardrail", + "GuardrailDecision", + "GuardrailEvaluating", + "GuardrailApplying", + "GuardrailHITLPresenting", + "WorkflowKind", + "WorkflowStartGuardrailApplying", + "ToolInvocationGuardrailApplying", + "AssistantContentGuardrailApplying", + "PendingHITLApprovalRow", + ], + persistenceMayBeSQLite: true, + uiMustNotImportMCPServer: true + ), ] public static func contract(for feature: AppLayerFeature) -> AppLayerFeatureContract { diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/EffectorAdmissionPolicy.swift b/packages/Structure/Sources/AppLayerServices/MCPService/EffectorAdmissionPolicy.swift deleted file mode 100644 index 5604b30b..00000000 --- a/packages/Structure/Sources/AppLayerServices/MCPService/EffectorAdmissionPolicy.swift +++ /dev/null @@ -1,27 +0,0 @@ -import Foundation - -/// MCP effector admission from signed `ExecutionContextWire` (not process-local flags). -public enum EffectorAdmissionPolicy: Sendable { - public static func allowsSyncWebCrawl( - context: ExecutionContextWire?, - principal: ServicePrincipal - ) -> Bool { - switch principal { - case .job, .agent: - return true - default: - break - } - if let context, context.capabilities.contains(.syncWebCrawl) { return true } - if let context, context.workflow?.kind == .pluginFactoryCreate { return true } - return false - } - - public static func parseContextJSON(_ json: String?) -> ExecutionContextWire? { - guard let json, - !json.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { - return nil - } - return try? ExecutionContextWire.decodeJSON(json) - } -} diff --git a/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift b/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift index 9257a43d..f67dedfe 100644 --- a/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift +++ b/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift @@ -11,15 +11,6 @@ public enum ExecutionContextCapability: String, Codable, Sendable, Hashable, Cas case hostReviewRetry = "host_review_retry" } -public enum WorkflowKind: String, Codable, Sendable, Hashable, CaseIterable { - case pluginFactoryCreate = "plugin_factory_create" - case pluginFactoryEdit = "plugin_factory_edit" - case connectorAuthDiscover = "connector_auth_discover" - case jobStep = "job_step" - case interactiveTool = "interactive_tool" - case none -} - public struct WorkflowContextWire: Codable, Sendable, Hashable { public let workflowID: String? public let kind: WorkflowKind @@ -105,6 +96,15 @@ public struct ExecutionContextWire: Codable, Sendable, Hashable { return try JSONDecoder.service.decode(ExecutionContextWire.self, from: data) } + /// Best-effort decode used at MCP / effector boundaries when context may be absent. + public static func parseOptionalJSON(_ json: String?) -> ExecutionContextWire? { + guard let json, + !json.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { + return nil + } + return try? decodeJSON(json) + } + public func withWorkflow( workflowID: String, kind: WorkflowKind, diff --git a/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift b/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift index a5346556..c4678ae8 100644 --- a/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift +++ b/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift @@ -37,6 +37,7 @@ public enum WorkflowRuntimeError: Error, LocalizedError, Sendable { case mcpUnavailable case workflowNotFound case unsupportedKind(WorkflowKind) + case deniedByGuardrail(String) public var errorDescription: String? { switch self { @@ -46,6 +47,8 @@ public enum WorkflowRuntimeError: Error, LocalizedError, Sendable { return "Workflow was not found." case .unsupportedKind(let kind): return "Unsupported workflow kind \(kind.rawValue)." + case .deniedByGuardrail(let reason): + return reason } } } diff --git a/packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift new file mode 100644 index 00000000..526cd78b --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift @@ -0,0 +1,51 @@ +import Foundation + +/// Applies a `GuardrailDecision` for assistant chunk / completion content. +public struct AssistantContentGuardrailApplying: Sendable { + public init() {} + + /// Streaming chunks: never modal mid-token. Confirm is deferred to completion. + public func apply( + _ decision: GuardrailDecision, + for chunk: AssistantChunkEvent + ) -> AssistantContentGuardrailOutput { + switch decision { + case .allow: + return .allowed(chunk.content) + case .deny(let reason): + return .denied(reason: reason) + case .redactContent(let pattern, let replacement): + let redacted = chunk.content.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + return .allowed(redacted) + case .confirmHITL, .requireWorkflow, .redactArgument: + return .allowed(chunk.content) + } + } + + public func apply( + _ decision: GuardrailDecision, + for completion: AssistantCompletionEvent + ) -> AssistantContentGuardrailOutput { + switch decision { + case .allow: + return .allowed(completion.fullCompletion) + case .deny(let reason): + return .denied(reason: reason) + case .redactContent(let pattern, let replacement): + let redacted = completion.fullCompletion.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + return .allowed(redacted) + case .confirmHITL(let hitl): + return .confirm(content: completion.fullCompletion, requiredFields: hitl.requiredFields) + case .requireWorkflow, .redactArgument: + return .denied(reason: "Content policy returned an unsupported control decision.") + } + } +} diff --git a/packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift new file mode 100644 index 00000000..6251703f --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift @@ -0,0 +1,8 @@ +import Foundation + +/// Applies a `GuardrailDecision` for one request kind. +public protocol GuardrailApplying: Sendable { + associatedtype Request: Sendable + associatedtype Output: Sendable + func apply(_ decision: GuardrailDecision, for request: Request) async throws -> Output +} diff --git a/packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift new file mode 100644 index 00000000..9b6dfe9b --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift @@ -0,0 +1,97 @@ +import Foundation + +/// Applies a `GuardrailDecision` for tool invocations. +public struct ToolInvocationGuardrailApplying: GuardrailApplying { + public typealias Request = ToolInvocationEvent + public typealias Output = ToolInvocationEvent + + private let hitl: (any GuardrailHITLPresenting)? + + public init(hitl: (any GuardrailHITLPresenting)? = nil) { + self.hitl = hitl + } + + public func apply( + _ decision: GuardrailDecision, + for request: ToolInvocationEvent + ) async throws -> ToolInvocationEvent { + switch decision { + case .allow: + return request + case .deny(let reason): + throw ToolInvocationGuardrailError.denied(reason: reason) + case .confirmHITL(let hitlRequest): + guard let hitl else { + throw ToolInvocationGuardrailError.denied( + reason: "Tool requires HITL approval but no presenter is configured." + ) + } + let presentation = GuardrailHITLPresentation( + sessionID: request.sessionID, + turnID: request.sessionID, + subject: request.toolName, + payloadJSON: request.argumentsJSON, + hitl: hitlRequest + ) + switch await hitl.resolve(presentation) { + case .approved(let editedPayloadJSON, _): + guard let editedPayloadJSON else { return request } + return ToolInvocationEvent( + sessionID: request.sessionID, + toolName: request.toolName, + argumentsJSON: editedPayloadJSON, + timestamp: request.timestamp + ) + case .cancelled(let actor): + let suffix = actor.map { " by \($0)" } ?? "" + throw ToolInvocationGuardrailError.cancelled( + reason: "User cancelled the approval request\(suffix)" + ) + } + case .redactArgument(let key, let pattern, let replacement): + return ToolInvocationEvent( + sessionID: request.sessionID, + toolName: request.toolName, + argumentsJSON: redactArgumentJSON( + request.argumentsJSON, + key: key, + pattern: pattern, + replacement: replacement + ), + timestamp: request.timestamp + ) + case .requireWorkflow(let kind): + throw ToolInvocationGuardrailError.denied( + reason: "Tool requires workflow '\(kind.rawValue)' before it can run." + ) + case .redactContent: + throw ToolInvocationGuardrailError.denied( + reason: "Tool policy returned an unsupported control decision." + ) + } + } +} + +private func redactArgumentJSON( + _ json: String, + key: String, + pattern: String, + replacement: String +) -> String { + guard let data = json.data(using: .utf8), + var object = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { + return json + } + if let stringValue = object[key] as? String { + object[key] = stringValue.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + } + guard let redacted = try? JSONSerialization.data(withJSONObject: object, options: [.sortedKeys]), + let redactedString = String(data: redacted, encoding: .utf8) else { + return json + } + return redactedString +} diff --git a/packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift new file mode 100644 index 00000000..4fd855a3 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift @@ -0,0 +1,50 @@ +import Foundation + +/// Applies a `GuardrailDecision` for workflow starts. +public struct WorkflowStartGuardrailApplying: GuardrailApplying { + public typealias Request = WorkflowStartRequest + public typealias Output = Void + + private let hitl: any GuardrailHITLPresenting + + public init(hitl: any GuardrailHITLPresenting) { + self.hitl = hitl + } + + public func apply(_ decision: GuardrailDecision, for request: WorkflowStartRequest) async throws { + switch decision { + case .allow: + return + case .deny(let reason): + throw WorkflowRuntimeError.deniedByGuardrail(reason) + case .confirmHITL(let hitlRequest): + let isJob: Bool = { + if case .job = request.principal { return true } + return false + }() + let presentation = GuardrailHITLPresentation( + sessionID: request.sessionID, + turnID: request.turnID ?? request.sessionID, + subject: "workflow_start:\(request.kind.rawValue)", + payloadJSON: request.inputJSON, + hitl: hitlRequest, + isJobContext: isJob + ) + switch await hitl.resolve(presentation) { + case .approved: + return + case .cancelled: + let detail = hitlRequest.message ?? hitlRequest.title ?? "Workflow start was not approved." + throw WorkflowRuntimeError.deniedByGuardrail(detail) + } + case .requireWorkflow: + throw WorkflowRuntimeError.deniedByGuardrail( + "requireWorkflow is not a valid decision for workflow start." + ) + case .redactContent, .redactArgument: + throw WorkflowRuntimeError.deniedByGuardrail( + "Redaction is not a valid decision for workflow start." + ) + } + } +} diff --git a/packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift b/packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift new file mode 100644 index 00000000..4a82fe70 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift @@ -0,0 +1,7 @@ +import Foundation + +/// Interprets a Policy rule set for one request kind into a `GuardrailDecision`. +public protocol GuardrailEvaluating: Sendable { + associatedtype Request: Sendable + func evaluate(_ request: Request) async throws -> GuardrailDecision +} diff --git a/packages/Structure/Sources/Guardrail/Guardrail.swift b/packages/Structure/Sources/Guardrail/Guardrail.swift new file mode 100644 index 00000000..e2b026cb --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Guardrail.swift @@ -0,0 +1,12 @@ +import Foundation + +/// Derrick's agent-control umbrella in Structure. +/// +/// Flow: **Policy evaluates rules → adapters apply `GuardrailDecision` → chokepoints only call those two.** +/// +/// Naming (no exceptions): +/// - `Guardrail*` — control-plane types +/// - `*Evaluating` — rule interpreters (`Request` → `GuardrailDecision`) +/// - `*Applying` — decision adapters (decision → effect) +/// - `StoreBacked*Evaluating` — SQLite-backed interpreters in PolicyRuntime +public enum Guardrail: Sendable {} diff --git a/packages/Structure/Sources/Guardrail/GuardrailDecision.swift b/packages/Structure/Sources/Guardrail/GuardrailDecision.swift new file mode 100644 index 00000000..ee5df253 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/GuardrailDecision.swift @@ -0,0 +1,33 @@ +import Foundation + +/// Canonical control-plane decision. Every `*Evaluating` type returns this; every `*Applying` type consumes it. +public enum GuardrailDecision: Hashable, Sendable { + case allow + case deny(reason: String) + case confirmHITL(GuardrailHITLRequest) + case requireWorkflow(WorkflowKind) + case redactContent(pattern: String, replacement: String) + case redactArgument(argumentKey: String, pattern: String, replacement: String) + + public var isAllow: Bool { + if case .allow = self { return true } + return false + } +} + +/// HITL details for a `confirmHITL` decision. +public struct GuardrailHITLRequest: Hashable, Sendable { + public let requiredFields: [String] + public let title: String? + public let message: String? + + public init( + requiredFields: [String] = ["user_approval"], + title: String? = nil, + message: String? = nil + ) { + self.requiredFields = requiredFields + self.title = title + self.message = message + } +} diff --git a/packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLApprovalPresentationWake.swift b/packages/Structure/Sources/Guardrail/HITL/DerrickHITLApprovalPresentationWake.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLApprovalPresentationWake.swift rename to packages/Structure/Sources/Guardrail/HITL/DerrickHITLApprovalPresentationWake.swift diff --git a/packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLNotificationSignal.swift b/packages/Structure/Sources/Guardrail/HITL/DerrickHITLNotificationSignal.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLNotificationSignal.swift rename to packages/Structure/Sources/Guardrail/HITL/DerrickHITLNotificationSignal.swift diff --git a/packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift b/packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift new file mode 100644 index 00000000..2b6a9465 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift @@ -0,0 +1,40 @@ +import Foundation + +/// Context for presenting a Policy `confirmHITL` decision. +public struct GuardrailHITLPresentation: Sendable, Hashable { + public let id: String + public let sessionID: String + public let turnID: String + public let subject: String + public let payloadJSON: String + public let hitl: GuardrailHITLRequest + public let isJobContext: Bool + + public init( + id: String = UUID().uuidString, + sessionID: String, + turnID: String, + subject: String, + payloadJSON: String, + hitl: GuardrailHITLRequest, + isJobContext: Bool = false + ) { + self.id = id + self.sessionID = sessionID + self.turnID = turnID + self.subject = subject + self.payloadJSON = payloadJSON + self.hitl = hitl + self.isJobContext = isJobContext + } +} + +public enum GuardrailHITLResolution: Equatable, Sendable { + case approved(editedPayloadJSON: String?, actor: String?) + case cancelled(actor: String?) +} + +/// Presents HITL and returns the human resolution. One implementation shared by all adapters. +public protocol GuardrailHITLPresenting: Sendable { + func resolve(_ presentation: GuardrailHITLPresentation) async -> GuardrailHITLResolution +} diff --git a/packages/Structure/Sources/DBRepository/HITLTypes.swift b/packages/Structure/Sources/Guardrail/HITL/HITLTypes.swift similarity index 100% rename from packages/Structure/Sources/DBRepository/HITLTypes.swift rename to packages/Structure/Sources/Guardrail/HITL/HITLTypes.swift diff --git a/packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift b/packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift new file mode 100644 index 00000000..60408353 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift @@ -0,0 +1,36 @@ +import Foundation + +/// Test / non-interactive HITL presenter that returns a fixed resolution. +public struct ImmediateGuardrailHITLPresenting: GuardrailHITLPresenting { + private let resolution: GuardrailHITLResolution + + public init(approved: Bool) { + self.resolution = approved + ? .approved(editedPayloadJSON: nil, actor: nil) + : .cancelled(actor: nil) + } + + public init(resolution: GuardrailHITLResolution) { + self.resolution = resolution + } + + public func resolve(_ presentation: GuardrailHITLPresentation) async -> GuardrailHITLResolution { + _ = presentation + return resolution + } +} + +/// Bridges a closure to `GuardrailHITLPresenting`. +public struct ClosureGuardrailHITLPresenting: GuardrailHITLPresenting { + private let handler: @Sendable (GuardrailHITLPresentation) async -> GuardrailHITLResolution + + public init( + _ handler: @escaping @Sendable (GuardrailHITLPresentation) async -> GuardrailHITLResolution + ) { + self.handler = handler + } + + public func resolve(_ presentation: GuardrailHITLPresentation) async -> GuardrailHITLResolution { + await handler(presentation) + } +} diff --git a/packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift b/packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift new file mode 100644 index 00000000..9f2efb4a --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift @@ -0,0 +1,72 @@ +import Foundation + +/// Baseline Guardrail Policy rules Derrick seeds on first launch. +public enum DefaultGuardrailPolicySeeds: Sendable { + /// `workflow_start` rules: allow Derrick-owned kinds; deny edit placeholder and `none`. + public static func workflowStartRules(applicationName: String) -> [PolicyRule] { + [ + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-plugin-factory-create", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"plugin_factory_create"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-connector-auth-discover", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"connector_auth_discover"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-job-step", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"job_step"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-interactive-tool", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"interactive_tool"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "deny-workflow-plugin-factory-edit", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"plugin_factory_edit"}"#, + outcomeJSON: #"{"action":"deny","reason":"Plugin edit workflow is reserved and not enabled yet."}"#, + priority: 1000 + ), + PolicyRule( + applicationName: applicationName, + name: "deny-workflow-none", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"none"}"#, + outcomeJSON: #"{"action":"deny","reason":"A workflow kind is required."}"#, + priority: 1000 + ), + ] + } + + /// Idempotently insert `workflow_start` rules into a Policy store. + public static func seedWorkflowStartRulesIfNeeded( + store: any PolicyStore, + applicationName: String + ) async throws { + for rule in workflowStartRules(applicationName: applicationName) { + let existing = try await store.loadRules(applicationName: applicationName, scope: rule.scope) + guard existing.contains(where: { $0.name == rule.name }) == false else { + continue + } + try await store.saveRule(rule) + } + } +} diff --git a/packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift b/packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift new file mode 100644 index 00000000..30dec3f5 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift @@ -0,0 +1,17 @@ +import Foundation + +/// Policy rule scopes. Interpreters load only their scope(s). +public enum GuardrailPolicyScope: String, Sendable, Hashable, CaseIterable { + case toolInvocation = "tool_invocation" + /// Legacy alias still accepted when loading tool rules. + case toolCall = "tool_call" + case workflowStart = "workflow_start" + case assistantChunk = "assistant_chunk" + case assistantCompletionContent = "assistant_completion_content" + case assistantCompletion = "assistant_completion" + + public static var toolInvocationScopes: [GuardrailPolicyScope] { [.toolInvocation, .toolCall] } + public static var assistantCompletionScopes: [GuardrailPolicyScope] { + [.assistantCompletionContent, .assistantCompletion] + } +} diff --git a/packages/Structure/Sources/Policy/PolicyEngine/Matchers.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Matchers.swift similarity index 93% rename from packages/Structure/Sources/Policy/PolicyEngine/Matchers.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Matchers.swift index 3b599b04..b66a5f4c 100644 --- a/packages/Structure/Sources/Policy/PolicyEngine/Matchers.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Matchers.swift @@ -157,7 +157,7 @@ public struct ToolRule: ToolPolicyRule { Self(matcher: matcher, outcome: .confirm(title: title, message: message)) } - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { guard matcher.matches(request.call) else { return nil } @@ -168,7 +168,13 @@ public struct ToolRule: ToolPolicyRule { case .deny(let reason): return .deny(reason: reason) case .confirm(let title, let message): - return .confirm(.init(title: title, message: message, call: request.call, context: request.context)) + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: ["user_approval"], + title: title, + message: message + ) + ) } } } diff --git a/packages/Structure/Sources/Policy/PolicyEngine/PolicyEngineTypes.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift similarity index 80% rename from packages/Structure/Sources/Policy/PolicyEngine/PolicyEngineTypes.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift index 43c84978..6215f304 100644 --- a/packages/Structure/Sources/Policy/PolicyEngine/PolicyEngineTypes.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift @@ -71,16 +71,19 @@ public struct PolicyConfirmationRequest: Hashable, Sendable { self.call = call self.context = context } -} -public enum PolicyDecision: Hashable, Sendable { - case allow - case deny(reason: String) - case confirm(PolicyConfirmationRequest) + public var guardrailHITL: GuardrailHITLRequest { + GuardrailHITLRequest( + requiredFields: ["user_approval"], + title: title, + message: message + ) + } } +/// In-memory tool rules used by `PolicyEngine`. Prefer store-backed `PolicyRule` for production. public protocol ToolPolicyRule: Sendable { - func evaluate(_ request: PolicyRequest) -> PolicyDecision? + func evaluate(_ request: PolicyRequest) -> GuardrailDecision? } public protocol PolicyConfirmationPresenting: Sendable { @@ -92,6 +95,8 @@ public enum PolicyError: Error, Sendable, Equatable { case cancelled } +/// In-memory rule list that produces `GuardrailDecision`. Production chat uses store-backed evaluators; +/// this engine shares the same decision vocabulary. public struct PolicyEngine: Sendable { public let rules: [any ToolPolicyRule] @@ -99,19 +104,18 @@ public struct PolicyEngine: Sendable { self.rules = rules } - public func decision(for request: PolicyRequest) -> PolicyDecision { + public func decision(for request: PolicyRequest) -> GuardrailDecision { for rule in rules { if let decision = rule.evaluate(request) { return decision } } - return .confirm( - PolicyConfirmationRequest( + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: ["user_approval"], title: "Confirm tool call", - message: "This tool may change state or trigger side effects.", - call: request.call, - context: request.context + message: "This tool may change state or trigger side effects." ) ) } diff --git a/packages/Structure/Sources/Policy/PolicyEngine/Rules.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Rules.swift similarity index 72% rename from packages/Structure/Sources/Policy/PolicyEngine/Rules.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Rules.swift index 8d798952..ce29f82f 100644 --- a/packages/Structure/Sources/Policy/PolicyEngine/Rules.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Rules.swift @@ -7,7 +7,7 @@ public struct AllowToolNamesRule: ToolPolicyRule { self.toolNames = Set(toolNames) } - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { toolNames.contains(request.call.name) ? .allow : nil } } @@ -21,7 +21,7 @@ public struct DenyToolNamesRule: ToolPolicyRule { self.reason = reason } - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { toolNames.contains(request.call.name) ? .deny(reason: reason) : nil } } @@ -29,17 +29,16 @@ public struct DenyToolNamesRule: ToolPolicyRule { public struct ConfirmMutationRule: ToolPolicyRule { public init() {} - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { guard request.call.effects.contains(.changesState) || request.call.effects.contains(.externalSideEffects) else { return nil } - return .confirm( - PolicyConfirmationRequest( + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: ["user_approval"], title: "Confirm action", - message: "This tool call can change state or trigger side effects. Confirm before proceeding.", - call: request.call, - context: request.context + message: "This tool call can change state or trigger side effects. Confirm before proceeding." ) ) } diff --git a/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift new file mode 100644 index 00000000..ce83f8f1 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift @@ -0,0 +1,9 @@ +import Foundation + +/// Result of applying a content `GuardrailDecision` (preserves deny reasons for UI). +public enum AssistantContentGuardrailOutput: Equatable, Sendable { + case allowed(String) + case denied(reason: String) + /// Policy requires user approval before this content is accepted as final. + case confirm(content: String, requiredFields: [String]) +} diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterceptionEvent.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterceptionEvent.swift similarity index 100% rename from packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterceptionEvent.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterceptionEvent.swift diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyStore.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyStore.swift similarity index 100% rename from packages/Structure/Sources/Policy/PolicyRuntime/PolicyStore.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyStore.swift diff --git a/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift new file mode 100644 index 00000000..d415e93f --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift @@ -0,0 +1,7 @@ +import Foundation + +/// Errors thrown while applying a tool `GuardrailDecision`. +public enum ToolInvocationGuardrailError: Error, Equatable, Sendable { + case denied(reason: String) + case cancelled(reason: String) +} diff --git a/packages/Structure/Sources/Policy/PolicyUserInteraction/PolicyUserEvent.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyUserInteraction/PolicyUserEvent.swift similarity index 100% rename from packages/Structure/Sources/Policy/PolicyUserInteraction/PolicyUserEvent.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyUserInteraction/PolicyUserEvent.swift diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/WorkflowChatProgress.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowChatProgress.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/MCPService/WorkflowChatProgress.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowChatProgress.swift diff --git a/packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift new file mode 100644 index 00000000..3047ed3b --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift @@ -0,0 +1,13 @@ +import Foundation + +/// Sequenced procedures the harness can run. Policy may require one; plugins never authorize them. +/// +/// `pluginFactoryEdit` is a reserved placeholder for future plugin editability. +public enum WorkflowKind: String, Codable, Sendable, Hashable, CaseIterable { + case pluginFactoryCreate = "plugin_factory_create" + case pluginFactoryEdit = "plugin_factory_edit" + case connectorAuthDiscover = "connector_auth_discover" + case jobStep = "job_step" + case interactiveTool = "interactive_tool" + case none +} diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeDTOs.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeDTOs.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeDTOs.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeDTOs.swift diff --git a/packages/Structure/Sources/DBRepository/WorkflowRuntimeRows.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeRows.swift similarity index 100% rename from packages/Structure/Sources/DBRepository/WorkflowRuntimeRows.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeRows.swift diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeXPC.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeXPC.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeXPC.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeXPC.swift diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterception.swift b/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterception.swift deleted file mode 100644 index cdb5814e..00000000 --- a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterception.swift +++ /dev/null @@ -1,26 +0,0 @@ -import Foundation - -public enum PolicyDecisionOutcome: Equatable, Sendable { - case allow - case deny(reason: String) - case confirm(requiredFields: [String]) - case redact(pattern: String, replacement: String) -} - -public protocol PolicyEvaluator: Sendable { - func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> PolicyDecisionOutcome - func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome -} - -/// Result of content policy interception (preserves deny reasons for UI). -public enum AssistantContentInterceptResult: Equatable, Sendable { - case allowed(String) - case denied(reason: String) - /// Policy requires user approval before this content is accepted as final. - case confirm(content: String, requiredFields: [String]) -} - -public protocol PolicyInterceptor: Sendable { - func interceptAssistantChunk(_ event: AssistantChunkEvent) async throws -> AssistantContentInterceptResult - func interceptAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> AssistantContentInterceptResult -} diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/ToolInterception.swift b/packages/Structure/Sources/Policy/PolicyRuntime/ToolInterception.swift deleted file mode 100644 index b42c9178..00000000 --- a/packages/Structure/Sources/Policy/PolicyRuntime/ToolInterception.swift +++ /dev/null @@ -1,49 +0,0 @@ -import Foundation - -public enum ToolGovernanceOutcome: Equatable, Sendable { - case allow - case deny(reason: String) - case confirm(requiredFields: [String]) - case redact(argumentKey: String, pattern: String, replacement: String) -} - -public enum ToolInterceptionDecision: Equatable, Sendable { - case allow(ToolInvocationEvent) - case deny(reason: String) - case confirm(ToolInvocationEvent, requiredFields: [String]) -} - -/// Errors thrown by confirm-before-proceed tool interception. -public enum ToolInvocationInterceptionError: Error, Equatable, Sendable { - case denied(reason: String) - case cancelled(reason: String) -} - -/// Result of asking the user to approve a tool that policy marked as `confirm`. -public enum ToolInvocationConfirmation: Equatable, Sendable { - /// Proceed with this (possibly edited) invocation event. - case approved(ToolInvocationEvent) - /// User (or system) cancelled confirmation. - case cancelled(actor: String?) -} - -public protocol ToolGovernancePolicy: Sendable { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome -} - -public protocol ToolRequestInterceptor: Sendable { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInterceptionDecision - func interceptToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInvocationEvent? - - /// Evaluate policy, optionally confirm with the user, then run `proceed` with the gated event. - /// - /// Control flow: - /// - allow / redact → `proceed(processedEvent)` - /// - deny → throws `ToolInvocationInterceptionError.denied` - /// - confirm → `confirm(...)`; on approve → `proceed`; on cancel → throws `.cancelled` - func interceptAndRun( - _ event: ToolInvocationEvent, - confirm: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent, [String]) async throws -> ToolInvocationConfirmation, - proceed: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent) async throws -> R - ) async throws -> R -} diff --git a/packages/Structure/Sources/Structure/Architecture.swift b/packages/Structure/Sources/Structure/Architecture.swift index 0f48f2ba..f2785397 100644 --- a/packages/Structure/Sources/Structure/Architecture.swift +++ b/packages/Structure/Sources/Structure/Architecture.swift @@ -1,3 +1,3 @@ /// Root marker for the Structure architecture module. -/// Domain types live in sibling folders (`AppLayerServices`, `Contract`, `Plugin`, `Policy`, …). +/// Domain types live in sibling folders (`AppLayerServices`, `Contract`, `Plugin`, `Guardrail`, …). public enum Architecture {} diff --git a/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift b/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift index 5297c61f..76c15915 100644 --- a/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift +++ b/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift @@ -1413,37 +1413,17 @@ import Testing #expect(fromMinutes.containerRunMaxTTLMinutes == 12) } - @Test func effectorAdmissionAllowsWorkflowCrawl() { + @Test func executionContextParseOptionalJSON() throws { let context = ExecutionContextWire( - sessionID: "s1", - principal: .agent(sessionID: "s1", agentID: "a1"), - workflow: WorkflowContextWire(workflowID: "w1", kind: .pluginFactoryCreate), - capabilities: [.syncWebCrawl, .hostReviewRetry] - ) - #expect( - EffectorAdmissionPolicy.allowsSyncWebCrawl( - context: context, - principal: .agent(sessionID: "s1", agentID: "a1") - ) - ) - } - - @Test func effectorAdmissionAllowsLiveChatWithoutContext() { - #expect( - EffectorAdmissionPolicy.allowsSyncWebCrawl( - context: nil, - principal: .agent(sessionID: "s1", agentID: "a1") - ) - ) - } - - @Test func effectorAdmissionAllowsJobsWithoutContext() { - #expect( - EffectorAdmissionPolicy.allowsSyncWebCrawl( - context: nil, - principal: .job(jobID: "j1") - ) + sessionID: "s", + principal: .ui, + capabilities: [.syncWebCrawl] ) + let json = try context.encodedJSON() + let parsed = ExecutionContextWire.parseOptionalJSON(json) + #expect(parsed?.capabilities.contains(.syncWebCrawl) == true) + #expect(ExecutionContextWire.parseOptionalJSON(nil) == nil) + #expect(ExecutionContextWire.parseOptionalJSON(" ") == nil) } @Test func executionContextWireRoundTrip() throws { diff --git a/packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift b/packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift new file mode 100644 index 00000000..a0fc0003 --- /dev/null +++ b/packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift @@ -0,0 +1,71 @@ +import Foundation +import Structure +import XCTest + +/// Applying-layer coverage formerly owned by MemorySystem interceptor tests. +final class GuardrailApplyingTests: XCTestCase { + func test_toolApplying_allow_returnsEvent() async throws { + let event = ToolInvocationEvent(sessionID: "s", toolName: "t", argumentsJSON: "{}") + let gated = try await ToolInvocationGuardrailApplying().apply(.allow, for: event) + XCTAssertEqual(gated.toolName, "t") + } + + func test_toolApplying_deny_throws() async { + let event = ToolInvocationEvent(sessionID: "s", toolName: "t", argumentsJSON: "{}") + do { + _ = try await ToolInvocationGuardrailApplying().apply(.deny(reason: "no"), for: event) + XCTFail("expected deny") + } catch let error as ToolInvocationGuardrailError { + guard case .denied(let reason) = error else { + return XCTFail("expected denied") + } + XCTAssertEqual(reason, "no") + } catch { + XCTFail("unexpected \(error)") + } + } + + func test_toolApplying_redactArgument() async throws { + let event = ToolInvocationEvent( + sessionID: "s", + toolName: "t", + argumentsJSON: #"{"token":"abc-secret"}"# + ) + let gated = try await ToolInvocationGuardrailApplying().apply( + .redactArgument(argumentKey: "token", pattern: "secret", replacement: "[x]"), + for: event + ) + XCTAssertTrue(gated.argumentsJSON.contains("[x]")) + XCTAssertFalse(gated.argumentsJSON.contains("secret")) + } + + func test_toolApplying_confirmHITL_approved() async throws { + let event = ToolInvocationEvent(sessionID: "s", toolName: "t", argumentsJSON: #"{"a":1}"#) + let hitl = ImmediateGuardrailHITLPresenting( + resolution: .approved(editedPayloadJSON: #"{"a":2}"#, actor: "user") + ) + let gated = try await ToolInvocationGuardrailApplying(hitl: hitl).apply( + .confirmHITL(GuardrailHITLRequest()), + for: event + ) + XCTAssertEqual(gated.argumentsJSON, #"{"a":2}"#) + } + + func test_contentApplying_chunk_softAllowsConfirm() { + let chunk = AssistantChunkEvent(sessionID: "s", chunkIndex: 0, content: "mid") + let out = AssistantContentGuardrailApplying().apply( + .confirmHITL(GuardrailHITLRequest()), + for: chunk + ) + XCTAssertEqual(out, .allowed("mid")) + } + + func test_contentApplying_completion_confirm() { + let completion = AssistantCompletionEvent(sessionID: "s", fullCompletion: "full", chunkCount: 1) + let out = AssistantContentGuardrailApplying().apply( + .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"])), + for: completion + ) + XCTAssertEqual(out, .confirm(content: "full", requiredFields: ["review"])) + } +} diff --git a/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift new file mode 100644 index 00000000..30679a3d --- /dev/null +++ b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift @@ -0,0 +1,133 @@ +import Testing +@testable import Structure + +@Suite("Guardrail") +struct GuardrailDecisionTests { + @Test func policyEngineSharesGuardrailVocabulary() { + let engine = PolicyEngine(rules: [ + DenyToolNamesRule(toolNames: ["x"], reason: "no") + ]) + let decision = engine.decision(for: .init(call: .init(name: "x"), context: .init(agentID: "a"))) + #expect({ + if case .deny = decision { return true } + return false + }()) + } + + @Test func pluginFactoryEditKindRemainsPlaceholder() { + #expect(WorkflowKind.pluginFactoryEdit.rawValue == "plugin_factory_edit") + #expect(WorkflowKind.allCases.contains(.pluginFactoryEdit)) + } + + @Test func workflowStartApplyingAllowAndDeny() async throws { + let applying = WorkflowStartGuardrailApplying(hitl: ImmediateGuardrailHITLPresenting(approved: true)) + let request = WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + try await applying.apply(.allow, for: request) + do { + try await applying.apply(.deny(reason: "blocked"), for: request) + Issue.record("deny must throw") + } catch let error as WorkflowRuntimeError { + if case .deniedByGuardrail(let reason) = error { + #expect(reason == "blocked") + } else { + Issue.record("Expected deniedByGuardrail") + } + } + } + + @Test func workflowStartApplyingConfirmHITL() async throws { + let request = WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + try await WorkflowStartGuardrailApplying(hitl: ImmediateGuardrailHITLPresenting(approved: true)) + .apply(.confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"])), for: request) + + do { + try await WorkflowStartGuardrailApplying(hitl: ImmediateGuardrailHITLPresenting(approved: false)) + .apply(.confirmHITL(GuardrailHITLRequest(title: "Need approval")), for: request) + Issue.record("cancelled HITL must throw") + } catch let error as WorkflowRuntimeError { + if case .deniedByGuardrail = error { + #expect(true) + } else { + Issue.record("Expected deniedByGuardrail") + } + } + } + + @Test func toolInvocationApplyingAllowDenyRedact() async throws { + let applying = ToolInvocationGuardrailApplying() + let event = ToolInvocationEvent( + sessionID: "s", + toolName: "script_exec", + argumentsJSON: #"{"code":"secret-token"}"# + ) + let allowed = try await applying.apply(.allow, for: event) + #expect(allowed.toolName == "script_exec") + + do { + _ = try await applying.apply(.deny(reason: "nope"), for: event) + Issue.record("deny must throw") + } catch let error as ToolInvocationGuardrailError { + if case .denied(let reason) = error { + #expect(reason == "nope") + } else { + Issue.record("Expected denied") + } + } + + let redacted = try await applying.apply( + .redactArgument(argumentKey: "code", pattern: "secret", replacement: "[REDACTED]"), + for: event + ) + #expect(redacted.argumentsJSON.contains("[REDACTED]")) + #expect(!redacted.argumentsJSON.contains("secret-token")) + } + + @Test func assistantContentApplyingChunkAndCompletion() { + let applying = AssistantContentGuardrailApplying() + let chunk = AssistantChunkEvent(sessionID: "s", chunkIndex: 0, content: "hello secret") + let chunkOut = applying.apply( + .redactContent(pattern: "secret", replacement: "[x]"), + for: chunk + ) + #expect(chunkOut == .allowed("hello [x]")) + + let completion = AssistantCompletionEvent(sessionID: "s", fullCompletion: "done", chunkCount: 1) + let confirmOut = applying.apply( + .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"])), + for: completion + ) + #expect(confirmOut == .confirm(content: "done", requiredFields: ["review"])) + } + + @Test func defaultWorkflowSeedsCoverCreateAndDenyEdit() { + let rules = DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: "ui") + #expect(rules.contains { $0.name == "allow-workflow-plugin-factory-create" }) + #expect(rules.contains { $0.name == "deny-workflow-plugin-factory-edit" }) + #expect(rules.allSatisfy { $0.scope == "workflow_start" }) + } + + @Test func executionContextParseOptionalJSON() throws { + let context = ExecutionContextWire( + sessionID: "s", + principal: .ui, + capabilities: [.syncWebCrawl] + ) + let json = try context.encodedJSON() + let parsed = ExecutionContextWire.parseOptionalJSON(json) + #expect(parsed?.capabilities.contains(.syncWebCrawl) == true) + #expect(ExecutionContextWire.parseOptionalJSON(nil) == nil) + #expect(ExecutionContextWire.parseOptionalJSON(" ") == nil) + } +} diff --git a/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift b/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift index 2f90f878..346cbbcc 100644 --- a/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift +++ b/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift @@ -25,8 +25,8 @@ struct PolicyEngineTests { let decision = engine.decision(for: .init(call: .init(name: "writeFile", effects: [.changesState]), context: .init(agentID: "a"))) #expect({ - if case .confirm(let request) = decision { - return request.title == "Confirm action" + if case .confirmHITL(let hitl) = decision { + return hitl.title == "Confirm action" } return false }()) diff --git a/readme.md b/readme.md index c642a268..55b50fc1 100644 --- a/readme.md +++ b/readme.md @@ -45,7 +45,7 @@ A native Swift macOS desktop agent: chat, just-in-time software (plugins), messa - **Docker** runs untrusted Go for `script_exec`, just-in-time plugin builds, and approved plugin invocations. - **HostUI** (`packages/HostUI`) paints plugin screens from a declared schema. The host does not invent a default inbox. -See [docs/adr-headless-backend.md](docs/adr-headless-backend.md) and [docs/services-plan.md](docs/services-plan.md). +See [docs/Design.md](docs/Design.md) for the current architecture notes. ## Security model @@ -58,7 +58,6 @@ Derrick treats model output and guest code as untrusted. - Canonical I/O types live in `packages/Structure/Sources/Contract/Resources/schemas/` and are mirrored to `workers/go/internal/contract/schemas/`. - No net/http, subprocess, filesystem access, or credentials inside the guest. - The host dispatches `http.request` envelopes, attaches secrets, and enforces egress policy. -- Historical Swift guest notes: [docs/adr-swift-script-runtime.md](docs/adr-swift-script-runtime.md). ### Script review agent @@ -95,15 +94,16 @@ Copy [.env.example](.env.example) — **never commit `.env`**. - Plugin credential collection (Keychain save) - Policy events (usage limits, content sensitivity) -### Policy +### Guardrail -- **`PolicyEngine`** — in-process tool-call rules for chat (`Structure/Policy/PolicyEngine`). -- **`PolicyRuntime`** — persisted rules from SQLite (`Structure/Policy/PolicyRuntime`); implemented by `StoreBacked*` evaluators in `packages/PolicyRuntime`. -- **`PolicyInterceptor`** / **`ToolRequestInterceptor`** (MemorySystem) — pipeline hooks that call those policies. +- **`Structure/Guardrail`** — control plane. Policy evaluates; adapters apply `GuardrailDecision`; chokepoints only call those two. +- Naming: `Guardrail*`, `*Evaluating`, `*Applying`, `StoreBacked*Evaluating`. +- Persisted rules in SQLite (`Guardrail/Policy`, `packages/PolicyRuntime`) produce `GuardrailDecision`. +- Workflow starts and MCP tools are admitted by Policy rules (`workflow_start` / `tool_invocation`); thin `*GuardrailApplying` adapters apply allow / deny / confirmHITL / redact. ### Messaging -Connectors are just-in-time software with `role: connector`. They must emit a HostUI tree (`ui.present`) — copy `messaging_inbox` or compose their own. Messages live in SQLite; sync and send still go through the guest. See [docs/messaging-design.md](docs/messaging-design.md). +Connectors are just-in-time software with `role: connector`. They must emit a HostUI tree (`ui.present`) — copy `messaging_inbox` or compose their own. Messages live in SQLite; sync and send still go through the guest. ## Repository layout @@ -112,12 +112,12 @@ Connectors are just-in-time software with `role: connector`. They must emit a Ho | `ui/` | macOS app, Login Item daemon, XPC services | | `packages/DBRepository` | SQLite schema, migrations, messaging tables | | `packages/MCPServer` | MCP bridge, script execution, plugin runtime | -| `packages/Structure` | Architecture map: wire types, protocols, JSON schemas (`AppLayerServices/`, `Policy/`, `Plugin/`, `Contract/`, …) | +| `packages/Structure` | Architecture map: wire types, protocols, JSON schemas (`AppLayerServices/`, `Guardrail/`, `Plugin/`, `Contract/`, …) | | `packages/HostUI` | Schema-driven screens for connectors and plugin present trees | | `packages/Plugin` | Just-in-time software factory (builder, reviewer, release) | | `packages/DerrickBackend` | Daemon runtime, notifications, HITL polling | | `packages/DockerRunnerXPC` | Constrained Docker helper | -| `packages/PolicyRuntime` | Store-backed policy evaluators (`Structure/Policy/PolicyRuntime`) | +| `packages/PolicyRuntime` | `StoreBacked*Evaluating` interpreters (`Structure/Guardrail`) | | `packages/LLMAgentClient` | Provider clients (OpenAI, Gemini, …) | ## Quick start @@ -141,11 +141,11 @@ Connectors are just-in-time software with `role: connector`. They must emit a Ho Or from the terminal: `./scripts/build.sh test` -See [CONTRIBUTING.md](CONTRIBUTING.md) and [docs/development.md](docs/development.md). +See [CONTRIBUTING.md](CONTRIBUTING.md) and [docs/Design.md](docs/Design.md). ## Open-sourcing checklist -We maintain [docs/opensource-plan.md](docs/opensource-plan.md) for pre-release cleanup (secret audit, personal reference removal, CI). +Pre-release cleanup: secret audit (`./scripts/verify-no-secrets.sh`), personal reference removal, and CI hygiene. Verify no secrets in git: @@ -155,10 +155,7 @@ Verify no secrets in git: ## Documentation -- [Headless backend ADR](docs/adr-headless-backend.md) -- [Swift Docker runtime ADR](docs/adr-swift-script-runtime.md) -- [Background services plan](docs/services-plan.md) -- [Messaging design](docs/messaging-design.md) +- [Design](docs/Design.md) - [Security policy](SECURITY.md) - [Third-party notices](THIRD_PARTY_NOTICES.md) - [Code of Conduct](CODE_OF_CONDUCT.md) diff --git a/scripts/reset-local-state.sh b/scripts/reset-local-state.sh index fb845c65..6c0c22df 100755 --- a/scripts/reset-local-state.sh +++ b/scripts/reset-local-state.sh @@ -2,23 +2,32 @@ set -euo pipefail ROOT="$(cd "$(dirname "$0")/.." && pwd)" -APP_GROUP="$HOME/Library/Group Containers/VUSK4B2YKQ.derrick.shared" + +# Same search order as DerrickAppSupport.preferredDatabaseParentDirectories(). +DB_ROOTS=( + "$HOME/Library/Group Containers/VUSK4B2YKQ.derrick.shared" + "$HOME/Library/Containers/derrick.ui/Data/Library/Application Support" + "$HOME/Library/Application Support" +) echo "==> Quit Derrick before resetting local state." echo "==> This removes all local SQLite data (chats, plugins, messaging, credentials in DB)." echo "==> Keychain plugin secrets are not removed." removed_dbs=0 -if [[ -d "$APP_GROUP" ]]; then +for root in "${DB_ROOTS[@]}"; do + if [[ ! -d "$root" ]]; then + continue + fi while IFS= read -r db; do rm -f "$db" "${db}-wal" "${db}-shm" echo "removed $(basename "$db") at ${db%/*}" removed_dbs=$((removed_dbs + 1)) - done < <(find "$APP_GROUP" -name 'derrick.sqlite3' 2>/dev/null) -fi + done < <(find "$root" -name 'derrick.sqlite3' 2>/dev/null) +done if [[ "$removed_dbs" -eq 0 ]]; then - echo "no derrick.sqlite3 files found under $APP_GROUP" + echo "no derrick.sqlite3 files found under preferred database roots" fi if command -v docker >/dev/null 2>&1; then @@ -44,5 +53,5 @@ else fi echo -echo "Done. Reopen Derrick from $ROOT (go-workers) to recreate an empty database." +echo "Done. Reopen Derrick from $ROOT to recreate an empty database." echo "Policy rules seed automatically on first UI launch." diff --git a/todo.rtf b/todo.rtf deleted file mode 100644 index e8e4ba7c..00000000 --- a/todo.rtf +++ /dev/null @@ -1,8 +0,0 @@ -{\rtf1\ansi\ansicpg1252\cocoartf2870 -\cocoatextscaling0\cocoaplatform0{\fonttbl\f0\fswiss\fcharset0 Helvetica;} -{\colortbl;\red255\green255\blue255;} -{\*\expandedcolortbl;;} -\margl1440\margr1440\vieww11520\viewh8400\viewkind0 -\pard\tx720\tx1440\tx2160\tx2880\tx3600\tx4320\tx5040\tx5760\tx6480\tx7200\tx7920\tx8640\pardirnatural\partightenfactor0 - -\f0\fs24 \cf0 - create a content side panel that shows a todo or table of contents for long ongoing discussions} \ No newline at end of file diff --git a/ui/MCPService/MCPServiceStore.swift b/ui/MCPService/MCPServiceStore.swift index 24eace28..a56d3998 100644 --- a/ui/MCPService/MCPServiceStore.swift +++ b/ui/MCPService/MCPServiceStore.swift @@ -7,9 +7,16 @@ actor MCPServiceStore { static let shared = MCPServiceStore() private var repository: DBRepository? + private var didSeedPolicy = false func sharedRepository() async throws -> DBRepository { - if let repository { return repository } + if let repository { + if !didSeedPolicy { + try await seedBaselinePolicyIfNeeded(repository) + didSeedPolicy = true + } + return repository + } let directory = try DerrickAppSupport.databaseDirectory() let repo = DBRepository( configuration: DBRepositoryConfiguration( @@ -21,6 +28,8 @@ actor MCPServiceStore { ) ) _ = try await repo.createEmptyDatabaseIfNeeded(username: "ui", password: "ui") + try await seedBaselinePolicyIfNeeded(repo) + didSeedPolicy = true repository = repo let path = await repo.databaseURL.path fputs("[MCPService] shared DB: \(path)\n", stderr) @@ -52,4 +61,38 @@ actor MCPServiceStore { func databasePath() async -> String? { try? await sharedRepository().databaseURL.path } + + /// Idempotent baseline so effector admission works even if UI has not launched yet. + private func seedBaselinePolicyIfNeeded(_ repository: DBRepository) async throws { + let app = DerrickAppSupport.defaultApplicationName + try await DefaultGuardrailPolicySeeds.seedWorkflowStartRulesIfNeeded( + store: repository, + applicationName: app + ) + var rules: [PolicyRule] = AllowedMCPTool.allCases.map { tool in + PolicyRule( + applicationName: app, + name: "allow-\(tool.rawValue)", + scope: "tool_invocation", + matcherJSON: #"{"tool_name":"\#(tool.rawValue)"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 1 + ) + } + rules += ["tool_search", "tool", "tool_batch"].map { name in + PolicyRule( + applicationName: app, + name: "allow-\(name)", + scope: "tool_invocation", + matcherJSON: #"{"tool_name":"\#(name)"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 1 + ) + } + for rule in rules { + let existing = try await repository.loadRules(applicationName: app, scope: rule.scope) + guard existing.contains(where: { $0.name == rule.name }) == false else { continue } + try await repository.saveRule(rule) + } + } } diff --git a/ui/MCPService/MCPServiceToolHost.swift b/ui/MCPService/MCPServiceToolHost.swift index 3063ce97..03bc0898 100644 --- a/ui/MCPService/MCPServiceToolHost.swift +++ b/ui/MCPService/MCPServiceToolHost.swift @@ -5,6 +5,7 @@ import MCPClient import MCPServer import MemorySystem import Plugin +import PolicyRuntime import Structure /// MCP effectors hosted in MCPService (`script_exec`, `web.crawl`, `web.search`, factory @@ -288,9 +289,9 @@ actor MCPServiceToolHost { helperReviewerModelJSON: request.helperReviewerModelJSON, memorySessionKey: sessionKey, pluginFactoryCreationActive: request.pluginFactoryCreationActive - || EffectorAdmissionPolicy.parseContextJSON(request.executionContextJSON)? + || ExecutionContextWire.parseOptionalJSON(request.executionContextJSON)? .capabilities.contains(.syncWebCrawl) == true, - workflowID: EffectorAdmissionPolicy.parseContextJSON(request.executionContextJSON)? + workflowID: ExecutionContextWire.parseOptionalJSON(request.executionContextJSON)? .workflow?.workflowID ) let jobID: String? @@ -301,11 +302,73 @@ actor MCPServiceToolHost { HostHTTPCallContext.shared.clear() } + // Policy decides effector/tool admission (same engine as chat). Evaluate then apply. + let policyRepo = try await MCPServiceStore.shared.sharedRepository() + let toolEvaluating = StoreBackedToolInvocationEvaluating( + store: policyRepo, + applicationName: DerrickAppSupport.defaultApplicationName + ) + let sessionIDForPolicy: String = { + if case .agent(let sessionID, _) = request.principal { return sessionID } + if case .job(let jobID) = request.principal { return jobID } + return "mcp-service" + }() + let invocationEvent = ToolInvocationEvent( + sessionID: sessionIDForPolicy, + toolName: toolName, + argumentsJSON: request.argumentsJSON + ) + let decision = try await toolEvaluating.evaluate(invocationEvent) + let hitl = ClosureGuardrailHITLPresenting { [self] presentation in + let resolved = await awaitMCPToolHITL( + sessionID: sessionIDForPolicy, + toolName: presentation.subject, + argumentsJSON: presentation.payloadJSON, + hitl: presentation.hitl, + principal: request.principal, + repository: policyRepo + ) + switch resolved { + case .approved(let edited, let actor): + return .approved(editedPayloadJSON: edited, actor: actor) + case .cancelled(let actor): + return .cancelled(actor: actor) + } + } + let argumentsJSON: String + do { + let gated = try await ToolInvocationGuardrailApplying(hitl: hitl) + .apply(decision, for: invocationEvent) + argumentsJSON = gated.argumentsJSON + } catch ToolInvocationGuardrailError.denied(let reason) { + await MCPServiceStore.shared.log( + level: .error, + message: "tool denied by Policy tool=\(toolName): \(reason)", + code: "tool_denied", + detailJSON: #"{"requestID":"\#(request.requestID)"}"# + ) + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: reason + ) + } catch ToolInvocationGuardrailError.cancelled(let reason) { + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: reason + ) + } + // Shared Lib parser (same as Agent policy path) — handles repaired model JSON. // `{}` is a valid empty object for tools with no required args. let args: [String: Value] do { - args = try parseToolArgumentsObject(request.argumentsJSON) + args = try parseToolArgumentsObject(argumentsJSON) } catch { await MCPServiceStore.shared.log( level: .error, @@ -378,6 +441,68 @@ actor MCPServiceToolHost { private static func skillIndex(from repo: DBRepository) async throws -> [PluginSkillDisclosure.IndexEntry] { try await repo.listPluginSkillIndex() } + + private static let toolHITLPollNanoseconds: UInt64 = 1_000_000_000 + private static let toolHITLTimeoutNanoseconds: UInt64 = 15 * 60 * 1_000_000_000 + + private func awaitMCPToolHITL( + sessionID: String, + toolName: String, + argumentsJSON: String, + hitl: GuardrailHITLRequest, + principal: ServicePrincipal, + repository: DBRepository + ) async -> ApprovalConfirmationDecision { + let approvalID = UUID().uuidString + let requiredJSON = (try? JSONEncoder().encode(hitl.requiredFields)) + .flatMap { String(data: $0, encoding: .utf8) } ?? "[]" + let isJob = isJobPrincipal(principal) + let row = PendingHITLApprovalRow( + id: approvalID, + turnID: sessionID, + sessionID: sessionID, + toolName: toolName, + argumentsJSON: argumentsJSON, + requiredFieldsJSON: requiredJSON, + isJobContext: isJob + ) + do { + try await repository.insertPendingHITLApproval(row) + } catch { + fputs("[MCPService] HITL persist failed: \(error.localizedDescription)\n", stderr) + return .cancelled(actor: "system-persist-failed") + } + DerrickHITLNotificationSignal.postPoll() + + let deadline = Date().addingTimeInterval( + Double(Self.toolHITLTimeoutNanoseconds) / 1_000_000_000 + ) + while Date() < deadline { + if Task.isCancelled { + return .cancelled(actor: "system-cancelled") + } + if let decision = try? await repository.fetchPendingHITLApproval(id: approvalID), + decision.status != .pending { + switch decision.status { + case .approved: + let args = decision.editedArgumentsJSON?.isEmpty == false + ? decision.editedArgumentsJSON! + : argumentsJSON + return .approved(editedArgumentsJSON: args, actor: decision.actor) + case .cancelled, .timeout, .pending: + return .cancelled(actor: decision.actor ?? decision.status.rawValue) + } + } + try? await Task.sleep(nanoseconds: Self.toolHITLPollNanoseconds) + } + try? await repository.resolveHITLApproval( + id: approvalID, + status: .timeout, + editedArgumentsJSON: nil, + actor: "system-timeout" + ) + return .cancelled(actor: "system-timeout") + } } private func pluginFactoryFailureDetail(for error: Error) -> String { diff --git a/ui/SharedAgentRuntime/Conversation/ConversationModel.swift b/ui/SharedAgentRuntime/Conversation/ConversationModel.swift index 501216c9..b0a76a25 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationModel.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationModel.swift @@ -89,8 +89,9 @@ final class ConversationModel { policy: TieredMemoryCompactionPolicy(), budget: budget ) - let interceptor = DefaultPolicyInterceptor( - policy: StoreBackedCompletionContentPolicy(store: repository, applicationName: "ui") + let contentEvaluating = StoreBackedAssistantContentEvaluating( + store: repository, + applicationName: "ui" ) let delegateAgentsHost = try await makeAgentsOrchestrationHost( @@ -143,7 +144,7 @@ final class ConversationModel { ragInstructions: ragInstructions, mcpToolInstructions: mcpToolInstructions, responseSchema: Self.defaultResponseSchema, - interceptor: interceptor + contentEvaluating: contentEvaluating ) await ProfileSubagentGate.shared.end(caller: caller) return result @@ -290,7 +291,7 @@ final class ConversationModel { let ragInstructions = self.ragInstructions let mcpToolInstructions = self.mcpToolInstructions let responseSchema = self.responseSchema - let interceptor = makeContentPolicyInterceptor() + let contentEvaluating = makeContentEvaluating() let orchestrator = self.orchestrator let workerModel = helperModelSettings.workerAgentModel let workerApiKey = await LLMProviderCredentialGate.resolveAPIKey(for: workerModel) ?? apiKey @@ -345,7 +346,7 @@ final class ConversationModel { ragInstructions: rag, mcpToolInstructions: mcpToolInstructions, responseSchema: responseSchema, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: nil ) var completeText = "" @@ -385,7 +386,7 @@ final class ConversationModel { ragInstructions: userRagBase, mcpToolInstructions: effectiveMcpToolInstructions, responseSchema: responseSchema, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: approvalPresenter, retrievalLimit: retrievalLimit ) @@ -606,7 +607,7 @@ final class ConversationModel { ragInstructions: String, mcpToolInstructions: String, responseSchema: AgentSchema, - interceptor: PolicyInterceptor, + contentEvaluating: StoreBackedAssistantContentEvaluating?, approvalPresenter: (any ApprovalConfirmationPresenting)?, retrievalLimit: Int = 5 ) async -> AsyncThrowingStream { @@ -630,7 +631,7 @@ final class ConversationModel { return await pipeline.streamWithPolicyInterception( prompt: prompt, sessionID: sessionKey.sessionID, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: approvalPresenter, responseSchema: responseSchema ) @@ -653,19 +654,16 @@ final class ConversationModel { return await pipeline.streamWithPolicyInterception( prompt: prompt, sessionID: sessionKey.sessionID, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: approvalPresenter, responseSchema: responseSchema ) } } - private func makeContentPolicyInterceptor() -> PolicyInterceptor { - guard let policyStore else { - return DefaultPolicyInterceptor() - } - let policy = StoreBackedCompletionContentPolicy(store: policyStore, applicationName: "ui") - return DefaultPolicyInterceptor(policy: policy) + private func makeContentEvaluating() -> StoreBackedAssistantContentEvaluating? { + guard let policyStore else { return nil } + return StoreBackedAssistantContentEvaluating(store: policyStore, applicationName: "ui") } /// In-process host for orchestration tools (`agents_*`, `jobs_*`). Not used for MCP effectors. @@ -970,7 +968,7 @@ final class ConversationModel { outcomeJSON: #"{"action":"allow"}"#, priority: 1 ) - } + [ + } + DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: applicationName) + [ PolicyRule( applicationName: applicationName, name: "allow-default-assistant-chunks", diff --git a/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift b/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift index 14e7d03b..020a172d 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift @@ -3,6 +3,7 @@ import LLMAgentClient import MCP import MCPClient import MemorySystem +import PolicyRuntime import PartialJSON import AppEvents import PolicyUserInteraction @@ -15,7 +16,7 @@ extension ConversationPipeline { parentAgentID: String? = nil, toolCalls: [ToolCallRecord] = [], scope: MemoryAccessibility = .private, - interceptor: PolicyInterceptor = DefaultPolicyInterceptor(), + contentEvaluating: StoreBackedAssistantContentEvaluating? = nil, approvalPresenter: (any ApprovalConfirmationPresenting)? = nil, responseSchema: AgentSchema? = nil ) async -> AsyncThrowingStream { @@ -104,7 +105,13 @@ extension ConversationPipeline { ) let chunkPolicyStarted = Date() - let chunkIntercept = try await interceptor.interceptAssistantChunk(event) + let chunkDecision: GuardrailDecision + if let contentEvaluating { + chunkDecision = try await contentEvaluating.evaluate(event) + } else { + chunkDecision = .allow + } + let chunkIntercept = AssistantContentGuardrailApplying().apply(chunkDecision, for: event) chunkPolicyMS += PipelineTiming.elapsedMS(from: chunkPolicyStarted) let interceptedContent: String switch chunkIntercept { @@ -329,7 +336,16 @@ extension ConversationPipeline { chunkCount: chunkIndex ) let completionPolicyStarted = Date() - let completionIntercept = try await interceptor.interceptAssistantCompletion(completionEvent) + let completionDecision: GuardrailDecision + if let contentEvaluating { + completionDecision = try await contentEvaluating.evaluate(completionEvent) + } else { + completionDecision = .allow + } + let completionIntercept = AssistantContentGuardrailApplying().apply( + completionDecision, + for: completionEvent + ) var completionPolicyMS = PipelineTiming.elapsedMS(from: completionPolicyStarted) let contentConfirmStarted = Date() let interceptedCompletion: String? diff --git a/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift b/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift index 6f19bae2..de08d3d7 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift @@ -14,7 +14,7 @@ extension ConversationPipeline { arguments: [String: Value], sessionID: String, userPrompt: String? = nil, - interceptor: ToolRequestInterceptor? = nil, + toolEvaluating: (any GuardrailEvaluating)? = nil, approvalPresenter: (any ApprovalConfirmationPresenting)? = nil ) async throws -> MCPToolResult { let toolOverallStarted = Date() @@ -45,7 +45,7 @@ extension ConversationPipeline { } - let effectiveInterceptor = makeToolInterceptor(override: interceptor) + let evaluating = makeToolEvaluating(override: toolEvaluating) await MainActor.run { debugLog("Policy rule processing: evaluating \(name)") } @@ -53,165 +53,162 @@ extension ConversationPipeline { do { let confirmMSBox = TimingAccumulator() let proceedMSBox = TimingAccumulator() - let result = try await effectiveInterceptor.interceptAndRun( - event, - confirm: { [self] confirmEvent, requiredFields in - let confirmStarted = Date() - defer { confirmMSBox.add(PipelineTiming.elapsedMS(from: confirmStarted)) } - await MainActor.run { - debugLog("Policy decision: confirm \(name)") - } - guard let approvalPresenter else { - try await policyStore?.saveApproval( - PolicyApproval( - applicationName: applicationName, - sessionID: sessionID, - ruleID: "runtime-confirmation", - requestType: "tool_invocation", - requestPayloadJSON: event.argumentsJSON, - editedPayloadJSON: nil, - decision: "cancelled", - actor: "system", - createdAt: .now, - acedAt: .now - ) - ) - try await persistPolicyDecision( + let hitl = ClosureGuardrailHITLPresenting { [self] presentation in + let confirmStarted = Date() + defer { confirmMSBox.add(PipelineTiming.elapsedMS(from: confirmStarted)) } + await MainActor.run { + debugLog("Policy decision: confirm \(name)") + } + guard let approvalPresenter else { + try? await policyStore?.saveApproval( + PolicyApproval( + applicationName: applicationName, sessionID: sessionID, + ruleID: "runtime-confirmation", + requestType: "tool_invocation", requestPayloadJSON: event.argumentsJSON, + editedPayloadJSON: nil, decision: "cancelled", - actor: "system" + actor: "system", + createdAt: .now, + acedAt: .now ) - throw MCPClientError.toolExecutionDenied( - toolName: name, - reason: "Tool execution requires user confirmation" - ) - } - - let confirmationRequest = ApprovalConfirmationRequest( - sessionID: sessionID, - toolName: confirmEvent.toolName, - argumentsJSON: confirmEvent.argumentsJSON, - requiredFields: requiredFields ) - let confirmation = await approvalPresenter.confirm(confirmationRequest) - await MainActor.run { - debugLog("Approval response received for \(name)") - } - let approvalRecord = PolicyApproval.fromApprovalDecision( - applicationName: applicationName, + try? await persistPolicyDecision( sessionID: sessionID, - requestPayloadJSON: confirmEvent.argumentsJSON, - decision: confirmation + requestPayloadJSON: event.argumentsJSON, + decision: "cancelled", + actor: "system" ) - try await policyStore?.saveApproval(approvalRecord) + return .cancelled(actor: "system") + } - switch confirmation { - case .approved(let editedArgumentsJSON, let actor): - await MainActor.run { - debugLog("Approval granted for \(name) by \(actor ?? "unknown")") - } - try await persistPolicyDecision( - sessionID: sessionID, - requestPayloadJSON: confirmEvent.argumentsJSON, - decision: "approved", - actor: actor - ) - return .approved( - ToolInvocationEvent( - sessionID: confirmEvent.sessionID, - toolName: confirmEvent.toolName, - argumentsJSON: editedArgumentsJSON, - timestamp: confirmEvent.timestamp - ) - ) - case .cancelled(let actor): - await MainActor.run { - debugLog("Approval cancelled for \(name) by \(actor ?? "unknown")") - } - try await persistPolicyDecision( - sessionID: sessionID, - requestPayloadJSON: confirmEvent.argumentsJSON, - decision: "cancelled", - actor: actor - ) - return .cancelled(actor: actor) - } - }, - proceed: { [self] interceptedEvent in - let proceedStarted = Date() - defer { proceedMSBox.add(PipelineTiming.elapsedMS(from: proceedStarted)) } + let confirmationRequest = ApprovalConfirmationRequest( + sessionID: sessionID, + toolName: presentation.subject, + argumentsJSON: presentation.payloadJSON, + requiredFields: presentation.hitl.requiredFields + ) + let confirmation = await approvalPresenter.confirm(confirmationRequest) + await MainActor.run { + debugLog("Approval response received for \(name)") + } + let approvalRecord = PolicyApproval.fromApprovalDecision( + applicationName: applicationName, + sessionID: sessionID, + requestPayloadJSON: presentation.payloadJSON, + decision: confirmation + ) + try? await policyStore?.saveApproval(approvalRecord) + + switch confirmation { + case .approved(let editedArgumentsJSON, let actor): await MainActor.run { - debugLog("Policy decision: allow \(interceptedEvent.toolName)") - } - let interceptedArguments = try toolArgumentsFromJSON(interceptedEvent.argumentsJSON) - guard let mcpClient else { - return MCPToolResult(content: [MCPToolContent.text("Tool client unavailable.")], isError: true) - } - if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { - let scriptAllowed = await UsageLimitsService.shared.allowScriptRun() - if !scriptAllowed { - throw MCPClientError.toolExecutionDenied( - toolName: interceptedEvent.toolName, - reason: "Usage limit: max script_exec runs for this message." - ) - } - let reviewerAllowed = await UsageLimitsService.shared.allowReviewerCall() - if !reviewerAllowed { - throw MCPClientError.toolExecutionDenied( - toolName: interceptedEvent.toolName, - reason: "Usage limit: max security reviewer calls for this message." - ) - } + debugLog("Approval granted for \(name) by \(actor ?? "unknown")") } + try? await persistPolicyDecision( + sessionID: sessionID, + requestPayloadJSON: presentation.payloadJSON, + decision: "approved", + actor: actor + ) + return .approved(editedPayloadJSON: editedArgumentsJSON, actor: actor) + case .cancelled(let actor): await MainActor.run { - debugLog("Executing tool: \(interceptedEvent.toolName)") + debugLog("Approval cancelled for \(name) by \(actor ?? "unknown")") } - let execStarted = Date() - let result = try await mcpClient.callTool( - named: interceptedEvent.toolName, - arguments: interceptedArguments + try? await persistPolicyDecision( + sessionID: sessionID, + requestPayloadJSON: presentation.payloadJSON, + decision: "cancelled", + actor: actor ) - let mcpCallMS = PipelineTiming.elapsedMS(from: execStarted) - PipelineTiming.log( - "tool=\(interceptedEvent.toolName) mcp_call_ms=\(mcpCallMS) isError=\(result.isError) result_chars=\(result.text.utf8.count)" + return .cancelled(actor: actor) + } + } + + let decision: GuardrailDecision + if let evaluating { + decision = try await evaluating.evaluate(event) + } else { + decision = .allow + } + let applying = ToolInvocationGuardrailApplying(hitl: hitl) + let interceptedEvent = try await applying.apply(decision, for: event) + + let proceedStarted = Date() + await MainActor.run { + debugLog("Policy decision: allow \(interceptedEvent.toolName)") + } + let interceptedArguments = try toolArgumentsFromJSON(interceptedEvent.argumentsJSON) + guard let mcpClient else { + proceedMSBox.add(PipelineTiming.elapsedMS(from: proceedStarted)) + return MCPToolResult(content: [MCPToolContent.text("Tool client unavailable.")], isError: true) + } + if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { + let scriptAllowed = await UsageLimitsService.shared.allowScriptRun() + if !scriptAllowed { + throw MCPClientError.toolExecutionDenied( + toolName: interceptedEvent.toolName, + reason: "Usage limit: max script_exec runs for this message." ) - // Attribute reviewer-ish tokens from phase timing when present. - if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { - await Self.recordReviewerTokensIfPresent(resultText: result.text) - } - await MainActor.run { - debugLog("Tool result: \(interceptedEvent.toolName) (isError=\(result.isError))") - ToolOutcomeLogger.log( - toolName: interceptedEvent.toolName, - rawText: result.text - ) - if interceptedEvent.toolName != AllowedMCPTool.pluginFactoryBuild.rawValue - && interceptedEvent.toolName != AllowedMCPTool.webCrawl.rawValue { - debugLog("Tool result content: \(Self.debugPayload(result.text))") - } - } - await Self.publishJobSchedulingFailureIfNeeded( + } + let reviewerAllowed = await UsageLimitsService.shared.allowReviewerCall() + if !reviewerAllowed { + throw MCPClientError.toolExecutionDenied( toolName: interceptedEvent.toolName, - resultText: result.text, - sessionID: sessionID + reason: "Usage limit: max security reviewer calls for this message." ) - if interceptedEvent.toolName == AllowedMCPTool.pluginFactoryBuild.rawValue { - return try await self.resolvePluginFactoryBuildResult( - initialResult: result, - arguments: interceptedArguments, - userPrompt: userPrompt, - mcpClient: mcpClient - ) - } - return result } + } + await MainActor.run { + debugLog("Executing tool: \(interceptedEvent.toolName)") + } + let execStarted = Date() + let toolResult = try await mcpClient.callTool( + named: interceptedEvent.toolName, + arguments: interceptedArguments ) + let mcpCallMS = PipelineTiming.elapsedMS(from: execStarted) + PipelineTiming.log( + "tool=\(interceptedEvent.toolName) mcp_call_ms=\(mcpCallMS) isError=\(toolResult.isError) result_chars=\(toolResult.text.utf8.count)" + ) + if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { + await Self.recordReviewerTokensIfPresent(resultText: toolResult.text) + } + await MainActor.run { + debugLog("Tool result: \(interceptedEvent.toolName) (isError=\(toolResult.isError))") + ToolOutcomeLogger.log( + toolName: interceptedEvent.toolName, + rawText: toolResult.text + ) + if interceptedEvent.toolName != AllowedMCPTool.pluginFactoryBuild.rawValue + && interceptedEvent.toolName != AllowedMCPTool.webCrawl.rawValue { + debugLog("Tool result content: \(Self.debugPayload(toolResult.text))") + } + } + await Self.publishJobSchedulingFailureIfNeeded( + toolName: interceptedEvent.toolName, + resultText: toolResult.text, + sessionID: sessionID + ) + let result: MCPToolResult + if interceptedEvent.toolName == AllowedMCPTool.pluginFactoryBuild.rawValue { + result = try await self.resolvePluginFactoryBuildResult( + initialResult: toolResult, + arguments: interceptedArguments, + userPrompt: userPrompt, + mcpClient: mcpClient + ) + } else { + result = toolResult + } + proceedMSBox.add(PipelineTiming.elapsedMS(from: proceedStarted)) PipelineTiming.log( "tool=\(name) encode_ms=\(encodeMS) confirm_ms=\(confirmMSBox.total) proceed_ms=\(proceedMSBox.total) total_ms=\(PipelineTiming.elapsedMS(from: toolOverallStarted))" ) return result - } catch ToolInvocationInterceptionError.denied(let reason) { + } catch ToolInvocationGuardrailError.denied(let reason) { PipelineTiming.log( "tool=\(name) denied encode_ms=\(encodeMS) total_ms=\(PipelineTiming.elapsedMS(from: toolOverallStarted)) reason=\(reason)" ) @@ -267,7 +264,7 @@ extension ConversationPipeline { ) ) throw MCPClientError.toolExecutionDenied(toolName: name, reason: reason) - } catch ToolInvocationInterceptionError.cancelled(let reason) { + } catch ToolInvocationGuardrailError.cancelled(let reason) { PipelineTiming.log( "tool=\(name) cancelled encode_ms=\(encodeMS) total_ms=\(PipelineTiming.elapsedMS(from: toolOverallStarted)) reason=\(reason)" ) @@ -301,7 +298,7 @@ extension ConversationPipeline { _ request: MCPToolBatchRequest, sessionID: String, userPrompt: String? = nil, - interceptor: ToolRequestInterceptor? = nil, + toolEvaluating: (any GuardrailEvaluating)? = nil, approvalPresenter: (any ApprovalConfirmationPresenting)? = nil ) async throws -> MCPToolBatchResult { // MA-3: run independent batch invocations concurrently (order of results preserved). @@ -319,7 +316,7 @@ extension ConversationPipeline { arguments: invocation.arguments, sessionID: sessionID, userPrompt: userPrompt, - interceptor: interceptor, + toolEvaluating: toolEvaluating, approvalPresenter: approvalPresenter ) return (index, result) @@ -354,15 +351,12 @@ extension ConversationPipeline { ) } - private func makeToolInterceptor(override: ToolRequestInterceptor?) -> ToolRequestInterceptor { - if let override { - return override - } - if let policyStore { - let policy = StoreBackedToolGovernancePolicy(store: policyStore, applicationName: applicationName) - return DefaultToolRequestInterceptor(policy: policy) - } - return DefaultToolRequestInterceptor() + private func makeToolEvaluating( + override: (any GuardrailEvaluating)? + ) -> (any GuardrailEvaluating)? { + if let override { return override } + guard let policyStore else { return nil } + return StoreBackedToolInvocationEvaluating(store: policyStore, applicationName: applicationName) } private func persistPolicyDecision( diff --git a/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift b/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift index 04f47b3e..9e624a7d 100644 --- a/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift +++ b/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift @@ -44,7 +44,7 @@ enum ProfileDelegateRunner { ragInstructions: String, mcpToolInstructions: String, responseSchema: AgentSchema, - interceptor: PolicyInterceptor + contentEvaluating: StoreBackedAssistantContentEvaluating? ) async throws -> String { let normalized = AgentProfileHandle.normalize( profileHandle.trimmingCharacters(in: .whitespacesAndNewlines).replacingOccurrences(of: "$", with: "") @@ -103,7 +103,7 @@ enum ProfileDelegateRunner { ragInstructions: userRagBase, mcpToolInstructions: mcpToolInstructions, responseSchema: responseSchema, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: nil, retrievalLimit: retrievalLimit )