From 85dcf9a1d20f429f983ba5888e92bf22636d8740 Mon Sep 17 00:00:00 2001 From: David Choi Date: Sat, 26 Sep 2026 18:31:05 -0400 Subject: [PATCH 1/3] Introduce Guardrail as Structure's control plane. Nest Policy, HITL, and workflows under Guardrail with one GuardrailDecision vocabulary, admit workflow starts and web.crawl through that policy, and drop stale scratch docs/scripts. Co-authored-by: Cursor --- CONTRIBUTING.md | 4 +- check_browser_deps.py | 6 -- check_playwright.py | 9 --- docs/Design.md | 12 +++- fetch_ms.py | 16 ----- master-todo.md | 3 +- .../WorkflowRuntimeEngine.swift | 1 + .../MemorySystem/PolicyInterceptor.swift | 12 ++-- .../MemorySystem/ToolRequestInterceptor.swift | 8 ++- .../PolicyInterceptionTests.swift | 4 +- .../ToolRequestInterceptionTests.swift | 14 ++-- .../StoreBackedPolicyEvaluators.swift | 27 +++++--- .../StoreBackedPolicyEvaluatorsTests.swift | 10 +-- .../AppLayerFeatureCatalog.swift | 14 ++++ .../ExecutionContextWire.swift | 9 --- .../DerrickBackend/DerrickBackendErrors.swift | 3 + .../Sources/Guardrail/Guardrail.swift | 12 ++++ .../Sources/Guardrail/GuardrailDecision.swift | 46 +++++++++++++ .../DerrickHITLApprovalPresentationWake.swift | 0 .../HITL}/DerrickHITLNotificationSignal.swift | 0 .../HITL}/HITLTypes.swift | 0 .../Policy/PolicyEngine/Matchers.swift | 10 ++- .../PolicyEngine/PolicyEngineTypes.swift | 24 +++---- .../Policy/PolicyEngine/Rules.swift | 15 ++--- .../PolicyRuntime/PolicyInterception.swift | 16 ++--- .../PolicyInterceptionEvent.swift | 0 .../Policy/PolicyRuntime/PolicyStore.swift | 0 .../PolicyRuntime/ToolInterception.swift | 18 +++-- .../PolicyUserEvent.swift | 0 .../Workflow}/EffectorAdmissionPolicy.swift | 15 ++++- .../Workflow/WorkflowAdmissionPolicy.swift | 40 +++++++++++ .../Workflow}/WorkflowChatProgress.swift | 0 .../Guardrail/Workflow/WorkflowKind.swift | 13 ++++ .../Workflow}/WorkflowRuntimeDTOs.swift | 0 .../Workflow}/WorkflowRuntimeRows.swift | 0 .../Workflow}/WorkflowRuntimeXPC.swift | 0 .../Sources/Structure/Architecture.swift | 2 +- .../GuardrailDecisionTests.swift | 66 +++++++++++++++++++ .../StructureTests/PolicyEngineTests.swift | 4 +- readme.md | 27 ++++---- scripts/reset-local-state.sh | 21 ++++-- todo.rtf | 8 --- ui/MCPService/MCPServiceToolHost.swift | 34 ++++++++++ 43 files changed, 371 insertions(+), 152 deletions(-) delete mode 100644 check_browser_deps.py delete mode 100644 check_playwright.py delete mode 100644 fetch_ms.py create mode 100644 packages/Structure/Sources/Guardrail/Guardrail.swift create mode 100644 packages/Structure/Sources/Guardrail/GuardrailDecision.swift rename packages/Structure/Sources/{AppLayerServices/AppServices => Guardrail/HITL}/DerrickHITLApprovalPresentationWake.swift (100%) rename packages/Structure/Sources/{AppLayerServices/AppServices => Guardrail/HITL}/DerrickHITLNotificationSignal.swift (100%) rename packages/Structure/Sources/{DBRepository => Guardrail/HITL}/HITLTypes.swift (100%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyEngine/Matchers.swift (93%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyEngine/PolicyEngineTypes.swift (82%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyEngine/Rules.swift (72%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyRuntime/PolicyInterception.swift (72%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyRuntime/PolicyInterceptionEvent.swift (100%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyRuntime/PolicyStore.swift (100%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyRuntime/ToolInterception.swift (81%) rename packages/Structure/Sources/{ => Guardrail}/Policy/PolicyUserInteraction/PolicyUserEvent.swift (100%) rename packages/Structure/Sources/{AppLayerServices/MCPService => Guardrail/Workflow}/EffectorAdmissionPolicy.swift (58%) create mode 100644 packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift rename packages/Structure/Sources/{AppLayerServices/MCPService => Guardrail/Workflow}/WorkflowChatProgress.swift (100%) create mode 100644 packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift rename packages/Structure/Sources/{AppLayerServices/MCPService => Guardrail/Workflow}/WorkflowRuntimeDTOs.swift (100%) rename packages/Structure/Sources/{DBRepository => Guardrail/Workflow}/WorkflowRuntimeRows.swift (100%) rename packages/Structure/Sources/{AppLayerServices/MCPService => Guardrail/Workflow}/WorkflowRuntimeXPC.swift (100%) create mode 100644 packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift delete mode 100644 todo.rtf diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 96feb39e..40bce045 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -26,7 +26,7 @@ Thank you for your interest in contributing. Please read the [Code of Conduct](C git config core.hooksPath .githooks ``` -See [docs/development.md](docs/development.md) for daemon bootstrap, Login Items, and database paths. +See [docs/Design.md](docs/Design.md) for architecture notes. Local SQLite lives under the Derrick app group; reset with `./scripts/reset-local-state.sh`. ## Build from the command line @@ -48,7 +48,7 @@ Do not commit provisioning profiles (`.mobileprovision`, `.p12`, `.pem`). - Run `./scripts/verify-no-secrets.sh --staged` before committing. - Run `./scripts/build.sh test` when you change build-affecting code. - Keep changes focused; match existing Swift style and module boundaries. -- Update README or ADRs when behavior or architecture changes. +- Update README or `docs/Design.md` when behavior or architecture changes. ## License diff --git a/check_browser_deps.py b/check_browser_deps.py deleted file mode 100644 index bb9af009..00000000 --- a/check_browser_deps.py +++ /dev/null @@ -1,6 +0,0 @@ -import subprocess -try: - result = subprocess.run(["apt-cache", "search", "chromium"], capture_output=True, text=True) - print(result.stdout[:500]) -except Exception as e: - print(str(e)) diff --git a/check_playwright.py b/check_playwright.py deleted file mode 100644 index 5c9a7b84..00000000 --- a/check_playwright.py +++ /dev/null @@ -1,9 +0,0 @@ -import subprocess -try: - # Try to check if chromium is installable in the container - # I will try to update apt cache first - this might take time/fail due to read-only container - print("Checking if chromium is available via apt-cache...") - result = subprocess.run(["apt-cache", "search", "chromium"], capture_output=True, text=True) - print(result.stdout[:500]) -except Exception as e: - print(str(e)) diff --git a/docs/Design.md b/docs/Design.md index 5a9171d9..1bb48bdd 100644 --- a/docs/Design.md +++ b/docs/Design.md @@ -8,6 +8,12 @@ Derrick does not provide many tools to agents. Instead Derrick provides a small ## Structure This application is Protocol first. All major features must have a Protocol and internally use GoF Design Patterns. No exceptions. The Protocols can be found in the Structure spm. -## Plugins -- Plugins are using the Agent Plugin standard -- Plugins and `script_exec` run in the unified Go worker Docker image (`derrick-worker:go-v1`) +## Guardrail +Guardrail is Derrick's control plane in Structure (`Sources/Guardrail`). + +- **Policy decides** allow / deny / confirm HITL / require workflow / redact (`GuardrailDecision`). +- **HITL** and **workflows** are enforcement shapes for those decisions. +- **Plugins** propose capabilities; they do not authorize control outcomes. +- Workflow starts are admitted by `WorkflowAdmissionPolicy` inside `WorkflowRuntimeEngine` (UI proposes; Guardrail authorizes). +- Effector admission (for example sync `web.crawl`) uses the same decision vocabulary and is enforced in MCPService before the tool runs. +- `WorkflowKind.pluginFactoryEdit` is reserved for future plugin editability and is denied until enabled. diff --git a/fetch_ms.py b/fetch_ms.py deleted file mode 100644 index 95156d5f..00000000 --- a/fetch_ms.py +++ /dev/null @@ -1,16 +0,0 @@ -import requests, json -from bs4 import BeautifulSoup -headers = { - "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0.0.0 Safari/537.36", - "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8", - "Accept-Language": "en-US,en;q=0.9", - "Referer": "https://www.google.com/" -} -try: - r = requests.get("https://microsoft.com", timeout=20, headers=headers) - r.raise_for_status() - soup = BeautifulSoup(r.text, "lxml") - out = {"url": r.url, "status_code": r.status_code, "title": soup.title.string.strip() if soup.title and soup.title.string else ""} - print(json.dumps(out)) -except Exception as e: - print(str(e)) diff --git a/master-todo.md b/master-todo.md index affeecf7..8564906d 100644 --- a/master-todo.md +++ b/master-todo.md @@ -118,8 +118,7 @@ In-use lease TTL (default 7 minutes) stays as the anti-hoard cap. No idle TTL (n ### Docs -- [ ] Sweep `docs/`: drop or mark stale ADRs, fix Python-vs-Swift guest, file extractor vs native reads, container recreate-on-handoff, messaging roadmap items that already shipped. -- [ ] Files to revisit: `docs/adr-swift-script-runtime.md`, `docs/development.md`, `docs/messaging-design.md` remaining table, `docs/opensource-plan.md`, `readme.md` if it still implies one-shot containers only. +- [x] Dropped stale ADRs / plan docs; `docs/Design.md` is the remaining architecture note. README/CONTRIBUTING no longer link to deleted files. ### Startup: crawler image build blocks the app diff --git a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift index c622c88c..141654a6 100644 --- a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift +++ b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift @@ -15,6 +15,7 @@ public actor WorkflowRuntimeEngine { _ request: WorkflowStartRequest, repositoryProvider: @escaping @Sendable () async throws -> DBRepository ) async throws -> WorkflowHandleDTO { + try WorkflowAdmissionPolicy.admitOrThrow(request) let repo = try await repositoryProvider() let idempotencyKey = WorkflowRuntimeIdempotency.key( sessionID: request.sessionID, diff --git a/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift b/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift index 8f766309..a8d45ef4 100644 --- a/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift +++ b/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift @@ -17,14 +17,14 @@ public struct DefaultPolicyInterceptor: PolicyInterceptor { return .allowed(event.content) case .deny(let reason): return .denied(reason: reason) - case .redact(let pattern, let replacement): + case .redactContent(let pattern, let replacement): let redacted = event.content.replacingOccurrences( of: pattern, with: replacement, options: .regularExpression ) return .allowed(redacted) - case .confirm: + case .confirmHITL, .requireWorkflow, .redactArgument: // Streaming chunks: never modal mid-token. Completion path enforces confirm. return .allowed(event.content) } @@ -39,15 +39,17 @@ public struct DefaultPolicyInterceptor: PolicyInterceptor { return .allowed(event.fullCompletion) case .deny(let reason): return .denied(reason: reason) - case .redact(let pattern, let replacement): + case .redactContent(let pattern, let replacement): let redacted = event.fullCompletion.replacingOccurrences( of: pattern, with: replacement, options: .regularExpression ) return .allowed(redacted) - case .confirm(let requiredFields): - return .confirm(content: event.fullCompletion, requiredFields: requiredFields) + case .confirmHITL(let hitl): + return .confirm(content: event.fullCompletion, requiredFields: hitl.requiredFields) + case .requireWorkflow, .redactArgument: + return .denied(reason: "Content policy returned an unsupported control decision.") } } } diff --git a/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift b/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift index 2043b5a7..4e867476 100644 --- a/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift +++ b/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift @@ -19,9 +19,9 @@ public struct DefaultToolRequestInterceptor: ToolRequestInterceptor { return .allow(event) case .deny(let reason): return .deny(reason: reason) - case .confirm(let requiredFields): - return .confirm(event, requiredFields: requiredFields) - case .redact(let key, let pattern, let replacement): + case .confirmHITL(let hitl): + return .confirm(event, requiredFields: hitl.requiredFields) + case .redactArgument(let key, let pattern, let replacement): guard let parsedArgs = try? JSONSerialization.jsonObject(with: event.argumentsJSON.data(using: .utf8) ?? Data(), options: []) as? [String: Any] else { return .allow(event) } @@ -48,6 +48,8 @@ public struct DefaultToolRequestInterceptor: ToolRequestInterceptor { timestamp: event.timestamp ) ) + case .requireWorkflow, .redactContent: + return .deny(reason: "Tool policy returned an unsupported control decision.") } } diff --git a/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift b/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift index 903058fb..e4bf6b91 100644 --- a/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift +++ b/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift @@ -150,14 +150,14 @@ struct MockResponseContentPolicy: PolicyEvaluator { if shouldDeny { return .deny(reason: denyReason) } - return shouldAllow ? .allow : .confirm(requiredFields: []) + return shouldAllow ? .allow : .confirmHITL(GuardrailHITLRequest(requiredFields: [])) } func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome { if shouldDeny { return .deny(reason: denyReason) } - return shouldAllow ? .allow : .confirm(requiredFields: []) + return shouldAllow ? .allow : .confirmHITL(GuardrailHITLRequest(requiredFields: [])) } } diff --git a/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift b/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift index eec6f240..e94be9b0 100644 --- a/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift +++ b/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift @@ -103,7 +103,7 @@ struct MockToolGovernancePolicy: ToolGovernancePolicy { if shouldDeny { return .deny(reason: denyReason) } - return shouldAllow ? .allow : .confirm(requiredFields: ["approval"]) + return shouldAllow ? .allow : .confirmHITL(GuardrailHITLRequest(requiredFields: ["approval"])) } } @@ -157,7 +157,7 @@ final class ToolRequestInterceptorTests: XCTestCase { func test_defaultInterceptor_redacts_arguments() async throws { struct RedactingPolicy: ToolGovernancePolicy { func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - return .redact(argumentKey: "password", pattern: ".+", replacement: "[REDACTED]") + return .redactArgument(argumentKey: "password", pattern: ".+", replacement: "[REDACTED]") } } @@ -199,7 +199,7 @@ final class ToolRequestInterceptorTests: XCTestCase { func test_confirm_outcome_preserves_tool_name() async throws { struct ConfirmingPolicy: ToolGovernancePolicy { func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - return .confirm(requiredFields: ["user_consent"]) + return .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_consent"])) } } @@ -218,7 +218,7 @@ final class ToolRequestInterceptorTests: XCTestCase { func test_evaluateToolInvocation_returns_confirm_with_requiredFields() async throws { struct ConfirmingPolicy: ToolGovernancePolicy { func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirm(requiredFields: ["user_consent", "ticket_id"]) + .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_consent", "ticket_id"])) } } @@ -314,7 +314,7 @@ final class ToolRequestInterceptorTests: XCTestCase { func test_interceptAndRun_confirm_approved_then_proceed() async throws { struct ConfirmPolicy: ToolGovernancePolicy { func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirm(requiredFields: ["user_approval"]) + .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_approval"])) } } @@ -352,7 +352,7 @@ final class ToolRequestInterceptorTests: XCTestCase { func test_interceptAndRun_confirm_cancelled_throws_without_proceed() async throws { struct ConfirmPolicy: ToolGovernancePolicy { func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirm(requiredFields: ["user_approval"]) + .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_approval"])) } } @@ -384,7 +384,7 @@ final class ToolRequestInterceptorTests: XCTestCase { func test_interceptAndRun_redact_then_proceed_with_redacted_event() async throws { struct RedactPolicy: ToolGovernancePolicy { func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .redact(argumentKey: "token", pattern: ".+", replacement: "[REDACTED]") + .redactArgument(argumentKey: "token", pattern: ".+", replacement: "[REDACTED]") } } diff --git a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift index 33846c0b..2a97eedf 100644 --- a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift +++ b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift @@ -12,7 +12,7 @@ public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { self.applicationName = applicationName } - public func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { + public func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> GuardrailDecision { let rules = try await loadRules(scopes: ["tool_invocation", "tool_call"]) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) @@ -66,7 +66,7 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { self.applicationName = applicationName } - public func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> PolicyDecisionOutcome { + public func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> GuardrailDecision { let rules = try await loadRules(scopes: ["assistant_chunk"]) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) @@ -89,7 +89,7 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { return .deny(reason: Self.noMatchingRuleReason) } - public func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome { + public func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> GuardrailDecision { let rules = try await loadRules(scopes: ["assistant_completion_content", "assistant_completion"]) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) @@ -415,30 +415,38 @@ private struct OutcomeRule: Decodable { case replacement } - var toolOutcome: ToolGovernanceOutcome { + var toolOutcome: GuardrailDecision { switch action.lowercased() { case "deny": return .deny(reason: reason ?? "Tool invocation denied by policy.") case "confirm": - return .confirm(requiredFields: requiredFields ?? ["user_approval"]) + return .confirmHITL( + GuardrailHITLRequest(requiredFields: requiredFields ?? ["user_approval"]) + ) case "allow": return .allow case "redact": guard let argumentKey, let pattern else { return .deny(reason: "Invalid redact outcome for tool rule (missing argument_key/pattern).") } - return .redact(argumentKey: argumentKey, pattern: pattern, replacement: replacement ?? "[REDACTED]") + return .redactArgument( + argumentKey: argumentKey, + pattern: pattern, + replacement: replacement ?? "[REDACTED]" + ) default: return .deny(reason: "Unknown tool policy action '\(action)'; denying by default.") } } - func contentOutcome(fallbackPattern: String?) -> PolicyDecisionOutcome { + func contentOutcome(fallbackPattern: String?) -> GuardrailDecision { switch action.lowercased() { case "deny": return .deny(reason: reason ?? "Assistant content denied by policy.") case "confirm": - return .confirm(requiredFields: requiredFields ?? ["review_confirmation"]) + return .confirmHITL( + GuardrailHITLRequest(requiredFields: requiredFields ?? ["review_confirmation"]) + ) case "allow": return .allow case "redact": @@ -446,13 +454,14 @@ private struct OutcomeRule: Decodable { guard let patternToUse else { return .deny(reason: "Invalid redact outcome for content rule (missing pattern).") } - return .redact(pattern: patternToUse, replacement: replacement ?? "[REDACTED]") + return .redactContent(pattern: patternToUse, replacement: replacement ?? "[REDACTED]") default: return .deny(reason: "Unknown content policy action '\(action)'; denying by default.") } } } + private func parseJSONObject(from json: String) -> [String: Any] { guard let data = json.data(using: .utf8), let object = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { diff --git a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift index 2fbc2869..a335a98d 100644 --- a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift +++ b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift @@ -159,7 +159,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { chunkCount: 1 ) ) - XCTAssertEqual(outcome, .confirm(requiredFields: ["review"])) + XCTAssertEqual(outcome, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) } func test_completionPolicy_skipsDisabledRule() async throws { @@ -227,7 +227,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { argumentsJSON: #"{"code":"print(1)"}"# ) ) - XCTAssertEqual(withCode, .confirm(requiredFields: ["review"])) + XCTAssertEqual(withCode, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) let withoutCode = try await policy.evaluateToolInvocation( ToolInvocationEvent( @@ -355,7 +355,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { argumentsJSON: #"{"code":"x"}"# ) ) - XCTAssertEqual(offlineScript, .confirm(requiredFields: ["offline_ok"])) + XCTAssertEqual(offlineScript, .confirmHITL(GuardrailHITLRequest(requiredFields: ["offline_ok"]))) let networkedScript = try await policy.evaluateToolInvocation( ToolInvocationEvent( @@ -396,7 +396,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { let sessionTool = try await policy.evaluateToolInvocation( ToolInvocationEvent(sessionID: "s1", toolName: "session_memory_search", argumentsJSON: "{}") ) - XCTAssertEqual(sessionTool, .confirm(requiredFields: ["ok"])) + XCTAssertEqual(sessionTool, .confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"]))) let otherTool = try await policy.evaluateToolInvocation( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") @@ -432,7 +432,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { chunkCount: 1 ) ) - XCTAssertEqual(ssn, .confirm(requiredFields: ["review"])) + XCTAssertEqual(ssn, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) let plain = try await policy.evaluateAssistantCompletion( AssistantCompletionEvent( diff --git a/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift b/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift index bccbeecd..db374a5d 100644 --- a/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift +++ b/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift @@ -10,6 +10,7 @@ public enum AppLayerFeature: String, CaseIterable, Sendable { case scriptReview case mcpToolCatalog case conversationMCPBridge + case guardrail } public struct AppLayerFeatureContract: Sendable, Hashable { @@ -78,6 +79,19 @@ public enum AppLayerFeatureCatalog { persistenceMayBeSQLite: false, uiMustNotImportMCPServer: false ), + AppLayerFeatureContract( + feature: .guardrail, + structureTypeNames: [ + "GuardrailDecision", + "PolicyEvaluator", + "ToolGovernancePolicy", + "WorkflowKind", + "WorkflowAdmissionPolicy", + "PendingHITLApprovalRow", + ], + persistenceMayBeSQLite: true, + uiMustNotImportMCPServer: true + ), ] public static func contract(for feature: AppLayerFeature) -> AppLayerFeatureContract { diff --git a/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift b/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift index 9257a43d..f299d07b 100644 --- a/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift +++ b/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift @@ -11,15 +11,6 @@ public enum ExecutionContextCapability: String, Codable, Sendable, Hashable, Cas case hostReviewRetry = "host_review_retry" } -public enum WorkflowKind: String, Codable, Sendable, Hashable, CaseIterable { - case pluginFactoryCreate = "plugin_factory_create" - case pluginFactoryEdit = "plugin_factory_edit" - case connectorAuthDiscover = "connector_auth_discover" - case jobStep = "job_step" - case interactiveTool = "interactive_tool" - case none -} - public struct WorkflowContextWire: Codable, Sendable, Hashable { public let workflowID: String? public let kind: WorkflowKind diff --git a/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift b/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift index a5346556..c4678ae8 100644 --- a/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift +++ b/packages/Structure/Sources/DerrickBackend/DerrickBackendErrors.swift @@ -37,6 +37,7 @@ public enum WorkflowRuntimeError: Error, LocalizedError, Sendable { case mcpUnavailable case workflowNotFound case unsupportedKind(WorkflowKind) + case deniedByGuardrail(String) public var errorDescription: String? { switch self { @@ -46,6 +47,8 @@ public enum WorkflowRuntimeError: Error, LocalizedError, Sendable { return "Workflow was not found." case .unsupportedKind(let kind): return "Unsupported workflow kind \(kind.rawValue)." + case .deniedByGuardrail(let reason): + return reason } } } diff --git a/packages/Structure/Sources/Guardrail/Guardrail.swift b/packages/Structure/Sources/Guardrail/Guardrail.swift new file mode 100644 index 00000000..120d5e1e --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Guardrail.swift @@ -0,0 +1,12 @@ +import Foundation + +/// Derrick's agent-control umbrella in Structure. +/// +/// **Policy decides.** HITL and workflows are enforcement shapes for those decisions. +/// Plugins propose actions; they do not authorize control outcomes. +/// +/// Folder layout: +/// - `Guardrail/Policy` — rules and interception contracts +/// - `Guardrail/HITL` — human confirmation wire types and presentation signals +/// - `Guardrail/Workflow` — workflow kinds, runtime DTOs, effector admission +public enum Guardrail: Sendable {} diff --git a/packages/Structure/Sources/Guardrail/GuardrailDecision.swift b/packages/Structure/Sources/Guardrail/GuardrailDecision.swift new file mode 100644 index 00000000..4481aa41 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/GuardrailDecision.swift @@ -0,0 +1,46 @@ +import Foundation + +/// Canonical control-plane decision. Every Policy evaluator maps to this. +public enum GuardrailDecision: Hashable, Sendable { + case allow + case deny(reason: String) + /// Require a human gate before proceeding. + case confirmHITL(GuardrailHITLRequest) + /// Require a sequenced workflow before / instead of the proposed action. + case requireWorkflow(WorkflowKind) + case redactContent(pattern: String, replacement: String) + case redactArgument(argumentKey: String, pattern: String, replacement: String) + + public var isAllow: Bool { + if case .allow = self { return true } + return false + } +} + +/// HITL details for a `confirmHITL` decision. +public struct GuardrailHITLRequest: Hashable, Sendable { + public let requiredFields: [String] + public let title: String? + public let message: String? + + public init( + requiredFields: [String] = ["user_approval"], + title: String? = nil, + message: String? = nil + ) { + self.requiredFields = requiredFields + self.title = title + self.message = message + } + + public init(_ confirmation: PolicyConfirmationRequest) { + self.requiredFields = ["user_approval"] + self.title = confirmation.title + self.message = confirmation.message + } +} + +/// Something Policy can evaluate into a `GuardrailDecision`. +public protocol GuardrailEvaluating: Sendable { + func guardrailDecision() async throws -> GuardrailDecision +} diff --git a/packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLApprovalPresentationWake.swift b/packages/Structure/Sources/Guardrail/HITL/DerrickHITLApprovalPresentationWake.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLApprovalPresentationWake.swift rename to packages/Structure/Sources/Guardrail/HITL/DerrickHITLApprovalPresentationWake.swift diff --git a/packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLNotificationSignal.swift b/packages/Structure/Sources/Guardrail/HITL/DerrickHITLNotificationSignal.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/AppServices/DerrickHITLNotificationSignal.swift rename to packages/Structure/Sources/Guardrail/HITL/DerrickHITLNotificationSignal.swift diff --git a/packages/Structure/Sources/DBRepository/HITLTypes.swift b/packages/Structure/Sources/Guardrail/HITL/HITLTypes.swift similarity index 100% rename from packages/Structure/Sources/DBRepository/HITLTypes.swift rename to packages/Structure/Sources/Guardrail/HITL/HITLTypes.swift diff --git a/packages/Structure/Sources/Policy/PolicyEngine/Matchers.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Matchers.swift similarity index 93% rename from packages/Structure/Sources/Policy/PolicyEngine/Matchers.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Matchers.swift index 3b599b04..b66a5f4c 100644 --- a/packages/Structure/Sources/Policy/PolicyEngine/Matchers.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Matchers.swift @@ -157,7 +157,7 @@ public struct ToolRule: ToolPolicyRule { Self(matcher: matcher, outcome: .confirm(title: title, message: message)) } - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { guard matcher.matches(request.call) else { return nil } @@ -168,7 +168,13 @@ public struct ToolRule: ToolPolicyRule { case .deny(let reason): return .deny(reason: reason) case .confirm(let title, let message): - return .confirm(.init(title: title, message: message, call: request.call, context: request.context)) + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: ["user_approval"], + title: title, + message: message + ) + ) } } } diff --git a/packages/Structure/Sources/Policy/PolicyEngine/PolicyEngineTypes.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift similarity index 82% rename from packages/Structure/Sources/Policy/PolicyEngine/PolicyEngineTypes.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift index 43c84978..51028dc3 100644 --- a/packages/Structure/Sources/Policy/PolicyEngine/PolicyEngineTypes.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift @@ -71,16 +71,15 @@ public struct PolicyConfirmationRequest: Hashable, Sendable { self.call = call self.context = context } -} -public enum PolicyDecision: Hashable, Sendable { - case allow - case deny(reason: String) - case confirm(PolicyConfirmationRequest) + public var guardrailHITL: GuardrailHITLRequest { + GuardrailHITLRequest(self) + } } +/// In-memory tool rules used by `PolicyEngine`. Prefer store-backed `PolicyRule` for production. public protocol ToolPolicyRule: Sendable { - func evaluate(_ request: PolicyRequest) -> PolicyDecision? + func evaluate(_ request: PolicyRequest) -> GuardrailDecision? } public protocol PolicyConfirmationPresenting: Sendable { @@ -92,6 +91,8 @@ public enum PolicyError: Error, Sendable, Equatable { case cancelled } +/// In-memory rule list that produces `GuardrailDecision`. Production chat uses store-backed evaluators; +/// this engine shares the same decision vocabulary. public struct PolicyEngine: Sendable { public let rules: [any ToolPolicyRule] @@ -99,19 +100,18 @@ public struct PolicyEngine: Sendable { self.rules = rules } - public func decision(for request: PolicyRequest) -> PolicyDecision { + public func decision(for request: PolicyRequest) -> GuardrailDecision { for rule in rules { if let decision = rule.evaluate(request) { return decision } } - return .confirm( - PolicyConfirmationRequest( + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: ["user_approval"], title: "Confirm tool call", - message: "This tool may change state or trigger side effects.", - call: request.call, - context: request.context + message: "This tool may change state or trigger side effects." ) ) } diff --git a/packages/Structure/Sources/Policy/PolicyEngine/Rules.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Rules.swift similarity index 72% rename from packages/Structure/Sources/Policy/PolicyEngine/Rules.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Rules.swift index 8d798952..ce29f82f 100644 --- a/packages/Structure/Sources/Policy/PolicyEngine/Rules.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/Rules.swift @@ -7,7 +7,7 @@ public struct AllowToolNamesRule: ToolPolicyRule { self.toolNames = Set(toolNames) } - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { toolNames.contains(request.call.name) ? .allow : nil } } @@ -21,7 +21,7 @@ public struct DenyToolNamesRule: ToolPolicyRule { self.reason = reason } - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { toolNames.contains(request.call.name) ? .deny(reason: reason) : nil } } @@ -29,17 +29,16 @@ public struct DenyToolNamesRule: ToolPolicyRule { public struct ConfirmMutationRule: ToolPolicyRule { public init() {} - public func evaluate(_ request: PolicyRequest) -> PolicyDecision? { + public func evaluate(_ request: PolicyRequest) -> GuardrailDecision? { guard request.call.effects.contains(.changesState) || request.call.effects.contains(.externalSideEffects) else { return nil } - return .confirm( - PolicyConfirmationRequest( + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: ["user_approval"], title: "Confirm action", - message: "This tool call can change state or trigger side effects. Confirm before proceeding.", - call: request.call, - context: request.context + message: "This tool call can change state or trigger side effects. Confirm before proceeding." ) ) } diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterception.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift similarity index 72% rename from packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterception.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift index cdb5814e..39a4241e 100644 --- a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterception.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift @@ -1,15 +1,9 @@ import Foundation -public enum PolicyDecisionOutcome: Equatable, Sendable { - case allow - case deny(reason: String) - case confirm(requiredFields: [String]) - case redact(pattern: String, replacement: String) -} - +/// Content Policy evaluator. Returns the canonical `GuardrailDecision`. public protocol PolicyEvaluator: Sendable { - func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> PolicyDecisionOutcome - func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome + func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> GuardrailDecision + func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> GuardrailDecision } /// Result of content policy interception (preserves deny reasons for UI). @@ -24,3 +18,7 @@ public protocol PolicyInterceptor: Sendable { func interceptAssistantChunk(_ event: AssistantChunkEvent) async throws -> AssistantContentInterceptResult func interceptAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> AssistantContentInterceptResult } + +/// Historical alias — prefer `GuardrailDecision`. +@available(*, deprecated, renamed: "GuardrailDecision") +public typealias PolicyDecisionOutcome = GuardrailDecision diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterceptionEvent.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterceptionEvent.swift similarity index 100% rename from packages/Structure/Sources/Policy/PolicyRuntime/PolicyInterceptionEvent.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterceptionEvent.swift diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/PolicyStore.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyStore.swift similarity index 100% rename from packages/Structure/Sources/Policy/PolicyRuntime/PolicyStore.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyStore.swift diff --git a/packages/Structure/Sources/Policy/PolicyRuntime/ToolInterception.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift similarity index 81% rename from packages/Structure/Sources/Policy/PolicyRuntime/ToolInterception.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift index b42c9178..7333a44f 100644 --- a/packages/Structure/Sources/Policy/PolicyRuntime/ToolInterception.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift @@ -1,12 +1,5 @@ import Foundation -public enum ToolGovernanceOutcome: Equatable, Sendable { - case allow - case deny(reason: String) - case confirm(requiredFields: [String]) - case redact(argumentKey: String, pattern: String, replacement: String) -} - public enum ToolInterceptionDecision: Equatable, Sendable { case allow(ToolInvocationEvent) case deny(reason: String) @@ -19,7 +12,7 @@ public enum ToolInvocationInterceptionError: Error, Equatable, Sendable { case cancelled(reason: String) } -/// Result of asking the user to approve a tool that policy marked as `confirm`. +/// Result of asking the user to approve a tool that policy marked as `confirmHITL`. public enum ToolInvocationConfirmation: Equatable, Sendable { /// Proceed with this (possibly edited) invocation event. case approved(ToolInvocationEvent) @@ -27,8 +20,9 @@ public enum ToolInvocationConfirmation: Equatable, Sendable { case cancelled(actor: String?) } +/// Tool Policy evaluator. Returns the canonical `GuardrailDecision`. public protocol ToolGovernancePolicy: Sendable { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome + func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> GuardrailDecision } public protocol ToolRequestInterceptor: Sendable { @@ -40,10 +34,14 @@ public protocol ToolRequestInterceptor: Sendable { /// Control flow: /// - allow / redact → `proceed(processedEvent)` /// - deny → throws `ToolInvocationInterceptionError.denied` - /// - confirm → `confirm(...)`; on approve → `proceed`; on cancel → throws `.cancelled` + /// - confirmHITL → `confirm(...)`; on approve → `proceed`; on cancel → throws `.cancelled` func interceptAndRun( _ event: ToolInvocationEvent, confirm: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent, [String]) async throws -> ToolInvocationConfirmation, proceed: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent) async throws -> R ) async throws -> R } + +/// Historical alias — prefer `GuardrailDecision`. +@available(*, deprecated, renamed: "GuardrailDecision") +public typealias ToolGovernanceOutcome = GuardrailDecision diff --git a/packages/Structure/Sources/Policy/PolicyUserInteraction/PolicyUserEvent.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyUserInteraction/PolicyUserEvent.swift similarity index 100% rename from packages/Structure/Sources/Policy/PolicyUserInteraction/PolicyUserEvent.swift rename to packages/Structure/Sources/Guardrail/Policy/PolicyUserInteraction/PolicyUserEvent.swift diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/EffectorAdmissionPolicy.swift b/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift similarity index 58% rename from packages/Structure/Sources/AppLayerServices/MCPService/EffectorAdmissionPolicy.swift rename to packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift index 5604b30b..3c544f5c 100644 --- a/packages/Structure/Sources/AppLayerServices/MCPService/EffectorAdmissionPolicy.swift +++ b/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift @@ -1,7 +1,20 @@ import Foundation -/// MCP effector admission from signed `ExecutionContextWire` (not process-local flags). +/// MCP effector admission under Guardrail. Policy vocabulary: allow or deny. +/// +/// Effectors are MCP-hosted side-effect tools (`web.crawl`, `script_exec`, …). +/// Admission belongs to Guardrail/Policy, not to plugins. public enum EffectorAdmissionPolicy: Sendable { + public static func syncWebCrawlDecision( + context: ExecutionContextWire?, + principal: ServicePrincipal + ) -> GuardrailDecision { + if allowsSyncWebCrawl(context: context, principal: principal) { + return .allow + } + return .deny(reason: "Sync web crawl is not admitted for this execution context.") + } + public static func allowsSyncWebCrawl( context: ExecutionContextWire?, principal: ServicePrincipal diff --git a/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift new file mode 100644 index 00000000..50f5e4a1 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift @@ -0,0 +1,40 @@ +import Foundation + +/// Guardrail admission for starting a workflow. Callers propose; Policy decides. +/// +/// Derrick-owned kinds are allowed. `pluginFactoryEdit` stays denied until editability ships. +public enum WorkflowAdmissionPolicy: Sendable { + public static func decision(for request: WorkflowStartRequest) -> GuardrailDecision { + decision(kind: request.kind) + } + + public static func decision(kind: WorkflowKind) -> GuardrailDecision { + switch kind { + case .pluginFactoryCreate, .connectorAuthDiscover, .jobStep, .interactiveTool: + return .allow + case .pluginFactoryEdit: + return .deny( + reason: "Plugin edit workflow is reserved and not enabled yet." + ) + case .none: + return .deny(reason: "A workflow kind is required.") + } + } + + /// Throws `WorkflowRuntimeError.deniedByGuardrail` unless the decision is `.allow`. + public static func admitOrThrow(_ request: WorkflowStartRequest) throws { + switch decision(for: request) { + case .allow: + return + case .deny(let reason): + throw WorkflowRuntimeError.deniedByGuardrail(reason) + case .confirmHITL(let hitl): + let detail = hitl.message ?? hitl.title ?? "Workflow start requires confirmation." + throw WorkflowRuntimeError.deniedByGuardrail(detail) + case .requireWorkflow, .redactContent, .redactArgument: + throw WorkflowRuntimeError.deniedByGuardrail( + "Workflow start returned an unsupported Guardrail decision." + ) + } + } +} diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/WorkflowChatProgress.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowChatProgress.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/MCPService/WorkflowChatProgress.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowChatProgress.swift diff --git a/packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift new file mode 100644 index 00000000..3047ed3b --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Workflow/WorkflowKind.swift @@ -0,0 +1,13 @@ +import Foundation + +/// Sequenced procedures the harness can run. Policy may require one; plugins never authorize them. +/// +/// `pluginFactoryEdit` is a reserved placeholder for future plugin editability. +public enum WorkflowKind: String, Codable, Sendable, Hashable, CaseIterable { + case pluginFactoryCreate = "plugin_factory_create" + case pluginFactoryEdit = "plugin_factory_edit" + case connectorAuthDiscover = "connector_auth_discover" + case jobStep = "job_step" + case interactiveTool = "interactive_tool" + case none +} diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeDTOs.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeDTOs.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeDTOs.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeDTOs.swift diff --git a/packages/Structure/Sources/DBRepository/WorkflowRuntimeRows.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeRows.swift similarity index 100% rename from packages/Structure/Sources/DBRepository/WorkflowRuntimeRows.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeRows.swift diff --git a/packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeXPC.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeXPC.swift similarity index 100% rename from packages/Structure/Sources/AppLayerServices/MCPService/WorkflowRuntimeXPC.swift rename to packages/Structure/Sources/Guardrail/Workflow/WorkflowRuntimeXPC.swift diff --git a/packages/Structure/Sources/Structure/Architecture.swift b/packages/Structure/Sources/Structure/Architecture.swift index 0f48f2ba..f2785397 100644 --- a/packages/Structure/Sources/Structure/Architecture.swift +++ b/packages/Structure/Sources/Structure/Architecture.swift @@ -1,3 +1,3 @@ /// Root marker for the Structure architecture module. -/// Domain types live in sibling folders (`AppLayerServices`, `Contract`, `Plugin`, `Policy`, …). +/// Domain types live in sibling folders (`AppLayerServices`, `Contract`, `Plugin`, `Guardrail`, …). public enum Architecture {} diff --git a/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift new file mode 100644 index 00000000..2a674722 --- /dev/null +++ b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift @@ -0,0 +1,66 @@ +import Testing +@testable import Structure + +@Suite("Guardrail") +struct GuardrailDecisionTests { + @Test func effectorAdmissionUsesGuardrailDecision() { + let denied = EffectorAdmissionPolicy.syncWebCrawlDecision( + context: nil, + principal: .ui + ) + #expect(!denied.isAllow) + if case .deny = denied { + #expect(true) + } else { + Issue.record("Expected deny for UI without crawl capability") + } + + let allowed = EffectorAdmissionPolicy.syncWebCrawlDecision( + context: nil, + principal: .job(jobID: "job-1") + ) + #expect(allowed == .allow) + } + + @Test func policyEngineSharesGuardrailVocabulary() { + let engine = PolicyEngine(rules: [ + DenyToolNamesRule(toolNames: ["x"], reason: "no") + ]) + let decision = engine.decision(for: .init(call: .init(name: "x"), context: .init(agentID: "a"))) + #expect({ + if case .deny = decision { return true } + return false + }()) + } + + @Test func pluginFactoryEditKindRemainsPlaceholder() { + #expect(WorkflowKind.pluginFactoryEdit.rawValue == "plugin_factory_edit") + #expect(WorkflowKind.allCases.contains(.pluginFactoryEdit)) + } + + @Test func workflowAdmissionAllowsCreateAndDeniesEditPlaceholder() throws { + #expect(WorkflowAdmissionPolicy.decision(kind: .pluginFactoryCreate) == .allow) + #expect(WorkflowAdmissionPolicy.decision(kind: .connectorAuthDiscover) == .allow) + + let edit = WorkflowAdmissionPolicy.decision(kind: .pluginFactoryEdit) + #expect(!edit.isAllow) + + let request = WorkflowStartRequest( + kind: .pluginFactoryEdit, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + do { + try WorkflowAdmissionPolicy.admitOrThrow(request) + Issue.record("edit placeholder must be denied") + } catch let error as WorkflowRuntimeError { + if case .deniedByGuardrail = error { + #expect(true) + } else { + Issue.record("Expected deniedByGuardrail, got \(error)") + } + } + } +} diff --git a/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift b/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift index 2f90f878..346cbbcc 100644 --- a/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift +++ b/packages/Structure/Tests/StructureTests/PolicyEngineTests.swift @@ -25,8 +25,8 @@ struct PolicyEngineTests { let decision = engine.decision(for: .init(call: .init(name: "writeFile", effects: [.changesState]), context: .init(agentID: "a"))) #expect({ - if case .confirm(let request) = decision { - return request.title == "Confirm action" + if case .confirmHITL(let hitl) = decision { + return hitl.title == "Confirm action" } return false }()) diff --git a/readme.md b/readme.md index c642a268..0b9c4b8f 100644 --- a/readme.md +++ b/readme.md @@ -45,7 +45,7 @@ A native Swift macOS desktop agent: chat, just-in-time software (plugins), messa - **Docker** runs untrusted Go for `script_exec`, just-in-time plugin builds, and approved plugin invocations. - **HostUI** (`packages/HostUI`) paints plugin screens from a declared schema. The host does not invent a default inbox. -See [docs/adr-headless-backend.md](docs/adr-headless-backend.md) and [docs/services-plan.md](docs/services-plan.md). +See [docs/Design.md](docs/Design.md) for the current architecture notes. ## Security model @@ -58,7 +58,6 @@ Derrick treats model output and guest code as untrusted. - Canonical I/O types live in `packages/Structure/Sources/Contract/Resources/schemas/` and are mirrored to `workers/go/internal/contract/schemas/`. - No net/http, subprocess, filesystem access, or credentials inside the guest. - The host dispatches `http.request` envelopes, attaches secrets, and enforces egress policy. -- Historical Swift guest notes: [docs/adr-swift-script-runtime.md](docs/adr-swift-script-runtime.md). ### Script review agent @@ -95,15 +94,16 @@ Copy [.env.example](.env.example) — **never commit `.env`**. - Plugin credential collection (Keychain save) - Policy events (usage limits, content sensitivity) -### Policy +### Guardrail -- **`PolicyEngine`** — in-process tool-call rules for chat (`Structure/Policy/PolicyEngine`). -- **`PolicyRuntime`** — persisted rules from SQLite (`Structure/Policy/PolicyRuntime`); implemented by `StoreBacked*` evaluators in `packages/PolicyRuntime`. -- **`PolicyInterceptor`** / **`ToolRequestInterceptor`** (MemorySystem) — pipeline hooks that call those policies. +- **`Structure/Guardrail`** — control plane umbrella. Policy decides; HITL and workflows enforce. +- Persisted rules in SQLite (`Guardrail/Policy`, `packages/PolicyRuntime`) produce `GuardrailDecision`. +- **`PolicyInterceptor`** / **`ToolRequestInterceptor`** (MemorySystem) — pipeline hooks that apply those decisions. +- Workflow starts are admitted by Guardrail (`WorkflowAdmissionPolicy`) before the runtime runs them. ### Messaging -Connectors are just-in-time software with `role: connector`. They must emit a HostUI tree (`ui.present`) — copy `messaging_inbox` or compose their own. Messages live in SQLite; sync and send still go through the guest. See [docs/messaging-design.md](docs/messaging-design.md). +Connectors are just-in-time software with `role: connector`. They must emit a HostUI tree (`ui.present`) — copy `messaging_inbox` or compose their own. Messages live in SQLite; sync and send still go through the guest. ## Repository layout @@ -112,12 +112,12 @@ Connectors are just-in-time software with `role: connector`. They must emit a Ho | `ui/` | macOS app, Login Item daemon, XPC services | | `packages/DBRepository` | SQLite schema, migrations, messaging tables | | `packages/MCPServer` | MCP bridge, script execution, plugin runtime | -| `packages/Structure` | Architecture map: wire types, protocols, JSON schemas (`AppLayerServices/`, `Policy/`, `Plugin/`, `Contract/`, …) | +| `packages/Structure` | Architecture map: wire types, protocols, JSON schemas (`AppLayerServices/`, `Guardrail/`, `Plugin/`, `Contract/`, …) | | `packages/HostUI` | Schema-driven screens for connectors and plugin present trees | | `packages/Plugin` | Just-in-time software factory (builder, reviewer, release) | | `packages/DerrickBackend` | Daemon runtime, notifications, HITL polling | | `packages/DockerRunnerXPC` | Constrained Docker helper | -| `packages/PolicyRuntime` | Store-backed policy evaluators (`Structure/Policy/PolicyRuntime`) | +| `packages/PolicyRuntime` | Store-backed policy evaluators (`Structure/Guardrail/Policy`) | | `packages/LLMAgentClient` | Provider clients (OpenAI, Gemini, …) | ## Quick start @@ -141,11 +141,11 @@ Connectors are just-in-time software with `role: connector`. They must emit a Ho Or from the terminal: `./scripts/build.sh test` -See [CONTRIBUTING.md](CONTRIBUTING.md) and [docs/development.md](docs/development.md). +See [CONTRIBUTING.md](CONTRIBUTING.md) and [docs/Design.md](docs/Design.md). ## Open-sourcing checklist -We maintain [docs/opensource-plan.md](docs/opensource-plan.md) for pre-release cleanup (secret audit, personal reference removal, CI). +Pre-release cleanup: secret audit (`./scripts/verify-no-secrets.sh`), personal reference removal, and CI hygiene. Verify no secrets in git: @@ -155,10 +155,7 @@ Verify no secrets in git: ## Documentation -- [Headless backend ADR](docs/adr-headless-backend.md) -- [Swift Docker runtime ADR](docs/adr-swift-script-runtime.md) -- [Background services plan](docs/services-plan.md) -- [Messaging design](docs/messaging-design.md) +- [Design](docs/Design.md) - [Security policy](SECURITY.md) - [Third-party notices](THIRD_PARTY_NOTICES.md) - [Code of Conduct](CODE_OF_CONDUCT.md) diff --git a/scripts/reset-local-state.sh b/scripts/reset-local-state.sh index fb845c65..6c0c22df 100755 --- a/scripts/reset-local-state.sh +++ b/scripts/reset-local-state.sh @@ -2,23 +2,32 @@ set -euo pipefail ROOT="$(cd "$(dirname "$0")/.." && pwd)" -APP_GROUP="$HOME/Library/Group Containers/VUSK4B2YKQ.derrick.shared" + +# Same search order as DerrickAppSupport.preferredDatabaseParentDirectories(). +DB_ROOTS=( + "$HOME/Library/Group Containers/VUSK4B2YKQ.derrick.shared" + "$HOME/Library/Containers/derrick.ui/Data/Library/Application Support" + "$HOME/Library/Application Support" +) echo "==> Quit Derrick before resetting local state." echo "==> This removes all local SQLite data (chats, plugins, messaging, credentials in DB)." echo "==> Keychain plugin secrets are not removed." removed_dbs=0 -if [[ -d "$APP_GROUP" ]]; then +for root in "${DB_ROOTS[@]}"; do + if [[ ! -d "$root" ]]; then + continue + fi while IFS= read -r db; do rm -f "$db" "${db}-wal" "${db}-shm" echo "removed $(basename "$db") at ${db%/*}" removed_dbs=$((removed_dbs + 1)) - done < <(find "$APP_GROUP" -name 'derrick.sqlite3' 2>/dev/null) -fi + done < <(find "$root" -name 'derrick.sqlite3' 2>/dev/null) +done if [[ "$removed_dbs" -eq 0 ]]; then - echo "no derrick.sqlite3 files found under $APP_GROUP" + echo "no derrick.sqlite3 files found under preferred database roots" fi if command -v docker >/dev/null 2>&1; then @@ -44,5 +53,5 @@ else fi echo -echo "Done. Reopen Derrick from $ROOT (go-workers) to recreate an empty database." +echo "Done. Reopen Derrick from $ROOT to recreate an empty database." echo "Policy rules seed automatically on first UI launch." diff --git a/todo.rtf b/todo.rtf deleted file mode 100644 index e8e4ba7c..00000000 --- a/todo.rtf +++ /dev/null @@ -1,8 +0,0 @@ -{\rtf1\ansi\ansicpg1252\cocoartf2870 -\cocoatextscaling0\cocoaplatform0{\fonttbl\f0\fswiss\fcharset0 Helvetica;} -{\colortbl;\red255\green255\blue255;} -{\*\expandedcolortbl;;} -\margl1440\margr1440\vieww11520\viewh8400\viewkind0 -\pard\tx720\tx1440\tx2160\tx2880\tx3600\tx4320\tx5040\tx5760\tx6480\tx7200\tx7920\tx8640\pardirnatural\partightenfactor0 - -\f0\fs24 \cf0 - create a content side panel that shows a todo or table of contents for long ongoing discussions} \ No newline at end of file diff --git a/ui/MCPService/MCPServiceToolHost.swift b/ui/MCPService/MCPServiceToolHost.swift index 3063ce97..b06f6779 100644 --- a/ui/MCPService/MCPServiceToolHost.swift +++ b/ui/MCPService/MCPServiceToolHost.swift @@ -301,6 +301,40 @@ actor MCPServiceToolHost { HostHTTPCallContext.shared.clear() } + if toolName == "web.crawl" { + let context = EffectorAdmissionPolicy.parseContextJSON(request.executionContextJSON) + switch EffectorAdmissionPolicy.syncWebCrawlDecision( + context: context, + principal: request.principal + ) { + case .allow: + break + case .deny(let reason): + await MCPServiceStore.shared.log( + level: .error, + message: "web.crawl denied by Guardrail: \(reason)", + code: "tool_denied", + detailJSON: #"{"requestID":"\#(request.requestID)"}"# + ) + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: reason + ) + case .confirmHITL, .requireWorkflow, .redactContent, .redactArgument: + let message = "web.crawl returned an unsupported Guardrail decision." + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: message + ) + } + } + // Shared Lib parser (same as Agent policy path) — handles repaired model JSON. // `{}` is a valid empty object for tools with no required args. let args: [String: Value] From 5fecf071c7e960a5a9b750100e7a0b453ac95bc2 Mon Sep 17 00:00:00 2001 From: David Choi Date: Sat, 26 Sep 2026 19:25:59 -0400 Subject: [PATCH 2/3] Route workflow and MCP gates through Policy rules. Replace hardcoded workflow/effector allowlists with store-backed Policy evaluation; thin adapters apply allow, deny, confirmHITL, and redact at WorkflowRuntimeEngine and MCPService. Co-authored-by: Cursor --- docs/Design.md | 17 +- packages/DerrickBackend/Package.swift | 4 +- .../WorkflowRuntimeEngine.swift | 71 ++++++- .../WorkflowIntegrationTests.swift | 86 +++++---- .../StoreBackedWorkflowAdmissionPolicy.swift | 118 ++++++++++++ ...reBackedWorkflowAdmissionPolicyTests.swift | 90 +++++++++ .../Policy/DefaultGuardrailPolicySeeds.swift | 72 +++++++ .../Workflow/EffectorAdmissionPolicy.swift | 30 +-- .../Workflow/WorkflowAdmissionPolicy.swift | 49 +++-- .../AppLayerServicesWireTests.swift | 38 +--- .../GuardrailDecisionTests.swift | 63 +++--- readme.md | 2 +- ui/MCPService/MCPServiceStore.swift | 45 ++++- ui/MCPService/MCPServiceToolHost.swift | 180 +++++++++++++++--- .../Conversation/ConversationModel.swift | 2 +- 15 files changed, 673 insertions(+), 194 deletions(-) create mode 100644 packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift create mode 100644 packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift create mode 100644 packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift diff --git a/docs/Design.md b/docs/Design.md index 1bb48bdd..4616e1f7 100644 --- a/docs/Design.md +++ b/docs/Design.md @@ -11,9 +11,14 @@ This application is Protocol first. All major features must have a Protocol and ## Guardrail Guardrail is Derrick's control plane in Structure (`Sources/Guardrail`). -- **Policy decides** allow / deny / confirm HITL / require workflow / redact (`GuardrailDecision`). -- **HITL** and **workflows** are enforcement shapes for those decisions. -- **Plugins** propose capabilities; they do not authorize control outcomes. -- Workflow starts are admitted by `WorkflowAdmissionPolicy` inside `WorkflowRuntimeEngine` (UI proposes; Guardrail authorizes). -- Effector admission (for example sync `web.crawl`) uses the same decision vocabulary and is enforced in MCPService before the tool runs. -- `WorkflowKind.pluginFactoryEdit` is reserved for future plugin editability and is denied until enabled. +Flow: + +1. **Guardrail** — logical container +2. **Policy engine** — store-backed rules decide allow / deny / confirm HITL / require workflow / redact +3. **Enforcement** — those decisions gate workflow starts, MCP effector/tool calls, HITL confirmations, and content redaction (e.g. PII) + +- Workflow starts: `StoreBackedWorkflowAdmissionPolicy` (`workflow_start` scope) → `WorkflowAdmissionPolicy.apply` in `WorkflowRuntimeEngine` (allow / deny / confirmHITL then resume or cancel). +- MCP tools/effectors: `StoreBackedToolGovernancePolicy` (`tool_invocation`) in the chat pipeline and again in MCPService before the tool runs. +- Content: same engine via `StoreBackedCompletionContentPolicy`. +- `WorkflowKind.pluginFactoryEdit` is denied by a Policy rule until editability ships. +- Plugins propose work; they do not authorize control outcomes. diff --git a/packages/DerrickBackend/Package.swift b/packages/DerrickBackend/Package.swift index e38d7ba7..6b7fea31 100644 --- a/packages/DerrickBackend/Package.swift +++ b/packages/DerrickBackend/Package.swift @@ -15,6 +15,7 @@ let package = Package( .package(path: "../DBRepository"), .package(path: "../DockerRunnerXPC"), .package(path: "../Plugin"), + .package(path: "../PolicyRuntime"), ], targets: [ .target( @@ -24,6 +25,7 @@ let package = Package( "DBRepository", "DockerRunnerXPC", "Plugin", + "PolicyRuntime", ], path: "Sources/DerrickBackend", swiftSettings: [ @@ -32,7 +34,7 @@ let package = Package( ), .testTarget( name: "DerrickBackendTests", - dependencies: ["DerrickBackend", "DBRepository", "Plugin", "Structure"], + dependencies: ["DerrickBackend", "DBRepository", "Plugin", "Structure", "PolicyRuntime"], path: "Tests/DerrickBackendTests", swiftSettings: [ .enableUpcomingFeature("ApproachableConcurrency") diff --git a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift index 141654a6..bb930d2c 100644 --- a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift +++ b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift @@ -1,5 +1,6 @@ import DBRepository import Foundation +import PolicyRuntime import Structure /// Durable workflow coordinator (Process Manager) running inside derrickd. @@ -15,8 +16,23 @@ public actor WorkflowRuntimeEngine { _ request: WorkflowStartRequest, repositoryProvider: @escaping @Sendable () async throws -> DBRepository ) async throws -> WorkflowHandleDTO { - try WorkflowAdmissionPolicy.admitOrThrow(request) let repo = try await repositoryProvider() + try await DefaultGuardrailPolicySeeds.seedWorkflowStartRulesIfNeeded( + store: repo, + applicationName: DerrickAppSupport.defaultApplicationName + ) + let decision = try await StoreBackedWorkflowAdmissionPolicy( + store: repo, + applicationName: DerrickAppSupport.defaultApplicationName + ).evaluate(request) + try await WorkflowAdmissionPolicy.apply(decision) { + guard case .confirmHITL(let hitl) = decision else { return true } + return await awaitWorkflowStartHITL( + request: request, + repository: repo, + hitl: hitl + ) + } let idempotencyKey = WorkflowRuntimeIdempotency.key( sessionID: request.sessionID, kind: request.kind, @@ -337,4 +353,57 @@ public actor WorkflowRuntimeEngine { message: message ) } + + private static let workflowHITLPollNanoseconds: UInt64 = 1_000_000_000 + private static let workflowHITLTimeoutNanoseconds: UInt64 = 15 * 60 * 1_000_000_000 + + /// Present HITL for a workflow_start confirm decision; returns true when approved. + private func awaitWorkflowStartHITL( + request: WorkflowStartRequest, + repository: DBRepository, + hitl: GuardrailHITLRequest + ) async -> Bool { + let approvalID = UUID().uuidString + let requiredJSON = (try? JSONEncoder().encode(hitl.requiredFields)) + .flatMap { String(data: $0, encoding: .utf8) } ?? "[]" + let isJob: Bool = { + if case .job = request.principal { return true } + return false + }() + let row = PendingHITLApprovalRow( + id: approvalID, + turnID: request.turnID ?? request.sessionID, + sessionID: request.sessionID, + toolName: "workflow_start:\(request.kind.rawValue)", + argumentsJSON: request.inputJSON, + requiredFieldsJSON: requiredJSON, + isJobContext: isJob + ) + do { + try await repository.insertPendingHITLApproval(row) + } catch { + fputs("[workflow] HITL persist failed: \(error.localizedDescription)\n", stderr) + return false + } + DerrickHITLNotificationSignal.postPoll() + + let deadline = Date().addingTimeInterval( + Double(Self.workflowHITLTimeoutNanoseconds) / 1_000_000_000 + ) + while Date() < deadline { + if Task.isCancelled { return false } + if let decision = try? await repository.fetchPendingHITLApproval(id: approvalID), + decision.status != .pending { + return decision.status == .approved + } + try? await Task.sleep(nanoseconds: Self.workflowHITLPollNanoseconds) + } + try? await repository.resolveHITLApproval( + id: approvalID, + status: .timeout, + editedArgumentsJSON: nil, + actor: "system-timeout" + ) + return false + } } diff --git a/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift b/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift index 85382f6b..c0b8338f 100644 --- a/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift +++ b/packages/DerrickBackend/Tests/DerrickBackendTests/WorkflowIntegrationTests.swift @@ -123,66 +123,70 @@ import Testing ) let repository = DBRepository(configuration: configuration) _ = try await repository.createEmptyDatabaseIfNeeded(username: "app-user", password: "app-secret") + try await DefaultGuardrailPolicySeeds.seedWorkflowStartRulesIfNeeded( + store: repository, + applicationName: "ui" + ) let previousMCP = InProcessServiceBridges.mcpCallTool defer { InProcessServiceBridges.mcpCallTool = previousMCP } - InProcessServiceBridges.mcpCallTool = { request in - switch request.toolName { - case "web.crawl": - let pages: String - if request.argumentsJSON.contains("agent-plugins.org") { - pages = #"{"pages":[{"url":"https://agent-plugins.org/specification","title":"Spec","text":"plugin.json and skills/SKILL.md are required."}]}"# - } else { - pages = #"{"pages":[{"url":"https://api.slack.com/docs","title":"Slack","text":"auth"}]}"# + // Hold the first run in `running` until after the second start (dedupe only matches running). + actor ReleaseGate { + private var released = false + private var waiters: [CheckedContinuation] = [] + + func wait() async { + if released { return } + await withCheckedContinuation { (c: CheckedContinuation) in + waiters.append(c) } - let outcome = try ToolExecutionOutcome.completed( - output: ToolExecutionOutcome.Output(format: .json, value: pages) - ).encodedJSON() - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: true, - isError: false, - text: outcome - ) - case "plugin_factory_build": - let receipt = """ - {"plugin_id":"slack-connector","version":"1.0.0","content_hash":"abc","review_summary":"ok","secrets":[]} - """ - let outcome = try ToolExecutionOutcome.completed( - output: ToolExecutionOutcome.Output(format: .json, value: receipt) - ).encodedJSON() - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: true, - isError: false, - text: outcome - ) - default: - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: false, - isError: true, - text: "", - message: "unexpected tool \(request.toolName)" - ) } + + func release() { + released = true + for waiter in waiters { + waiter.resume() + } + waiters.removeAll() + } + } + let gate = ReleaseGate() + + InProcessServiceBridges.mcpCallTool = { request in + await gate.wait() + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: "held for dedupe test" + ) } + let inputJSON = try PluginFactoryCreateInput.makeConnector( + vendor: .slack, + scope: .fullSync, + userDescription: "Post alerts." + ).encodedJSON() let request = WorkflowStartRequest( kind: .pluginFactoryCreate, - sessionID: "session-1", + sessionID: "session-dedupe", agentID: "ui", - inputJSON: "slack connector", - principal: .agent(sessionID: "session-1", agentID: "ui") + inputJSON: inputJSON, + principal: .agent(sessionID: "session-dedupe", agentID: "ui") ) let provider: @Sendable () async throws -> DBRepository = { repository } let first = try await WorkflowRuntimeEngine.shared.startWorkflow(request, repositoryProvider: provider) + // Give the run task a turn to reach the gated MCP call while still running. + try await Task.sleep(nanoseconds: 50_000_000) let second = try await WorkflowRuntimeEngine.shared.startWorkflow(request, repositoryProvider: provider) #expect(first.deduplicated == false) #expect(second.deduplicated == true) #expect(first.workflowID == second.workflowID) + await gate.release() + var status = WorkflowRunStatus.running for _ in 0..<50 { try await Task.sleep(nanoseconds: 100_000_000) diff --git a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift new file mode 100644 index 00000000..0ae3c9fd --- /dev/null +++ b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift @@ -0,0 +1,118 @@ +import Foundation +import Structure + +/// Store-backed workflow start admission. Scope: `workflow_start`. +public struct StoreBackedWorkflowAdmissionPolicy: Sendable { + private let store: any PolicyStore + private let applicationName: String + + public init(store: any PolicyStore, applicationName: String) { + self.store = store + self.applicationName = applicationName + } + + public func evaluate(_ request: WorkflowStartRequest) async throws -> GuardrailDecision { + let rules = try await loadRules() + guard !rules.isEmpty else { + return .deny(reason: Self.noRulesConfiguredReason) + } + + for rule in rules { + guard rule.enabled else { continue } + guard let matcher = try? decode(WorkflowMatcher.self, from: rule.matcherJSON) else { + continue + } + guard matcher.matches(kind: request.kind) else { + continue + } + guard let outcome = try? decode(WorkflowOutcomeRule.self, from: rule.outcomeJSON) else { + continue + } + return outcome.decision + } + + return .deny(reason: Self.noMatchingRuleReason) + } + + public static let noRulesConfiguredReason = + "No workflow_start rules are configured; denying by default." + + public static let noMatchingRuleReason = + "No workflow_start rule matched this kind; denying by default." + + private func loadRules() async throws -> [PolicyRule] { + try await store.loadRules(applicationName: applicationName, scope: "workflow_start") + .filter(\.enabled) + .sorted { lhs, rhs in + if lhs.priority != rhs.priority { return lhs.priority > rhs.priority } + return lhs.createdAt > rhs.createdAt + } + } + + private func decode(_ type: T.Type, from json: String) throws -> T { + guard let data = json.data(using: .utf8) else { + throw DecodingError.dataCorrupted( + .init(codingPath: [], debugDescription: "Invalid UTF-8 in policy JSON.") + ) + } + return try JSONDecoder().decode(T.self, from: data) + } +} + +private struct WorkflowMatcher: Decodable { + let workflowKind: String? + let workflowKindAny: [String]? + + enum CodingKeys: String, CodingKey { + case workflowKind = "workflow_kind" + case workflowKindAny = "workflow_kind_any" + } + + func matches(kind: WorkflowKind) -> Bool { + if let workflowKind, kind.rawValue != workflowKind { + return false + } + if let workflowKindAny { + guard workflowKindAny.contains(kind.rawValue) else { return false } + } + return true + } +} + +private struct WorkflowOutcomeRule: Decodable { + let action: String + let reason: String? + let requiredFields: [String]? + let title: String? + let message: String? + + enum CodingKeys: String, CodingKey { + case action + case reason + case requiredFields = "required_fields" + case title + case message + } + + var decision: GuardrailDecision { + switch action.lowercased() { + case "allow": + return .allow + case "deny": + return .deny(reason: reason ?? "Workflow start denied by policy.") + case "confirm": + return .confirmHITL( + GuardrailHITLRequest( + requiredFields: requiredFields ?? ["user_approval"], + title: title, + message: message + ) + ) + case "require_workflow": + // Already starting a workflow; treat as deny. + return .deny(reason: reason ?? "require_workflow is not valid for workflow_start.") + default: + return .deny(reason: "Unknown workflow_start action '\(action)'; denying by default.") + } + } +} diff --git a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift new file mode 100644 index 00000000..2301835e --- /dev/null +++ b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift @@ -0,0 +1,90 @@ +import XCTest +import Structure +@testable import PolicyRuntime + +final class StoreBackedWorkflowAdmissionPolicyTests: XCTestCase { + func test_allowsSeededCreateAndDeniesEdit() async throws { + let store = MockWorkflowPolicyStore() + for rule in DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: "ui") { + try await store.saveRule(rule) + } + let policy = StoreBackedWorkflowAdmissionPolicy(store: store, applicationName: "ui") + + let create = try await policy.evaluate( + WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + ) + XCTAssertEqual(create, .allow) + + let edit = try await policy.evaluate( + WorkflowStartRequest( + kind: .pluginFactoryEdit, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + ) + guard case .deny = edit else { + return XCTFail("edit placeholder must be denied by Policy rule") + } + } + + func test_confirmOutcomeReturnsHITL() async throws { + let store = MockWorkflowPolicyStore(rulesByScope: [ + "workflow_start": [ + PolicyRule( + applicationName: "ui", + name: "confirm-create", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"plugin_factory_create"}"#, + outcomeJSON: #"{"action":"confirm","required_fields":["review"],"title":"Approve create"}"# + ) + ] + ]) + let policy = StoreBackedWorkflowAdmissionPolicy(store: store, applicationName: "ui") + let decision = try await policy.evaluate( + WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + ) + guard case .confirmHITL(let hitl) = decision else { + return XCTFail("expected confirmHITL") + } + XCTAssertEqual(hitl.requiredFields, ["review"]) + XCTAssertEqual(hitl.title, "Approve create") + } +} + +private final class MockWorkflowPolicyStore: PolicyStore, @unchecked Sendable { + private var rulesByScope: [String: [PolicyRule]] + + init(rulesByScope: [String: [PolicyRule]] = [:]) { + self.rulesByScope = rulesByScope + } + + func loadRules(applicationName: String, scope: String) async throws -> [PolicyRule] { + rulesByScope[scope] ?? [] + } + + func saveRule(_ rule: PolicyRule) async throws { + var rules = rulesByScope[rule.scope] ?? [] + rules.removeAll { $0.name == rule.name } + rules.append(rule) + rulesByScope[rule.scope] = rules + } + + func saveApproval(_ approval: PolicyApproval) async throws {} + func loadApprovals(sessionID: String, limit: Int) async throws -> [PolicyApproval] { [] } + func logAuditEntry(_ entry: PolicyAuditLogEntry) async throws {} + func auditLog(sessionID: String, limit: Int, page: Int) async throws -> [PolicyAuditLogEntry] { [] } +} diff --git a/packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift b/packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift new file mode 100644 index 00000000..9f2efb4a --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Policy/DefaultGuardrailPolicySeeds.swift @@ -0,0 +1,72 @@ +import Foundation + +/// Baseline Guardrail Policy rules Derrick seeds on first launch. +public enum DefaultGuardrailPolicySeeds: Sendable { + /// `workflow_start` rules: allow Derrick-owned kinds; deny edit placeholder and `none`. + public static func workflowStartRules(applicationName: String) -> [PolicyRule] { + [ + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-plugin-factory-create", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"plugin_factory_create"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-connector-auth-discover", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"connector_auth_discover"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-job-step", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"job_step"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "allow-workflow-interactive-tool", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"interactive_tool"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 100 + ), + PolicyRule( + applicationName: applicationName, + name: "deny-workflow-plugin-factory-edit", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"plugin_factory_edit"}"#, + outcomeJSON: #"{"action":"deny","reason":"Plugin edit workflow is reserved and not enabled yet."}"#, + priority: 1000 + ), + PolicyRule( + applicationName: applicationName, + name: "deny-workflow-none", + scope: "workflow_start", + matcherJSON: #"{"workflow_kind":"none"}"#, + outcomeJSON: #"{"action":"deny","reason":"A workflow kind is required."}"#, + priority: 1000 + ), + ] + } + + /// Idempotently insert `workflow_start` rules into a Policy store. + public static func seedWorkflowStartRulesIfNeeded( + store: any PolicyStore, + applicationName: String + ) async throws { + for rule in workflowStartRules(applicationName: applicationName) { + let existing = try await store.loadRules(applicationName: applicationName, scope: rule.scope) + guard existing.contains(where: { $0.name == rule.name }) == false else { + continue + } + try await store.saveRule(rule) + } + } +} diff --git a/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift b/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift index 3c544f5c..5b4bf0e0 100644 --- a/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift +++ b/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift @@ -1,35 +1,7 @@ import Foundation -/// MCP effector admission under Guardrail. Policy vocabulary: allow or deny. -/// -/// Effectors are MCP-hosted side-effect tools (`web.crawl`, `script_exec`, …). -/// Admission belongs to Guardrail/Policy, not to plugins. +/// Helpers for effector execution context. Admission itself is Policy (`tool_invocation` rules). public enum EffectorAdmissionPolicy: Sendable { - public static func syncWebCrawlDecision( - context: ExecutionContextWire?, - principal: ServicePrincipal - ) -> GuardrailDecision { - if allowsSyncWebCrawl(context: context, principal: principal) { - return .allow - } - return .deny(reason: "Sync web crawl is not admitted for this execution context.") - } - - public static func allowsSyncWebCrawl( - context: ExecutionContextWire?, - principal: ServicePrincipal - ) -> Bool { - switch principal { - case .job, .agent: - return true - default: - break - } - if let context, context.capabilities.contains(.syncWebCrawl) { return true } - if let context, context.workflow?.kind == .pluginFactoryCreate { return true } - return false - } - public static func parseContextJSON(_ json: String?) -> ExecutionContextWire? { guard let json, !json.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { diff --git a/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift index 50f5e4a1..14934035 100644 --- a/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift +++ b/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift @@ -1,39 +1,38 @@ import Foundation -/// Guardrail admission for starting a workflow. Callers propose; Policy decides. +/// Thin Guardrail adapter for workflow starts. /// -/// Derrick-owned kinds are allowed. `pluginFactoryEdit` stays denied until editability ships. +/// Policy owns the decision (`StoreBackedWorkflowAdmissionPolicy` / seeded `workflow_start` rules). +/// This type only applies that decision: allow, deny, or confirm HITL then continue/cancel. public enum WorkflowAdmissionPolicy: Sendable { - public static func decision(for request: WorkflowStartRequest) -> GuardrailDecision { - decision(kind: request.kind) - } - - public static func decision(kind: WorkflowKind) -> GuardrailDecision { - switch kind { - case .pluginFactoryCreate, .connectorAuthDiscover, .jobStep, .interactiveTool: - return .allow - case .pluginFactoryEdit: - return .deny( - reason: "Plugin edit workflow is reserved and not enabled yet." - ) - case .none: - return .deny(reason: "A workflow kind is required.") - } - } - - /// Throws `WorkflowRuntimeError.deniedByGuardrail` unless the decision is `.allow`. - public static func admitOrThrow(_ request: WorkflowStartRequest) throws { - switch decision(for: request) { + /// Apply a Policy decision for starting a workflow. + /// + /// - Parameters: + /// - decision: From the Policy engine. + /// - confirm: Invoked on `.confirmHITL`. Return `true` to proceed, `false` to cancel. + public static func apply( + _ decision: GuardrailDecision, + confirm: () async -> Bool + ) async throws { + switch decision { case .allow: return case .deny(let reason): throw WorkflowRuntimeError.deniedByGuardrail(reason) case .confirmHITL(let hitl): - let detail = hitl.message ?? hitl.title ?? "Workflow start requires confirmation." + let approved = await confirm() + if approved { + return + } + let detail = hitl.message ?? hitl.title ?? "Workflow start was not approved." throw WorkflowRuntimeError.deniedByGuardrail(detail) - case .requireWorkflow, .redactContent, .redactArgument: + case .requireWorkflow: + throw WorkflowRuntimeError.deniedByGuardrail( + "requireWorkflow is not a valid decision for workflow start." + ) + case .redactContent, .redactArgument: throw WorkflowRuntimeError.deniedByGuardrail( - "Workflow start returned an unsupported Guardrail decision." + "Redaction is not a valid decision for workflow start." ) } } diff --git a/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift b/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift index 5297c61f..d5f6115e 100644 --- a/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift +++ b/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift @@ -1413,37 +1413,17 @@ import Testing #expect(fromMinutes.containerRunMaxTTLMinutes == 12) } - @Test func effectorAdmissionAllowsWorkflowCrawl() { + @Test func effectorAdmissionParseContextJSON() throws { let context = ExecutionContextWire( - sessionID: "s1", - principal: .agent(sessionID: "s1", agentID: "a1"), - workflow: WorkflowContextWire(workflowID: "w1", kind: .pluginFactoryCreate), - capabilities: [.syncWebCrawl, .hostReviewRetry] - ) - #expect( - EffectorAdmissionPolicy.allowsSyncWebCrawl( - context: context, - principal: .agent(sessionID: "s1", agentID: "a1") - ) - ) - } - - @Test func effectorAdmissionAllowsLiveChatWithoutContext() { - #expect( - EffectorAdmissionPolicy.allowsSyncWebCrawl( - context: nil, - principal: .agent(sessionID: "s1", agentID: "a1") - ) - ) - } - - @Test func effectorAdmissionAllowsJobsWithoutContext() { - #expect( - EffectorAdmissionPolicy.allowsSyncWebCrawl( - context: nil, - principal: .job(jobID: "j1") - ) + sessionID: "s", + principal: .ui, + capabilities: [.syncWebCrawl] ) + let json = try context.encodedJSON() + let parsed = EffectorAdmissionPolicy.parseContextJSON(json) + #expect(parsed?.capabilities.contains(.syncWebCrawl) == true) + #expect(EffectorAdmissionPolicy.parseContextJSON(nil) == nil) + #expect(EffectorAdmissionPolicy.parseContextJSON(" ") == nil) } @Test func executionContextWireRoundTrip() throws { diff --git a/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift index 2a674722..ece3acea 100644 --- a/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift +++ b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift @@ -3,25 +3,6 @@ import Testing @Suite("Guardrail") struct GuardrailDecisionTests { - @Test func effectorAdmissionUsesGuardrailDecision() { - let denied = EffectorAdmissionPolicy.syncWebCrawlDecision( - context: nil, - principal: .ui - ) - #expect(!denied.isAllow) - if case .deny = denied { - #expect(true) - } else { - Issue.record("Expected deny for UI without crawl capability") - } - - let allowed = EffectorAdmissionPolicy.syncWebCrawlDecision( - context: nil, - principal: .job(jobID: "job-1") - ) - #expect(allowed == .allow) - } - @Test func policyEngineSharesGuardrailVocabulary() { let engine = PolicyEngine(rules: [ DenyToolNamesRule(toolNames: ["x"], reason: "no") @@ -38,29 +19,43 @@ struct GuardrailDecisionTests { #expect(WorkflowKind.allCases.contains(.pluginFactoryEdit)) } - @Test func workflowAdmissionAllowsCreateAndDeniesEditPlaceholder() throws { - #expect(WorkflowAdmissionPolicy.decision(kind: .pluginFactoryCreate) == .allow) - #expect(WorkflowAdmissionPolicy.decision(kind: .connectorAuthDiscover) == .allow) + @Test func workflowAdmissionApplyAllowAndDeny() async throws { + try await WorkflowAdmissionPolicy.apply(.allow) { true } + do { + try await WorkflowAdmissionPolicy.apply(.deny(reason: "blocked")) { true } + Issue.record("deny must throw") + } catch let error as WorkflowRuntimeError { + if case .deniedByGuardrail(let reason) = error { + #expect(reason == "blocked") + } else { + Issue.record("Expected deniedByGuardrail") + } + } + } - let edit = WorkflowAdmissionPolicy.decision(kind: .pluginFactoryEdit) - #expect(!edit.isAllow) + @Test func workflowAdmissionApplyConfirmHITL() async throws { + try await WorkflowAdmissionPolicy.apply( + .confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"])) + ) { true } - let request = WorkflowStartRequest( - kind: .pluginFactoryEdit, - sessionID: "s", - agentID: "ui", - inputJSON: "{}", - principal: .ui - ) do { - try WorkflowAdmissionPolicy.admitOrThrow(request) - Issue.record("edit placeholder must be denied") + try await WorkflowAdmissionPolicy.apply( + .confirmHITL(GuardrailHITLRequest(title: "Need approval")) + ) { false } + Issue.record("cancelled HITL must throw") } catch let error as WorkflowRuntimeError { if case .deniedByGuardrail = error { #expect(true) } else { - Issue.record("Expected deniedByGuardrail, got \(error)") + Issue.record("Expected deniedByGuardrail") } } } + + @Test func defaultWorkflowSeedsCoverCreateAndDenyEdit() { + let rules = DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: "ui") + #expect(rules.contains { $0.name == "allow-workflow-plugin-factory-create" }) + #expect(rules.contains { $0.name == "deny-workflow-plugin-factory-edit" }) + #expect(rules.allSatisfy { $0.scope == "workflow_start" }) + } } diff --git a/readme.md b/readme.md index 0b9c4b8f..86c833e2 100644 --- a/readme.md +++ b/readme.md @@ -99,7 +99,7 @@ Copy [.env.example](.env.example) — **never commit `.env`**. - **`Structure/Guardrail`** — control plane umbrella. Policy decides; HITL and workflows enforce. - Persisted rules in SQLite (`Guardrail/Policy`, `packages/PolicyRuntime`) produce `GuardrailDecision`. - **`PolicyInterceptor`** / **`ToolRequestInterceptor`** (MemorySystem) — pipeline hooks that apply those decisions. -- Workflow starts are admitted by Guardrail (`WorkflowAdmissionPolicy`) before the runtime runs them. +- Workflow starts and MCP tools are admitted by Policy rules (`workflow_start` / `tool_invocation`); thin adapters apply allow / deny / confirmHITL / redact. ### Messaging diff --git a/ui/MCPService/MCPServiceStore.swift b/ui/MCPService/MCPServiceStore.swift index 24eace28..a56d3998 100644 --- a/ui/MCPService/MCPServiceStore.swift +++ b/ui/MCPService/MCPServiceStore.swift @@ -7,9 +7,16 @@ actor MCPServiceStore { static let shared = MCPServiceStore() private var repository: DBRepository? + private var didSeedPolicy = false func sharedRepository() async throws -> DBRepository { - if let repository { return repository } + if let repository { + if !didSeedPolicy { + try await seedBaselinePolicyIfNeeded(repository) + didSeedPolicy = true + } + return repository + } let directory = try DerrickAppSupport.databaseDirectory() let repo = DBRepository( configuration: DBRepositoryConfiguration( @@ -21,6 +28,8 @@ actor MCPServiceStore { ) ) _ = try await repo.createEmptyDatabaseIfNeeded(username: "ui", password: "ui") + try await seedBaselinePolicyIfNeeded(repo) + didSeedPolicy = true repository = repo let path = await repo.databaseURL.path fputs("[MCPService] shared DB: \(path)\n", stderr) @@ -52,4 +61,38 @@ actor MCPServiceStore { func databasePath() async -> String? { try? await sharedRepository().databaseURL.path } + + /// Idempotent baseline so effector admission works even if UI has not launched yet. + private func seedBaselinePolicyIfNeeded(_ repository: DBRepository) async throws { + let app = DerrickAppSupport.defaultApplicationName + try await DefaultGuardrailPolicySeeds.seedWorkflowStartRulesIfNeeded( + store: repository, + applicationName: app + ) + var rules: [PolicyRule] = AllowedMCPTool.allCases.map { tool in + PolicyRule( + applicationName: app, + name: "allow-\(tool.rawValue)", + scope: "tool_invocation", + matcherJSON: #"{"tool_name":"\#(tool.rawValue)"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 1 + ) + } + rules += ["tool_search", "tool", "tool_batch"].map { name in + PolicyRule( + applicationName: app, + name: "allow-\(name)", + scope: "tool_invocation", + matcherJSON: #"{"tool_name":"\#(name)"}"#, + outcomeJSON: #"{"action":"allow"}"#, + priority: 1 + ) + } + for rule in rules { + let existing = try await repository.loadRules(applicationName: app, scope: rule.scope) + guard existing.contains(where: { $0.name == rule.name }) == false else { continue } + try await repository.saveRule(rule) + } + } } diff --git a/ui/MCPService/MCPServiceToolHost.swift b/ui/MCPService/MCPServiceToolHost.swift index b06f6779..3cf510af 100644 --- a/ui/MCPService/MCPServiceToolHost.swift +++ b/ui/MCPService/MCPServiceToolHost.swift @@ -5,6 +5,7 @@ import MCPClient import MCPServer import MemorySystem import Plugin +import PolicyRuntime import Structure /// MCP effectors hosted in MCPService (`script_exec`, `web.crawl`, `web.search`, factory @@ -301,30 +302,57 @@ actor MCPServiceToolHost { HostHTTPCallContext.shared.clear() } - if toolName == "web.crawl" { - let context = EffectorAdmissionPolicy.parseContextJSON(request.executionContextJSON) - switch EffectorAdmissionPolicy.syncWebCrawlDecision( - context: context, - principal: request.principal - ) { - case .allow: - break - case .deny(let reason): - await MCPServiceStore.shared.log( - level: .error, - message: "web.crawl denied by Guardrail: \(reason)", - code: "tool_denied", - detailJSON: #"{"requestID":"\#(request.requestID)"}"# - ) - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: false, - isError: true, - text: "", - message: reason - ) - case .confirmHITL, .requireWorkflow, .redactContent, .redactArgument: - let message = "web.crawl returned an unsupported Guardrail decision." + // Policy decides effector/tool admission (same engine as chat). Apply allow / deny / + // confirmHITL / redact before the tool runs. + let policyRepo = try await MCPServiceStore.shared.sharedRepository() + let toolPolicy = StoreBackedToolGovernancePolicy( + store: policyRepo, + applicationName: DerrickAppSupport.defaultApplicationName + ) + let sessionIDForPolicy: String = { + if case .agent(let sessionID, _) = request.principal { return sessionID } + if case .job(let jobID) = request.principal { return jobID } + return "mcp-service" + }() + var argumentsJSON = request.argumentsJSON + let decision = try await toolPolicy.evaluateToolInvocation( + ToolInvocationEvent( + sessionID: sessionIDForPolicy, + toolName: toolName, + argumentsJSON: argumentsJSON + ) + ) + switch decision { + case .allow: + break + case .deny(let reason): + await MCPServiceStore.shared.log( + level: .error, + message: "tool denied by Policy tool=\(toolName): \(reason)", + code: "tool_denied", + detailJSON: #"{"requestID":"\#(request.requestID)"}"# + ) + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: reason + ) + case .confirmHITL(let hitl): + let approved = await awaitMCPToolHITL( + sessionID: sessionIDForPolicy, + toolName: toolName, + argumentsJSON: argumentsJSON, + hitl: hitl, + principal: request.principal, + repository: policyRepo + ) + switch approved { + case .approved(let edited, _): + argumentsJSON = edited + case .cancelled(let actor): + let message = "Tool \(toolName) was not approved\(actor.map { " by \($0)" } ?? "")." return MCPToolCallResultDTO( requestID: request.requestID, ok: false, @@ -333,13 +361,29 @@ actor MCPServiceToolHost { message: message ) } + case .redactArgument(let key, let pattern, let replacement): + argumentsJSON = redactToolArgumentJSON( + argumentsJSON, + key: key, + pattern: pattern, + replacement: replacement + ) + case .requireWorkflow, .redactContent: + let message = "Tool \(toolName) returned an unsupported Policy decision." + return MCPToolCallResultDTO( + requestID: request.requestID, + ok: false, + isError: true, + text: "", + message: message + ) } // Shared Lib parser (same as Agent policy path) — handles repaired model JSON. // `{}` is a valid empty object for tools with no required args. let args: [String: Value] do { - args = try parseToolArgumentsObject(request.argumentsJSON) + args = try parseToolArgumentsObject(argumentsJSON) } catch { await MCPServiceStore.shared.log( level: .error, @@ -412,6 +456,92 @@ actor MCPServiceToolHost { private static func skillIndex(from repo: DBRepository) async throws -> [PluginSkillDisclosure.IndexEntry] { try await repo.listPluginSkillIndex() } + + private static let toolHITLPollNanoseconds: UInt64 = 1_000_000_000 + private static let toolHITLTimeoutNanoseconds: UInt64 = 15 * 60 * 1_000_000_000 + + private func awaitMCPToolHITL( + sessionID: String, + toolName: String, + argumentsJSON: String, + hitl: GuardrailHITLRequest, + principal: ServicePrincipal, + repository: DBRepository + ) async -> ApprovalConfirmationDecision { + let approvalID = UUID().uuidString + let requiredJSON = (try? JSONEncoder().encode(hitl.requiredFields)) + .flatMap { String(data: $0, encoding: .utf8) } ?? "[]" + let isJob = isJobPrincipal(principal) + let row = PendingHITLApprovalRow( + id: approvalID, + turnID: sessionID, + sessionID: sessionID, + toolName: toolName, + argumentsJSON: argumentsJSON, + requiredFieldsJSON: requiredJSON, + isJobContext: isJob + ) + do { + try await repository.insertPendingHITLApproval(row) + } catch { + fputs("[MCPService] HITL persist failed: \(error.localizedDescription)\n", stderr) + return .cancelled(actor: "system-persist-failed") + } + DerrickHITLNotificationSignal.postPoll() + + let deadline = Date().addingTimeInterval( + Double(Self.toolHITLTimeoutNanoseconds) / 1_000_000_000 + ) + while Date() < deadline { + if Task.isCancelled { + return .cancelled(actor: "system-cancelled") + } + if let decision = try? await repository.fetchPendingHITLApproval(id: approvalID), + decision.status != .pending { + switch decision.status { + case .approved: + let args = decision.editedArgumentsJSON?.isEmpty == false + ? decision.editedArgumentsJSON! + : argumentsJSON + return .approved(editedArgumentsJSON: args, actor: decision.actor) + case .cancelled, .timeout, .pending: + return .cancelled(actor: decision.actor ?? decision.status.rawValue) + } + } + try? await Task.sleep(nanoseconds: Self.toolHITLPollNanoseconds) + } + try? await repository.resolveHITLApproval( + id: approvalID, + status: .timeout, + editedArgumentsJSON: nil, + actor: "system-timeout" + ) + return .cancelled(actor: "system-timeout") + } +} + +private func redactToolArgumentJSON( + _ json: String, + key: String, + pattern: String, + replacement: String +) -> String { + guard let data = json.data(using: .utf8), + var object = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { + return json + } + if let stringValue = object[key] as? String { + object[key] = stringValue.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + } + guard let redacted = try? JSONSerialization.data(withJSONObject: object, options: [.sortedKeys]), + let redactedString = String(data: redacted, encoding: .utf8) else { + return json + } + return redactedString } private func pluginFactoryFailureDetail(for error: Error) -> String { diff --git a/ui/SharedAgentRuntime/Conversation/ConversationModel.swift b/ui/SharedAgentRuntime/Conversation/ConversationModel.swift index 501216c9..61619771 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationModel.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationModel.swift @@ -970,7 +970,7 @@ final class ConversationModel { outcomeJSON: #"{"action":"allow"}"#, priority: 1 ) - } + [ + } + DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: applicationName) + [ PolicyRule( applicationName: applicationName, name: "allow-default-assistant-chunks", From cfe3278375157b6d0be3fc5dd25f0097285a713e Mon Sep 17 00:00:00 2001 From: David Choi Date: Sat, 26 Sep 2026 21:17:14 -0400 Subject: [PATCH 3/3] Unify Guardrail on Evaluating/Applying naming and purge leftovers. Chokepoints now only evaluate Policy rules then apply GuardrailDecision; remove interceptor and admission-policy types that duplicated that path. Co-authored-by: Cursor --- docs/Design.md | 18 +- .../WorkflowRuntimeEngine.swift | 13 +- .../MemorySystem/PolicyInterceptor.swift | 55 --- .../MemorySystem/ToolRequestInterceptor.swift | 89 ---- .../PolicyInterceptionTests.swift | 234 ---------- .../ToolRequestInterceptionTests.swift | 409 ------------------ ...t => StoreBackedGuardrailEvaluating.swift} | 27 +- ... StoreBackedWorkflowStartEvaluating.swift} | 4 +- ...StoreBackedGuardrailEvaluatingTests.swift} | 80 ++-- ...eBackedWorkflowStartEvaluatingTests.swift} | 6 +- .../AppLayerFeatureCatalog.swift | 10 +- .../ExecutionContextWire.swift | 9 + .../AssistantContentGuardrailApplying.swift | 51 +++ .../Applying/GuardrailApplying.swift | 8 + .../ToolInvocationGuardrailApplying.swift | 97 +++++ .../WorkflowStartGuardrailApplying.swift | 50 +++ .../Evaluating/GuardrailEvaluating.swift | 7 + .../Sources/Guardrail/Guardrail.swift | 12 +- .../Sources/Guardrail/GuardrailDecision.swift | 15 +- .../HITL/GuardrailHITLPresenting.swift | 40 ++ .../ImmediateGuardrailHITLPresenting.swift | 36 ++ .../Policy/GuardrailPolicyScope.swift | 17 + .../PolicyEngine/PolicyEngineTypes.swift | 6 +- .../PolicyRuntime/PolicyInterception.swift | 19 +- .../PolicyRuntime/ToolInterception.swift | 44 +- .../Workflow/EffectorAdmissionPolicy.swift | 12 - .../Workflow/WorkflowAdmissionPolicy.swift | 39 -- .../AppLayerServicesWireTests.swift | 8 +- .../GuardrailApplyingTests.swift | 71 +++ .../GuardrailDecisionTests.swift | 92 +++- readme.md | 8 +- ui/MCPService/MCPServiceToolHost.swift | 103 ++--- .../Conversation/ConversationModel.swift | 28 +- .../ConversationPipelinePolicy.swift | 22 +- ...ConversationPipelineToolInterception.swift | 296 +++++++------ .../Conversation/ProfileDelegateRunner.swift | 4 +- 36 files changed, 787 insertions(+), 1252 deletions(-) delete mode 100644 packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift delete mode 100644 packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift delete mode 100644 packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift delete mode 100644 packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift rename packages/PolicyRuntime/Sources/PolicyRuntime/{StoreBackedPolicyEvaluators.swift => StoreBackedGuardrailEvaluating.swift} (93%) rename packages/PolicyRuntime/Sources/PolicyRuntime/{StoreBackedWorkflowAdmissionPolicy.swift => StoreBackedWorkflowStartEvaluating.swift} (96%) rename packages/PolicyRuntime/Tests/PolicyRuntimeTests/{StoreBackedPolicyEvaluatorsTests.swift => StoreBackedGuardrailEvaluatingTests.swift} (84%) rename packages/PolicyRuntime/Tests/PolicyRuntimeTests/{StoreBackedWorkflowAdmissionPolicyTests.swift => StoreBackedWorkflowStartEvaluatingTests.swift} (94%) create mode 100644 packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift create mode 100644 packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift create mode 100644 packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift create mode 100644 packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift create mode 100644 packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift create mode 100644 packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift create mode 100644 packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift create mode 100644 packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift delete mode 100644 packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift delete mode 100644 packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift create mode 100644 packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift diff --git a/docs/Design.md b/docs/Design.md index 4616e1f7..3d097e29 100644 --- a/docs/Design.md +++ b/docs/Design.md @@ -11,14 +11,18 @@ This application is Protocol first. All major features must have a Protocol and ## Guardrail Guardrail is Derrick's control plane in Structure (`Sources/Guardrail`). -Flow: +Flow: **Policy evaluates rules → adapters apply `GuardrailDecision` → chokepoints only call those two.** -1. **Guardrail** — logical container -2. **Policy engine** — store-backed rules decide allow / deny / confirm HITL / require workflow / redact -3. **Enforcement** — those decisions gate workflow starts, MCP effector/tool calls, HITL confirmations, and content redaction (e.g. PII) +Naming (no exceptions): -- Workflow starts: `StoreBackedWorkflowAdmissionPolicy` (`workflow_start` scope) → `WorkflowAdmissionPolicy.apply` in `WorkflowRuntimeEngine` (allow / deny / confirmHITL then resume or cancel). -- MCP tools/effectors: `StoreBackedToolGovernancePolicy` (`tool_invocation`) in the chat pipeline and again in MCPService before the tool runs. -- Content: same engine via `StoreBackedCompletionContentPolicy`. +- `Guardrail*` — control-plane types +- `*Evaluating` — rule interpreters (`Request` → `GuardrailDecision`) +- `*Applying` — decision adapters (decision → effect) +- `StoreBacked*Evaluating` — SQLite-backed interpreters in `packages/PolicyRuntime` + +- Workflow starts: `StoreBackedWorkflowStartEvaluating` (`workflow_start`) → `WorkflowStartGuardrailApplying` in `WorkflowRuntimeEngine`. +- MCP tools/effectors: `StoreBackedToolInvocationEvaluating` (`tool_invocation`) → `ToolInvocationGuardrailApplying` in the chat pipeline and MCPService. +- Content: `StoreBackedAssistantContentEvaluating` → `AssistantContentGuardrailApplying`. +- HITL: `GuardrailHITLPresenting` (shared); adapters take a presenter, chokepoints do not switch on decisions. - `WorkflowKind.pluginFactoryEdit` is denied by a Policy rule until editability ships. - Plugins propose work; they do not authorize control outcomes. diff --git a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift index bb930d2c..3334a510 100644 --- a/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift +++ b/packages/DerrickBackend/Sources/DerrickBackend/WorkflowRuntimeEngine.swift @@ -21,18 +21,21 @@ public actor WorkflowRuntimeEngine { store: repo, applicationName: DerrickAppSupport.defaultApplicationName ) - let decision = try await StoreBackedWorkflowAdmissionPolicy( + let decision = try await StoreBackedWorkflowStartEvaluating( store: repo, applicationName: DerrickAppSupport.defaultApplicationName ).evaluate(request) - try await WorkflowAdmissionPolicy.apply(decision) { - guard case .confirmHITL(let hitl) = decision else { return true } - return await awaitWorkflowStartHITL( + let hitl = ClosureGuardrailHITLPresenting { [self] presentation in + let approved = await awaitWorkflowStartHITL( request: request, repository: repo, - hitl: hitl + hitl: presentation.hitl ) + return approved + ? .approved(editedPayloadJSON: nil, actor: nil) + : .cancelled(actor: nil) } + try await WorkflowStartGuardrailApplying(hitl: hitl).apply(decision, for: request) let idempotencyKey = WorkflowRuntimeIdempotency.key( sessionID: request.sessionID, kind: request.kind, diff --git a/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift b/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift deleted file mode 100644 index a8d45ef4..00000000 --- a/packages/MemorySystem/Sources/MemorySystem/PolicyInterceptor.swift +++ /dev/null @@ -1,55 +0,0 @@ -import Foundation -import Structure - -public struct DefaultPolicyInterceptor: PolicyInterceptor { - private let policy: PolicyEvaluator? - - public init(policy: PolicyEvaluator? = nil) { - self.policy = policy - } - - public func interceptAssistantChunk(_ event: AssistantChunkEvent) async throws -> AssistantContentInterceptResult { - guard let policy else { return .allowed(event.content) } - - let outcome = try await policy.evaluateAssistantChunk(event) - switch outcome { - case .allow: - return .allowed(event.content) - case .deny(let reason): - return .denied(reason: reason) - case .redactContent(let pattern, let replacement): - let redacted = event.content.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - return .allowed(redacted) - case .confirmHITL, .requireWorkflow, .redactArgument: - // Streaming chunks: never modal mid-token. Completion path enforces confirm. - return .allowed(event.content) - } - } - - public func interceptAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> AssistantContentInterceptResult { - guard let policy else { return .allowed(event.fullCompletion) } - - let outcome = try await policy.evaluateAssistantCompletion(event) - switch outcome { - case .allow: - return .allowed(event.fullCompletion) - case .deny(let reason): - return .denied(reason: reason) - case .redactContent(let pattern, let replacement): - let redacted = event.fullCompletion.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - return .allowed(redacted) - case .confirmHITL(let hitl): - return .confirm(content: event.fullCompletion, requiredFields: hitl.requiredFields) - case .requireWorkflow, .redactArgument: - return .denied(reason: "Content policy returned an unsupported control decision.") - } - } -} diff --git a/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift b/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift deleted file mode 100644 index 4e867476..00000000 --- a/packages/MemorySystem/Sources/MemorySystem/ToolRequestInterceptor.swift +++ /dev/null @@ -1,89 +0,0 @@ -import Foundation -import Structure - -public struct DefaultToolRequestInterceptor: ToolRequestInterceptor { - private let policy: ToolGovernancePolicy? - - public init(policy: ToolGovernancePolicy? = nil) { - self.policy = policy - } - - public func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInterceptionDecision { - guard let policy else { - return .allow(event) - } - - let outcome = try await policy.evaluateToolInvocation(event) - switch outcome { - case .allow: - return .allow(event) - case .deny(let reason): - return .deny(reason: reason) - case .confirmHITL(let hitl): - return .confirm(event, requiredFields: hitl.requiredFields) - case .redactArgument(let key, let pattern, let replacement): - guard let parsedArgs = try? JSONSerialization.jsonObject(with: event.argumentsJSON.data(using: .utf8) ?? Data(), options: []) as? [String: Any] else { - return .allow(event) - } - - var mutableArgs = parsedArgs - if let stringValue = mutableArgs[key] as? String { - mutableArgs[key] = stringValue.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - } - - guard let redactedJSON = try? JSONSerialization.data(withJSONObject: mutableArgs, options: [.sortedKeys]), - let redactedString = String(data: redactedJSON, encoding: .utf8) else { - return .allow(event) - } - - return .allow( - ToolInvocationEvent( - sessionID: event.sessionID, - toolName: event.toolName, - argumentsJSON: redactedString, - timestamp: event.timestamp - ) - ) - case .requireWorkflow, .redactContent: - return .deny(reason: "Tool policy returned an unsupported control decision.") - } - } - - public func interceptToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInvocationEvent? { - switch try await evaluateToolInvocation(event) { - case .allow(let processedEvent): - return processedEvent - case .deny: - return nil - case .confirm(let processedEvent, _): - return processedEvent - } - } - - public func interceptAndRun( - _ event: ToolInvocationEvent, - confirm: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent, [String]) async throws -> ToolInvocationConfirmation, - proceed: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent) async throws -> R - ) async throws -> R { - switch try await evaluateToolInvocation(event) { - case .allow(let allowedEvent): - return try await proceed(allowedEvent) - case .deny(let reason): - throw ToolInvocationInterceptionError.denied(reason: reason) - case .confirm(let confirmEvent, let requiredFields): - switch try await confirm(confirmEvent, requiredFields) { - case .approved(let approvedEvent): - return try await proceed(approvedEvent) - case .cancelled(let actor): - let suffix = actor.map { " by \($0)" } ?? "" - throw ToolInvocationInterceptionError.cancelled( - reason: "User cancelled the approval request\(suffix)" - ) - } - } - } -} diff --git a/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift b/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift deleted file mode 100644 index e4bf6b91..00000000 --- a/packages/MemorySystem/Tests/MemorySystemTests/PolicyInterceptionTests.swift +++ /dev/null @@ -1,234 +0,0 @@ -import XCTest -import Structure -@testable import MemorySystem - -final class PolicyInterceptionTests: XCTestCase { - func test_assistantChunkEvent_creation() { - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Hello, " - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.sessionID, "session-1") - XCTAssertEqual(event.chunkIndex, 0) - XCTAssertEqual(event.content, "Hello, ") - } - - func test_assistantCompletionEvent_creation() { - let event = AssistantCompletionEvent( - sessionID: "session-1", - fullCompletion: "Hello, world!", - chunkCount: 2 - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.fullCompletion, "Hello, world!") - XCTAssertEqual(event.chunkCount, 2) - } - - func test_toolInvocationEvent_creation() { - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/etc/config.txt"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.toolName, "read_file") - XCTAssertTrue(event.argumentsJSON.contains("path")) - } - - func test_toolResultEvent_creation() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: #"{"content": "config data"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.toolName, "read_file") - XCTAssertNil(event.error) - } - - func test_statusUpdateEvent_creation() { - let event = StatusUpdateEvent( - sessionID: "session-1", - message: "Processing..." - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.message, "Processing...") - } - - func test_policyInterceptionEvent_timestamp() { - let now = Date() - let chunkEvent = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test", - timestamp: now - ) - let event = PolicyInterceptionEvent.assistantChunk(chunkEvent) - - XCTAssertEqual(event.timestamp, now) - } - - func test_defaultPolicyInterceptor_allowsContent() async throws { - let interceptor = DefaultPolicyInterceptor() - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Safe content" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("Safe content")) - } - - func test_defaultPolicyInterceptor_withoutPolicy_passesThrough() async throws { - let interceptor = DefaultPolicyInterceptor(policy: nil) - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Any content" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("Any content")) - } - - func test_events_are_hashable() { - let event1 = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test" - ) - let event2 = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test" - ) - - XCTAssertNotEqual(event1, event2) - } - - func test_events_are_sendable() async { - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "test" - ) - - let task = Task { - return event.content - } - - let result = await task.value - XCTAssertEqual(result, "test") - } - - func test_toolResultEvent_withError() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: "{}", - error: "File not found" - ) - - XCTAssertEqual(event.error, "File not found") - } -} - -struct MockResponseContentPolicy: PolicyEvaluator { - var shouldAllow = true - var shouldDeny = false - var denyReason = "Policy rejected" - - func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> PolicyDecisionOutcome { - if shouldDeny { - return .deny(reason: denyReason) - } - return shouldAllow ? .allow : .confirmHITL(GuardrailHITLRequest(requiredFields: [])) - } - - func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> PolicyDecisionOutcome { - if shouldDeny { - return .deny(reason: denyReason) - } - return shouldAllow ? .allow : .confirmHITL(GuardrailHITLRequest(requiredFields: [])) - } -} - -final class PolicyInterceptorTests: XCTestCase { - func test_interceptor_allows_with_policy() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = true - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Hello" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("Hello")) - } - - func test_interceptor_denies_with_policy() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = false - policy.shouldDeny = true - policy.denyReason = "Policy violation" - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantChunkEvent( - sessionID: "session-1", - chunkIndex: 0, - content: "Unsafe content" - ) - - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .denied(reason: "Policy violation")) - } - - func test_interceptor_redacts_content() async throws { - let interceptor = DefaultPolicyInterceptor() - let event = AssistantCompletionEvent( - sessionID: "session-1", - fullCompletion: "Email is test@example.com here", - chunkCount: 1 - ) - - let result = try await interceptor.interceptAssistantCompletion(event) - XCTAssertEqual(result, .allowed("Email is test@example.com here")) - } - - func test_interceptor_completion_confirm_does_not_silently_allow() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = false - policy.shouldDeny = false - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantCompletionEvent( - sessionID: "session-1", - fullCompletion: "Contact me at a@b.co", - chunkCount: 1 - ) - let result = try await interceptor.interceptAssistantCompletion(event) - XCTAssertEqual(result, .confirm(content: "Contact me at a@b.co", requiredFields: [])) - } - - func test_interceptor_chunk_confirm_still_soft_allows() async throws { - var policy = MockResponseContentPolicy() - policy.shouldAllow = false - policy.shouldDeny = false - - let interceptor = DefaultPolicyInterceptor(policy: policy) - let event = AssistantChunkEvent(sessionID: "s", chunkIndex: 0, content: "partial") - let result = try await interceptor.interceptAssistantChunk(event) - XCTAssertEqual(result, .allowed("partial")) - } -} diff --git a/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift b/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift deleted file mode 100644 index e94be9b0..00000000 --- a/packages/MemorySystem/Tests/MemorySystemTests/ToolRequestInterceptionTests.swift +++ /dev/null @@ -1,409 +0,0 @@ -import XCTest -import Structure -@testable import MemorySystem - -final class ToolRequestInterceptionTests: XCTestCase { - func test_toolInvocationEvent_creation() { - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/etc/config.txt"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.sessionID, "session-1") - XCTAssertEqual(event.toolName, "read_file") - XCTAssertTrue(event.argumentsJSON.contains("path")) - } - - func test_toolResultEvent_creation() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: #"{"content": "file content"}"# - ) - - XCTAssertNotNil(event.eventID) - XCTAssertEqual(event.toolName, "read_file") - XCTAssertNil(event.error) - } - - func test_toolResultEvent_with_error() { - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: "{}", - error: "File not found" - ) - - XCTAssertEqual(event.error, "File not found") - } - - func test_toolInvocationEvent_timestamp() { - let now = Date() - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}", - timestamp: now - ) - - XCTAssertEqual(event.timestamp, now) - } - - func test_toolResultEvent_timestamp() { - let now = Date() - let event = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: "{}", - timestamp: now - ) - - XCTAssertEqual(event.timestamp, now) - } - - func test_toolInvocationEvent_sendable() async { - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - - let task = Task { - return event.toolName - } - - let result = await task.value - XCTAssertEqual(result, "read_file") - } - - func test_toolRequestEvent_hashable() { - let event1 = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - let event2 = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - - XCTAssertNotEqual(event1, event2) - } -} - -struct MockToolGovernancePolicy: ToolGovernancePolicy { - var shouldAllow = true - var shouldDeny = false - var denyReason = "Tool not allowed" - - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - if shouldDeny { - return .deny(reason: denyReason) - } - return shouldAllow ? .allow : .confirmHITL(GuardrailHITLRequest(requiredFields: ["approval"])) - } -} - -final class ToolRequestInterceptorTests: XCTestCase { - func test_defaultInterceptor_allows_tool() async throws { - var policy = MockToolGovernancePolicy() - policy.shouldAllow = true - - let interceptor = DefaultToolRequestInterceptor(policy: policy) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/etc/config.txt"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNotNil(result) - XCTAssertEqual(result?.toolName, "read_file") - } - - func test_defaultInterceptor_denies_tool() async throws { - var policy = MockToolGovernancePolicy() - policy.shouldAllow = false - policy.shouldDeny = true - policy.denyReason = "Tool is restricted" - - let interceptor = DefaultToolRequestInterceptor(policy: policy) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "shell_exec", - argumentsJSON: #"{"cmd": "rm -rf /"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNil(result) - } - - func test_defaultInterceptor_without_policy() async throws { - let interceptor = DefaultToolRequestInterceptor(policy: nil) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{"path": "/home/user/file.txt"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertEqual(result?.toolName, "read_file") - XCTAssertEqual(result?.argumentsJSON, event.argumentsJSON) - } - - func test_defaultInterceptor_redacts_arguments() async throws { - struct RedactingPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - return .redactArgument(argumentKey: "password", pattern: ".+", replacement: "[REDACTED]") - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: RedactingPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "authenticate", - argumentsJSON: #"{"username": "admin", "password": "secret123"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNotNil(result) - XCTAssertTrue(result!.argumentsJSON.contains("[REDACTED]")) - XCTAssertFalse(result!.argumentsJSON.contains("secret123")) - } - - func test_interceptionEvent_in_enum() { - let invocationEvent = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: "{}" - ) - let event = PolicyInterceptionEvent.toolInvocation(invocationEvent) - - XCTAssertEqual(event.timestamp, invocationEvent.timestamp) - } - - func test_toolResult_interceptionEvent() { - let resultEvent = ToolResultEvent( - sessionID: "session-1", - toolName: "read_file", - resultJSON: #"{"content": "data"}"# - ) - let event = PolicyInterceptionEvent.toolResult(resultEvent) - - XCTAssertEqual(event.timestamp, resultEvent.timestamp) - } - - func test_confirm_outcome_preserves_tool_name() async throws { - struct ConfirmingPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - return .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_consent"])) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmingPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{"path": "/tmp/file.txt"}"# - ) - - let result = try await interceptor.interceptToolInvocation(event) - XCTAssertNotNil(result) - XCTAssertEqual(result?.toolName, "delete_file") - } - - func test_evaluateToolInvocation_returns_confirm_with_requiredFields() async throws { - struct ConfirmingPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_consent", "ticket_id"])) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmingPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{"path": "/tmp/file.txt"}"# - ) - - let decision = try await interceptor.evaluateToolInvocation(event) - switch decision { - case .confirm(let confirmedEvent, let requiredFields): - XCTAssertEqual(confirmedEvent.toolName, "delete_file") - XCTAssertEqual(requiredFields, ["user_consent", "ticket_id"]) - default: - XCTFail("Expected confirm decision") - } - } - - // MARK: - interceptAndRun (item 4: confirm-before-proceed) - - private final class CallProbe: @unchecked Sendable { - var confirmCalled = false - var proceedCalled = false - var proceedTool: String? - var seenFields: [String] = [] - } - - func test_interceptAndRun_allow_calls_proceed_only() async throws { - struct AllowPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .allow - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: AllowPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "read_file", - argumentsJSON: #"{}"# - ) - - let probe = CallProbe() - let result = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in - probe.confirmCalled = true - return .cancelled(actor: nil) - }, - proceed: { gated in - probe.proceedTool = gated.toolName - return "ok" - } - ) - - XCTAssertEqual(result, "ok") - XCTAssertFalse(probe.confirmCalled) - XCTAssertEqual(probe.proceedTool, "read_file") - } - - func test_interceptAndRun_deny_throws_without_proceed() async throws { - struct DenyPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .deny(reason: "blocked by policy") - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: DenyPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "shell_exec", - argumentsJSON: #"{}"# - ) - - let probe = CallProbe() - do { - _ = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in .cancelled(actor: nil) }, - proceed: { _ -> String in - probe.proceedCalled = true - return "should not run" - } - ) - XCTFail("Expected deny error") - } catch ToolInvocationInterceptionError.denied(let reason) { - XCTAssertEqual(reason, "blocked by policy") - } - XCTAssertFalse(probe.proceedCalled) - } - - func test_interceptAndRun_confirm_approved_then_proceed() async throws { - struct ConfirmPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_approval"])) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{"path":"/tmp/a"}"# - ) - - let probe = CallProbe() - let result = try await interceptor.interceptAndRun( - event, - confirm: { confirmEvent, fields in - probe.seenFields = fields - return .approved( - ToolInvocationEvent( - sessionID: confirmEvent.sessionID, - toolName: confirmEvent.toolName, - argumentsJSON: #"{"path":"/tmp/edited"}"#, - timestamp: confirmEvent.timestamp - ) - ) - }, - proceed: { gated in - XCTAssertEqual(gated.argumentsJSON, #"{"path":"/tmp/edited"}"#) - return gated.argumentsJSON - } - ) - - XCTAssertEqual(probe.seenFields, ["user_approval"]) - XCTAssertEqual(result, #"{"path":"/tmp/edited"}"#) - } - - func test_interceptAndRun_confirm_cancelled_throws_without_proceed() async throws { - struct ConfirmPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .confirmHITL(GuardrailHITLRequest(requiredFields: ["user_approval"])) - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: ConfirmPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "delete_file", - argumentsJSON: #"{}"# - ) - - let probe = CallProbe() - do { - _ = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in .cancelled(actor: "tester") }, - proceed: { _ -> String in - probe.proceedCalled = true - return "no" - } - ) - XCTFail("Expected cancelled error") - } catch ToolInvocationInterceptionError.cancelled(let reason) { - XCTAssertTrue(reason.contains("cancelled")) - XCTAssertTrue(reason.contains("tester")) - } - XCTAssertFalse(probe.proceedCalled) - } - - func test_interceptAndRun_redact_then_proceed_with_redacted_event() async throws { - struct RedactPolicy: ToolGovernancePolicy { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolGovernanceOutcome { - .redactArgument(argumentKey: "token", pattern: ".+", replacement: "[REDACTED]") - } - } - - let interceptor = DefaultToolRequestInterceptor(policy: RedactPolicy()) - let event = ToolInvocationEvent( - sessionID: "session-1", - toolName: "auth", - argumentsJSON: #"{"token":"secret"}"# - ) - - let result = try await interceptor.interceptAndRun( - event, - confirm: { _, _ in .cancelled(actor: nil) }, - proceed: { gated in - gated.argumentsJSON - } - ) - - XCTAssertTrue(result.contains("[REDACTED]")) - XCTAssertFalse(result.contains("secret")) - } -} diff --git a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedGuardrailEvaluating.swift similarity index 93% rename from packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift rename to packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedGuardrailEvaluating.swift index 2a97eedf..d4d11bd7 100644 --- a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedPolicyEvaluators.swift +++ b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedGuardrailEvaluating.swift @@ -1,9 +1,8 @@ import Foundation -import MemorySystem import Structure /// Store-backed tool governance: loads rules from `PolicyStore`, matches, returns first enabled hit. -public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { +public struct StoreBackedToolInvocationEvaluating: GuardrailEvaluating { private let store: any PolicyStore private let applicationName: String @@ -12,19 +11,19 @@ public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { self.applicationName = applicationName } - public func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> GuardrailDecision { - let rules = try await loadRules(scopes: ["tool_invocation", "tool_call"]) + public func evaluate(_ request: ToolInvocationEvent) async throws -> GuardrailDecision { + let rules = try await loadRules(scopes: GuardrailPolicyScope.toolInvocationScopes.map(\.rawValue)) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) } - let argumentsObject = parseJSONObject(from: event.argumentsJSON) + let argumentsObject = parseJSONObject(from: request.argumentsJSON) for rule in rules { guard rule.enabled else { continue } guard let matcher = try? decode(ToolMatcher.self, from: rule.matcherJSON) else { continue } - guard matcher.matches(event: event, arguments: argumentsObject) else { + guard matcher.matches(event: request, arguments: argumentsObject) else { continue } guard let outcome = try? decode(OutcomeRule.self, from: rule.outcomeJSON) else { @@ -57,7 +56,7 @@ public struct StoreBackedToolGovernancePolicy: ToolGovernancePolicy { } /// Store-backed assistant content policy for chunks and full completions. -public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { +public struct StoreBackedAssistantContentEvaluating: Sendable { private let store: any PolicyStore private let applicationName: String @@ -66,8 +65,8 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { self.applicationName = applicationName } - public func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> GuardrailDecision { - let rules = try await loadRules(scopes: ["assistant_chunk"]) + public func evaluate(_ request: AssistantChunkEvent) async throws -> GuardrailDecision { + let rules = try await loadRules(scopes: [GuardrailPolicyScope.assistantChunk.rawValue]) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) } @@ -77,7 +76,7 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { guard let matcher = try? decode(ContentMatcher.self, from: rule.matcherJSON) else { continue } - guard matcher.matches(content: event.content) else { + guard matcher.matches(content: request.content) else { continue } guard let outcome = try? decode(OutcomeRule.self, from: rule.outcomeJSON) else { @@ -89,19 +88,19 @@ public struct StoreBackedCompletionContentPolicy: PolicyEvaluator { return .deny(reason: Self.noMatchingRuleReason) } - public func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> GuardrailDecision { - let rules = try await loadRules(scopes: ["assistant_completion_content", "assistant_completion"]) + public func evaluate(_ request: AssistantCompletionEvent) async throws -> GuardrailDecision { + let rules = try await loadRules(scopes: GuardrailPolicyScope.assistantCompletionScopes.map(\.rawValue)) guard !rules.isEmpty else { return .deny(reason: Self.noRulesConfiguredReason) } - let detectedPatterns = detectSensitivePatterns(in: event.fullCompletion) + let detectedPatterns = detectSensitivePatterns(in: request.fullCompletion) for rule in rules { guard rule.enabled else { continue } guard let matcher = try? decode(CompletionMatcher.self, from: rule.matcherJSON) else { continue } - guard matcher.matches(content: event.fullCompletion, detectedPatterns: detectedPatterns) else { + guard matcher.matches(content: request.fullCompletion, detectedPatterns: detectedPatterns) else { continue } guard let outcome = try? decode(OutcomeRule.self, from: rule.outcomeJSON) else { diff --git a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowStartEvaluating.swift similarity index 96% rename from packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift rename to packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowStartEvaluating.swift index 0ae3c9fd..747e5aaa 100644 --- a/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowAdmissionPolicy.swift +++ b/packages/PolicyRuntime/Sources/PolicyRuntime/StoreBackedWorkflowStartEvaluating.swift @@ -1,8 +1,8 @@ import Foundation import Structure -/// Store-backed workflow start admission. Scope: `workflow_start`. -public struct StoreBackedWorkflowAdmissionPolicy: Sendable { +/// Store-backed `workflow_start` interpreter. Implements `GuardrailEvaluating`. +public struct StoreBackedWorkflowStartEvaluating: GuardrailEvaluating { private let store: any PolicyStore private let applicationName: String diff --git a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedGuardrailEvaluatingTests.swift similarity index 84% rename from packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift rename to packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedGuardrailEvaluatingTests.swift index a335a98d..dbe0353b 100644 --- a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedPolicyEvaluatorsTests.swift +++ b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedGuardrailEvaluatingTests.swift @@ -3,7 +3,7 @@ import MemorySystem @testable import PolicyRuntime import Structure -final class StoreBackedPolicyEvaluatorsTests: XCTestCase { +final class StoreBackedGuardrailEvaluatingTests: XCTestCase { func test_toolPolicy_loadsRelevantScopeAndDeniesMatch() async throws { let store = MockPolicyStore(rulesByScope: [ "tool_invocation": [ @@ -16,13 +16,13 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") let event = ToolInvocationEvent( sessionID: "s1", toolName: "delete_file", argumentsJSON: #"{"path":"/tmp/a"}"# ) - let outcome = try await policy.evaluateToolInvocation(event) + let outcome = try await policy.evaluate(event) XCTAssertEqual(outcome, .deny(reason: "blocked")) } @@ -39,8 +39,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent( sessionID: "factory-1", toolName: "tool_search", @@ -73,8 +73,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) XCTAssertEqual(outcome, .allow) @@ -82,11 +82,11 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { func test_toolPolicy_deniesWhenNoRulesConfigured() async throws { let store = MockPolicyStore(rulesByScope: [:]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) - XCTAssertEqual(outcome, .deny(reason: StoreBackedToolGovernancePolicy.noRulesConfiguredReason)) + XCTAssertEqual(outcome, .deny(reason: StoreBackedToolInvocationEvaluating.noRulesConfiguredReason)) } func test_toolPolicy_deniesWhenNoRuleMatches() async throws { @@ -102,11 +102,11 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) - XCTAssertEqual(outcome, .deny(reason: StoreBackedToolGovernancePolicy.noMatchingRuleReason)) + XCTAssertEqual(outcome, .deny(reason: StoreBackedToolInvocationEvaluating.noMatchingRuleReason)) } func test_toolPolicy_prefersHigherPriorityAcrossScopes() async throws { @@ -132,8 +132,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "file_write", argumentsJSON: "{}") ) XCTAssertEqual(outcome, .deny(reason: "high priority")) @@ -151,8 +151,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedCompletionContentPolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateAssistantCompletion( + let policy = StoreBackedAssistantContentEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "Please email me at hi@example.com", @@ -185,8 +185,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedCompletionContentPolicy(store: store, applicationName: "ui") - let outcome = try await policy.evaluateAssistantCompletion( + let policy = StoreBackedAssistantContentEvaluating(store: store, applicationName: "ui") + let outcome = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "hi@example.com", @@ -218,9 +218,9 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let withCode = try await policy.evaluateToolInvocation( + let withCode = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", @@ -229,7 +229,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(withCode, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) - let withoutCode = try await policy.evaluateToolInvocation( + let withoutCode = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", @@ -238,7 +238,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(withoutCode, .allow) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "read_file", @@ -268,19 +268,19 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let fileWrite = try await policy.evaluateToolInvocation( + let fileWrite = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "file_write", argumentsJSON: "{}") ) XCTAssertEqual(fileWrite, .deny(reason: "mutating tool")) - let deleteFile = try await policy.evaluateToolInvocation( + let deleteFile = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "delete_file", argumentsJSON: "{}") ) XCTAssertEqual(deleteFile, .deny(reason: "mutating tool")) - let readFile = try await policy.evaluateToolInvocation( + let readFile = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "read_file", argumentsJSON: "{}") ) XCTAssertEqual(readFile, .allow) @@ -306,14 +306,14 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let sessionTool = try await policy.evaluateToolInvocation( + let sessionTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "session_memory_search", argumentsJSON: "{}") ) XCTAssertEqual(sessionTool, .allow) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) XCTAssertEqual(otherTool, .deny(reason: "only session memory allowed")) @@ -346,9 +346,9 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") - let offlineScript = try await policy.evaluateToolInvocation( + let offlineScript = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", @@ -357,7 +357,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(offlineScript, .confirmHITL(GuardrailHITLRequest(requiredFields: ["offline_ok"]))) - let networkedScript = try await policy.evaluateToolInvocation( + let networkedScript = try await policy.evaluate( ToolInvocationEvent( sessionID: "s1", toolName: "script_exec", @@ -366,7 +366,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(networkedScript, .allow) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "read_file", argumentsJSON: "{}") ) XCTAssertEqual(otherTool, .allow) @@ -392,13 +392,13 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedToolGovernancePolicy(store: store, applicationName: "ui") - let sessionTool = try await policy.evaluateToolInvocation( + let policy = StoreBackedToolInvocationEvaluating(store: store, applicationName: "ui") + let sessionTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "session_memory_search", argumentsJSON: "{}") ) XCTAssertEqual(sessionTool, .confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"]))) - let otherTool = try await policy.evaluateToolInvocation( + let otherTool = try await policy.evaluate( ToolInvocationEvent(sessionID: "s1", toolName: "script_exec", argumentsJSON: "{}") ) XCTAssertEqual(otherTool, .allow) @@ -424,8 +424,8 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) ] ]) - let policy = StoreBackedCompletionContentPolicy(store: store, applicationName: "ui") - let ssn = try await policy.evaluateAssistantCompletion( + let policy = StoreBackedAssistantContentEvaluating(store: store, applicationName: "ui") + let ssn = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "SSN 123-45-6789", @@ -434,7 +434,7 @@ final class StoreBackedPolicyEvaluatorsTests: XCTestCase { ) XCTAssertEqual(ssn, .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"]))) - let plain = try await policy.evaluateAssistantCompletion( + let plain = try await policy.evaluate( AssistantCompletionEvent( sessionID: "s1", fullCompletion: "hello world", diff --git a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowStartEvaluatingTests.swift similarity index 94% rename from packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift rename to packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowStartEvaluatingTests.swift index 2301835e..d38c1c0f 100644 --- a/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowAdmissionPolicyTests.swift +++ b/packages/PolicyRuntime/Tests/PolicyRuntimeTests/StoreBackedWorkflowStartEvaluatingTests.swift @@ -2,13 +2,13 @@ import XCTest import Structure @testable import PolicyRuntime -final class StoreBackedWorkflowAdmissionPolicyTests: XCTestCase { +final class StoreBackedWorkflowStartEvaluatingTests: XCTestCase { func test_allowsSeededCreateAndDeniesEdit() async throws { let store = MockWorkflowPolicyStore() for rule in DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: "ui") { try await store.saveRule(rule) } - let policy = StoreBackedWorkflowAdmissionPolicy(store: store, applicationName: "ui") + let policy = StoreBackedWorkflowStartEvaluating(store: store, applicationName: "ui") let create = try await policy.evaluate( WorkflowStartRequest( @@ -47,7 +47,7 @@ final class StoreBackedWorkflowAdmissionPolicyTests: XCTestCase { ) ] ]) - let policy = StoreBackedWorkflowAdmissionPolicy(store: store, applicationName: "ui") + let policy = StoreBackedWorkflowStartEvaluating(store: store, applicationName: "ui") let decision = try await policy.evaluate( WorkflowStartRequest( kind: .pluginFactoryCreate, diff --git a/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift b/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift index db374a5d..e1cb2b73 100644 --- a/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift +++ b/packages/Structure/Sources/AppLayerServices/AppLayerFeatureCatalog.swift @@ -82,11 +82,15 @@ public enum AppLayerFeatureCatalog { AppLayerFeatureContract( feature: .guardrail, structureTypeNames: [ + "Guardrail", "GuardrailDecision", - "PolicyEvaluator", - "ToolGovernancePolicy", + "GuardrailEvaluating", + "GuardrailApplying", + "GuardrailHITLPresenting", "WorkflowKind", - "WorkflowAdmissionPolicy", + "WorkflowStartGuardrailApplying", + "ToolInvocationGuardrailApplying", + "AssistantContentGuardrailApplying", "PendingHITLApprovalRow", ], persistenceMayBeSQLite: true, diff --git a/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift b/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift index f299d07b..f67dedfe 100644 --- a/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift +++ b/packages/Structure/Sources/AppLayerServices/SharedAgentRuntime/ExecutionContextWire.swift @@ -96,6 +96,15 @@ public struct ExecutionContextWire: Codable, Sendable, Hashable { return try JSONDecoder.service.decode(ExecutionContextWire.self, from: data) } + /// Best-effort decode used at MCP / effector boundaries when context may be absent. + public static func parseOptionalJSON(_ json: String?) -> ExecutionContextWire? { + guard let json, + !json.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { + return nil + } + return try? decodeJSON(json) + } + public func withWorkflow( workflowID: String, kind: WorkflowKind, diff --git a/packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift new file mode 100644 index 00000000..526cd78b --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/AssistantContentGuardrailApplying.swift @@ -0,0 +1,51 @@ +import Foundation + +/// Applies a `GuardrailDecision` for assistant chunk / completion content. +public struct AssistantContentGuardrailApplying: Sendable { + public init() {} + + /// Streaming chunks: never modal mid-token. Confirm is deferred to completion. + public func apply( + _ decision: GuardrailDecision, + for chunk: AssistantChunkEvent + ) -> AssistantContentGuardrailOutput { + switch decision { + case .allow: + return .allowed(chunk.content) + case .deny(let reason): + return .denied(reason: reason) + case .redactContent(let pattern, let replacement): + let redacted = chunk.content.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + return .allowed(redacted) + case .confirmHITL, .requireWorkflow, .redactArgument: + return .allowed(chunk.content) + } + } + + public func apply( + _ decision: GuardrailDecision, + for completion: AssistantCompletionEvent + ) -> AssistantContentGuardrailOutput { + switch decision { + case .allow: + return .allowed(completion.fullCompletion) + case .deny(let reason): + return .denied(reason: reason) + case .redactContent(let pattern, let replacement): + let redacted = completion.fullCompletion.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + return .allowed(redacted) + case .confirmHITL(let hitl): + return .confirm(content: completion.fullCompletion, requiredFields: hitl.requiredFields) + case .requireWorkflow, .redactArgument: + return .denied(reason: "Content policy returned an unsupported control decision.") + } + } +} diff --git a/packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift new file mode 100644 index 00000000..6251703f --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/GuardrailApplying.swift @@ -0,0 +1,8 @@ +import Foundation + +/// Applies a `GuardrailDecision` for one request kind. +public protocol GuardrailApplying: Sendable { + associatedtype Request: Sendable + associatedtype Output: Sendable + func apply(_ decision: GuardrailDecision, for request: Request) async throws -> Output +} diff --git a/packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift new file mode 100644 index 00000000..9b6dfe9b --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/ToolInvocationGuardrailApplying.swift @@ -0,0 +1,97 @@ +import Foundation + +/// Applies a `GuardrailDecision` for tool invocations. +public struct ToolInvocationGuardrailApplying: GuardrailApplying { + public typealias Request = ToolInvocationEvent + public typealias Output = ToolInvocationEvent + + private let hitl: (any GuardrailHITLPresenting)? + + public init(hitl: (any GuardrailHITLPresenting)? = nil) { + self.hitl = hitl + } + + public func apply( + _ decision: GuardrailDecision, + for request: ToolInvocationEvent + ) async throws -> ToolInvocationEvent { + switch decision { + case .allow: + return request + case .deny(let reason): + throw ToolInvocationGuardrailError.denied(reason: reason) + case .confirmHITL(let hitlRequest): + guard let hitl else { + throw ToolInvocationGuardrailError.denied( + reason: "Tool requires HITL approval but no presenter is configured." + ) + } + let presentation = GuardrailHITLPresentation( + sessionID: request.sessionID, + turnID: request.sessionID, + subject: request.toolName, + payloadJSON: request.argumentsJSON, + hitl: hitlRequest + ) + switch await hitl.resolve(presentation) { + case .approved(let editedPayloadJSON, _): + guard let editedPayloadJSON else { return request } + return ToolInvocationEvent( + sessionID: request.sessionID, + toolName: request.toolName, + argumentsJSON: editedPayloadJSON, + timestamp: request.timestamp + ) + case .cancelled(let actor): + let suffix = actor.map { " by \($0)" } ?? "" + throw ToolInvocationGuardrailError.cancelled( + reason: "User cancelled the approval request\(suffix)" + ) + } + case .redactArgument(let key, let pattern, let replacement): + return ToolInvocationEvent( + sessionID: request.sessionID, + toolName: request.toolName, + argumentsJSON: redactArgumentJSON( + request.argumentsJSON, + key: key, + pattern: pattern, + replacement: replacement + ), + timestamp: request.timestamp + ) + case .requireWorkflow(let kind): + throw ToolInvocationGuardrailError.denied( + reason: "Tool requires workflow '\(kind.rawValue)' before it can run." + ) + case .redactContent: + throw ToolInvocationGuardrailError.denied( + reason: "Tool policy returned an unsupported control decision." + ) + } + } +} + +private func redactArgumentJSON( + _ json: String, + key: String, + pattern: String, + replacement: String +) -> String { + guard let data = json.data(using: .utf8), + var object = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { + return json + } + if let stringValue = object[key] as? String { + object[key] = stringValue.replacingOccurrences( + of: pattern, + with: replacement, + options: .regularExpression + ) + } + guard let redacted = try? JSONSerialization.data(withJSONObject: object, options: [.sortedKeys]), + let redactedString = String(data: redacted, encoding: .utf8) else { + return json + } + return redactedString +} diff --git a/packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift b/packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift new file mode 100644 index 00000000..4fd855a3 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Applying/WorkflowStartGuardrailApplying.swift @@ -0,0 +1,50 @@ +import Foundation + +/// Applies a `GuardrailDecision` for workflow starts. +public struct WorkflowStartGuardrailApplying: GuardrailApplying { + public typealias Request = WorkflowStartRequest + public typealias Output = Void + + private let hitl: any GuardrailHITLPresenting + + public init(hitl: any GuardrailHITLPresenting) { + self.hitl = hitl + } + + public func apply(_ decision: GuardrailDecision, for request: WorkflowStartRequest) async throws { + switch decision { + case .allow: + return + case .deny(let reason): + throw WorkflowRuntimeError.deniedByGuardrail(reason) + case .confirmHITL(let hitlRequest): + let isJob: Bool = { + if case .job = request.principal { return true } + return false + }() + let presentation = GuardrailHITLPresentation( + sessionID: request.sessionID, + turnID: request.turnID ?? request.sessionID, + subject: "workflow_start:\(request.kind.rawValue)", + payloadJSON: request.inputJSON, + hitl: hitlRequest, + isJobContext: isJob + ) + switch await hitl.resolve(presentation) { + case .approved: + return + case .cancelled: + let detail = hitlRequest.message ?? hitlRequest.title ?? "Workflow start was not approved." + throw WorkflowRuntimeError.deniedByGuardrail(detail) + } + case .requireWorkflow: + throw WorkflowRuntimeError.deniedByGuardrail( + "requireWorkflow is not a valid decision for workflow start." + ) + case .redactContent, .redactArgument: + throw WorkflowRuntimeError.deniedByGuardrail( + "Redaction is not a valid decision for workflow start." + ) + } + } +} diff --git a/packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift b/packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift new file mode 100644 index 00000000..4a82fe70 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Evaluating/GuardrailEvaluating.swift @@ -0,0 +1,7 @@ +import Foundation + +/// Interprets a Policy rule set for one request kind into a `GuardrailDecision`. +public protocol GuardrailEvaluating: Sendable { + associatedtype Request: Sendable + func evaluate(_ request: Request) async throws -> GuardrailDecision +} diff --git a/packages/Structure/Sources/Guardrail/Guardrail.swift b/packages/Structure/Sources/Guardrail/Guardrail.swift index 120d5e1e..e2b026cb 100644 --- a/packages/Structure/Sources/Guardrail/Guardrail.swift +++ b/packages/Structure/Sources/Guardrail/Guardrail.swift @@ -2,11 +2,11 @@ import Foundation /// Derrick's agent-control umbrella in Structure. /// -/// **Policy decides.** HITL and workflows are enforcement shapes for those decisions. -/// Plugins propose actions; they do not authorize control outcomes. +/// Flow: **Policy evaluates rules → adapters apply `GuardrailDecision` → chokepoints only call those two.** /// -/// Folder layout: -/// - `Guardrail/Policy` — rules and interception contracts -/// - `Guardrail/HITL` — human confirmation wire types and presentation signals -/// - `Guardrail/Workflow` — workflow kinds, runtime DTOs, effector admission +/// Naming (no exceptions): +/// - `Guardrail*` — control-plane types +/// - `*Evaluating` — rule interpreters (`Request` → `GuardrailDecision`) +/// - `*Applying` — decision adapters (decision → effect) +/// - `StoreBacked*Evaluating` — SQLite-backed interpreters in PolicyRuntime public enum Guardrail: Sendable {} diff --git a/packages/Structure/Sources/Guardrail/GuardrailDecision.swift b/packages/Structure/Sources/Guardrail/GuardrailDecision.swift index 4481aa41..ee5df253 100644 --- a/packages/Structure/Sources/Guardrail/GuardrailDecision.swift +++ b/packages/Structure/Sources/Guardrail/GuardrailDecision.swift @@ -1,12 +1,10 @@ import Foundation -/// Canonical control-plane decision. Every Policy evaluator maps to this. +/// Canonical control-plane decision. Every `*Evaluating` type returns this; every `*Applying` type consumes it. public enum GuardrailDecision: Hashable, Sendable { case allow case deny(reason: String) - /// Require a human gate before proceeding. case confirmHITL(GuardrailHITLRequest) - /// Require a sequenced workflow before / instead of the proposed action. case requireWorkflow(WorkflowKind) case redactContent(pattern: String, replacement: String) case redactArgument(argumentKey: String, pattern: String, replacement: String) @@ -32,15 +30,4 @@ public struct GuardrailHITLRequest: Hashable, Sendable { self.title = title self.message = message } - - public init(_ confirmation: PolicyConfirmationRequest) { - self.requiredFields = ["user_approval"] - self.title = confirmation.title - self.message = confirmation.message - } -} - -/// Something Policy can evaluate into a `GuardrailDecision`. -public protocol GuardrailEvaluating: Sendable { - func guardrailDecision() async throws -> GuardrailDecision } diff --git a/packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift b/packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift new file mode 100644 index 00000000..2b6a9465 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/HITL/GuardrailHITLPresenting.swift @@ -0,0 +1,40 @@ +import Foundation + +/// Context for presenting a Policy `confirmHITL` decision. +public struct GuardrailHITLPresentation: Sendable, Hashable { + public let id: String + public let sessionID: String + public let turnID: String + public let subject: String + public let payloadJSON: String + public let hitl: GuardrailHITLRequest + public let isJobContext: Bool + + public init( + id: String = UUID().uuidString, + sessionID: String, + turnID: String, + subject: String, + payloadJSON: String, + hitl: GuardrailHITLRequest, + isJobContext: Bool = false + ) { + self.id = id + self.sessionID = sessionID + self.turnID = turnID + self.subject = subject + self.payloadJSON = payloadJSON + self.hitl = hitl + self.isJobContext = isJobContext + } +} + +public enum GuardrailHITLResolution: Equatable, Sendable { + case approved(editedPayloadJSON: String?, actor: String?) + case cancelled(actor: String?) +} + +/// Presents HITL and returns the human resolution. One implementation shared by all adapters. +public protocol GuardrailHITLPresenting: Sendable { + func resolve(_ presentation: GuardrailHITLPresentation) async -> GuardrailHITLResolution +} diff --git a/packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift b/packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift new file mode 100644 index 00000000..60408353 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/HITL/ImmediateGuardrailHITLPresenting.swift @@ -0,0 +1,36 @@ +import Foundation + +/// Test / non-interactive HITL presenter that returns a fixed resolution. +public struct ImmediateGuardrailHITLPresenting: GuardrailHITLPresenting { + private let resolution: GuardrailHITLResolution + + public init(approved: Bool) { + self.resolution = approved + ? .approved(editedPayloadJSON: nil, actor: nil) + : .cancelled(actor: nil) + } + + public init(resolution: GuardrailHITLResolution) { + self.resolution = resolution + } + + public func resolve(_ presentation: GuardrailHITLPresentation) async -> GuardrailHITLResolution { + _ = presentation + return resolution + } +} + +/// Bridges a closure to `GuardrailHITLPresenting`. +public struct ClosureGuardrailHITLPresenting: GuardrailHITLPresenting { + private let handler: @Sendable (GuardrailHITLPresentation) async -> GuardrailHITLResolution + + public init( + _ handler: @escaping @Sendable (GuardrailHITLPresentation) async -> GuardrailHITLResolution + ) { + self.handler = handler + } + + public func resolve(_ presentation: GuardrailHITLPresentation) async -> GuardrailHITLResolution { + await handler(presentation) + } +} diff --git a/packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift b/packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift new file mode 100644 index 00000000..30dec3f5 --- /dev/null +++ b/packages/Structure/Sources/Guardrail/Policy/GuardrailPolicyScope.swift @@ -0,0 +1,17 @@ +import Foundation + +/// Policy rule scopes. Interpreters load only their scope(s). +public enum GuardrailPolicyScope: String, Sendable, Hashable, CaseIterable { + case toolInvocation = "tool_invocation" + /// Legacy alias still accepted when loading tool rules. + case toolCall = "tool_call" + case workflowStart = "workflow_start" + case assistantChunk = "assistant_chunk" + case assistantCompletionContent = "assistant_completion_content" + case assistantCompletion = "assistant_completion" + + public static var toolInvocationScopes: [GuardrailPolicyScope] { [.toolInvocation, .toolCall] } + public static var assistantCompletionScopes: [GuardrailPolicyScope] { + [.assistantCompletionContent, .assistantCompletion] + } +} diff --git a/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift index 51028dc3..6215f304 100644 --- a/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyEngine/PolicyEngineTypes.swift @@ -73,7 +73,11 @@ public struct PolicyConfirmationRequest: Hashable, Sendable { } public var guardrailHITL: GuardrailHITLRequest { - GuardrailHITLRequest(self) + GuardrailHITLRequest( + requiredFields: ["user_approval"], + title: title, + message: message + ) } } diff --git a/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift index 39a4241e..ce83f8f1 100644 --- a/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/PolicyInterception.swift @@ -1,24 +1,9 @@ import Foundation -/// Content Policy evaluator. Returns the canonical `GuardrailDecision`. -public protocol PolicyEvaluator: Sendable { - func evaluateAssistantChunk(_ event: AssistantChunkEvent) async throws -> GuardrailDecision - func evaluateAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> GuardrailDecision -} - -/// Result of content policy interception (preserves deny reasons for UI). -public enum AssistantContentInterceptResult: Equatable, Sendable { +/// Result of applying a content `GuardrailDecision` (preserves deny reasons for UI). +public enum AssistantContentGuardrailOutput: Equatable, Sendable { case allowed(String) case denied(reason: String) /// Policy requires user approval before this content is accepted as final. case confirm(content: String, requiredFields: [String]) } - -public protocol PolicyInterceptor: Sendable { - func interceptAssistantChunk(_ event: AssistantChunkEvent) async throws -> AssistantContentInterceptResult - func interceptAssistantCompletion(_ event: AssistantCompletionEvent) async throws -> AssistantContentInterceptResult -} - -/// Historical alias — prefer `GuardrailDecision`. -@available(*, deprecated, renamed: "GuardrailDecision") -public typealias PolicyDecisionOutcome = GuardrailDecision diff --git a/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift index 7333a44f..d415e93f 100644 --- a/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift +++ b/packages/Structure/Sources/Guardrail/Policy/PolicyRuntime/ToolInterception.swift @@ -1,47 +1,7 @@ import Foundation -public enum ToolInterceptionDecision: Equatable, Sendable { - case allow(ToolInvocationEvent) - case deny(reason: String) - case confirm(ToolInvocationEvent, requiredFields: [String]) -} - -/// Errors thrown by confirm-before-proceed tool interception. -public enum ToolInvocationInterceptionError: Error, Equatable, Sendable { +/// Errors thrown while applying a tool `GuardrailDecision`. +public enum ToolInvocationGuardrailError: Error, Equatable, Sendable { case denied(reason: String) case cancelled(reason: String) } - -/// Result of asking the user to approve a tool that policy marked as `confirmHITL`. -public enum ToolInvocationConfirmation: Equatable, Sendable { - /// Proceed with this (possibly edited) invocation event. - case approved(ToolInvocationEvent) - /// User (or system) cancelled confirmation. - case cancelled(actor: String?) -} - -/// Tool Policy evaluator. Returns the canonical `GuardrailDecision`. -public protocol ToolGovernancePolicy: Sendable { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> GuardrailDecision -} - -public protocol ToolRequestInterceptor: Sendable { - func evaluateToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInterceptionDecision - func interceptToolInvocation(_ event: ToolInvocationEvent) async throws -> ToolInvocationEvent? - - /// Evaluate policy, optionally confirm with the user, then run `proceed` with the gated event. - /// - /// Control flow: - /// - allow / redact → `proceed(processedEvent)` - /// - deny → throws `ToolInvocationInterceptionError.denied` - /// - confirmHITL → `confirm(...)`; on approve → `proceed`; on cancel → throws `.cancelled` - func interceptAndRun( - _ event: ToolInvocationEvent, - confirm: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent, [String]) async throws -> ToolInvocationConfirmation, - proceed: nonisolated(nonsending) @escaping @Sendable (ToolInvocationEvent) async throws -> R - ) async throws -> R -} - -/// Historical alias — prefer `GuardrailDecision`. -@available(*, deprecated, renamed: "GuardrailDecision") -public typealias ToolGovernanceOutcome = GuardrailDecision diff --git a/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift b/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift deleted file mode 100644 index 5b4bf0e0..00000000 --- a/packages/Structure/Sources/Guardrail/Workflow/EffectorAdmissionPolicy.swift +++ /dev/null @@ -1,12 +0,0 @@ -import Foundation - -/// Helpers for effector execution context. Admission itself is Policy (`tool_invocation` rules). -public enum EffectorAdmissionPolicy: Sendable { - public static func parseContextJSON(_ json: String?) -> ExecutionContextWire? { - guard let json, - !json.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { - return nil - } - return try? ExecutionContextWire.decodeJSON(json) - } -} diff --git a/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift b/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift deleted file mode 100644 index 14934035..00000000 --- a/packages/Structure/Sources/Guardrail/Workflow/WorkflowAdmissionPolicy.swift +++ /dev/null @@ -1,39 +0,0 @@ -import Foundation - -/// Thin Guardrail adapter for workflow starts. -/// -/// Policy owns the decision (`StoreBackedWorkflowAdmissionPolicy` / seeded `workflow_start` rules). -/// This type only applies that decision: allow, deny, or confirm HITL then continue/cancel. -public enum WorkflowAdmissionPolicy: Sendable { - /// Apply a Policy decision for starting a workflow. - /// - /// - Parameters: - /// - decision: From the Policy engine. - /// - confirm: Invoked on `.confirmHITL`. Return `true` to proceed, `false` to cancel. - public static func apply( - _ decision: GuardrailDecision, - confirm: () async -> Bool - ) async throws { - switch decision { - case .allow: - return - case .deny(let reason): - throw WorkflowRuntimeError.deniedByGuardrail(reason) - case .confirmHITL(let hitl): - let approved = await confirm() - if approved { - return - } - let detail = hitl.message ?? hitl.title ?? "Workflow start was not approved." - throw WorkflowRuntimeError.deniedByGuardrail(detail) - case .requireWorkflow: - throw WorkflowRuntimeError.deniedByGuardrail( - "requireWorkflow is not a valid decision for workflow start." - ) - case .redactContent, .redactArgument: - throw WorkflowRuntimeError.deniedByGuardrail( - "Redaction is not a valid decision for workflow start." - ) - } - } -} diff --git a/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift b/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift index d5f6115e..76c15915 100644 --- a/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift +++ b/packages/Structure/Tests/StructureTests/AppLayerServicesWireTests.swift @@ -1413,17 +1413,17 @@ import Testing #expect(fromMinutes.containerRunMaxTTLMinutes == 12) } - @Test func effectorAdmissionParseContextJSON() throws { + @Test func executionContextParseOptionalJSON() throws { let context = ExecutionContextWire( sessionID: "s", principal: .ui, capabilities: [.syncWebCrawl] ) let json = try context.encodedJSON() - let parsed = EffectorAdmissionPolicy.parseContextJSON(json) + let parsed = ExecutionContextWire.parseOptionalJSON(json) #expect(parsed?.capabilities.contains(.syncWebCrawl) == true) - #expect(EffectorAdmissionPolicy.parseContextJSON(nil) == nil) - #expect(EffectorAdmissionPolicy.parseContextJSON(" ") == nil) + #expect(ExecutionContextWire.parseOptionalJSON(nil) == nil) + #expect(ExecutionContextWire.parseOptionalJSON(" ") == nil) } @Test func executionContextWireRoundTrip() throws { diff --git a/packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift b/packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift new file mode 100644 index 00000000..a0fc0003 --- /dev/null +++ b/packages/Structure/Tests/StructureTests/GuardrailApplyingTests.swift @@ -0,0 +1,71 @@ +import Foundation +import Structure +import XCTest + +/// Applying-layer coverage formerly owned by MemorySystem interceptor tests. +final class GuardrailApplyingTests: XCTestCase { + func test_toolApplying_allow_returnsEvent() async throws { + let event = ToolInvocationEvent(sessionID: "s", toolName: "t", argumentsJSON: "{}") + let gated = try await ToolInvocationGuardrailApplying().apply(.allow, for: event) + XCTAssertEqual(gated.toolName, "t") + } + + func test_toolApplying_deny_throws() async { + let event = ToolInvocationEvent(sessionID: "s", toolName: "t", argumentsJSON: "{}") + do { + _ = try await ToolInvocationGuardrailApplying().apply(.deny(reason: "no"), for: event) + XCTFail("expected deny") + } catch let error as ToolInvocationGuardrailError { + guard case .denied(let reason) = error else { + return XCTFail("expected denied") + } + XCTAssertEqual(reason, "no") + } catch { + XCTFail("unexpected \(error)") + } + } + + func test_toolApplying_redactArgument() async throws { + let event = ToolInvocationEvent( + sessionID: "s", + toolName: "t", + argumentsJSON: #"{"token":"abc-secret"}"# + ) + let gated = try await ToolInvocationGuardrailApplying().apply( + .redactArgument(argumentKey: "token", pattern: "secret", replacement: "[x]"), + for: event + ) + XCTAssertTrue(gated.argumentsJSON.contains("[x]")) + XCTAssertFalse(gated.argumentsJSON.contains("secret")) + } + + func test_toolApplying_confirmHITL_approved() async throws { + let event = ToolInvocationEvent(sessionID: "s", toolName: "t", argumentsJSON: #"{"a":1}"#) + let hitl = ImmediateGuardrailHITLPresenting( + resolution: .approved(editedPayloadJSON: #"{"a":2}"#, actor: "user") + ) + let gated = try await ToolInvocationGuardrailApplying(hitl: hitl).apply( + .confirmHITL(GuardrailHITLRequest()), + for: event + ) + XCTAssertEqual(gated.argumentsJSON, #"{"a":2}"#) + } + + func test_contentApplying_chunk_softAllowsConfirm() { + let chunk = AssistantChunkEvent(sessionID: "s", chunkIndex: 0, content: "mid") + let out = AssistantContentGuardrailApplying().apply( + .confirmHITL(GuardrailHITLRequest()), + for: chunk + ) + XCTAssertEqual(out, .allowed("mid")) + } + + func test_contentApplying_completion_confirm() { + let completion = AssistantCompletionEvent(sessionID: "s", fullCompletion: "full", chunkCount: 1) + let out = AssistantContentGuardrailApplying().apply( + .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"])), + for: completion + ) + XCTAssertEqual(out, .confirm(content: "full", requiredFields: ["review"])) + } +} diff --git a/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift index ece3acea..30679a3d 100644 --- a/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift +++ b/packages/Structure/Tests/StructureTests/GuardrailDecisionTests.swift @@ -19,10 +19,18 @@ struct GuardrailDecisionTests { #expect(WorkflowKind.allCases.contains(.pluginFactoryEdit)) } - @Test func workflowAdmissionApplyAllowAndDeny() async throws { - try await WorkflowAdmissionPolicy.apply(.allow) { true } + @Test func workflowStartApplyingAllowAndDeny() async throws { + let applying = WorkflowStartGuardrailApplying(hitl: ImmediateGuardrailHITLPresenting(approved: true)) + let request = WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + try await applying.apply(.allow, for: request) do { - try await WorkflowAdmissionPolicy.apply(.deny(reason: "blocked")) { true } + try await applying.apply(.deny(reason: "blocked"), for: request) Issue.record("deny must throw") } catch let error as WorkflowRuntimeError { if case .deniedByGuardrail(let reason) = error { @@ -33,15 +41,20 @@ struct GuardrailDecisionTests { } } - @Test func workflowAdmissionApplyConfirmHITL() async throws { - try await WorkflowAdmissionPolicy.apply( - .confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"])) - ) { true } + @Test func workflowStartApplyingConfirmHITL() async throws { + let request = WorkflowStartRequest( + kind: .pluginFactoryCreate, + sessionID: "s", + agentID: "ui", + inputJSON: "{}", + principal: .ui + ) + try await WorkflowStartGuardrailApplying(hitl: ImmediateGuardrailHITLPresenting(approved: true)) + .apply(.confirmHITL(GuardrailHITLRequest(requiredFields: ["ok"])), for: request) do { - try await WorkflowAdmissionPolicy.apply( - .confirmHITL(GuardrailHITLRequest(title: "Need approval")) - ) { false } + try await WorkflowStartGuardrailApplying(hitl: ImmediateGuardrailHITLPresenting(approved: false)) + .apply(.confirmHITL(GuardrailHITLRequest(title: "Need approval")), for: request) Issue.record("cancelled HITL must throw") } catch let error as WorkflowRuntimeError { if case .deniedByGuardrail = error { @@ -52,10 +65,69 @@ struct GuardrailDecisionTests { } } + @Test func toolInvocationApplyingAllowDenyRedact() async throws { + let applying = ToolInvocationGuardrailApplying() + let event = ToolInvocationEvent( + sessionID: "s", + toolName: "script_exec", + argumentsJSON: #"{"code":"secret-token"}"# + ) + let allowed = try await applying.apply(.allow, for: event) + #expect(allowed.toolName == "script_exec") + + do { + _ = try await applying.apply(.deny(reason: "nope"), for: event) + Issue.record("deny must throw") + } catch let error as ToolInvocationGuardrailError { + if case .denied(let reason) = error { + #expect(reason == "nope") + } else { + Issue.record("Expected denied") + } + } + + let redacted = try await applying.apply( + .redactArgument(argumentKey: "code", pattern: "secret", replacement: "[REDACTED]"), + for: event + ) + #expect(redacted.argumentsJSON.contains("[REDACTED]")) + #expect(!redacted.argumentsJSON.contains("secret-token")) + } + + @Test func assistantContentApplyingChunkAndCompletion() { + let applying = AssistantContentGuardrailApplying() + let chunk = AssistantChunkEvent(sessionID: "s", chunkIndex: 0, content: "hello secret") + let chunkOut = applying.apply( + .redactContent(pattern: "secret", replacement: "[x]"), + for: chunk + ) + #expect(chunkOut == .allowed("hello [x]")) + + let completion = AssistantCompletionEvent(sessionID: "s", fullCompletion: "done", chunkCount: 1) + let confirmOut = applying.apply( + .confirmHITL(GuardrailHITLRequest(requiredFields: ["review"])), + for: completion + ) + #expect(confirmOut == .confirm(content: "done", requiredFields: ["review"])) + } + @Test func defaultWorkflowSeedsCoverCreateAndDenyEdit() { let rules = DefaultGuardrailPolicySeeds.workflowStartRules(applicationName: "ui") #expect(rules.contains { $0.name == "allow-workflow-plugin-factory-create" }) #expect(rules.contains { $0.name == "deny-workflow-plugin-factory-edit" }) #expect(rules.allSatisfy { $0.scope == "workflow_start" }) } + + @Test func executionContextParseOptionalJSON() throws { + let context = ExecutionContextWire( + sessionID: "s", + principal: .ui, + capabilities: [.syncWebCrawl] + ) + let json = try context.encodedJSON() + let parsed = ExecutionContextWire.parseOptionalJSON(json) + #expect(parsed?.capabilities.contains(.syncWebCrawl) == true) + #expect(ExecutionContextWire.parseOptionalJSON(nil) == nil) + #expect(ExecutionContextWire.parseOptionalJSON(" ") == nil) + } } diff --git a/readme.md b/readme.md index 86c833e2..55b50fc1 100644 --- a/readme.md +++ b/readme.md @@ -96,10 +96,10 @@ Copy [.env.example](.env.example) — **never commit `.env`**. ### Guardrail -- **`Structure/Guardrail`** — control plane umbrella. Policy decides; HITL and workflows enforce. +- **`Structure/Guardrail`** — control plane. Policy evaluates; adapters apply `GuardrailDecision`; chokepoints only call those two. +- Naming: `Guardrail*`, `*Evaluating`, `*Applying`, `StoreBacked*Evaluating`. - Persisted rules in SQLite (`Guardrail/Policy`, `packages/PolicyRuntime`) produce `GuardrailDecision`. -- **`PolicyInterceptor`** / **`ToolRequestInterceptor`** (MemorySystem) — pipeline hooks that apply those decisions. -- Workflow starts and MCP tools are admitted by Policy rules (`workflow_start` / `tool_invocation`); thin adapters apply allow / deny / confirmHITL / redact. +- Workflow starts and MCP tools are admitted by Policy rules (`workflow_start` / `tool_invocation`); thin `*GuardrailApplying` adapters apply allow / deny / confirmHITL / redact. ### Messaging @@ -117,7 +117,7 @@ Connectors are just-in-time software with `role: connector`. They must emit a Ho | `packages/Plugin` | Just-in-time software factory (builder, reviewer, release) | | `packages/DerrickBackend` | Daemon runtime, notifications, HITL polling | | `packages/DockerRunnerXPC` | Constrained Docker helper | -| `packages/PolicyRuntime` | Store-backed policy evaluators (`Structure/Guardrail/Policy`) | +| `packages/PolicyRuntime` | `StoreBacked*Evaluating` interpreters (`Structure/Guardrail`) | | `packages/LLMAgentClient` | Provider clients (OpenAI, Gemini, …) | ## Quick start diff --git a/ui/MCPService/MCPServiceToolHost.swift b/ui/MCPService/MCPServiceToolHost.swift index 3cf510af..03bc0898 100644 --- a/ui/MCPService/MCPServiceToolHost.swift +++ b/ui/MCPService/MCPServiceToolHost.swift @@ -289,9 +289,9 @@ actor MCPServiceToolHost { helperReviewerModelJSON: request.helperReviewerModelJSON, memorySessionKey: sessionKey, pluginFactoryCreationActive: request.pluginFactoryCreationActive - || EffectorAdmissionPolicy.parseContextJSON(request.executionContextJSON)? + || ExecutionContextWire.parseOptionalJSON(request.executionContextJSON)? .capabilities.contains(.syncWebCrawl) == true, - workflowID: EffectorAdmissionPolicy.parseContextJSON(request.executionContextJSON)? + workflowID: ExecutionContextWire.parseOptionalJSON(request.executionContextJSON)? .workflow?.workflowID ) let jobID: String? @@ -302,10 +302,9 @@ actor MCPServiceToolHost { HostHTTPCallContext.shared.clear() } - // Policy decides effector/tool admission (same engine as chat). Apply allow / deny / - // confirmHITL / redact before the tool runs. + // Policy decides effector/tool admission (same engine as chat). Evaluate then apply. let policyRepo = try await MCPServiceStore.shared.sharedRepository() - let toolPolicy = StoreBackedToolGovernancePolicy( + let toolEvaluating = StoreBackedToolInvocationEvaluating( store: policyRepo, applicationName: DerrickAppSupport.defaultApplicationName ) @@ -314,18 +313,34 @@ actor MCPServiceToolHost { if case .job(let jobID) = request.principal { return jobID } return "mcp-service" }() - var argumentsJSON = request.argumentsJSON - let decision = try await toolPolicy.evaluateToolInvocation( - ToolInvocationEvent( + let invocationEvent = ToolInvocationEvent( + sessionID: sessionIDForPolicy, + toolName: toolName, + argumentsJSON: request.argumentsJSON + ) + let decision = try await toolEvaluating.evaluate(invocationEvent) + let hitl = ClosureGuardrailHITLPresenting { [self] presentation in + let resolved = await awaitMCPToolHITL( sessionID: sessionIDForPolicy, - toolName: toolName, - argumentsJSON: argumentsJSON + toolName: presentation.subject, + argumentsJSON: presentation.payloadJSON, + hitl: presentation.hitl, + principal: request.principal, + repository: policyRepo ) - ) - switch decision { - case .allow: - break - case .deny(let reason): + switch resolved { + case .approved(let edited, let actor): + return .approved(editedPayloadJSON: edited, actor: actor) + case .cancelled(let actor): + return .cancelled(actor: actor) + } + } + let argumentsJSON: String + do { + let gated = try await ToolInvocationGuardrailApplying(hitl: hitl) + .apply(decision, for: invocationEvent) + argumentsJSON = gated.argumentsJSON + } catch ToolInvocationGuardrailError.denied(let reason) { await MCPServiceStore.shared.log( level: .error, message: "tool denied by Policy tool=\(toolName): \(reason)", @@ -339,43 +354,13 @@ actor MCPServiceToolHost { text: "", message: reason ) - case .confirmHITL(let hitl): - let approved = await awaitMCPToolHITL( - sessionID: sessionIDForPolicy, - toolName: toolName, - argumentsJSON: argumentsJSON, - hitl: hitl, - principal: request.principal, - repository: policyRepo - ) - switch approved { - case .approved(let edited, _): - argumentsJSON = edited - case .cancelled(let actor): - let message = "Tool \(toolName) was not approved\(actor.map { " by \($0)" } ?? "")." - return MCPToolCallResultDTO( - requestID: request.requestID, - ok: false, - isError: true, - text: "", - message: message - ) - } - case .redactArgument(let key, let pattern, let replacement): - argumentsJSON = redactToolArgumentJSON( - argumentsJSON, - key: key, - pattern: pattern, - replacement: replacement - ) - case .requireWorkflow, .redactContent: - let message = "Tool \(toolName) returned an unsupported Policy decision." + } catch ToolInvocationGuardrailError.cancelled(let reason) { return MCPToolCallResultDTO( requestID: request.requestID, ok: false, isError: true, text: "", - message: message + message: reason ) } @@ -520,30 +505,6 @@ actor MCPServiceToolHost { } } -private func redactToolArgumentJSON( - _ json: String, - key: String, - pattern: String, - replacement: String -) -> String { - guard let data = json.data(using: .utf8), - var object = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { - return json - } - if let stringValue = object[key] as? String { - object[key] = stringValue.replacingOccurrences( - of: pattern, - with: replacement, - options: .regularExpression - ) - } - guard let redacted = try? JSONSerialization.data(withJSONObject: object, options: [.sortedKeys]), - let redactedString = String(data: redacted, encoding: .utf8) else { - return json - } - return redactedString -} - private func pluginFactoryFailureDetail(for error: Error) -> String { if let factoryError = error as? PluginFactoryError { return factoryError.localizedDescription diff --git a/ui/SharedAgentRuntime/Conversation/ConversationModel.swift b/ui/SharedAgentRuntime/Conversation/ConversationModel.swift index 61619771..b0a76a25 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationModel.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationModel.swift @@ -89,8 +89,9 @@ final class ConversationModel { policy: TieredMemoryCompactionPolicy(), budget: budget ) - let interceptor = DefaultPolicyInterceptor( - policy: StoreBackedCompletionContentPolicy(store: repository, applicationName: "ui") + let contentEvaluating = StoreBackedAssistantContentEvaluating( + store: repository, + applicationName: "ui" ) let delegateAgentsHost = try await makeAgentsOrchestrationHost( @@ -143,7 +144,7 @@ final class ConversationModel { ragInstructions: ragInstructions, mcpToolInstructions: mcpToolInstructions, responseSchema: Self.defaultResponseSchema, - interceptor: interceptor + contentEvaluating: contentEvaluating ) await ProfileSubagentGate.shared.end(caller: caller) return result @@ -290,7 +291,7 @@ final class ConversationModel { let ragInstructions = self.ragInstructions let mcpToolInstructions = self.mcpToolInstructions let responseSchema = self.responseSchema - let interceptor = makeContentPolicyInterceptor() + let contentEvaluating = makeContentEvaluating() let orchestrator = self.orchestrator let workerModel = helperModelSettings.workerAgentModel let workerApiKey = await LLMProviderCredentialGate.resolveAPIKey(for: workerModel) ?? apiKey @@ -345,7 +346,7 @@ final class ConversationModel { ragInstructions: rag, mcpToolInstructions: mcpToolInstructions, responseSchema: responseSchema, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: nil ) var completeText = "" @@ -385,7 +386,7 @@ final class ConversationModel { ragInstructions: userRagBase, mcpToolInstructions: effectiveMcpToolInstructions, responseSchema: responseSchema, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: approvalPresenter, retrievalLimit: retrievalLimit ) @@ -606,7 +607,7 @@ final class ConversationModel { ragInstructions: String, mcpToolInstructions: String, responseSchema: AgentSchema, - interceptor: PolicyInterceptor, + contentEvaluating: StoreBackedAssistantContentEvaluating?, approvalPresenter: (any ApprovalConfirmationPresenting)?, retrievalLimit: Int = 5 ) async -> AsyncThrowingStream { @@ -630,7 +631,7 @@ final class ConversationModel { return await pipeline.streamWithPolicyInterception( prompt: prompt, sessionID: sessionKey.sessionID, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: approvalPresenter, responseSchema: responseSchema ) @@ -653,19 +654,16 @@ final class ConversationModel { return await pipeline.streamWithPolicyInterception( prompt: prompt, sessionID: sessionKey.sessionID, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: approvalPresenter, responseSchema: responseSchema ) } } - private func makeContentPolicyInterceptor() -> PolicyInterceptor { - guard let policyStore else { - return DefaultPolicyInterceptor() - } - let policy = StoreBackedCompletionContentPolicy(store: policyStore, applicationName: "ui") - return DefaultPolicyInterceptor(policy: policy) + private func makeContentEvaluating() -> StoreBackedAssistantContentEvaluating? { + guard let policyStore else { return nil } + return StoreBackedAssistantContentEvaluating(store: policyStore, applicationName: "ui") } /// In-process host for orchestration tools (`agents_*`, `jobs_*`). Not used for MCP effectors. diff --git a/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift b/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift index 14e7d03b..020a172d 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationPipelinePolicy.swift @@ -3,6 +3,7 @@ import LLMAgentClient import MCP import MCPClient import MemorySystem +import PolicyRuntime import PartialJSON import AppEvents import PolicyUserInteraction @@ -15,7 +16,7 @@ extension ConversationPipeline { parentAgentID: String? = nil, toolCalls: [ToolCallRecord] = [], scope: MemoryAccessibility = .private, - interceptor: PolicyInterceptor = DefaultPolicyInterceptor(), + contentEvaluating: StoreBackedAssistantContentEvaluating? = nil, approvalPresenter: (any ApprovalConfirmationPresenting)? = nil, responseSchema: AgentSchema? = nil ) async -> AsyncThrowingStream { @@ -104,7 +105,13 @@ extension ConversationPipeline { ) let chunkPolicyStarted = Date() - let chunkIntercept = try await interceptor.interceptAssistantChunk(event) + let chunkDecision: GuardrailDecision + if let contentEvaluating { + chunkDecision = try await contentEvaluating.evaluate(event) + } else { + chunkDecision = .allow + } + let chunkIntercept = AssistantContentGuardrailApplying().apply(chunkDecision, for: event) chunkPolicyMS += PipelineTiming.elapsedMS(from: chunkPolicyStarted) let interceptedContent: String switch chunkIntercept { @@ -329,7 +336,16 @@ extension ConversationPipeline { chunkCount: chunkIndex ) let completionPolicyStarted = Date() - let completionIntercept = try await interceptor.interceptAssistantCompletion(completionEvent) + let completionDecision: GuardrailDecision + if let contentEvaluating { + completionDecision = try await contentEvaluating.evaluate(completionEvent) + } else { + completionDecision = .allow + } + let completionIntercept = AssistantContentGuardrailApplying().apply( + completionDecision, + for: completionEvent + ) var completionPolicyMS = PipelineTiming.elapsedMS(from: completionPolicyStarted) let contentConfirmStarted = Date() let interceptedCompletion: String? diff --git a/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift b/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift index 6f19bae2..de08d3d7 100644 --- a/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift +++ b/ui/SharedAgentRuntime/Conversation/ConversationPipelineToolInterception.swift @@ -14,7 +14,7 @@ extension ConversationPipeline { arguments: [String: Value], sessionID: String, userPrompt: String? = nil, - interceptor: ToolRequestInterceptor? = nil, + toolEvaluating: (any GuardrailEvaluating)? = nil, approvalPresenter: (any ApprovalConfirmationPresenting)? = nil ) async throws -> MCPToolResult { let toolOverallStarted = Date() @@ -45,7 +45,7 @@ extension ConversationPipeline { } - let effectiveInterceptor = makeToolInterceptor(override: interceptor) + let evaluating = makeToolEvaluating(override: toolEvaluating) await MainActor.run { debugLog("Policy rule processing: evaluating \(name)") } @@ -53,165 +53,162 @@ extension ConversationPipeline { do { let confirmMSBox = TimingAccumulator() let proceedMSBox = TimingAccumulator() - let result = try await effectiveInterceptor.interceptAndRun( - event, - confirm: { [self] confirmEvent, requiredFields in - let confirmStarted = Date() - defer { confirmMSBox.add(PipelineTiming.elapsedMS(from: confirmStarted)) } - await MainActor.run { - debugLog("Policy decision: confirm \(name)") - } - guard let approvalPresenter else { - try await policyStore?.saveApproval( - PolicyApproval( - applicationName: applicationName, - sessionID: sessionID, - ruleID: "runtime-confirmation", - requestType: "tool_invocation", - requestPayloadJSON: event.argumentsJSON, - editedPayloadJSON: nil, - decision: "cancelled", - actor: "system", - createdAt: .now, - acedAt: .now - ) - ) - try await persistPolicyDecision( + let hitl = ClosureGuardrailHITLPresenting { [self] presentation in + let confirmStarted = Date() + defer { confirmMSBox.add(PipelineTiming.elapsedMS(from: confirmStarted)) } + await MainActor.run { + debugLog("Policy decision: confirm \(name)") + } + guard let approvalPresenter else { + try? await policyStore?.saveApproval( + PolicyApproval( + applicationName: applicationName, sessionID: sessionID, + ruleID: "runtime-confirmation", + requestType: "tool_invocation", requestPayloadJSON: event.argumentsJSON, + editedPayloadJSON: nil, decision: "cancelled", - actor: "system" + actor: "system", + createdAt: .now, + acedAt: .now ) - throw MCPClientError.toolExecutionDenied( - toolName: name, - reason: "Tool execution requires user confirmation" - ) - } - - let confirmationRequest = ApprovalConfirmationRequest( - sessionID: sessionID, - toolName: confirmEvent.toolName, - argumentsJSON: confirmEvent.argumentsJSON, - requiredFields: requiredFields ) - let confirmation = await approvalPresenter.confirm(confirmationRequest) - await MainActor.run { - debugLog("Approval response received for \(name)") - } - let approvalRecord = PolicyApproval.fromApprovalDecision( - applicationName: applicationName, + try? await persistPolicyDecision( sessionID: sessionID, - requestPayloadJSON: confirmEvent.argumentsJSON, - decision: confirmation + requestPayloadJSON: event.argumentsJSON, + decision: "cancelled", + actor: "system" ) - try await policyStore?.saveApproval(approvalRecord) + return .cancelled(actor: "system") + } - switch confirmation { - case .approved(let editedArgumentsJSON, let actor): - await MainActor.run { - debugLog("Approval granted for \(name) by \(actor ?? "unknown")") - } - try await persistPolicyDecision( - sessionID: sessionID, - requestPayloadJSON: confirmEvent.argumentsJSON, - decision: "approved", - actor: actor - ) - return .approved( - ToolInvocationEvent( - sessionID: confirmEvent.sessionID, - toolName: confirmEvent.toolName, - argumentsJSON: editedArgumentsJSON, - timestamp: confirmEvent.timestamp - ) - ) - case .cancelled(let actor): - await MainActor.run { - debugLog("Approval cancelled for \(name) by \(actor ?? "unknown")") - } - try await persistPolicyDecision( - sessionID: sessionID, - requestPayloadJSON: confirmEvent.argumentsJSON, - decision: "cancelled", - actor: actor - ) - return .cancelled(actor: actor) - } - }, - proceed: { [self] interceptedEvent in - let proceedStarted = Date() - defer { proceedMSBox.add(PipelineTiming.elapsedMS(from: proceedStarted)) } + let confirmationRequest = ApprovalConfirmationRequest( + sessionID: sessionID, + toolName: presentation.subject, + argumentsJSON: presentation.payloadJSON, + requiredFields: presentation.hitl.requiredFields + ) + let confirmation = await approvalPresenter.confirm(confirmationRequest) + await MainActor.run { + debugLog("Approval response received for \(name)") + } + let approvalRecord = PolicyApproval.fromApprovalDecision( + applicationName: applicationName, + sessionID: sessionID, + requestPayloadJSON: presentation.payloadJSON, + decision: confirmation + ) + try? await policyStore?.saveApproval(approvalRecord) + + switch confirmation { + case .approved(let editedArgumentsJSON, let actor): await MainActor.run { - debugLog("Policy decision: allow \(interceptedEvent.toolName)") - } - let interceptedArguments = try toolArgumentsFromJSON(interceptedEvent.argumentsJSON) - guard let mcpClient else { - return MCPToolResult(content: [MCPToolContent.text("Tool client unavailable.")], isError: true) - } - if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { - let scriptAllowed = await UsageLimitsService.shared.allowScriptRun() - if !scriptAllowed { - throw MCPClientError.toolExecutionDenied( - toolName: interceptedEvent.toolName, - reason: "Usage limit: max script_exec runs for this message." - ) - } - let reviewerAllowed = await UsageLimitsService.shared.allowReviewerCall() - if !reviewerAllowed { - throw MCPClientError.toolExecutionDenied( - toolName: interceptedEvent.toolName, - reason: "Usage limit: max security reviewer calls for this message." - ) - } + debugLog("Approval granted for \(name) by \(actor ?? "unknown")") } + try? await persistPolicyDecision( + sessionID: sessionID, + requestPayloadJSON: presentation.payloadJSON, + decision: "approved", + actor: actor + ) + return .approved(editedPayloadJSON: editedArgumentsJSON, actor: actor) + case .cancelled(let actor): await MainActor.run { - debugLog("Executing tool: \(interceptedEvent.toolName)") + debugLog("Approval cancelled for \(name) by \(actor ?? "unknown")") } - let execStarted = Date() - let result = try await mcpClient.callTool( - named: interceptedEvent.toolName, - arguments: interceptedArguments + try? await persistPolicyDecision( + sessionID: sessionID, + requestPayloadJSON: presentation.payloadJSON, + decision: "cancelled", + actor: actor ) - let mcpCallMS = PipelineTiming.elapsedMS(from: execStarted) - PipelineTiming.log( - "tool=\(interceptedEvent.toolName) mcp_call_ms=\(mcpCallMS) isError=\(result.isError) result_chars=\(result.text.utf8.count)" + return .cancelled(actor: actor) + } + } + + let decision: GuardrailDecision + if let evaluating { + decision = try await evaluating.evaluate(event) + } else { + decision = .allow + } + let applying = ToolInvocationGuardrailApplying(hitl: hitl) + let interceptedEvent = try await applying.apply(decision, for: event) + + let proceedStarted = Date() + await MainActor.run { + debugLog("Policy decision: allow \(interceptedEvent.toolName)") + } + let interceptedArguments = try toolArgumentsFromJSON(interceptedEvent.argumentsJSON) + guard let mcpClient else { + proceedMSBox.add(PipelineTiming.elapsedMS(from: proceedStarted)) + return MCPToolResult(content: [MCPToolContent.text("Tool client unavailable.")], isError: true) + } + if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { + let scriptAllowed = await UsageLimitsService.shared.allowScriptRun() + if !scriptAllowed { + throw MCPClientError.toolExecutionDenied( + toolName: interceptedEvent.toolName, + reason: "Usage limit: max script_exec runs for this message." ) - // Attribute reviewer-ish tokens from phase timing when present. - if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { - await Self.recordReviewerTokensIfPresent(resultText: result.text) - } - await MainActor.run { - debugLog("Tool result: \(interceptedEvent.toolName) (isError=\(result.isError))") - ToolOutcomeLogger.log( - toolName: interceptedEvent.toolName, - rawText: result.text - ) - if interceptedEvent.toolName != AllowedMCPTool.pluginFactoryBuild.rawValue - && interceptedEvent.toolName != AllowedMCPTool.webCrawl.rawValue { - debugLog("Tool result content: \(Self.debugPayload(result.text))") - } - } - await Self.publishJobSchedulingFailureIfNeeded( + } + let reviewerAllowed = await UsageLimitsService.shared.allowReviewerCall() + if !reviewerAllowed { + throw MCPClientError.toolExecutionDenied( toolName: interceptedEvent.toolName, - resultText: result.text, - sessionID: sessionID + reason: "Usage limit: max security reviewer calls for this message." ) - if interceptedEvent.toolName == AllowedMCPTool.pluginFactoryBuild.rawValue { - return try await self.resolvePluginFactoryBuildResult( - initialResult: result, - arguments: interceptedArguments, - userPrompt: userPrompt, - mcpClient: mcpClient - ) - } - return result } + } + await MainActor.run { + debugLog("Executing tool: \(interceptedEvent.toolName)") + } + let execStarted = Date() + let toolResult = try await mcpClient.callTool( + named: interceptedEvent.toolName, + arguments: interceptedArguments ) + let mcpCallMS = PipelineTiming.elapsedMS(from: execStarted) + PipelineTiming.log( + "tool=\(interceptedEvent.toolName) mcp_call_ms=\(mcpCallMS) isError=\(toolResult.isError) result_chars=\(toolResult.text.utf8.count)" + ) + if AllowedMCPTool.isScriptExec(interceptedEvent.toolName) { + await Self.recordReviewerTokensIfPresent(resultText: toolResult.text) + } + await MainActor.run { + debugLog("Tool result: \(interceptedEvent.toolName) (isError=\(toolResult.isError))") + ToolOutcomeLogger.log( + toolName: interceptedEvent.toolName, + rawText: toolResult.text + ) + if interceptedEvent.toolName != AllowedMCPTool.pluginFactoryBuild.rawValue + && interceptedEvent.toolName != AllowedMCPTool.webCrawl.rawValue { + debugLog("Tool result content: \(Self.debugPayload(toolResult.text))") + } + } + await Self.publishJobSchedulingFailureIfNeeded( + toolName: interceptedEvent.toolName, + resultText: toolResult.text, + sessionID: sessionID + ) + let result: MCPToolResult + if interceptedEvent.toolName == AllowedMCPTool.pluginFactoryBuild.rawValue { + result = try await self.resolvePluginFactoryBuildResult( + initialResult: toolResult, + arguments: interceptedArguments, + userPrompt: userPrompt, + mcpClient: mcpClient + ) + } else { + result = toolResult + } + proceedMSBox.add(PipelineTiming.elapsedMS(from: proceedStarted)) PipelineTiming.log( "tool=\(name) encode_ms=\(encodeMS) confirm_ms=\(confirmMSBox.total) proceed_ms=\(proceedMSBox.total) total_ms=\(PipelineTiming.elapsedMS(from: toolOverallStarted))" ) return result - } catch ToolInvocationInterceptionError.denied(let reason) { + } catch ToolInvocationGuardrailError.denied(let reason) { PipelineTiming.log( "tool=\(name) denied encode_ms=\(encodeMS) total_ms=\(PipelineTiming.elapsedMS(from: toolOverallStarted)) reason=\(reason)" ) @@ -267,7 +264,7 @@ extension ConversationPipeline { ) ) throw MCPClientError.toolExecutionDenied(toolName: name, reason: reason) - } catch ToolInvocationInterceptionError.cancelled(let reason) { + } catch ToolInvocationGuardrailError.cancelled(let reason) { PipelineTiming.log( "tool=\(name) cancelled encode_ms=\(encodeMS) total_ms=\(PipelineTiming.elapsedMS(from: toolOverallStarted)) reason=\(reason)" ) @@ -301,7 +298,7 @@ extension ConversationPipeline { _ request: MCPToolBatchRequest, sessionID: String, userPrompt: String? = nil, - interceptor: ToolRequestInterceptor? = nil, + toolEvaluating: (any GuardrailEvaluating)? = nil, approvalPresenter: (any ApprovalConfirmationPresenting)? = nil ) async throws -> MCPToolBatchResult { // MA-3: run independent batch invocations concurrently (order of results preserved). @@ -319,7 +316,7 @@ extension ConversationPipeline { arguments: invocation.arguments, sessionID: sessionID, userPrompt: userPrompt, - interceptor: interceptor, + toolEvaluating: toolEvaluating, approvalPresenter: approvalPresenter ) return (index, result) @@ -354,15 +351,12 @@ extension ConversationPipeline { ) } - private func makeToolInterceptor(override: ToolRequestInterceptor?) -> ToolRequestInterceptor { - if let override { - return override - } - if let policyStore { - let policy = StoreBackedToolGovernancePolicy(store: policyStore, applicationName: applicationName) - return DefaultToolRequestInterceptor(policy: policy) - } - return DefaultToolRequestInterceptor() + private func makeToolEvaluating( + override: (any GuardrailEvaluating)? + ) -> (any GuardrailEvaluating)? { + if let override { return override } + guard let policyStore else { return nil } + return StoreBackedToolInvocationEvaluating(store: policyStore, applicationName: applicationName) } private func persistPolicyDecision( diff --git a/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift b/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift index 04f47b3e..9e624a7d 100644 --- a/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift +++ b/ui/SharedAgentRuntime/Conversation/ProfileDelegateRunner.swift @@ -44,7 +44,7 @@ enum ProfileDelegateRunner { ragInstructions: String, mcpToolInstructions: String, responseSchema: AgentSchema, - interceptor: PolicyInterceptor + contentEvaluating: StoreBackedAssistantContentEvaluating? ) async throws -> String { let normalized = AgentProfileHandle.normalize( profileHandle.trimmingCharacters(in: .whitespacesAndNewlines).replacingOccurrences(of: "$", with: "") @@ -103,7 +103,7 @@ enum ProfileDelegateRunner { ragInstructions: userRagBase, mcpToolInstructions: mcpToolInstructions, responseSchema: responseSchema, - interceptor: interceptor, + contentEvaluating: contentEvaluating, approvalPresenter: nil, retrievalLimit: retrievalLimit )