From 6d9da74d8beeb1ba8e575b636e1c06ca3bbdcbd1 Mon Sep 17 00:00:00 2001 From: chhhee10 Date: Tue, 29 Sep 2026 15:26:50 +0530 Subject: [PATCH 01/22] feat(hermes): link the native plugin into each profile instead of copying it Hermes has no configurable plugin search path, so the plugin was copied into /plugins/failproofai and went stale on every npm upgrade until someone reran setup. Install now symlinks that path to the package's own hermes-plugin/ (the OpenClaw plugins.load.paths model), so an upgrade applies with no reinstall. Hermes' scan_directory uses Path.is_dir(), which follows the link, and skips a dangling one silently. Ownership stays strict: only a marked copy (1.0.6-1.0.8 installs) or a link into a FailproofAI hermes-plugin/ (checked by basename and manifest name) is ever replaced or removed; an unmarked dir, a file, or a link anywhere else is refused. A copy is migrated to a link; where a link cannot be created the install falls back to the marked copy. Uninstall unlinks, never touching the package directory. Health gains pluginMode and cronUnchecked: a profile still on legacy config.yaml shell hooks is reported UNHEALTHY with "Hermes cron jobs are not checked", since cron fires build their own hook scope that config shell hooks never join. Adds migrateHermesProfiles(), the per-profile state machine `update` uses. The Hermes tests now copy hermes-plugin/ into a temp package root (a test editing files through the installed plugin would otherwise edit the repo's plugin via the link) and refuse to run when HOME overrides are not honoured by os.homedir() (Bun caches it), which would point them at a real ~/.hermes. Co-Authored-By: Claude Opus 5.5 --- __tests__/hooks/integrations.test.ts | 41 ++- src/hooks/integrations.ts | 404 +++++++++++++++++++++++---- 2 files changed, 385 insertions(+), 60 deletions(-) diff --git a/__tests__/hooks/integrations.test.ts b/__tests__/hooks/integrations.test.ts index 32f013d47..3155f96c3 100644 --- a/__tests__/hooks/integrations.test.ts +++ b/__tests__/hooks/integrations.test.ts @@ -18,6 +18,9 @@ import { writeFileSync, mkdirSync, readdirSync, + cpSync, + lstatSync, + readlinkSync, } from "node:fs"; import { tmpdir } from "node:os"; import { resolve, join } from "node:path"; @@ -571,6 +574,7 @@ describe("Hermes integration", () => { let origHome: string | undefined; let origHermesHome: string | undefined; let origPackageRoot: string | undefined; + let packageRoot: string; beforeEach(() => { origHome = process.env.HOME; process.env.HOME = tempDir; @@ -578,8 +582,21 @@ describe("Hermes integration", () => { // with a profile-scoped shell doesn't get their real config.yaml touched. origHermesHome = process.env.HERMES_HOME; delete process.env.HERMES_HOME; + // Every test below writes under `homedir()`. A runtime that caches the home + // directory at startup (Bun does) would ignore the HOME override and point + // them at the developer's REAL ~/.hermes — refuse rather than touch it. + if (homedir() !== tempDir) { + throw new Error( + `HOME override not honoured (homedir() is ${homedir()}); refusing to run Hermes tests against a real home`, + ); + } + // Install now LINKS profiles to the package's hermes-plugin/, so a test + // that edits files through the installed plugin would edit the package. + // Give every test its own throwaway package copy. origPackageRoot = process.env.FAILPROOFAI_PACKAGE_ROOT; - process.env.FAILPROOFAI_PACKAGE_ROOT = ORIG_CWD; + packageRoot = resolve(tempDir, "failproofai-package"); + cpSync(resolve(ORIG_CWD, "hermes-plugin"), resolve(packageRoot, "hermes-plugin"), { recursive: true }); + process.env.FAILPROOFAI_PACKAGE_ROOT = packageRoot; }); afterEach(() => { if (origHome === undefined) delete process.env.HOME; @@ -610,6 +627,15 @@ describe("Hermes integration", () => { return resolve(dirname(settingsPath), "plugins", "failproofai"); } + /** What a 1.0.6–1.0.8 install left behind: a marked COPY of the plugin. */ + function makeManagedCopy(destination: string): void { + mkdirSync(destination, { recursive: true }); + for (const file of ["plugin.yaml", "__init__.py", "client.py", "ledger.py"]) { + cpSync(resolve(ORIG_CWD, "hermes-plugin", file), resolve(destination, file)); + } + writeFileSync(resolve(destination, ".failproofai-managed"), "Managed by failproofai.\n"); + } + function installAt(settingsPath: string): void { hermes.prepareInstall!(settingsPath); const settings = hermes.readSettings(settingsPath); @@ -641,7 +667,7 @@ describe("Hermes integration", () => { it("buildHookEntry identifies the shipped native plugin", () => { const entry = hermes.buildHookEntry("/usr/bin/failproofai", "pre_tool_call", "user") as Record; expect(entry[FAILPROOFAI_HOOK_MARKER]).toBe(true); - expect(entry._hermesPluginPath).toBe(resolve(ORIG_CWD, "hermes-plugin")); + expect(entry._hermesPluginPath).toBe(resolve(packageRoot, "hermes-plugin")); }); it("installs the native plugin and enables it in config", () => { @@ -654,10 +680,14 @@ describe("Hermes integration", () => { expect(parsed.plugins?.enabled).toEqual(["failproofai"]); expect(parsed.hooks).toBeUndefined(); + // A symlink into the package, so npm upgrades apply with no reinstall. const installedPlugin = pluginPath(path); - for (const file of ["plugin.yaml", "__init__.py", "client.py", "ledger.py", ".failproofai-managed"]) { + expect(lstatSync(installedPlugin).isSymbolicLink()).toBe(true); + expect(readlinkSync(installedPlugin)).toBe(resolve(packageRoot, "hermes-plugin")); + for (const file of ["plugin.yaml", "__init__.py", "client.py", "ledger.py"]) { expect(existsSync(resolve(installedPlugin, file))).toBe(true); } + expect(hermesProfileHealth()[0]).toMatchObject({ pluginMode: "link", healthy: true, cronUnchecked: false }); expect(hermesProfileStatusRows()).toEqual([ ["hermes/default", "native plugin enabled"], ]); @@ -816,15 +846,16 @@ describe("Hermes integration", () => { ]); }); - it("atomically refreshes a managed plugin directory", () => { + it("replaces a managed plugin copy (1.0.6–1.0.8 install) with a link, leaving no temp entries", () => { const path = hermes.getSettingsPath("user"); - hermes.prepareInstall!(path); const destination = pluginPath(path); + makeManagedCopy(destination); writeFileSync(resolve(destination, "client.py"), "stale\n"); writeFileSync(resolve(destination, "obsolete.py"), "remove me\n"); hermes.prepareInstall!(path); + expect(lstatSync(destination).isSymbolicLink()).toBe(true); expect(readFileSync(resolve(destination, "client.py"), "utf8")).not.toBe("stale\n"); expect(existsSync(resolve(destination, "obsolete.py"))).toBe(false); expect( diff --git a/src/hooks/integrations.ts b/src/hooks/integrations.ts index 62e134871..444e1b01d 100644 --- a/src/hooks/integrations.ts +++ b/src/hooks/integrations.ts @@ -14,11 +14,14 @@ import { existsSync, lstatSync, mkdirSync, + readlinkSync, + realpathSync, renameSync, rmSync, + symlinkSync, unlinkSync, } from "node:fs"; -import { resolve, dirname } from "node:path"; +import { resolve, dirname, basename } from "node:path"; import { fileURLToPath } from "node:url"; import { homedir } from "node:os"; import { parseDocument, type Document } from "yaml"; @@ -1388,9 +1391,13 @@ function makePiProjectRelativeEntry(extPath: string): string { // Hermes loads trusted Python plugins from `/plugins//`. // The shipped plugin registers Hermes-native hooks and talks directly to the // local failproofaid socket, avoiding a fresh CLI process per event. Config is -// YAML and profile-scoped; install copies the plugin into every profile and +// YAML and profile-scoped; install LINKS every profile's `plugins/failproofai` +// to the package's `hermes-plugin/` (so npm upgrades apply with no reinstall, +// like OpenClaw's `plugins.load.paths`), falling back to a marked copy, and // adds `failproofai` to `plugins.enabled`. Legacy shell-hook entries are removed -// during migration so one tool call is never evaluated twice. +// during migration so one tool call is never evaluated twice — and because +// config.yaml shell hooks never reach Hermes cron jobs: each cron fire builds +// its own hook scope, which discovered plugins join and config hooks do not. /** One hook entry as stored under a `hooks:` event key in config.yaml. */ interface HermesHookEntry { @@ -1409,38 +1416,137 @@ interface HermesPluginsConfig { const HERMES_PLUGIN_ID = "failproofai"; const HERMES_PLUGIN_MARKER = ".failproofai-managed"; const HERMES_PLUGIN_FILES = ["plugin.yaml", "__init__.py", "client.py", "ledger.py"] as const; +/** Basename of the plugin directory inside the npm package. */ +const HERMES_PLUGIN_SOURCE_DIR = "hermes-plugin"; function getHermesPluginSourcePath(): string { const fromEnv = process.env.FAILPROOFAI_PACKAGE_ROOT; - if (fromEnv) return resolve(fromEnv, "hermes-plugin"); - return resolve(fileURLToPath(import.meta.url), "..", "..", "..", "hermes-plugin"); + if (fromEnv) return resolve(fromEnv, HERMES_PLUGIN_SOURCE_DIR); + return resolve(fileURLToPath(import.meta.url), "..", "..", "..", HERMES_PLUGIN_SOURCE_DIR); } export function hermesPluginPathForSettings(settingsPath: string): string { return resolve(dirname(settingsPath), "plugins", HERMES_PLUGIN_ID); } -function hasManagedHermesPlugin(settingsPath: string): boolean { - const pluginPath = hermesPluginPathForSettings(settingsPath); +/** `existsSync` follows symlinks, so a dangling link reads as absent. This does not. */ +function pathPresent(path: string): boolean { try { - if (lstatSync(pluginPath).isSymbolicLink()) return false; - if (!existsSync(resolve(pluginPath, HERMES_PLUGIN_MARKER))) return false; - return HERMES_PLUGIN_FILES.every((file) => existsSync(resolve(pluginPath, file))); + lstatSync(path); + return true; } catch { return false; } } -function hasHermesPluginMarker(settingsPath: string): boolean { - const pluginPath = hermesPluginPathForSettings(settingsPath); +function hasAllHermesPluginFiles(dir: string): boolean { + return HERMES_PLUGIN_FILES.every((file) => existsSync(resolve(dir, file))); +} + +/** + * Whether `dir` is a FailproofAI `hermes-plugin` directory (or was one: a link + * whose target vanished during an npm upgrade or uninstall is still ours). + * The basename is the package layout; the manifest name confirms it when the + * target is readable, so a link to someone else's plugin is never claimed. + */ +function isFailproofaiHermesPluginSource(dir: string): boolean { + if (basename(dir) !== HERMES_PLUGIN_SOURCE_DIR) return false; + if (!existsSync(dir)) return true; try { - return !lstatSync(pluginPath).isSymbolicLink() && existsSync(resolve(pluginPath, HERMES_PLUGIN_MARKER)); + const manifest = readFileSync(resolve(dir, "plugin.yaml"), "utf8"); + return /^name:\s*["']?failproofai["']?\s*$/m.test(manifest); } catch { return false; } } -function installHermesPlugin(settingsPath: string): void { +function samePath(a: string, b: string): boolean { + if (resolve(a) === resolve(b)) return true; + try { + return realpathSync(a) === realpathSync(b); + } catch { + return false; + } +} + +/** + * What occupies `/plugins/failproofai`, and whether FailproofAI owns it. + * + * • `link` — a symlink into a FailproofAI package's `hermes-plugin/` (the + * current install shape: npm upgrades apply with no reinstall). + * • `copy` — a copied directory carrying the `.failproofai-managed` + * marker (1.0.6–1.0.8 installs, and the fallback where a + * symlink cannot be created). + * • `foreign` — anything else: an unmarked directory, a file, or a symlink + * somewhere that is not a FailproofAI plugin. Never touched. + */ +type HermesPluginState = + | { kind: "absent" } + | { kind: "link"; target: string; complete: boolean; current: boolean } + | { kind: "copy"; complete: boolean } + | { kind: "foreign" }; + +function hermesPluginState(settingsPath: string): HermesPluginState { + const pluginPath = hermesPluginPathForSettings(settingsPath); + let isLink: boolean; + let isDirectory: boolean; + try { + const stat = lstatSync(pluginPath); + isLink = stat.isSymbolicLink(); + isDirectory = stat.isDirectory(); + } catch { + return { kind: "absent" }; + } + if (isLink) { + let target: string; + try { + target = resolve(dirname(pluginPath), readlinkSync(pluginPath)); + } catch { + return { kind: "foreign" }; + } + if (!isFailproofaiHermesPluginSource(target)) return { kind: "foreign" }; + return { + kind: "link", + target, + complete: hasAllHermesPluginFiles(target), + current: samePath(target, getHermesPluginSourcePath()), + }; + } + if (isDirectory && existsSync(resolve(pluginPath, HERMES_PLUGIN_MARKER))) { + return { kind: "copy", complete: hasAllHermesPluginFiles(pluginPath) }; + } + return { kind: "foreign" }; +} + +function hasManagedHermesPlugin(settingsPath: string): boolean { + const state = hermesPluginState(settingsPath); + return (state.kind === "link" || state.kind === "copy") && state.complete; +} + +/** Test seam: lets a test make symlink creation fail to exercise the copy fallback. */ +export interface HermesPluginInstallDeps { + symlink?: (target: string, path: string) => void; +} + +function defaultHermesSymlink(target: string, path: string): void { + // A junction needs no privilege on Windows; the type is ignored elsewhere. + symlinkSync(target, path, process.platform === "win32" ? "junction" : "dir"); +} + +/** + * Put the plugin at `/plugins/failproofai`, preferring a symlink to + * the package's own `hermes-plugin/` so every npm upgrade updates it in place — + * the same reason OpenClaw's config points into the package. Hermes' plugin + * scan uses `Path.is_dir()`, which follows symlinks, so the link is discovered + * like a directory. Falls back to a marked copy when a link cannot be created. + * + * Only a path FailproofAI owns (a marked copy, or a link into a FailproofAI + * `hermes-plugin/`) is ever replaced; anything else throws. + */ +export function installHermesPlugin( + settingsPath: string, + deps: HermesPluginInstallDeps = {}, +): "linked" | "copied" | "unchanged" { const source = getHermesPluginSourcePath(); for (const file of HERMES_PLUGIN_FILES) { if (!existsSync(resolve(source, file))) { @@ -1449,50 +1555,88 @@ function installHermesPlugin(settingsPath: string): void { } const destination = hermesPluginPathForSettings(settingsPath); - if (existsSync(destination)) { - if (lstatSync(destination).isSymbolicLink() || !existsSync(resolve(destination, HERMES_PLUGIN_MARKER))) { - throw new Error( - `Refusing to overwrite an unmanaged Hermes plugin at ${destination}. ` + - `Move it aside or install FailproofAI under a different Hermes profile.`, - ); - } + const existing = hermesPluginState(settingsPath); + if (existing.kind === "foreign") { + throw new Error( + `Refusing to overwrite an unmanaged Hermes plugin at ${destination}. ` + + `Move it aside or install FailproofAI under a different Hermes profile.`, + ); } + if (existing.kind === "link" && existing.current) return "unchanged"; mkdirSync(dirname(destination), { recursive: true }); const suffix = `${process.pid}-${Date.now()}`; const temporary = `${destination}.install-${suffix}`; - const backup = `${destination}.backup-${suffix}`; - let movedExisting = false; + + let linked = false; try { + (deps.symlink ?? defaultHermesSymlink)(source, temporary); + linked = true; + } catch { + rmSync(temporary, { recursive: true, force: true }); + } + + if (!linked) { mkdirSync(temporary, { recursive: true }); - for (const file of HERMES_PLUGIN_FILES) { - cpSync(resolve(source, file), resolve(temporary, file)); + try { + for (const file of HERMES_PLUGIN_FILES) { + cpSync(resolve(source, file), resolve(temporary, file)); + } + writeFileSync( + resolve(temporary, HERMES_PLUGIN_MARKER), + "Managed by failproofai. Remove with `failproofai policies --uninstall --cli hermes`.\n", + { encoding: "utf8", mode: 0o600 }, + ); + } catch (err) { + rmSync(temporary, { recursive: true, force: true }); + throw err; } - writeFileSync( - resolve(temporary, HERMES_PLUGIN_MARKER), - "Managed by failproofai. Remove with `failproofai policies --uninstall --cli hermes`.\n", - { encoding: "utf8", mode: 0o600 }, - ); - if (existsSync(destination)) { + } + + // A link replacing a link is one atomic rename. Anything replacing a + // directory (or a link replaced by a directory) cannot rename over it, so + // the old entry is moved aside first and restored if the swap fails. + if (linked && (existing.kind === "absent" || existing.kind === "link")) { + try { + renameSync(temporary, destination); + } catch (err) { + rmSync(temporary, { force: true }); + throw err; + } + return "linked"; + } + + const backup = `${destination}.backup-${suffix}`; + let movedExisting = false; + try { + if (pathPresent(destination)) { renameSync(destination, backup); movedExisting = true; } renameSync(temporary, destination); + // fs.rm never follows a symlink, so a backed-up link removes only itself. if (movedExisting) rmSync(backup, { recursive: true, force: true }); } catch (err) { rmSync(temporary, { recursive: true, force: true }); - if (movedExisting && !existsSync(destination) && existsSync(backup)) { + if (movedExisting && !pathPresent(destination) && pathPresent(backup)) { renameSync(backup, destination); } throw err; } + return linked ? "linked" : "copied"; } function removeManagedHermesPlugin(settingsPath: string): boolean { const pluginPath = hermesPluginPathForSettings(settingsPath); + const state = hermesPluginState(settingsPath); + if (state.kind === "link") { + // Only the link goes; the package directory it points at is npm's. + unlinkSync(pluginPath); + return true; + } // The marker is the ownership boundary. Remove even an incomplete managed - // install so an interrupted copy can always be repaired or uninstalled. - if (!hasHermesPluginMarker(settingsPath)) return false; + // copy so an interrupted install can always be repaired or uninstalled. + if (state.kind !== "copy") return false; rmSync(pluginPath, { recursive: true, force: true }); return true; } @@ -1549,13 +1693,163 @@ function hermesConfigState(settingsPath: string): { }; } +/** + * Enable the native plugin in one profile's config and drop FailproofAI's + * legacy shell hooks. A native plugin and the legacy shell bridge running + * together would evaluate every event twice. Only failproofai-owned shell + * entries are removed; unrelated operator hooks remain intact. + */ +function enableHermesPluginInDoc(doc: Document): void { + removeLegacyHermesHooks(doc); + + const js = (doc.toJS() ?? {}) as { plugins?: HermesPluginsConfig }; + const plugins: HermesPluginsConfig = + js.plugins && typeof js.plugins === "object" ? js.plugins : {}; + const enabled = Array.isArray(plugins.enabled) ? [...plugins.enabled] : []; + if (!enabled.includes(HERMES_PLUGIN_ID)) enabled.push(HERMES_PLUGIN_ID); + plugins.enabled = enabled; + if (Array.isArray(plugins.disabled)) { + plugins.disabled = plugins.disabled.filter((name) => name !== HERMES_PLUGIN_ID); + if (plugins.disabled.length === 0) delete plugins.disabled; + } + doc.set("plugins", plugins); +} + +export type HermesMigrationStatus = + /** Brought to the linked-plugin state by this run. */ + | "migrated" + /** Already linked to this package, enabled, and free of shell hooks. */ + | "current" + /** No FailproofAI integration in this profile — left alone. */ + | "untouched" + /** Needs migrating, but the daemon cannot serve the native plugin. Nothing changed. */ + | "blocked" + /** Could not be migrated (unmanaged plugin in the way, unreadable config, I/O error). */ + | "failed"; + +export interface HermesProfileMigration { + name: string; + home: string; + status: HermesMigrationStatus; + detail: string; + /** Legacy config.yaml shell hooks are still in place after this run. */ + legacyShellHooksRemain: boolean; +} + +/** + * The upgrade half `failproofai update` owes Hermes: bring every profile that + * already uses FailproofAI (legacy shell hooks, a copied plugin, or a link) to + * the linked-plugin state. Profiles with no FailproofAI integration are never + * touched — an update must not opt a profile in. + * + * `daemonSupportsPolicyEvaluation` is asked at most once, and only when some + * profile needs changing. When it says no, NOTHING is changed: removing the + * shell hooks in favour of a plugin whose daemon cannot answer would fail every + * tool call closed (the same gate `policies --install` applies). + */ +export async function migrateHermesProfiles(opts: { + daemonSupportsPolicyEvaluation: () => Promise; + installDeps?: HermesPluginInstallDeps; +}): Promise { + const results: HermesProfileMigration[] = []; + let daemonOk: boolean | undefined; + for (const profile of listHermesProfiles()) { + if (!existsSync(profile.home)) continue; + const settingsPath = resolve(profile.home, "config.yaml"); + const pluginPath = hermesPluginPathForSettings(settingsPath); + const config = hermesConfigState(settingsPath); + const plugin = hermesPluginState(settingsPath); + const ours = plugin.kind === "link" || plugin.kind === "copy"; + const base = { name: profile.name, home: profile.home }; + const legacy = config.legacyShellHookPresent; + + if (!legacy && !ours && !config.pluginEnabled) { + results.push({ ...base, status: "untouched", detail: "no failproofai integration", legacyShellHooksRemain: false }); + continue; + } + if (plugin.kind === "link" && plugin.current && plugin.complete && config.pluginEnabled && !legacy) { + results.push({ ...base, status: "current", detail: "already linked", legacyShellHooksRemain: false }); + continue; + } + if (plugin.kind === "foreign") { + results.push({ + ...base, + status: "failed", + detail: + `an unmanaged plugin occupies ${pluginPath}` + + (legacy ? "; shell hooks left in place (cron jobs are not checked)" : ""), + legacyShellHooksRemain: legacy, + }); + continue; + } + if (existsSync(settingsPath)) { + const errors = parseDocument(readFileSync(settingsPath, "utf8")).errors; + if (errors.length > 0) { + results.push({ + ...base, + status: "failed", + detail: `${settingsPath} does not parse as YAML; left untouched`, + legacyShellHooksRemain: legacy, + }); + continue; + } + } + + if (daemonOk === undefined) daemonOk = await opts.daemonSupportsPolicyEvaluation(); + if (!daemonOk) { + results.push({ + ...base, + status: "blocked", + detail: legacy + ? "daemon lacks native policy evaluation; shell hooks left in place (cron jobs are not checked)" + : "daemon lacks native policy evaluation; the plugin here cannot get verdicts", + legacyShellHooksRemain: legacy, + }); + continue; + } + + try { + // Plugin first, config second: the shell hooks are only removed once the + // plugin that replaces them is on disk. + const mode = installHermesPlugin(settingsPath, opts.installDeps); + const doc = readYamlDoc(settingsPath); + enableHermesPluginInDoc(doc); + writeYamlDoc(settingsPath, doc); + const from = legacy + ? "shell hooks" + : plugin.kind === "copy" + ? "copied plugin" + : plugin.kind === "link" + ? "old plugin link" + : "enabled but missing plugin"; + const to = mode === "copied" ? "plugin copy (symlink not possible here)" : "linked plugin"; + results.push({ ...base, status: "migrated", detail: `${from} → ${to}`, legacyShellHooksRemain: false }); + } catch (err) { + results.push({ + ...base, + status: "failed", + detail: err instanceof Error ? err.message : String(err), + legacyShellHooksRemain: hermesConfigState(settingsPath).legacyShellHookPresent, + }); + } + } + return results; +} + export interface HermesProfileHealth { name: string; home: string; settingsPath: string; pluginInstalled: boolean; + /** How the plugin is installed: a link into the package, a copy, or neither. */ + pluginMode: "link" | "copy" | null; pluginEnabled: boolean; legacyShellHookPresent: boolean; + /** + * Still enforced only by legacy config.yaml shell hooks. Hermes cron jobs run + * with none of those, so their tool calls go unchecked. + */ + cronUnchecked: boolean; healthy: boolean; } @@ -1564,15 +1858,19 @@ export function hermesProfileHealth(): HermesProfileHealth[] { return listHermesProfiles().map((profile) => { const settingsPath = resolve(profile.home, "config.yaml"); const config = hermesConfigState(settingsPath); - const pluginInstalled = hasManagedHermesPlugin(settingsPath); + const plugin = hermesPluginState(settingsPath); + const pluginInstalled = (plugin.kind === "link" || plugin.kind === "copy") && plugin.complete; + const pluginWorks = pluginInstalled && config.pluginEnabled; return { name: profile.name, home: profile.home, settingsPath, pluginInstalled, + pluginMode: plugin.kind === "link" || plugin.kind === "copy" ? plugin.kind : null, pluginEnabled: config.pluginEnabled, legacyShellHookPresent: config.legacyShellHookPresent, - healthy: pluginInstalled && config.pluginEnabled && !config.legacyShellHookPresent, + cronUnchecked: config.legacyShellHookPresent && !pluginWorks, + healthy: pluginWorks && !config.legacyShellHookPresent, }; }); } @@ -1582,13 +1880,25 @@ export function hermesProfileStatusRows(): Array<[string, string]> { return hermesProfileHealth() .filter((profile) => existsSync(profile.home)) .map((profile) => { + const label = "hermes/" + profile.name; + if (profile.cronUnchecked) { + // Not a cosmetic leftover: the shell-hook integration does not reach + // cron jobs at all, so this profile has an unenforced path. + return [ + label, + "UNHEALTHY — legacy shell hooks: Hermes cron jobs are not checked. Run `failproofai update`", + ]; + } const problems: string[] = []; if (!profile.pluginInstalled) problems.push("plugin files missing or incomplete"); if (!profile.pluginEnabled) problems.push("plugin not enabled"); if (profile.legacyShellHookPresent) problems.push("legacy shell hook also present"); + if (problems.length > 0) return [label, "UNHEALTHY — " + problems.join("; ")]; return [ - "hermes/" + profile.name, - problems.length === 0 ? "native plugin enabled" : "UNHEALTHY — " + problems.join("; "), + label, + profile.pluginMode === "copy" + ? "native plugin enabled (copied — `failproofai update` links it so upgrades apply)" + : "native plugin enabled", ]; }); } @@ -1641,23 +1951,7 @@ export const hermes: Integration = { isFailproofaiHook: isMarkedHook, writeHookEntries(settings) { - const doc = settings as unknown as Document; - // A native plugin and the legacy shell bridge running together would - // evaluate every event twice. Remove only failproofai-owned shell entries; - // unrelated operator hooks remain intact. - removeLegacyHermesHooks(doc); - - const js = (doc.toJS() ?? {}) as { plugins?: HermesPluginsConfig }; - const plugins: HermesPluginsConfig = - js.plugins && typeof js.plugins === "object" ? js.plugins : {}; - const enabled = Array.isArray(plugins.enabled) ? [...plugins.enabled] : []; - if (!enabled.includes(HERMES_PLUGIN_ID)) enabled.push(HERMES_PLUGIN_ID); - plugins.enabled = enabled; - if (Array.isArray(plugins.disabled)) { - plugins.disabled = plugins.disabled.filter((name) => name !== HERMES_PLUGIN_ID); - if (plugins.disabled.length === 0) delete plugins.disabled; - } - doc.set("plugins", plugins); + enableHermesPluginInDoc(settings as unknown as Document); }, removeHooksFromFile(settingsPath) { From 5a34c87a85b5c182ef6ba4f08b4cf563efa50b06 Mon Sep 17 00:00:00 2001 From: chhhee10 Date: Tue, 29 Sep 2026 15:26:50 +0530 Subject: [PATCH 02/22] fix(update): migrate Hermes profiles to the linked plugin and fail when it cannot `update` never looked at Hermes, so a machine that installed Hermes enforcement with <=1.0.5 kept its config.yaml shell hooks through every npm upgrade - and those hooks are never run for Hermes cron jobs, so scheduled jobs went unchecked while `update` reported success. After the daemon step, every Hermes profile that already uses FailproofAI (legacy shell hooks, a copied plugin, or a link) is brought to the linked plugin: plugin first, then plugins.enabled, then the legacy hooks are removed. Profiles with no FailproofAI integration are left alone, and a config.yaml that does not parse is never rewritten. The running daemon is probed for policyEvaluation (once, only when a profile needs changing); when it cannot serve the plugin - e.g. a sudo system daemon update could not replace - nothing is changed, the shell hooks stay, and update exits 1 telling the user to run `failproofai config`. A per-profile report is printed. Exit status is now non-zero when the daemon swap failed, a layout migration failed, or any Hermes profile could not be brought current. Co-Authored-By: Claude Opus 5.5 --- __tests__/hooks/hermes-update.test.ts | 355 ++++++++++++++++++++++++++ bin/failproofai.mjs | 32 ++- src/hooks/hermes-update.ts | 74 ++++++ 3 files changed, 458 insertions(+), 3 deletions(-) create mode 100644 __tests__/hooks/hermes-update.test.ts create mode 100644 src/hooks/hermes-update.ts diff --git a/__tests__/hooks/hermes-update.test.ts b/__tests__/hooks/hermes-update.test.ts new file mode 100644 index 000000000..1f7045aca --- /dev/null +++ b/__tests__/hooks/hermes-update.test.ts @@ -0,0 +1,355 @@ +/** + * `failproofai update`'s Hermes half, and the linked-plugin install it moves + * profiles to. + * + * Legacy (≤1.0.5) Hermes enforcement is config.yaml shell hooks, which Hermes + * cron jobs never run — each cron fire builds its own hook scope that only + * discovered plugins join. `update` never looked at Hermes, so an upgraded + * machine kept that gap silently. These tests pin the migration, the daemon + * gate that keeps shell hooks when the plugin could not get verdicts, and the + * ownership rules for `/plugins/failproofai`. + */ +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import { + cpSync, + existsSync, + lstatSync, + mkdirSync, + mkdtempSync, + readFileSync, + readlinkSync, + rmSync, + symlinkSync, + writeFileSync, +} from "node:fs"; +import { homedir, tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { parse } from "yaml"; +import { + hermes, + hermesProfileHealth, + hermesProfileStatusRows, + installHermesPlugin, +} from "../../src/hooks/integrations"; +import { runHermesUpdateMigration } from "../../src/hooks/hermes-update"; +import { FAILPROOFAI_HOOK_MARKER } from "../../src/hooks/types"; + +const ORIG_CWD = process.cwd(); +const PLUGIN_FILES = ["plugin.yaml", "__init__.py", "client.py", "ledger.py"]; + +let tempDir: string; +let packageRoot: string; +let source: string; +const saved: Record = {}; + +beforeEach(() => { + tempDir = mkdtempSync(join(tmpdir(), "fp-hermes-update-")); + for (const key of ["HOME", "HERMES_HOME", "FAILPROOFAI_PACKAGE_ROOT"]) saved[key] = process.env[key]; + process.env.HOME = tempDir; + delete process.env.HERMES_HOME; + // Bun caches homedir() at startup and would ignore the override, sending + // every write below into the developer's real ~/.hermes. Refuse instead. + if (homedir() !== tempDir) { + throw new Error(`HOME override not honoured (homedir() is ${homedir()}); refusing to run`); + } + packageRoot = resolve(tempDir, "npm-global", "failproofai"); + source = resolve(packageRoot, "hermes-plugin"); + cpSync(resolve(ORIG_CWD, "hermes-plugin"), source, { recursive: true }); + process.env.FAILPROOFAI_PACKAGE_ROOT = packageRoot; +}); + +afterEach(() => { + for (const [key, value] of Object.entries(saved)) { + if (value === undefined) delete process.env[key]; + else process.env[key] = value; + } + rmSync(tempDir, { recursive: true, force: true }); +}); + +function home(profile = "default"): string { + return profile === "default" + ? resolve(tempDir, ".hermes") + : resolve(tempDir, ".hermes", "profiles", profile); +} +function configPath(profile = "default"): string { + return resolve(home(profile), "config.yaml"); +} +function pluginPath(profile = "default"): string { + return resolve(home(profile), "plugins", "failproofai"); +} +function writeConfig(profile: string, body: string): void { + mkdirSync(home(profile), { recursive: true }); + writeFileSync(configPath(profile), body); +} +interface HermesConfig { + model?: string; + plugins?: { enabled?: string[] }; + hooks?: Record | undefined>; +} +function readConfig(profile = "default"): HermesConfig { + return parse(readFileSync(configPath(profile), "utf8")) as HermesConfig; +} + +/** A ≤1.0.5 install: failproofai shell hooks beside an operator's own hook. */ +const LEGACY_CONFIG = [ + "model: gpt-5", + "hooks_auto_accept: true", + "hooks:", + " pre_tool_call:", + " - command: operator-check --tool", + " - command: failproofai --hook pre_tool_call --cli hermes", + " " + FAILPROOFAI_HOOK_MARKER + ": true", + " post_tool_call:", + " - command: failproofai --hook post_tool_call --cli hermes", + " " + FAILPROOFAI_HOOK_MARKER + ": true", + "", +].join("\n"); + +/** A 1.0.6–1.0.8 install: a marked COPY of the plugin, enabled in config. */ +function makeManagedCopy(profile = "default"): void { + writeConfig(profile, "model: gpt-5\nplugins:\n enabled: [failproofai]\n"); + const dest = pluginPath(profile); + mkdirSync(dest, { recursive: true }); + for (const file of PLUGIN_FILES) cpSync(resolve(ORIG_CWD, "hermes-plugin", file), resolve(dest, file)); + writeFileSync(resolve(dest, ".failproofai-managed"), "Managed by failproofai.\n"); +} + +const daemonYes = () => vi.fn(async () => true); +const daemonNo = () => vi.fn(async () => false); + +describe("linked Hermes plugin install", () => { + it("links the profile's plugin dir to the package's hermes-plugin/", () => { + writeConfig("default", "model: gpt-5\n"); + expect(installHermesPlugin(configPath())).toBe("linked"); + expect(lstatSync(pluginPath()).isSymbolicLink()).toBe(true); + expect(readlinkSync(pluginPath())).toBe(source); + // Hermes reads the manifest through the link, exactly as from a directory. + expect(readFileSync(resolve(pluginPath(), "plugin.yaml"), "utf8")).toMatch(/^name: failproofai$/m); + expect(installHermesPlugin(configPath())).toBe("unchanged"); + }); + + it("an npm upgrade is picked up with no reinstall", () => { + installHermesPlugin(configPath()); + writeFileSync(resolve(source, "client.py"), "# new release\n"); + expect(readFileSync(resolve(pluginPath(), "client.py"), "utf8")).toBe("# new release\n"); + }); + + it("relinks a dangling link left by a removed install prefix", () => { + const oldPrefix = resolve(tempDir, "old-node", "lib", "node_modules", "failproofai", "hermes-plugin"); + mkdirSync(resolve(home(), "plugins"), { recursive: true }); + symlinkSync(oldPrefix, pluginPath()); // target never existed: dangling + expect(installHermesPlugin(configPath())).toBe("linked"); + expect(readlinkSync(pluginPath())).toBe(source); + }); + + it("refuses to replace a symlink to somebody else's plugin", () => { + const other = resolve(tempDir, "operator-plugins", "guard"); + mkdirSync(other, { recursive: true }); + writeFileSync(resolve(other, "plugin.yaml"), "name: guard\n"); + mkdirSync(resolve(home(), "plugins"), { recursive: true }); + symlinkSync(other, pluginPath()); + + expect(() => installHermesPlugin(configPath())).toThrow(/Refusing to overwrite an unmanaged Hermes plugin/); + expect(readlinkSync(pluginPath())).toBe(other); + }); + + it("refuses a link that is named hermes-plugin but is not FailproofAI's", () => { + const other = resolve(tempDir, "someone-else", "hermes-plugin"); + mkdirSync(other, { recursive: true }); + writeFileSync(resolve(other, "plugin.yaml"), "name: another-plugin\n"); + mkdirSync(resolve(home(), "plugins"), { recursive: true }); + symlinkSync(other, pluginPath()); + + expect(() => installHermesPlugin(configPath())).toThrow(/Refusing to overwrite/); + expect(readlinkSync(pluginPath())).toBe(other); + }); + + it("falls back to a marked copy when a symlink cannot be created", () => { + writeConfig("default", "model: gpt-5\n"); + const symlink = () => { + throw Object.assign(new Error("EPERM: operation not permitted"), { code: "EPERM" }); + }; + expect(installHermesPlugin(configPath(), { symlink })).toBe("copied"); + expect(lstatSync(pluginPath()).isDirectory()).toBe(true); + expect(existsSync(resolve(pluginPath(), ".failproofai-managed"))).toBe(true); + for (const file of PLUGIN_FILES) expect(existsSync(resolve(pluginPath(), file))).toBe(true); + + const settings = hermes.readSettings(configPath()); + hermes.writeHookEntries(settings, "/usr/bin/failproofai", "user"); + hermes.writeSettings(configPath(), settings); + expect(hermesProfileHealth()[0]).toMatchObject({ pluginMode: "copy", healthy: true }); + expect(hermesProfileStatusRows()[0][1]).toMatch(/^native plugin enabled \(copied/); + }); + + it("uninstall removes the link and never the package directory", () => { + writeConfig("default", "model: gpt-5\n"); + hermes.prepareInstall!(configPath()); + const settings = hermes.readSettings(configPath()); + hermes.writeHookEntries(settings, "/usr/bin/failproofai", "user"); + hermes.writeSettings(configPath(), settings); + + expect(hermes.removeHooksFromFile(configPath())).toBe(2); + expect(existsSync(pluginPath())).toBe(false); + expect(() => lstatSync(pluginPath())).toThrow(); + for (const file of PLUGIN_FILES) expect(existsSync(resolve(source, file))).toBe(true); + expect(readConfig().plugins).toBeUndefined(); + }); + + it("uninstall still removes an old managed copy", () => { + makeManagedCopy(); + expect(hermes.removeHooksFromFile(configPath())).toBe(2); + expect(existsSync(pluginPath())).toBe(false); + }); +}); + +describe("Hermes health flags legacy shell hooks", () => { + it("a profile only on shell hooks is unhealthy because cron is unchecked", () => { + writeConfig("default", LEGACY_CONFIG); + expect(hermesProfileHealth()[0]).toMatchObject({ + legacyShellHookPresent: true, + cronUnchecked: true, + healthy: false, + pluginMode: null, + }); + expect(hermesProfileStatusRows()).toEqual([ + [ + "hermes/default", + "UNHEALTHY — legacy shell hooks: Hermes cron jobs are not checked. Run `failproofai update`", + ], + ]); + }); +}); + +describe("failproofai update → Hermes migration", () => { + it("migrates a legacy shell-hook profile to the linked plugin, keeping operator hooks", async () => { + writeConfig("default", LEGACY_CONFIG); + const probe = daemonYes(); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: probe }); + + expect(result.ok).toBe(true); + expect(result.profiles[0]).toMatchObject({ status: "migrated", detail: "shell hooks → linked plugin" }); + expect(probe).toHaveBeenCalledTimes(1); + expect(readlinkSync(pluginPath())).toBe(source); + const config = readConfig(); + expect(config.plugins?.enabled).toEqual(["failproofai"]); + expect(config.hooks?.pre_tool_call).toEqual([{ command: "operator-check --tool" }]); + expect(config.hooks?.post_tool_call).toBeUndefined(); + expect(config.model).toBe("gpt-5"); + expect(hermesProfileHealth()[0]).toMatchObject({ healthy: true, pluginMode: "link" }); + expect(result.lines.join("\n")).toMatch(/hermes\/default\s+migrated — shell hooks → linked plugin/); + expect(result.lines.join("\n")).toMatch(/Cron jobs load the plugin on their next run/); + }); + + it("migrates a copied plugin (1.0.6–1.0.8) to the link", async () => { + makeManagedCopy(); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: daemonYes() }); + expect(result.ok).toBe(true); + expect(result.profiles[0]).toMatchObject({ status: "migrated", detail: "copied plugin → linked plugin" }); + expect(lstatSync(pluginPath()).isSymbolicLink()).toBe(true); + expect(readConfig().plugins?.enabled).toEqual(["failproofai"]); + }); + + it("leaves profiles without any failproofai integration alone and asks the daemon nothing", async () => { + writeConfig("default", "model: gpt-5\n"); + writeConfig("work", "model: other\nplugins:\n enabled: [operator-plugin]\n"); + const before = readFileSync(configPath("work"), "utf8"); + const probe = daemonYes(); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: probe }); + + expect(result).toMatchObject({ ok: true, lines: [] }); + expect(result.profiles.map((p) => p.status)).toEqual(["untouched", "untouched"]); + expect(probe).not.toHaveBeenCalled(); + expect(readFileSync(configPath("work"), "utf8")).toBe(before); + expect(existsSync(pluginPath("work"))).toBe(false); + expect(existsSync(pluginPath())).toBe(false); + }); + + it("migrates only the profiles that use failproofai", async () => { + writeConfig("default", LEGACY_CONFIG); + writeConfig("work", "model: other\n"); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: daemonYes() }); + expect(result.profiles.map((p) => [p.name, p.status])).toEqual([ + ["default", "migrated"], + ["work", "untouched"], + ]); + expect(existsSync(pluginPath("work"))).toBe(false); + expect(result.lines.join("\n")).toMatch(/hermes\/work\s+skipped — no failproofai integration/); + }); + + it("reports an already-linked profile as current without probing the daemon", async () => { + writeConfig("default", "model: gpt-5\n"); + hermes.prepareInstall!(configPath()); + const settings = hermes.readSettings(configPath()); + hermes.writeHookEntries(settings, "/usr/bin/failproofai", "user"); + hermes.writeSettings(configPath(), settings); + + const probe = daemonYes(); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: probe }); + expect(result.ok).toBe(true); + expect(result.profiles[0].status).toBe("current"); + expect(probe).not.toHaveBeenCalled(); + expect(result.lines.join("\n")).toMatch(/hermes\/default\s+already current/); + }); + + it("keeps the shell hooks and fails when the daemon cannot do policyEvaluation", async () => { + writeConfig("default", LEGACY_CONFIG); + makeManagedCopy("work"); + const legacyBefore = readFileSync(configPath(), "utf8"); + const probe = daemonNo(); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: probe }); + + expect(result.ok).toBe(false); // → `failproofai update` exits 1 + expect(probe).toHaveBeenCalledTimes(1); + expect(result.profiles.map((p) => [p.name, p.status, p.legacyShellHooksRemain])).toEqual([ + ["default", "blocked", true], + ["work", "blocked", false], + ]); + // Nothing changed anywhere: the hooks are the only enforcement left. + expect(readFileSync(configPath(), "utf8")).toBe(legacyBefore); + expect(existsSync(pluginPath())).toBe(false); + expect(lstatSync(pluginPath("work")).isDirectory()).toBe(true); + const text = result.lines.join("\n"); + expect(text).toMatch(/hermes\/default\s+NOT migrated — daemon lacks native policy evaluation; shell hooks left in place/); + expect(text).toMatch(/Hermes is NOT migrated/); + expect(text).toMatch(/failproofai config/); + }); + + it("fails without touching hooks when an unmanaged plugin holds the name", async () => { + writeConfig("default", LEGACY_CONFIG); + mkdirSync(pluginPath(), { recursive: true }); + writeFileSync(resolve(pluginPath(), "plugin.yaml"), "name: operator-owned\n"); + const before = readFileSync(configPath(), "utf8"); + + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: daemonYes() }); + expect(result.ok).toBe(false); + expect(result.profiles[0]).toMatchObject({ status: "failed", legacyShellHooksRemain: true }); + expect(result.profiles[0].detail).toMatch(/unmanaged plugin occupies/); + expect(readFileSync(configPath(), "utf8")).toBe(before); + expect(readFileSync(resolve(pluginPath(), "plugin.yaml"), "utf8")).toBe("name: operator-owned\n"); + }); + + it("never rewrites a config.yaml that does not parse", async () => { + const broken = LEGACY_CONFIG + " bad: [unclosed\n"; + writeConfig("default", broken); + const result = await runHermesUpdateMigration({ daemonSupportsPolicyEvaluation: daemonYes() }); + expect(result.ok).toBe(false); + expect(result.profiles[0].status).toBe("failed"); + expect(readFileSync(configPath(), "utf8")).toBe(broken); + expect(existsSync(pluginPath())).toBe(false); + }); + + it("uses the copy fallback during migration when links are impossible", async () => { + writeConfig("default", LEGACY_CONFIG); + const symlink = () => { + throw new Error("EPERM"); + }; + const result = await runHermesUpdateMigration({ + daemonSupportsPolicyEvaluation: daemonYes(), + installDeps: { symlink }, + }); + expect(result.ok).toBe(true); + expect(result.profiles[0].detail).toBe("shell hooks → plugin copy (symlink not possible here)"); + expect(existsSync(resolve(pluginPath(), ".failproofai-managed"))).toBe(true); + expect(readConfig().hooks?.pre_tool_call).toEqual([{ command: "operator-check --tool" }]); + }); +}); diff --git a/bin/failproofai.mjs b/bin/failproofai.mjs index 49cfc2e65..5876f64db 100755 --- a/bin/failproofai.mjs +++ b/bin/failproofai.mjs @@ -1180,6 +1180,9 @@ async function runCli() { "", "This does the rest of the upgrade: runs any pending layout migrations,", "puts the matching daemon binary in place, and restarts the service.", + "Hermes profiles already using failproofai are moved to the linked", + "native plugin (legacy shell hooks never checked Hermes cron jobs).", + "Exits non-zero when any half could not be brought current.", ], }, { @@ -1270,17 +1273,40 @@ async function runCli() { } } + // Hermes is migrated AFTER the daemon, and gated on what the daemon that is + // now running can do: the native plugin needs `policyEvaluation`, which + // ≤1.0.5 daemons lack, so when the swap above could not happen (a sudo + // system service with no sudo) the shell hooks stay and this fails loudly. + let hermesOk = true; + let hermesMigrated = 0; + try { + const { runHermesUpdateMigration } = await import("../src/hooks/hermes-update"); + const svc = await import("../src/hooks/daemon-service"); + const hermesResult = await runHermesUpdateMigration({ + daemonSupportsPolicyEvaluation: () => svc.probeDaemonPolicyEvaluation(), + }); + if (hermesResult.lines.length > 0) report.push("", ...hermesResult.lines); + hermesOk = hermesResult.ok; + hermesMigrated = hermesResult.profiles.filter((p) => p.status === "migrated").length; + } catch (err) { + report.push("", `Hermes migration failed: ${err instanceof Error ? err.message : String(err)}`); + hermesOk = false; + } + + const updateOk = daemonOk && !migrationFailed && hermesOk; await printReport("update", report, { - ok: daemonOk && !migrationFailed, + ok: updateOk, meta: `v${version}`, }); await track("cli_update", { - ok: daemonOk && !migrationFailed, + ok: updateOk, migrations: migrationsRan, migration_failed: migrationFailed, + hermes_ok: hermesOk, + hermes_migrated: hermesMigrated, }); lastSubcommand = null; - await exitAfterFlush(daemonOk && !migrationFailed ? 0 : 1); + await exitAfterFlush(updateOk ? 0 : 1); return; } diff --git a/src/hooks/hermes-update.ts b/src/hooks/hermes-update.ts new file mode 100644 index 000000000..182d017ae --- /dev/null +++ b/src/hooks/hermes-update.ts @@ -0,0 +1,74 @@ +/** + * The Hermes half of `failproofai update`. + * + * npm replaces the CLI and nothing else, and until this ran `update` never + * looked at Hermes — so a machine that installed Hermes enforcement with ≤1.0.5 + * kept its config.yaml shell hooks through every upgrade. Those hooks never + * reach Hermes cron jobs (each cron fire builds its own hook scope, which only + * discovered plugins join), so every scheduled job on such a machine ran + * unchecked while everything looked configured. + * + * This migrates each profile that already uses FailproofAI to the linked native + * plugin, and reports per profile. It fails (ok: false → non-zero exit) whenever + * a profile that uses FailproofAI is left on shell hooks or otherwise could not + * be brought current, because a green `update` is exactly how the gap stayed + * invisible. + */ +import { migrateHermesProfiles, type HermesPluginInstallDeps, type HermesProfileMigration } from "./integrations"; + +export interface HermesUpdateResult { + ok: boolean; + lines: string[]; + profiles: HermesProfileMigration[]; +} + +const STATUS_LABEL: Record = { + migrated: "migrated", + current: "already current", + untouched: "skipped", + blocked: "NOT migrated", + failed: "FAILED", +}; + +export async function runHermesUpdateMigration(deps: { + daemonSupportsPolicyEvaluation: () => Promise; + installDeps?: HermesPluginInstallDeps; +}): Promise { + const profiles = await migrateHermesProfiles(deps); + const relevant = profiles.filter((p) => p.status !== "untouched"); + // A machine without FailproofAI on Hermes hears nothing about Hermes. + if (relevant.length === 0) return { ok: true, lines: [], profiles }; + + const width = Math.max(...profiles.map((p) => ("hermes/" + p.name).length)); + const lines = ["Hermes:"]; + for (const p of profiles) { + const label = ("hermes/" + p.name).padEnd(width); + const detail = p.status === "current" ? "" : ` — ${p.detail}`; + lines.push(` ${label} ${STATUS_LABEL[p.status]}${detail}`); + } + + const blocked = profiles.filter((p) => p.status === "blocked"); + const failed = profiles.filter((p) => p.status === "failed"); + if (profiles.some((p) => p.status === "migrated")) { + lines.push( + " Cron jobs load the plugin on their next run; restart running Hermes gateways and sessions to load it there.", + ); + } + if (blocked.length > 0) { + lines.push( + "", + `Hermes is NOT migrated (${blocked.map((p) => p.name).join(", ")}): the running failproofaid cannot`, + "answer the native plugin (no policyEvaluation), so nothing was changed. Profiles still on", + "shell hooks do NOT check Hermes cron jobs. Run `failproofai config` (it asks for sudo to", + "replace the daemon), then `failproofai update` again.", + ); + } + if (failed.length > 0) { + lines.push( + "", + `Hermes could not be migrated for: ${failed.map((p) => p.name).join(", ")}. Fix the reason above,`, + "then run `failproofai update` again.", + ); + } + return { ok: blocked.length === 0 && failed.length === 0, lines, profiles }; +} From 53fc1261dbe7444872d2ad62b943693d1a2e3214 Mon Sep 17 00:00:00 2001 From: chhhee10 Date: Tue, 29 Sep 2026 15:27:49 +0530 Subject: [PATCH 03/22] docs(hermes): linked plugin, update migration, and the cron gap Harness and CLI reference, the plugin README, CLAUDE.md and the 1.0.9-beta.0 changelog now describe the linked install, `update` migrating Hermes profiles (and exiting non-zero when it cannot), and why legacy shell hooks leave Hermes cron jobs unchecked. English pages only; translations not updated. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 4 ++++ CLAUDE.md | 15 ++++++++++----- docs/reference/failproof-cli.mdx | 2 +- docs/reference/harnesses.mdx | 23 ++++++++++++++++++----- hermes-plugin/README.md | 26 ++++++++++++++++++-------- 5 files changed, 51 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bfc1ce19b..e8e4451bd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## 1.0.9-beta.0 — 2026-09-29 +### Fixes + +- **Hermes cron jobs ran unchecked after an upgrade.** Legacy Hermes shell hooks (≤1.0.5) are never run for cron jobs — each cron fire builds its own hook scope that only discovered plugins join — and `failproofai update` never touched Hermes, so upgraded machines stayed on them silently. `update` now moves every Hermes profile that already uses FailproofAI (shell hooks or a copied plugin) to the native plugin, prints a per-profile report, and leaves profiles without FailproofAI alone. When the running daemon cannot serve the plugin (no `policyEvaluation`, e.g. a sudo system daemon `update` could not replace) the shell hooks are kept and `update` exits 1 pointing at `failproofai config`; it also exits 1 whenever the daemon swap or a layout migration failed. The plugin is now **linked** into each profile (`plugins/failproofai` → the package's `hermes-plugin/`, a marked copy where symlinks are unavailable), so npm upgrades apply with no reinstall; uninstall removes only the link. `failproofai config --status` reports a profile still on shell hooks as unhealthy: "Hermes cron jobs are not checked". + ### Docs - Split Jev documentation into session evaluations under Find failures, live policy review under Prevent failures, and provider/configuration detail under Reference. Add a Use Jev page after Core concepts in Start with eval and policy setup tabs, plus a short quickstart link, dashboard screenshots, and CLI steps. Move sentiment analysis into Find failures and show its Jev-scored dashboard flow. Clarify shadow-mode verification and the Cloud machine key's `jev:evaluate` permission. diff --git a/CLAUDE.md b/CLAUDE.md index 65ab6fa13..fd43f6c11 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -323,10 +323,15 @@ we run in-repo. Hermes is a **dual-pillar** integration: an **audit** adapter Hermes enforcement uses the shipped **native Python plugin** under each profile's `plugins/failproofai/` directory. The profile's YAML config enables it through -`plugins.enabled: [failproofai]`. `integrations.ts` copies the plugin atomically, -marks the directory as FailproofAI-managed, refuses to overwrite an unmanaged -directory with the same name, and uses the `yaml` package's comment-preserving -`Document` API for config changes. +`plugins.enabled: [failproofai]`. `integrations.ts` symlinks that directory to the +package's `hermes-plugin/` (Hermes' scan follows symlinks, so npm upgrades apply +with no reinstall — the OpenClaw `plugins.load.paths` model), falls back to an +atomic copy marked `.failproofai-managed` where a link cannot be created, replaces +only a marked copy or a link into a FailproofAI `hermes-plugin/`, refuses anything +else with the same name, and uses the `yaml` package's comment-preserving +`Document` API for config changes. `failproofai update` migrates profiles already +using FailproofAI (legacy shell hooks or a copy) to the link, gated on the daemon +answering `policyEvaluation`; legacy shell hooks never run for Hermes cron jobs. Settings file paths: @@ -335,7 +340,7 @@ Settings file paths: | user | `~/.hermes/config.yaml` | Hermes is **user-scope only** — there is no project config, so `getSettingsPath` -ignores scope/cwd. Every default and named profile receives its own plugin copy and +ignores scope/cwd. Every default and named profile receives its own plugin link and enablement entry. Installed-state detection requires both the complete managed plugin directory and the config entry; a missing file, disabled plugin, newly-created profile, or leftover legacy shell hook is reported as unhealthy. diff --git a/docs/reference/failproof-cli.mdx b/docs/reference/failproof-cli.mdx index 8ee65be6c..3aa1838b6 100644 --- a/docs/reference/failproof-cli.mdx +++ b/docs/reference/failproof-cli.mdx @@ -116,7 +116,7 @@ Local pauses suspend builtin, custom, convention, and pack policies for one sess | `migrate` | `--dry-run` | | `uninstall` | `--purge`, `--dry-run`, `--yes` | -`failproofai update` should be run after `npm install -g failproofai@latest`; it performs home-layout migrations, installs the matching daemon binary, and restarts the service. `--no-daemon` performs only the layout migration. +`failproofai update` should be run after `npm install -g failproofai@latest`; it performs home-layout migrations, installs the matching daemon binary, and restarts the service. It then moves every Hermes profile that already uses FailproofAI to the linked native plugin and prints one line per profile. `--no-daemon` skips the daemon step. `update` exits non-zero when the daemon could not be replaced, a migration failed, or a Hermes profile could not be migrated (for example because the running daemon cannot serve the native plugin, in which case its shell hooks are left in place). ## Harness paths diff --git a/docs/reference/harnesses.mdx b/docs/reference/harnesses.mdx index 8d52d8952..b24b63614 100644 --- a/docs/reference/harnesses.mdx +++ b/docs/reference/harnesses.mdx @@ -46,16 +46,29 @@ Capabilities are version-sensitive. Re-test after upgrading an agent CLI, especi ### Hermes native plugin Hermes is integrated through a profile-local native plugin rather than a shell -command. Installation copies the plugin into every default and named Hermes -profile, enables it in that profile's `config.yaml`, and migrates only legacy -FailproofAI shell-hook entries. This avoids a process spawn on each hook and -lets `instruct()` reach the model through Hermes' native blocked-tool result. +command. Installation links every default and named Hermes profile's +`plugins/failproofai` to the plugin shipped in the npm package (a copy where a +symlink cannot be created), enables it in that profile's `config.yaml`, and +migrates only legacy FailproofAI shell-hook entries. Because the plugin is +linked, `npm install -g failproofai@latest` updates it with no reinstall. This +avoids a process spawn on each hook and lets `instruct()` reach the model +through Hermes' native blocked-tool result. + +Legacy shell hooks (installed by 1.0.5 and earlier) do **not** check Hermes +cron jobs: each cron run builds its own hook scope, which the native plugin +joins and `config.yaml` shell hooks do not. `failproofai update` migrates every +profile that already uses FailproofAI to the linked plugin. If the running +daemon cannot serve the plugin, `update` leaves the shell hooks in place and +exits non-zero; run `failproofai config` to update the daemon, then +`failproofai update` again. Cron jobs load the plugin on their next run; restart +running gateways and interactive sessions to load it there. The first matching instruction blocks the pending call. The same API request stays blocked; a later model iteration may retry. A persistent, profile-scoped ledger and a per-turn cap prevent an advisory instruction from becoming an unbounded loop. `deny()` remains a hard block. Run `failproofai config --status` -to detect a disabled, incomplete, duplicated, or newly unconfigured profile. +to detect a disabled, incomplete, duplicated, or newly unconfigured profile, or +one still on legacy shell hooks (reported as "Hermes cron jobs are not checked"). ## Install capture and policy hooks diff --git a/hermes-plugin/README.md b/hermes-plugin/README.md index d33d39806..05388a67d 100644 --- a/hermes-plugin/README.md +++ b/hermes-plugin/README.md @@ -21,11 +21,20 @@ Then run: failproofai policies --install --cli hermes --scope user ``` -FailproofAI copies this directory to every discovered profile at -`/plugins/failproofai/` and adds `failproofai` to -`plugins.enabled` in that profile's `config.yaml`. Reinstall replaces only a -directory carrying `.failproofai-managed`; an unrelated plugin with the same -directory name is never overwritten. +The installer symlinks `/plugins/failproofai` in every discovered +profile to this directory and adds `failproofai` to `plugins.enabled` in that +profile's `config.yaml`. Hermes' plugin scan follows the link, so an npm +upgrade updates the plugin with no reinstall. Where a symlink cannot be +created, the files are copied instead and marked `.failproofai-managed`. +Reinstall replaces only such a marked copy or a link into a FailproofAI +`hermes-plugin/`; an unrelated plugin with the same name is never overwritten. +Python may write `__pycache__/` here when the package directory is writable; +it goes with the package on upgrade and is skipped when it cannot be written. + +`failproofai update` moves profiles that already use FailproofAI (legacy shell +hooks or a copied plugin) to the link. Legacy shell hooks never ran for Hermes +cron jobs, so `update` exits non-zero and keeps the shell hooks when the +running daemon cannot serve the plugin. Legacy FailproofAI shell hooks are removed during migration. Operator-owned hooks and unrelated plugin settings are preserved. No dashboard deployment or @@ -118,7 +127,8 @@ untrusted evaluation failures. ## Diagnostics and rollback `failproofai config --status` reports each existing Hermes profile as healthy, -disabled, incomplete, or duplicated with a legacy shell hook. Hermes-side load +disabled, incomplete, duplicated with a legacy shell hook, or still on legacy +shell hooks alone (Hermes cron jobs are not checked). Hermes-side load errors are available through: ```bash @@ -134,5 +144,5 @@ the integration with: failproofai policies --uninstall --cli hermes --scope user ``` -Uninstall removes the config registration and only the plugin directory marked -as FailproofAI-managed. +Uninstall removes the config registration and the profile's link (or marked +copy); it never removes this package directory. From ff959594d7dc979cec8c6f6c893c289b0b032d83 Mon Sep 17 00:00:00 2001 From: chhhee10 Date: Tue, 29 Sep 2026 15:32:33 +0530 Subject: [PATCH 04/22] collector: read Jev's jevMode "observe" and ship one value MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Jev's log-only mode is being renamed from "shadow" to "observe". Hook rows written by an older worker or daemon still say "shadow", and the activity store is never rewritten, so the collector accepts both and normalizes "shadow" to "observe" before emitting jev_mode — the Cloud sees a single spelling per mode. Co-Authored-By: Claude Opus 5.5 --- .../src/sources/hooks/transform.rs | 17 +++-- crates/fpai-collect/tests/hooks_jev.rs | 63 +++++++++++++------ 2 files changed, 56 insertions(+), 24 deletions(-) diff --git a/crates/fpai-collect/src/sources/hooks/transform.rs b/crates/fpai-collect/src/sources/hooks/transform.rs index 365e8ef44..6bf8768f1 100644 --- a/crates/fpai-collect/src/sources/hooks/transform.rs +++ b/crates/fpai-collect/src/sources/hooks/transform.rs @@ -114,7 +114,7 @@ pub struct HookRow { pub pause_expires_at: Option, /// Verdicts from observe-mode policies: evaluated, then discarded. The - /// whole measurement a trial exists to produce. Jev in shadow mode files + /// whole measurement a trial exists to produce. Jev in observe mode files /// its own deny/instruct here too (`policyId: "semantic/"`, the Jev /// model id as `version`), and ships the same way: whole, un-rolled-up. pub observed: Option, @@ -141,7 +141,8 @@ pub struct HookRow { pub jev_latency_ms: Option, #[serde(rename = "jevModel", default, deserialize_with = "lenient")] pub jev_model: Option, - /// `shadow` | `enforce`. + /// `observe` | `enforce` — or `shadow`, what builds before the rename + /// wrote for `observe`. [`JevFacts::of`] emits `observe` for both. #[serde(rename = "jevMode", default, deserialize_with = "lenient")] pub jev_mode: Option, } @@ -408,10 +409,18 @@ impl JevFacts { .as_deref() .filter(|e| matches!(*e, "jev" | "jev-fallback"))? .to_string(); + // `shadow` is `observe` under the name it had before the rename. A + // worker or daemon from an older build still writes it, and the store + // is never rewritten, so both are read — and one value, `observe`, is + // emitted, so the Cloud sees a single spelling per mode. let mode = row .jev_mode .as_deref() - .filter(|m| matches!(*m, "shadow" | "enforce")) + .and_then(|m| match m { + "observe" | "shadow" => Some("observe"), + "enforce" => Some("enforce"), + _ => None, + }) .map(str::to_string); let fallback_reason = row.jev_fallback_reason.as_deref().and_then(jev_reason_code); let decision = row @@ -472,7 +481,7 @@ impl JevFacts { /// True when this row must be shipped on its own rather than rolled into /// an allow aggregate: Jev overruled a regex deny/instruct (a clear), or - /// Jev's own verdict was stricter than the outcome (shadow mode, where the + /// Jev's own verdict was stricter than the outcome (observe mode, where the /// regex result was enforced). Either is a decision someone will want to /// find, and a count cannot show it. pub fn is_notable(&self, final_decision: &str) -> bool { diff --git a/crates/fpai-collect/tests/hooks_jev.rs b/crates/fpai-collect/tests/hooks_jev.rs index 3f84b9755..f695f32f0 100644 --- a/crates/fpai-collect/tests/hooks_jev.rs +++ b/crates/fpai-collect/tests/hooks_jev.rs @@ -394,16 +394,16 @@ async fn a_jev_clear_is_never_rolled_into_an_allow_count() { } #[tokio::test(flavor = "multi_thread")] -async fn a_shadow_allow_jev_would_have_blocked_is_shipped_on_its_own() { +async fn an_observe_allow_jev_would_have_blocked_is_shipped_on_its_own() { let (store, state, spool) = (tmpdir("shd-s"), tmpdir("shd-st"), tmpdir("shd-sp")); - // Shadow mode enforced the regex allow; Jev said deny. Like an observe-mode + // Observe mode enforced the regex allow; Jev said deny. Like an observe-mode // verdict, that disagreement is the measurement. - let shadow = jev_row( + let observe = jev_row( 1785740912000, "allow", - json!({ "jevMode": "shadow", "jevDecision": "deny" }), + json!({ "jevMode": "observe", "jevDecision": "deny" }), ); - write_rows(&store, &[shadow]); + write_rows(&store, &[observe]); run_once(&store, &state, &spool, HooksVerbosity::Decisions).await; let events = spooled(&spool); @@ -411,7 +411,30 @@ async fn a_shadow_allow_jev_would_have_blocked_is_shipped_on_its_own() { let end = completed(&events); assert_eq!(end["outcome"], "allow"); assert_eq!(end["jev_decision"], "deny"); - assert_eq!(end["jev_mode"], "shadow"); + assert_eq!(end["jev_mode"], "observe"); + + cleanup(&[&store, &state, &spool]); +} + +#[tokio::test(flavor = "multi_thread")] +async fn a_row_an_older_build_wrote_with_jev_mode_shadow_ships_as_observe() { + let (store, state, spool) = (tmpdir("leg-s"), tmpdir("leg-st"), tmpdir("leg-sp")); + // `shadow` is what `observe` was called before the rename. An older worker + // may still write it, and the store is never rewritten; the Cloud is sent + // one spelling. + let legacy = jev_row( + 1785740912000, + "allow", + json!({ "jevMode": "shadow", "jevDecision": "deny" }), + ); + write_rows(&store, &[legacy]); + run_once(&store, &state, &spool, HooksVerbosity::Decisions).await; + + let events = spooled(&spool); + assert_eq!(events.len(), 2, "still notable: a pair, not an aggregate"); + let end = completed(&events); + assert_eq!(end["jev_decision"], "deny"); + assert_eq!(end["jev_mode"], "observe"); cleanup(&[&store, &state, &spool]); } @@ -644,7 +667,7 @@ fn every_golden_row_parses_and_maps() { } #[tokio::test(flavor = "multi_thread")] -async fn golden_rows_ship_clears_and_shadow_disagreements_individually() { +async fn golden_rows_ship_clears_and_observe_disagreements_individually() { let (store, state, spool) = (tmpdir("gold-s"), tmpdir("gold-st"), tmpdir("gold-sp")); fs::write(store.join("current.jsonl"), golden()).unwrap(); run_once(&store, &state, &spool, HooksVerbosity::Decisions).await; @@ -654,7 +677,7 @@ async fn golden_rows_ship_clears_and_shadow_disagreements_individually() { .iter() .filter(|e| e["type"] == "hook_completed") .collect(); - // Two denies, the enforce-mode clear and the shadow disagreement ship as + // Two denies, the enforce-mode clear and the observe disagreement ship as // pairs; only the plain regex allow rolls up. assert_eq!(completions.len(), 5, "{events:#?}"); let aggs: Vec<&&Value> = completions @@ -671,7 +694,7 @@ async fn golden_rows_ship_clears_and_shadow_disagreements_individually() { assert!( completions .iter() - .any(|e| e["jev_mode"] == "shadow" && e["jev_decision"] == "deny") + .any(|e| e["jev_mode"] == "observe" && e["jev_decision"] == "deny") ); assert!(!spooled_text(&spool).contains("zebra")); @@ -733,7 +756,7 @@ fn a_call_jev_was_not_consulted_on_claims_no_answer() { ); assert_eq!( completed(&transform::to_events(&rows[1], 1, "local"))["jev_mode"], - "shadow" + "observe" ); } @@ -956,7 +979,7 @@ fn a_call_jev_sent_no_request_for_claims_no_answer() { ); assert_eq!( completed(&transform::to_events(&rows[1], 1, "local"))["jev_mode"], - "shadow" + "observe" ); } @@ -1101,18 +1124,18 @@ async fn a_rollup_keeps_every_jev_outcome_and_the_regex_apart() { } // --------------------------------------------------------------------------- -// Shadow-mode and fallback verdicts stricter than the outcome +// Observe-mode and fallback verdicts stricter than the outcome // --------------------------------------------------------------------------- #[tokio::test(flavor = "multi_thread")] -async fn a_shadow_allow_jev_would_have_instructed_on_is_shipped_on_its_own() { +async fn an_observe_allow_jev_would_have_instructed_on_is_shipped_on_its_own() { let (store, state, spool) = (tmpdir("shi-s"), tmpdir("shi-st"), tmpdir("shi-sp")); - let shadow = jev_row( + let observe = jev_row( 1785740912000, "allow", - json!({ "jevMode": "shadow", "jevDecision": "instruct" }), + json!({ "jevMode": "observe", "jevDecision": "instruct" }), ); - write_rows(&store, &[shadow]); + write_rows(&store, &[observe]); run_once(&store, &state, &spool, HooksVerbosity::Decisions).await; let events = spooled(&spool); @@ -1125,7 +1148,7 @@ async fn a_shadow_allow_jev_would_have_instructed_on_is_shipped_on_its_own() { let end = completed(&events); assert_eq!(end["outcome"], "allow"); assert_eq!(end["jev_decision"], "instruct"); - assert_eq!(end["jev_mode"], "shadow"); + assert_eq!(end["jev_mode"], "observe"); cleanup(&[&store, &state, &spool]); } @@ -1436,7 +1459,7 @@ async fn a_rollups_latency_is_the_mean_and_max_of_valid_values_only() { // A. A call Jev's own verdict decided (enforce) is attributed // `policySource: "jev"`, which must reach the server as `policy_source` — // the chart's source bucket — rather than leaving the block "unattributed". -// B. In shadow mode, Jev's own deny / instruct is a "would have" in the row's +// B. In observe mode, Jev's own deny / instruct is a "would have" in the row's // `observed` list, which must reach the server whole in the observed // payload key. Those rows are ALLOWS, so an allow roll-up would fold them // into a count and the "would have" would vanish. @@ -1476,7 +1499,7 @@ fn a_jev_decided_row_ships_attributed_to_jev() { } #[test] -fn a_shadow_would_have_ships_whole_in_observed() { +fn an_observe_would_have_ships_whole_in_observed() { let rows: Vec = policy_page_golden() .lines() .map(|l| serde_json::from_str(l).expect("a store-written row must parse")) @@ -1495,7 +1518,7 @@ fn a_shadow_would_have_ships_whole_in_observed() { end.get("policy_source").is_none(), "nothing decided: {end:#}" ); - assert_eq!(end["jev_mode"], "shadow"); + assert_eq!(end["jev_mode"], "observe"); let observed = end[OBSERVED_KEY] .as_array() .expect("observed ships as an array"); From f7a58b67acc29f6501560064f97d2f0950f71286 Mon Sep 17 00:00:00 2001 From: chhhee10 Date: Tue, 29 Sep 2026 15:32:49 +0530 Subject: [PATCH 05/22] jev: rename the "shadow" mode to "observe", reading "shadow" as an alias Jev's log-only mode (asked and logged, regex decides) is now "observe", the word already used for a policy rollout that is evaluated but not enforced. Every writer emits "observe": jev.json (CLI setup, config --token's initial Cloud file, the dashboard actions), hook-activity rows' jevMode, verdicts.jsonl's applied, and the jev status stats (observeClearsByPolicy, modes.observe). CLI help, status text, dashboard panel and the activity pill ("jev observe") say observe. "shadow" stays a read alias wherever a mode is parsed: normalizeJevMode in jev-config (so an existing jev.json, a Cloud-written file, --mode shadow, and the dashboard server actions all accept it), sanitizeJevActivity for rows an older worker wrote, and the stats renderers for the pre-rename keys. --mode shadow saves "observe" and prints a one-line note; any save rewrites an older file's "shadow" as "observe". Internal names follow: JevMode, ObserveVerdict/observeVerdict, CLOUD_JEV_INITIAL_MODE. The collector's golden fixtures are regenerated from the store. Co-Authored-By: Claude Opus 5.5 --- __tests__/actions/jev-mode-action.test.ts | 18 ++++- __tests__/actions/update-jev-config.test.ts | 62 ++++++++------- .../jev-notices-no-request.test.tsx | 2 +- .../jev-notices-not-consulted.test.tsx | 4 +- __tests__/components/jev-notices.test.tsx | 16 ++-- .../components/jev-settings-panel.test.tsx | 26 +++---- __tests__/fixtures/jev-activity-rows.ts | 2 +- __tests__/fixtures/jev-no-request-rows.ts | 2 +- __tests__/fixtures/jev-not-consulted-rows.ts | 2 +- __tests__/fixtures/jev-policy-page-rows.ts | 10 +-- __tests__/hooks/cloud-connect-jev.test.ts | 42 +++++----- __tests__/hooks/hook-telemetry-jev.test.ts | 4 +- __tests__/hooks/jev-activity.test.ts | 14 +++- __tests__/hooks/jev-cli-bin.test.ts | 8 +- __tests__/hooks/jev-cli-cloud.test.ts | 77 +++++++++++-------- __tests__/hooks/jev-cli-hardening.test.ts | 16 ++-- __tests__/hooks/jev-cli-review.test.ts | 6 +- __tests__/hooks/jev-cli-status-stats.test.ts | 14 ++-- __tests__/hooks/jev-cli-url-token.test.ts | 10 +-- __tests__/hooks/jev-cli.test.ts | 53 ++++++++++--- .../hooks/jev-cloud-disconnect-race.test.ts | 8 +- __tests__/hooks/jev-env-key.test.ts | 4 +- __tests__/hooks/jev-field-shapes.test.ts | 4 +- __tests__/hooks/jev-no-request.test.ts | 10 +-- __tests__/hooks/jev-not-consulted.test.ts | 10 +-- .../hooks/jev-policy-page-golden.test.ts | 4 +- __tests__/hooks/jev-telemetry-privacy.test.ts | 10 +-- ...est.ts => combine-observe-verdict.test.ts} | 52 ++++++------- __tests__/hooks/semantic/combine.test.ts | 22 +++--- .../semantic/jev-client-hardening.test.ts | 16 ++-- .../semantic/jev-client-redirect.test.ts | 4 +- .../hooks/semantic/jev-cloud-config.test.ts | 16 ++-- .../semantic/jev-cloud-transport.test.ts | 6 +- __tests__/hooks/semantic/jev-config.test.ts | 26 ++++++- __tests__/hooks/semantic/jev-review.test.ts | 16 ++-- .../semantic/jev-stats-hardening.test.ts | 4 +- __tests__/hooks/semantic/jev-stats.test.ts | 39 ++++++++-- .../semantic/truncation-severity.test.ts | 4 +- __tests__/hooks/two-tier-handler.test.ts | 64 +++++++-------- app/actions/get-jev-config.ts | 5 +- app/actions/update-jev-config.ts | 23 +++--- app/components/jev-notices.tsx | 20 ++--- app/settings/jev-panel.tsx | 22 +++--- bin/failproofai.mjs | 8 +- .../hook-activity-jev-no-request.jsonl | 2 +- .../hook-activity-jev-not-consulted.jsonl | 2 +- .../hook-activity-jev-policy-page.jsonl | 4 +- .../tests/fixtures/hook-activity-jev.jsonl | 2 +- src/hooks/cloud-connection.ts | 12 +-- src/hooks/handler.ts | 18 ++--- src/hooks/hook-activity-store.ts | 10 ++- src/hooks/jev-activity.ts | 14 +++- src/hooks/jev-cli.ts | 52 ++++++++----- src/hooks/jev-cloud-connection.ts | 8 +- src/hooks/policy-evaluator.ts | 10 +-- src/hooks/semantic/combine.ts | 26 +++---- src/hooks/semantic/evaluator.ts | 5 +- src/hooks/semantic/jev-client.ts | 2 +- src/hooks/semantic/jev-config.ts | 45 ++++++++--- src/hooks/semantic/jev-review.ts | 6 +- src/hooks/semantic/jev-stats.ts | 54 +++++++++---- 61 files changed, 625 insertions(+), 432 deletions(-) rename __tests__/hooks/semantic/{combine-shadow-verdict.test.ts => combine-observe-verdict.test.ts} (61%) diff --git a/__tests__/actions/jev-mode-action.test.ts b/__tests__/actions/jev-mode-action.test.ts index 7aafccca4..7301d9fa4 100644 --- a/__tests__/actions/jev-mode-action.test.ts +++ b/__tests__/actions/jev-mode-action.test.ts @@ -1,7 +1,7 @@ // @vitest-environment node /** * The /settings Jev panel's FailproofAI Cloud controls, against the real - * loader: `setJevModeAction` (the on/off switch and shadow/enforce) and the + * loader: `setJevModeAction` (the on/off switch and observe/enforce) and the * Cloud connection row in `getJevSettingsAction`. * * 1. **It rewrites `mode` and nothing else** — every other byte-level field @@ -33,7 +33,7 @@ const INGEST_KEY = ["fp", "ingest", "0badc0ffee123456"].join("-"); const POLICY_KEY = ["fp", "policy", "feedfacecafe7890"].join("-"); const BYOK_KEY = ["ts", "byok", "0123456789abcdef"].join("-"); const ORIGIN = "https://app.befailproof.ai"; -const CLOUD_FILE = { provider: "failproofai", baseUrl: `${ORIGIN}/enforcement/v1/jev`, mode: "shadow" }; +const CLOUD_FILE = { provider: "failproofai", baseUrl: `${ORIGIN}/enforcement/v1/jev`, mode: "observe" }; let home: string; let prevHome: string | undefined; @@ -93,7 +93,7 @@ describe("setJevModeAction", () => { connect(); // A field this build does not know, and one it does not show: both must survive. seed({ ...CLOUD_FILE, timeoutMs: 2500, fromANewerBuild: { x: 1 } }); - for (const mode of ["enforce", "off", "shadow"] as const) { + for (const mode of ["enforce", "off", "observe"] as const) { const res = await setJevModeAction(mode); expect(res.ok).toBe(true); expect(onDisk()).toEqual({ ...CLOUD_FILE, timeoutMs: 2500, fromANewerBuild: { x: 1 }, mode }); @@ -122,7 +122,7 @@ describe("setJevModeAction", () => { expect(res.ok).toBe(true); expect(onDisk()).toEqual({ provider: "typesafe", apiKey: BYOK_KEY, mode: "off" }); secretFree(res); - expect((await setJevModeAction("shadow")).ok).toBe(true); + expect((await setJevModeAction("observe")).ok).toBe(true); expect(loadJevConfig()?.apiKey).toBe(BYOK_KEY); }); @@ -134,6 +134,16 @@ describe("setJevModeAction", () => { expect(readFileSync(jevConfigFile(), "utf8")).toBe(before); }); + it("takes `shadow`, the old name, as observe — and shows an older file's `shadow` as observe", async () => { + connect(); + seed({ ...CLOUD_FILE, mode: "shadow" }); + expect((await getJevSettingsAction()).mode).toBe("observe"); + await setJevModeAction("enforce"); + const res = await setJevModeAction("shadow"); + expect(res.ok && res.view).toMatchObject({ on: true, mode: "observe" }); + expect(onDisk().mode).toBe("observe"); + }); + it("refuses anything that is not a mode, and a missing file", async () => { connect(); expect((await setJevModeAction("disabled")).ok).toBe(false); diff --git a/__tests__/actions/update-jev-config.test.ts b/__tests__/actions/update-jev-config.test.ts index d04b32355..c04fbcb2c 100644 --- a/__tests__/actions/update-jev-config.test.ts +++ b/__tests__/actions/update-jev-config.test.ts @@ -150,9 +150,9 @@ describe("saving writes the file the hooks read", () => { it("carries the stored token across a mode change, without it being re-typed", async () => { await saveJevConfigAction(input()); - const res = await saveJevConfigAction(input({ mode: "shadow", token: "" })); + const res = await saveJevConfigAction(input({ mode: "observe", token: "" })); expect(res.ok).toBe(true); - expect(loadJevConfig()?.mode).toBe("shadow"); + expect(loadJevConfig()?.mode).toBe("observe"); expect(loadJevConfig()?.apiKey).toBe(TOKEN); }); @@ -367,23 +367,29 @@ describe("validation is the loader's, not a second copy of it", () => { expect(loadJevConfig()).toBeNull(); }); - it("refuses plain http in enforce mode, and accepts loopback http in shadow", async () => { + it("refuses plain http in enforce mode, and accepts loopback http in observe", async () => { const enforced = await saveJevConfigAction( input({ provider: "custom", baseUrl: "http://localhost:9999", mode: "enforce" }), ); expect(enforced.ok).toBe(false); expect(loadJevConfig()).toBeNull(); - const shadowed = await saveJevConfigAction( - input({ provider: "custom", baseUrl: "http://localhost:9999", mode: "shadow" }), + const observed = await saveJevConfigAction( + input({ provider: "custom", baseUrl: "http://localhost:9999", mode: "observe" }), ); - expect(shadowed.ok).toBe(true); - expect(loadJevConfig()?.mode).toBe("shadow"); + expect(observed.ok).toBe(true); + expect(loadJevConfig()?.mode).toBe("observe"); + }); + + it("takes `shadow`, the old name for observe, and writes observe", async () => { + const res = await saveJevConfigAction(input({ mode: "shadow" })); + expect(res.ok && res.view.mode).toBe("observe"); + expect(JSON.parse(readFileSync(configPath(), "utf8")).mode).toBe("observe"); }); it("refuses plain http to anywhere but loopback, in either mode", async () => { const res = await saveJevConfigAction( - input({ provider: "custom", baseUrl: "http://jev.example", mode: "shadow" }), + input({ provider: "custom", baseUrl: "http://jev.example", mode: "observe" }), ); expect(res.ok).toBe(false); expect(loadJevConfig()).toBeNull(); @@ -443,10 +449,10 @@ describe("validation is the loader's, not a second copy of it", () => { it("still re-saves an older mismatched file untouched, to switch its mode", async () => { seedConfig({ provider: "openrouter", apiKey: TOKEN, baseUrl: "https://ai-gateway.vercel.sh/v1" }); const res = await saveJevConfigAction( - input({ provider: "openrouter", baseUrl: "https://ai-gateway.vercel.sh/v1", mode: "shadow", token: "" }), + input({ provider: "openrouter", baseUrl: "https://ai-gateway.vercel.sh/v1", mode: "observe", token: "" }), ); expect(res.ok).toBe(true); - expect(onDisk().mode).toBe("shadow"); + expect(onDisk().mode).toBe("observe"); }); // The client never appends a second /systemone, so "adds /systemone to the @@ -472,21 +478,21 @@ describe("validation is the loader's, not a second copy of it", () => { // CLI likewise checks only a --base-url it was given. seedConfig({ provider: "custom", apiKey: TOKEN, baseUrl: "https://proxy.example/v1/systemone" }); const res = await saveJevConfigAction( - input({ provider: "custom", baseUrl: "https://proxy.example/v1/systemone", mode: "shadow", token: "" }), + input({ provider: "custom", baseUrl: "https://proxy.example/v1/systemone", mode: "observe", token: "" }), ); expect(res.ok).toBe(true); expect(onDisk().baseUrl).toBe("https://proxy.example/v1/systemone"); - expect(onDisk().mode).toBe("shadow"); + expect(onDisk().mode).toBe("observe"); }); // `setJevModeAction` and `jev setup --mode` refuse these; the save skipped // them, keeping the old mode — or, on a fresh machine, writing a file with no // mode, which loads as enforce. it.each(["yolo", "ENFORCE", "", null])("refuses mode %j and writes nothing", async (mode) => { - await saveJevConfigAction(input({ mode: "shadow" })); + await saveJevConfigAction(input({ mode: "observe" })); const before = readFileSync(configPath(), "utf8"); const res = await saveJevConfigAction(input({ mode: mode as string, token: "" })); - expect(res).toEqual({ ok: false, problem: 'mode must be "off", "shadow" or "enforce".' }); + expect(res).toEqual({ ok: false, problem: 'mode must be "off", "observe" or "enforce".' }); expect(readFileSync(configPath(), "utf8")).toBe(before); }); @@ -601,13 +607,13 @@ describe("a save keeps the fields the form does not show", () => { accountId: CLOUDFLARE_ACCOUNT, model: "typesafe/jev-1.13", timeoutMs: 4500, - mode: "shadow", + mode: "observe", }); // Exactly what the panel sends for that file with nothing touched: the form // holds the four values the view gave it, and a blank token. const res = await saveJevConfigAction( - input({ provider: "cloudflare", accountId: CLOUDFLARE_ACCOUNT, mode: "shadow", token: "" }), + input({ provider: "cloudflare", accountId: CLOUDFLARE_ACCOUNT, mode: "observe", token: "" }), ); expect(res.ok).toBe(true); @@ -615,7 +621,7 @@ describe("a save keeps the fields the form does not show", () => { expect(loaded?.model).toBe("typesafe/jev-1.13"); expect(loaded?.accountId).toBe(CLOUDFLARE_ACCOUNT); expect(loaded?.timeoutMs).toBe(4500); - expect(loaded?.mode).toBe("shadow"); + expect(loaded?.mode).toBe("observe"); expect(loaded?.apiKey).toBe(TOKEN); }); @@ -625,7 +631,7 @@ describe("a save keeps the fields the form does not show", () => { apiKey: TOKEN, model: "typesafe/jev-1.13", timeoutMs: 4500, - mode: "shadow", + mode: "observe", }); const res = await saveJevConfigAction(input({ mode: "enforce", token: "" })); @@ -640,7 +646,7 @@ describe("a save keeps the fields the form does not show", () => { it("keeps a field a newer failproofai wrote, which this form has never heard of", async () => { seedConfig({ provider: "typesafe", apiKey: TOKEN, futureField: { weights: [1, 2] } }); - expect((await saveJevConfigAction(input({ mode: "shadow", token: "" }))).ok).toBe(true); + expect((await saveJevConfigAction(input({ mode: "observe", token: "" }))).ok).toBe(true); expect(onDisk().futureField).toEqual({ weights: [1, 2] }); }); @@ -658,14 +664,14 @@ describe("a save keeps the fields the form does not show", () => { }; seedConfig(seed); - expect((await saveJevConfigAction(input({ mode: "shadow", token: "" }))).ok).toBe(true); + expect((await saveJevConfigAction(input({ mode: "observe", token: "" }))).ok).toBe(true); const viaPanel = onDisk(); seedConfig(seed); const { runJevCommand } = await import("../../src/hooks/jev-cli"); - // `jev setup --mode shadow`: the same change, named the same way, with no + // `jev setup --mode observe`: the same change, named the same way, with no // terminal to prompt on — the key is kept from the existing config. - const cli = await runJevCommand(["setup", "--mode", "shadow"], { + const cli = await runJevCommand(["setup", "--mode", "observe"], { stdinIsTTY: false, // No provider is reached from a unit test; see jev-cli-contracts.test.ts. readModelList: async () => ({ ok: false, reason: "no list read in tests" }), @@ -697,18 +703,18 @@ describe("a config whose key lives in the environment", () => { }); it("takes a provider change too, and stays keyless", async () => { - seedConfig({ provider: "typesafe", mode: "shadow" }); + seedConfig({ provider: "typesafe", mode: "observe" }); - const res = await saveJevConfigAction(input({ provider: "openrouter", mode: "shadow", token: "" })); + const res = await saveJevConfigAction(input({ provider: "openrouter", mode: "observe", token: "" })); expect(res.ok).toBe(true); - expect(onDisk()).toEqual({ provider: "openrouter", mode: "shadow" }); + expect(onDisk()).toEqual({ provider: "openrouter", mode: "observe" }); }); it("is still on after such a save, with the key read from the environment", async () => { process.env.FAILPROOFAI_JEV_API_KEY = TOKEN; seedConfig({ provider: "typesafe" }); - const res = await saveJevConfigAction(input({ mode: "shadow", token: "" })); + const res = await saveJevConfigAction(input({ mode: "observe", token: "" })); expect(res.ok).toBe(true); if (!res.ok) return; expect(res.view.on).toBe(true); @@ -733,7 +739,7 @@ describe("a config whose key lives in the environment", () => { seedConfig({ provider: "typesafe" }); chmodSync(configPath(), 0o644); - const res = await saveJevConfigAction(input({ mode: "shadow", token: "" })); + const res = await saveJevConfigAction(input({ mode: "observe", token: "" })); expect(res.ok).toBe(true); expect(statSync(configPath()).mode & 0o777).toBe(0o600); expect(onDisk().apiKey).toBeUndefined(); @@ -807,7 +813,7 @@ describe("the stored model, which the panel shows but does not offer", () => { it("survives a save, which is the whole point of not offering the field", async () => { seedConfig({ provider: "typesafe", apiKey: TOKEN, model: "typesafe/jev-1.13" }); - const res = await saveJevConfigAction(input({ mode: "shadow", token: "" })); + const res = await saveJevConfigAction(input({ mode: "observe", token: "" })); expect(res.ok).toBe(true); if (!res.ok) return; expect(res.view.model).toEqual({ kind: "id", id: "typesafe/jev-1.13" }); diff --git a/__tests__/components/jev-notices-no-request.test.tsx b/__tests__/components/jev-notices-no-request.test.tsx index 4639decbc..8fdff33fd 100644 --- a/__tests__/components/jev-notices-no-request.test.tsx +++ b/__tests__/components/jev-notices-no-request.test.tsx @@ -10,7 +10,7 @@ const NO_REQUEST = { decision: "allow", evaluator: "jev" as const, jevDecision: describe("a call Jev sent no request for", () => { it("gets no pill", () => { expect(jevPillKind(NO_REQUEST)).toBeNull(); - expect(jevPillKind({ ...NO_REQUEST, jevMode: "shadow" })).toBeNull(); + expect(jevPillKind({ ...NO_REQUEST, jevMode: "observe" })).toBeNull(); const { container } = render(); expect(container).toBeEmptyDOMElement(); }); diff --git a/__tests__/components/jev-notices-not-consulted.test.tsx b/__tests__/components/jev-notices-not-consulted.test.tsx index 679261404..c6f9e6f0a 100644 --- a/__tests__/components/jev-notices-not-consulted.test.tsx +++ b/__tests__/components/jev-notices-not-consulted.test.tsx @@ -6,12 +6,12 @@ import { JEV_NOT_CONSULTED_FACT } from "@/src/hooks/jev-activity"; // Exactly what the two-tier combine rules record for a hard deny: Jev was // aborted and its answer never read. const HARD_DENY = { decision: "deny", evaluator: "jev" as const, jevMode: "enforce" as const }; -const HARD_DENY_SHADOW = { decision: "deny", evaluator: "jev" as const, jevMode: "shadow" as const }; +const HARD_DENY_OBSERVE = { decision: "deny", evaluator: "jev" as const, jevMode: "observe" as const }; describe("a hard deny Jev was not consulted on", () => { it("gets no pill: it is an ordinary regex deny", () => { expect(jevPillKind(HARD_DENY)).toBeNull(); - expect(jevPillKind(HARD_DENY_SHADOW)).toBeNull(); + expect(jevPillKind(HARD_DENY_OBSERVE)).toBeNull(); const { container } = render(); expect(container).toBeEmptyDOMElement(); }); diff --git a/__tests__/components/jev-notices.test.tsx b/__tests__/components/jev-notices.test.tsx index 08bf1562d..b2544f67f 100644 --- a/__tests__/components/jev-notices.test.tsx +++ b/__tests__/components/jev-notices.test.tsx @@ -11,18 +11,24 @@ describe("jevPillKind", () => { expect(jevPillKind({ decision: "allow", evaluator: "jev", jevDecision: "allow", jevCleared: [], jevMode: "enforce" })).toBeNull(); }); - it("marks a clear, a fallback and a shadow disagreement", () => { + it("marks a clear, a fallback and an observe disagreement", () => { expect(jevPillKind({ decision: "allow", evaluator: "jev", jevCleared: ["block-env-files"], jevMode: "enforce" })).toBe( "cleared", ); expect(jevPillKind({ decision: "deny", evaluator: "jev-fallback", jevFallbackReason: "timeout" })).toBe("fallback"); expect( - jevPillKind({ decision: "deny", evaluator: "jev", jevCleared: ["block-env-files"], jevMode: "shadow" }), + jevPillKind({ decision: "deny", evaluator: "jev", jevCleared: ["block-env-files"], jevMode: "observe" }), ).toBe("would-clear"); - expect(jevPillKind({ decision: "allow", evaluator: "jev", jevDecision: "deny", jevMode: "shadow" })).toBe( - "shadow-stricter", + expect(jevPillKind({ decision: "allow", evaluator: "jev", jevDecision: "deny", jevMode: "observe" })).toBe( + "observe-stricter", ); - expect(jevPillKind({ decision: "deny", evaluator: "jev", jevDecision: "deny", jevMode: "shadow" })).toBeNull(); + expect(jevPillKind({ decision: "deny", evaluator: "jev", jevDecision: "deny", jevMode: "observe" })).toBeNull(); + }); + + it("reads a row an older build wrote with jevMode `shadow` as observe", () => { + const legacy = "shadow" as never; + expect(jevPillKind({ decision: "deny", evaluator: "jev", jevCleared: ["block-env-files"], jevMode: legacy })).toBe("would-clear"); + expect(jevPillKind({ decision: "allow", evaluator: "jev", jevDecision: "deny", jevMode: legacy })).toBe("observe-stricter"); }); }); diff --git a/__tests__/components/jev-settings-panel.test.tsx b/__tests__/components/jev-settings-panel.test.tsx index c8e259170..4ab33c40c 100644 --- a/__tests__/components/jev-settings-panel.test.tsx +++ b/__tests__/components/jev-settings-panel.test.tsx @@ -110,8 +110,8 @@ describe("what it says about the machine", () => { expect(screen.getByRole("button", { name: /turn jev on/i })).toBeInTheDocument(); }); - it("distinguishes enforce from shadow, because they are different guarantees", async () => { - renderPanel(configured({ mode: "shadow" })); + it("distinguishes enforce from observe, because they are different guarantees", async () => { + renderPanel(configured({ mode: "observe" })); expect(screen.getByText(/the regex result is what gets enforced/i)).toBeInTheDocument(); cleanup(); renderPanel(configured({ mode: "enforce" })); @@ -267,13 +267,13 @@ describe("the token field is write-only", () => { }); it("sends a blank token when nothing was typed, which the server reads as keep", async () => { - saveMock.mockResolvedValue({ ok: true, view: configured({ mode: "shadow" }) }); + saveMock.mockResolvedValue({ ok: true, view: configured({ mode: "observe" }) }); renderPanel(configured()); - fireEvent.change(screen.getByLabelText("mode"), { target: { value: "shadow" } }); + fireEvent.change(screen.getByLabelText("mode"), { target: { value: "observe" } }); fireEvent.click(screen.getByRole("button", { name: /save changes/i })); await waitFor(() => expect(saveMock).toHaveBeenCalledTimes(1)); expect(saveMock).toHaveBeenCalledWith( - expect.objectContaining({ provider: "typesafe", mode: "shadow", token: "" }), + expect.objectContaining({ provider: "typesafe", mode: "observe", token: "" }), ); }); @@ -293,9 +293,9 @@ describe("the token field is write-only", () => { describe("the form", () => { it("sends no model at all, so a save cannot clear the stored one", async () => { - saveMock.mockResolvedValue({ ok: true, view: configured({ mode: "shadow" }) }); + saveMock.mockResolvedValue({ ok: true, view: configured({ mode: "observe" }) }); renderPanel(configured({ model: { kind: "id", id: "typesafe/jev-1.13" } })); - fireEvent.change(screen.getByLabelText("mode"), { target: { value: "shadow" } }); + fireEvent.change(screen.getByLabelText("mode"), { target: { value: "observe" } }); fireEvent.click(screen.getByRole("button", { name: /save changes/i })); await waitFor(() => expect(saveMock).toHaveBeenCalledTimes(1)); // Not "sends an empty model": an empty string is what the server used to @@ -419,7 +419,7 @@ function cloudView(over: Partial = {}): JevSettingsView { baseUrl: "https://app.befailproof.ai/enforcement/v1/jev", endpoint: "https://app.befailproof.ai/enforcement/v1/jev/systemone", token: { source: "cloud" }, - mode: "shadow", + mode: "observe", timeoutMs: 3000, cloud: CONNECTED, ...over, @@ -455,21 +455,21 @@ describe("the FailproofAI Cloud route", () => { expect(screen.getByRole("button", { name: /turn jev on/i })).toBeInTheDocument(); }); - it("switches back on in shadow mode", async () => { + it("switches back on in observe mode", async () => { modeMock.mockResolvedValue({ ok: true, view: cloudView() }); renderPanel(cloudView({ status: "off", on: false, mode: "off" })); // Nothing to pick while it is off. expect(screen.getByLabelText(/^mode$/i)).toBeDisabled(); fireEvent.click(screen.getByRole("button", { name: /turn jev on/i })); - await waitFor(() => expect(modeMock).toHaveBeenCalledWith("shadow")); + await waitFor(() => expect(modeMock).toHaveBeenCalledWith("observe")); }); - it("switches shadow to enforce with the mode control, and says a refusal", async () => { - modeMock.mockResolvedValue({ ok: false, problem: "plain http is accepted only with mode shadow" }); + it("switches observe to enforce with the mode control, and says a refusal", async () => { + modeMock.mockResolvedValue({ ok: false, problem: "plain http is accepted only with mode observe" }); renderPanel(cloudView()); fireEvent.change(screen.getByLabelText(/^mode$/i), { target: { value: "enforce" } }); await waitFor(() => expect(modeMock).toHaveBeenCalledWith("enforce")); - await waitFor(() => expect(screen.getByText(/plain http is accepted only with mode shadow/)).toBeInTheDocument()); + await waitFor(() => expect(screen.getByText(/plain http is accepted only with mode observe/)).toBeInTheDocument()); }); it("not connected: says so, and what fixes it", async () => { diff --git a/__tests__/fixtures/jev-activity-rows.ts b/__tests__/fixtures/jev-activity-rows.ts index 222ee4624..9226a78b6 100644 --- a/__tests__/fixtures/jev-activity-rows.ts +++ b/__tests__/fixtures/jev-activity-rows.ts @@ -75,7 +75,7 @@ export const JEV_ACTIVITY_ROWS: HookActivityEntry[] = [ jevCleared: [], jevLatencyMs: 44, jevModel: "typesafe/jev", - jevMode: "shadow", + jevMode: "observe", }, { ...base, diff --git a/__tests__/fixtures/jev-no-request-rows.ts b/__tests__/fixtures/jev-no-request-rows.ts index 49f2c1223..2c479d91e 100644 --- a/__tests__/fixtures/jev-no-request-rows.ts +++ b/__tests__/fixtures/jev-no-request-rows.ts @@ -43,6 +43,6 @@ export const JEV_NO_REQUEST_ROWS: HookActivityEntry[] = [ durationMs: 2, evaluator: "jev", jevDecision: "allow", - jevMode: "shadow", + jevMode: "observe", }, ]; diff --git a/__tests__/fixtures/jev-not-consulted-rows.ts b/__tests__/fixtures/jev-not-consulted-rows.ts index 9ce146263..0d84cf200 100644 --- a/__tests__/fixtures/jev-not-consulted-rows.ts +++ b/__tests__/fixtures/jev-not-consulted-rows.ts @@ -38,6 +38,6 @@ export const JEV_NOT_CONSULTED_ROWS: HookActivityEntry[] = [ reason: "Recursive force deletes are blocked", durationMs: 3, evaluator: "jev", - jevMode: "shadow", + jevMode: "observe", }, ]; diff --git a/__tests__/fixtures/jev-policy-page-rows.ts b/__tests__/fixtures/jev-policy-page-rows.ts index 88709d557..0ffb012cb 100644 --- a/__tests__/fixtures/jev-policy-page-rows.ts +++ b/__tests__/fixtures/jev-policy-page-rows.ts @@ -5,7 +5,7 @@ * A. enforce mode, Jev's own verdict decided the call: `policyName` is * `semantic/` and `policySource` is `jev` (it used to be omitted, * so the chart filed every Jev block under "unattributed"); - * B. shadow mode, Jev's own verdict was deny / instruct while the regex + * B. observe mode, Jev's own verdict was deny / instruct while the regex * result (allow) was enforced: the verdict is a "would have" in * `observed`, `{policyId, version, decision, reason}`, the list * observe-mode cloud and pack policies already use. @@ -55,7 +55,7 @@ export const JEV_POLICY_PAGE_ROWS: HookActivityEntry[] = [ jevMode: "enforce", policySource: "jev", }, - // B — shadow: Jev would have denied; the regex result (allow) was enforced. + // B — observe: Jev would have denied; the regex result (allow) was enforced. { ...base, timestamp: 1785740915100, @@ -68,10 +68,10 @@ export const JEV_POLICY_PAGE_ROWS: HookActivityEntry[] = [ jevDecision: "deny", jevLatencyMs: 761, jevModel: "jev-1.13.0", - jevMode: "shadow", + jevMode: "observe", observed: [{ policyId: "semantic/destructive-deletion", version: "jev-1.13.0", decision: "deny", reason: DELETION_REASON }], }, - // B — shadow: Jev would have warned. + // B — observe: Jev would have warned. { ...base, timestamp: 1785740915200, @@ -84,7 +84,7 @@ export const JEV_POLICY_PAGE_ROWS: HookActivityEntry[] = [ jevDecision: "instruct", jevLatencyMs: 688, jevModel: "jev-1.13.0", - jevMode: "shadow", + jevMode: "observe", observed: [{ policyId: "semantic/system-modification", version: "jev-1.13.0", decision: "instruct", reason: SYSTEM_REASON }], }, ]; diff --git a/__tests__/hooks/cloud-connect-jev.test.ts b/__tests__/hooks/cloud-connect-jev.test.ts index 448750214..0a37be0dd 100644 --- a/__tests__/hooks/cloud-connect-jev.test.ts +++ b/__tests__/hooks/cloud-connect-jev.test.ts @@ -5,7 +5,7 @@ * - A key whose introspect lists `jev:evaluate` stores itself in the `jev` * slot of credentials.json, under the origin it was verified against, and — * only when there is NO jev.json — writes one that turns Jev on through - * FailproofAI Cloud in shadow mode. + * FailproofAI Cloud in observe mode. * - An existing jev.json is never overwritten, whatever it names. * - A key introspect says lacks `jev:evaluate` writes no Jev state at all, * and drops the Jev key a previous connection left. An introspect that @@ -82,14 +82,14 @@ function seedJev(obj: unknown, mode = 0o600): string { } describe("connecting with a key that carries jev:evaluate", () => { - it("stores the key under the verified origin and turns Jev on in shadow mode", async () => { + it("stores the key under the verified origin and turns Jev on in observe mode", async () => { const outcome = await connect(withPermissions(...MACHINE_PRESET)); expect(outcome.jev?.ok).toBe(true); expect(outcome.jev?.config?.status).toBe("written"); expect(readCredentials().jev).toEqual({ url: URL_, key: TOKEN }); const onDisk = JSON.parse(readFileSync(jevConfigFile(), "utf8")); - expect(onDisk).toEqual({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "shadow" }); + expect(onDisk).toEqual({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "observe" }); // No key in jev.json: the Cloud key has one home. expect(readFileSync(jevConfigFile(), "utf8")).not.toContain(TOKEN); if (posix) { @@ -100,12 +100,12 @@ describe("connecting with a key that carries jev:evaluate", () => { // What the hooks will now read. const cfg = loadJevConfig(); - expect(cfg).toMatchObject({ provider: "failproofai", apiKey: TOKEN, mode: "shadow" }); + expect(cfg).toMatchObject({ provider: "failproofai", apiKey: TOKEN, mode: "observe" }); const r = inspectJevConfig(); expect(r.status === "ok" && r.keySource).toBe("cloud"); const text = describeOutcome(outcome, "machine-1", URL_).join("\n"); - expect(text).toMatch(/Jev\s+on through FailproofAI Cloud, in shadow mode/); + expect(text).toMatch(/Jev\s+on through FailproofAI Cloud, in observe mode/); expect(text).toContain("--mode enforce"); expect(text).not.toContain(TOKEN); expect(configuredPaths(outcome)).toContain(credentialsFile()); @@ -115,12 +115,12 @@ describe("connecting with a key that carries jev:evaluate", () => { const outcome = await connect(withPermissions(...MACHINE_PRESET), "http://localhost:8080/fp"); expect(readCredentials().jev?.url).toBe("http://localhost:8080"); expect(JSON.parse(readFileSync(jevConfigFile(), "utf8")).baseUrl).toBe("http://localhost:8080/fp/enforcement/v1/jev"); - // Plain http to loopback is fine in the shadow mode connect writes. + // Plain http to loopback is fine in the observe mode connect writes. expect(loadJevConfig()?.baseUrl).toBe("http://localhost:8080/fp/enforcement/v1/jev"); // …and only there: `jev setup --mode enforce` (and the dashboard switch) // refuses plain http, so the output must not name it as the next step. const text = describeOutcome(outcome, "machine-1", "http://localhost:8080/fp").join("\n"); - expect(text).toMatch(/Jev\s+on through FailproofAI Cloud, in shadow mode/); + expect(text).toMatch(/Jev\s+on through FailproofAI Cloud, in observe mode/); expect(text).not.toContain("--mode enforce"); expect(text).toContain("Enforce needs an https FailproofAI Cloud URL"); // The command it would have named really is refused, and the refusal names the step that works. @@ -169,11 +169,11 @@ describe("connecting with a key that carries jev:evaluate", () => { const outcome = await connect(withPermissions(...MACHINE_PRESET)); const text = describeOutcome(outcome, "machine-1", URL_).join("\n"); expect(text).toContain("switched off"); - expect(text).toContain("jev setup --mode shadow"); + expect(text).toContain("jev setup --mode observe"); }); it("names the other origin when the Cloud jev.json on disk points somewhere else", async () => { - seedJev({ provider: "failproofai", baseUrl: "https://staging.befailproof.ai/enforcement/v1/jev", mode: "shadow" }); + seedJev({ provider: "failproofai", baseUrl: "https://staging.befailproof.ai/enforcement/v1/jev", mode: "observe" }); const outcome = await connect(withPermissions(...MACHINE_PRESET)); expect(outcome.jev?.config).toMatchObject({ status: "kept", otherOrigin: "https://staging.befailproof.ai" }); const text = describeOutcome(outcome, "machine-1", URL_).join("\n"); @@ -193,11 +193,11 @@ describe("connecting with a key that carries jev:evaluate", () => { expect(text).toContain("https://staging.befailproof.ai"); expect(text).toContain("switched off"); const cmds = [...text.matchAll(/`failproofai (jev setup[^`]*)`/g)].map((m) => m[1]); - expect(cmds).toEqual(["jev setup --provider failproofai --mode shadow"]); + expect(cmds).toEqual(["jev setup --provider failproofai --mode observe"]); const r = await runJevCommand(cmds[0].split(" ").slice(1), { render: { cols: 120, color: false } }); expect(r.exitCode).toBe(0); - expect(inspectJevConfig()).toMatchObject({ status: "ok", config: { mode: "shadow", baseUrl: `${URL_}/enforcement/v1/jev` } }); + expect(inspectJevConfig()).toMatchObject({ status: "ok", config: { mode: "observe", baseUrl: `${URL_}/enforcement/v1/jev` } }); }); it("the no-clobber write loses to a file that appears first", () => { @@ -242,7 +242,7 @@ describe("connecting with --no-transcripts (sessions !== true)", () => { expect(jevLines[0]).toContain("`failproofai jev setup --provider failproofai`"); const text = lines.join("\n"); expect(text).not.toMatch(/Jev\s+on\b/); - expect(text).not.toContain("shadow mode"); + expect(text).not.toContain("observe mode"); expect(text).not.toContain(TOKEN); // The key file is named in the closing note: a key WAS stored. expect(configuredPaths(outcome)).toContain(credentialsFile()); @@ -259,8 +259,8 @@ describe("connecting with --no-transcripts (sessions !== true)", () => { expect(text).not.toContain("available on this key"); }); - it("a Cloud jev.json already on (shadow or enforce) is left alone — and the output says Jev still sends, and how to stop it", async () => { - for (const mode of ["shadow", "enforce"] as const) { + it("a Cloud jev.json already on (observe or enforce) is left alone — and the output says Jev still sends, and how to stop it", async () => { + for (const mode of ["observe", "enforce"] as const) { const before = seedJev({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode }); const outcome = await decisionsOnly(); // Never overwritten (decision 16, invariant 7)… @@ -279,7 +279,7 @@ describe("connecting with --no-transcripts (sessions !== true)", () => { }); it("…the same line on the --connect path, above \"Decisions only.\"", async () => { - seedJev({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "shadow" }); + seedJev({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "observe" }); const r = await runConnectCommand({ url: URL_, token: TOKEN, @@ -291,7 +291,7 @@ describe("connecting with --no-transcripts (sessions !== true)", () => { daemonStatus: () => "running", }); const text = r.lines.join("\n"); - expect(text).toContain("Jev is still on through FailproofAI Cloud (shadow mode)"); + expect(text).toContain("Jev is still on through FailproofAI Cloud (observe mode)"); expect(text).toContain("Decisions only."); expect(text.indexOf("still on through")).toBeLessThan(text.indexOf("Decisions only.")); }); @@ -299,7 +299,7 @@ describe("connecting with --no-transcripts (sessions !== true)", () => { it("no such line when the Cloud jev.json does not send: switched off, or pointing at another Cloud", async () => { for (const file of [ { provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "off" }, - { provider: "failproofai", baseUrl: "https://staging.befailproof.ai/enforcement/v1/jev", mode: "shadow" }, + { provider: "failproofai", baseUrl: "https://staging.befailproof.ai/enforcement/v1/jev", mode: "observe" }, ]) { seedJev(file); const outcome = await decisionsOnly(); @@ -333,7 +333,7 @@ describe("connecting with --no-transcripts (sessions !== true)", () => { const { runJevCommand } = await import("../../src/hooks/jev-cli"); const r = await runJevCommand(["setup", "--provider", "failproofai"], { render: { cols: 120, color: false } }); expect(r.exitCode).toBe(0); - expect(loadJevConfig()).toMatchObject({ provider: "failproofai", apiKey: TOKEN, mode: "shadow" }); + expect(loadJevConfig()).toMatchObject({ provider: "failproofai", apiKey: TOKEN, mode: "observe" }); }); }); @@ -463,7 +463,7 @@ describe("disconnecting", () => { }); it("keeps a BYOK jev.json exactly as it was, and says whose it is", async () => { - const before = seedJev({ provider: "openrouter", apiKey: BYOK_KEY, mode: "shadow" }); + const before = seedJev({ provider: "openrouter", apiKey: BYOK_KEY, mode: "observe" }); await connect(withPermissions(...MACHINE_PRESET)); const r = runDisconnectCommand(); expect(readFileSync(jevConfigFile(), "utf8")).toBe(before); @@ -476,7 +476,7 @@ describe("disconnecting", () => { }); // `--mode off` is "the switch that lasts" (jev-cloud.mdx): deleting it here - // made the next connect write a fresh shadow file, and Jev came back on. + // made the next connect write a fresh observe file, and Jev came back on. it("keeps a Cloud jev.json switched off, so reconnecting leaves Jev off", async () => { await connect(withPermissions(...MACHINE_PRESET)); const before = seedJev({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "off" }); @@ -496,7 +496,7 @@ describe("disconnecting", () => { it("removes a Cloud jev.json even with no key left to clear", () => { mkdirSync(home, { recursive: true }); - seedJev({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "shadow" }); + seedJev({ provider: "failproofai", baseUrl: `${URL_}/enforcement/v1/jev`, mode: "observe" }); const r = runDisconnectCommand(); expect(existsSync(jevConfigFile())).toBe(false); expect(r.lines[0]).toBe("Disconnected from FailproofAI Cloud."); diff --git a/__tests__/hooks/hook-telemetry-jev.test.ts b/__tests__/hooks/hook-telemetry-jev.test.ts index fe56f7efe..ecbb7902a 100644 --- a/__tests__/hooks/hook-telemetry-jev.test.ts +++ b/__tests__/hooks/hook-telemetry-jev.test.ts @@ -39,11 +39,11 @@ describe("jevTelemetryProperties", () => { evaluator: "jev-fallback", jevFallbackReason: "error: connect ECONNREFUSED", jevLatencyMs: 3, - jevMode: "shadow", + jevMode: "observe", }), ).toEqual({ jev_evaluator: "jev-fallback", - jev_mode: "shadow", + jev_mode: "observe", jev_fallback_reason: "error", jev_latency_ms: 3, }); diff --git a/__tests__/hooks/jev-activity.test.ts b/__tests__/hooks/jev-activity.test.ts index 0cc4f0d51..6968c0862 100644 --- a/__tests__/hooks/jev-activity.test.ts +++ b/__tests__/hooks/jev-activity.test.ts @@ -91,6 +91,14 @@ describe("sanitizeJevActivity", () => { expect(sanitizeJevActivity(e)).toEqual(e); }); + it("reads jevMode `shadow`, what builds before the rename wrote, as observe", () => { + const out = sanitizeJevActivity(entry({ evaluator: "jev", jevMode: "shadow" as never, jevCleared: ["block-env-files"] })); + expect(out.jevMode).toBe("observe"); + expect(describeJevActivity(entry({ evaluator: "jev", jevMode: "shadow" as never, jevDecision: "allow", jevLatencyMs: 38 }))).toContain( + "observe mode: the regex result was enforced", + ); + }); + it("drops values outside each field's closed set, field by field", () => { const out = sanitizeJevActivity( entry({ @@ -185,14 +193,14 @@ describe("describeJevActivity", () => { ).toEqual(["Jev verdict: allow", "cleared block-read-outside-cwd", "38 ms", "jev-1.13.0"]); }); - it("says shadow mode enforced the regex result", () => { + it("says observe mode enforced the regex result", () => { const facts = describeJevActivity( - entry({ evaluator: "jev", jevMode: "shadow", jevDecision: "allow", jevCleared: ["block-env-files"] }), + entry({ evaluator: "jev", jevMode: "observe", jevDecision: "allow", jevCleared: ["block-env-files"] }), ); expect(facts).toEqual([ "Jev verdict: allow", "would have cleared block-env-files", - "shadow mode: the regex result was enforced", + "observe mode: the regex result was enforced", ]); }); diff --git a/__tests__/hooks/jev-cli-bin.test.ts b/__tests__/hooks/jev-cli-bin.test.ts index 07ab8caf0..305028975 100644 --- a/__tests__/hooks/jev-cli-bin.test.ts +++ b/__tests__/hooks/jev-cli-bin.test.ts @@ -56,24 +56,24 @@ describe("failproofai jev (real binary)", () => { }); it("setup → status → remove, with the key piped on stdin and never printed", () => { - const setup = cli(["jev", "setup", "--provider", "typesafe", "--mode", "shadow", "--key-stdin"], `${KEY}\n`); + const setup = cli(["jev", "setup", "--provider", "typesafe", "--mode", "observe", "--key-stdin"], `${KEY}\n`); expect(setup.exitCode).toBe(0); expect(setup.stdout + setup.stderr).not.toContain(KEY); expect(setup.stdout).toContain("jev.json"); expect(existsSync(CONFIG)).toBe(true); if (process.platform !== "win32") expect(statSync(CONFIG).mode & 0o777).toBe(0o600); - expect(JSON.parse(readFileSync(CONFIG, "utf8"))).toEqual({ provider: "typesafe", apiKey: KEY, mode: "shadow" }); + expect(JSON.parse(readFileSync(CONFIG, "utf8"))).toEqual({ provider: "typesafe", apiKey: KEY, mode: "observe" }); const status = cli(["jev", "status"]); expect(status.exitCode).toBe(0); expect(status.stdout + status.stderr).not.toContain(KEY); expect(status.stdout).toContain("typesafe"); - expect(status.stdout).toContain("shadow"); + expect(status.stdout).toContain("observe"); const json = cli(["jev", "status", "--json"]); expect(json.exitCode).toBe(0); expect(json.stdout).not.toContain(KEY); - expect(JSON.parse(json.stdout)).toMatchObject({ status: "ok", provider: "typesafe", mode: "shadow", keySource: "file" }); + expect(JSON.parse(json.stdout)).toMatchObject({ status: "ok", provider: "typesafe", mode: "observe", keySource: "file" }); const removed = cli(["jev", "remove"]); expect(removed.exitCode).toBe(0); diff --git a/__tests__/hooks/jev-cli-cloud.test.ts b/__tests__/hooks/jev-cli-cloud.test.ts index 08f99dc6a..9926a3f8b 100644 --- a/__tests__/hooks/jev-cli-cloud.test.ts +++ b/__tests__/hooks/jev-cli-cloud.test.ts @@ -73,11 +73,11 @@ describe("jev CLI: FailproofAI Cloud", () => { describe("status", () => { it("on: FailproofAI Cloud, host only, key from the connection", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); const human = await runJevCommand(["status"], RENDER); expect(human.exitCode).toBe(0); const t = text(human); - expect(t).toContain("on · shadow"); + expect(t).toContain("on · observe"); expect(t).toContain("FailproofAI Cloud"); expect(t).toContain("app.befailproof.ai"); expect(t).not.toContain("/enforcement/v1/jev"); @@ -91,7 +91,7 @@ describe("jev CLI: FailproofAI Cloud", () => { providerLabel: "FailproofAI Cloud", endpoint: "app.befailproof.ai", model: "jev-1.13.0", - mode: "shadow", + mode: "observe", keySource: "cloud", keySourceLabel: "FailproofAI Cloud connection", cloudConnected: true, @@ -120,7 +120,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("connected with a key that has no Jev: says so, never \"not connected\"", async () => { writeCredentials({ ingest: { url: `${ORIGIN}/v1/events`, key: KEY } }); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); const human = await runJevCommand(["status"], RENDER); expect(human.exitCode).toBe(0); expect(text(human)).toContain("off — no Jev key is stored for this machine's FailproofAI Cloud connection"); @@ -131,7 +131,7 @@ describe("jev CLI: FailproofAI Cloud", () => { status: "key-lacks-jev", provider: "failproofai", endpoint: "app.befailproof.ai", - mode: "shadow", + mode: "observe", keySource: "cloud", cloudConnected: true, keyCarriesJev: false, @@ -162,7 +162,7 @@ describe("jev CLI: FailproofAI Cloud", () => { ["not-connected", () => undefined, "failproofai jev remove"], ])("%s: the reconnect and the off switch are separate steps", async (_state, arrange, offCmd) => { arrange(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); const lines = (await runJevCommand(["status"], RENDER)).lines; const i = lines.findIndex((l) => l.trim() === "failproofai config --token "); expect(i).toBeGreaterThan(0); @@ -194,7 +194,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("a loose credentials.json: refused, with the chmod that fixes it", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); chmodSync(join(fpHome, "credentials.json"), 0o644); const human = await runJevCommand(["status"], RENDER); expect(human.exitCode).toBe(1); @@ -206,7 +206,7 @@ describe("jev CLI: FailproofAI Cloud", () => { // budget), not tampering: the note must name the bits that are set. it.each([["0644", 0o644], ["0640", 0o640], ["0604", 0o604]])("credentials.json at %s: says others can read it, not change it", async (_octal, mode) => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); chmodSync(join(fpHome, "credentials.json"), mode); const t = text(await runJevCommand(["status"], RENDER)); expect(t).toMatch(/can read it/); @@ -216,9 +216,9 @@ describe("jev CLI: FailproofAI Cloud", () => { // A Cloud jev.json has no key (it is in credentials.json), and neither has a // --key-from-env one: refused as firmly, but not for disclosing a key. it.each([ - ["Cloud", { provider: "failproofai", baseUrl: BASE, mode: "shadow" }, false], - ["BYOK key-from-env", { provider: "typesafe", mode: "shadow" }, false], - ["BYOK with a stored key", { provider: "typesafe", apiKey: BYOK_KEY, mode: "shadow" }, true], + ["Cloud", { provider: "failproofai", baseUrl: BASE, mode: "observe" }, false], + ["BYOK key-from-env", { provider: "typesafe", mode: "observe" }, false], + ["BYOK with a stored key", { provider: "typesafe", apiKey: BYOK_KEY, mode: "observe" }, true], ])("%s jev.json at 0644: says it holds a key only when it does", async (_kind, file, holdsKey) => { connect(); writeJev(file); @@ -234,14 +234,14 @@ describe("jev CLI: FailproofAI Cloud", () => { it("credentials.json group-writable: says others could change it", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); chmodSync(join(fpHome, "credentials.json"), 0o620); expect(text(await runJevCommand(["status"], RENDER))).toContain("could change it"); }); it("credentials.json that is not JSON: no permissions claim, reconnect instead", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); writeFileSync(join(fpHome, "credentials.json"), "{not json", { mode: 0o600 }); const human = await runJevCommand(["status"], RENDER); expect(human.exitCode).toBe(1); @@ -293,7 +293,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("refused credentials.json --json: the Cloud facts, and which file's permissions are which", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); chmodSync(join(fpHome, "credentials.json"), 0o640); const machine = await runJevCommand(["status", "--json"], RENDER); expect(machine.exitCode).toBe(1); @@ -320,14 +320,14 @@ describe("jev CLI: FailproofAI Cloud", () => { }); describe("setup --provider failproofai", () => { - it("builds the file from the connection: shadow, no key, 0600", async () => { + it("builds the file from the connection: observe, no key, 0600", async () => { connect(); const r = await runJevCommand(["setup", "--provider", "failproofai"], RENDER); expect(r.exitCode, text(r)).toBe(0); - expect(onDisk()).toEqual({ provider: "failproofai", mode: "shadow", baseUrl: BASE }); + expect(onDisk()).toEqual({ provider: "failproofai", mode: "observe", baseUrl: BASE }); if (process.platform !== "win32") expect(statSync(jevConfigPath()).mode & 0o777).toBe(0o600); expect(loadJevConfig()).toMatchObject({ provider: "failproofai", apiKey: KEY }); - expect(text(r)).toContain("saved · FailproofAI Cloud · shadow"); + expect(text(r)).toContain("saved · FailproofAI Cloud · observe"); noKey(r); }); @@ -373,7 +373,7 @@ describe("jev CLI: FailproofAI Cloud", () => { }); it("a mode switch over a Cloud file rewrites the mode and keeps the rest — connected or not", async () => { - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow", timeoutMs: 2500 }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe", timeoutMs: 2500 }); const r = await runJevCommand(["setup", "--mode", "enforce"], RENDER); expect(r.exitCode, text(r)).toBe(0); expect(onDisk()).toEqual({ provider: "failproofai", baseUrl: BASE, mode: "enforce", timeoutMs: 2500 }); @@ -381,12 +381,25 @@ describe("jev CLI: FailproofAI Cloud", () => { expect(onDisk().mode).toBe("off"); }); + it("reads a Cloud file's old mode name `shadow` as observe, and writes observe", async () => { + connect(); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + expect(loadJevConfig()?.mode).toBe("observe"); + const status = await runJevCommand(["status"], RENDER); + expect(text(status)).toContain("on · observe"); + // `--mode shadow` is taken too, and saved under the new name. + const r = await runJevCommand(["setup", "--provider", "failproofai", "--mode", "shadow"], RENDER); + expect(r.exitCode, text(r)).toBe(0); + expect(onDisk().mode).toBe("observe"); + expect(text(r)).toContain('"shadow" is now called "observe"'); + }); + it("a mode switch says which key state the machine is in, and offers `jev test` only when there is a key to test", async () => { writeJev({ provider: "failproofai", baseUrl: BASE, mode: "off" }); // Connected, with a key that has no Jev: said so — never "not connected". writeCredentials({ ingest: { url: `${ORIGIN}/v1/events`, key: KEY } }); - const lacks = await runJevCommand(["setup", "--mode", "shadow"], RENDER); + const lacks = await runJevCommand(["setup", "--mode", "observe"], RENDER); expect(lacks.exitCode, text(lacks)).toBe(0); expect(text(lacks)).toContain("connected, no Jev key stored for it"); expect(text(lacks)).not.toContain("not connected"); @@ -404,7 +417,7 @@ describe("jev CLI: FailproofAI Cloud", () => { // Connected with a Jev key: the live check is the next step. connect(); - const on = await runJevCommand(["setup", "--mode", "shadow"], RENDER); + const on = await runJevCommand(["setup", "--mode", "observe"], RENDER); expect(on.exitCode, text(on)).toBe(0); expect(text(on)).toContain("failproofai jev test"); expect(text(on)).not.toContain("config --token "); @@ -412,7 +425,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("drops a key someone put in a Cloud file, which is what makes it valid again", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow", apiKey: BYOK_KEY }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe", apiKey: BYOK_KEY }); expect(loadJevConfig()).toBeNull(); const r = await runJevCommand(["setup", "--provider", "failproofai"], RENDER); expect(r.exitCode).toBe(0); @@ -430,12 +443,12 @@ describe("jev CLI: FailproofAI Cloud", () => { expect(loadJevConfig()?.baseUrl).toBe(BASE); }); - it("switching from BYOK is explicit, starts in shadow, and carries no BYOK key over", async () => { + it("switching from BYOK is explicit, starts in observe, and carries no BYOK key over", async () => { connect(); writeJev({ provider: "typesafe", apiKey: BYOK_KEY, mode: "enforce" }); const r = await runJevCommand(["setup", "--provider", "failproofai"], RENDER); expect(r.exitCode).toBe(0); - expect(onDisk()).toEqual({ provider: "failproofai", mode: "shadow", baseUrl: BASE }); + expect(onDisk()).toEqual({ provider: "failproofai", mode: "observe", baseUrl: BASE }); expect(readFileSync(jevConfigPath(), "utf8")).not.toContain(BYOK_KEY); }); }); @@ -450,7 +463,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("jev models has nothing to read for it", async () => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); const r = await runJevCommand(["models"], RENDER); expect(r.exitCode).toBe(1); expect(text(r)).toContain("FailproofAI Cloud serves no model list"); @@ -469,12 +482,12 @@ describe("jev CLI: FailproofAI Cloud", () => { ["a group-writable directory", () => chmodSync(fpHome, 0o770), () => `chmod 700 ${fpHome}`], [ "a Cloud file on another origin", - () => writeJev({ provider: "failproofai", baseUrl: "https://staging.befailproof.ai/enforcement/v1/jev", mode: "shadow" }), + () => writeJev({ provider: "failproofai", baseUrl: "https://staging.befailproof.ai/enforcement/v1/jev", mode: "observe" }), () => "failproofai jev setup --provider failproofai", ], ])("refused (%s): the same fix as status", async (_label, breakIt, fix) => { connect(); - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); breakIt(); const t = text(await runJevCommand(["test"], RENDER)); chmodSync(fpHome, 0o700); @@ -483,7 +496,7 @@ describe("jev CLI: FailproofAI Cloud", () => { }); it("not connected / switched off: not run, with a code for each", async () => { - writeJev({ provider: "failproofai", baseUrl: BASE, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: BASE, mode: "observe" }); const nc = await runJevCommand(["test", "--json"], RENDER); expect(nc.exitCode).toBe(1); expect(json(nc)).toMatchObject({ ok: false, error: { code: "not-connected" } }); @@ -514,7 +527,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("says what to do in FailproofAI Cloud's terms", async () => { const origin = `http://127.0.0.1:${port}`; connect(origin); - writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "observe" }); status = 403; body = { error: "forbidden", message: "this key does not carry jev:evaluate" }; const refused = await runJevCommand(["test"], { ...RENDER, testTimeoutMs: 5_000 }); @@ -534,7 +547,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("a 429 names the daily limit when the body says so, and the per-minute one otherwise", async () => { const origin = `http://127.0.0.1:${port}`; connect(origin); - writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "observe" }); status = 429; body = { error: "daily_limit_reached" }; @@ -555,7 +568,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("a 503 names who fixes it, not a wait", async () => { const origin = `http://127.0.0.1:${port}`; connect(origin); - writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "observe" }); status = 503; body = { error: "jev_unavailable" }; const r = await runJevCommand(["test"], { ...RENDER, testTimeoutMs: 5_000 }); @@ -568,7 +581,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("a 422 request_rejected is that call's own, never an outage to wait out", async () => { const origin = `http://127.0.0.1:${port}`; connect(origin); - writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "observe" }); status = 422; body = { error: "request_rejected" }; const rejected = await runJevCommand(["test"], { ...RENDER, testTimeoutMs: 5_000 }); @@ -583,7 +596,7 @@ describe("jev CLI: FailproofAI Cloud", () => { it("a redirect is advice about the connection, never a --base-url this route refuses", async () => { const origin = `http://127.0.0.1:${port}`; connect(origin); - writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "shadow" }); + writeJev({ provider: "failproofai", baseUrl: `${origin}/enforcement/v1/jev`, mode: "observe" }); status = 302; body = {}; const redirected = await runJevCommand(["test"], { ...RENDER, testTimeoutMs: 5_000 }); diff --git a/__tests__/hooks/jev-cli-hardening.test.ts b/__tests__/hooks/jev-cli-hardening.test.ts index c29044523..e33d64d5b 100644 --- a/__tests__/hooks/jev-cli-hardening.test.ts +++ b/__tests__/hooks/jev-cli-hardening.test.ts @@ -207,24 +207,24 @@ describe("failproofai jev — review hardening", () => { }); }); - describe("setup refuses plain-http loopback outside shadow mode", () => { - it("enforce (the default) is refused with the reason; shadow is saved", async () => { + describe("setup refuses plain-http loopback outside observe mode", () => { + it("enforce (the default) is refused with the reason; observe is saved", async () => { const enforce = await runJevCommand(["setup", "--provider", "custom", "--base-url", "http://localhost:8787/v1", "--key-stdin"], withKey(KEY)); expect(enforce.exitCode).toBe(1); - expect(text(enforce)).toContain("shadow"); + expect(text(enforce)).toContain("observe"); expect(existsSync(jevConfigPath())).toBe(false); - const shadow = await runJevCommand( - ["setup", "--provider", "custom", "--base-url", "http://localhost:8787/v1", "--mode", "shadow", "--key-stdin"], + const observe = await runJevCommand( + ["setup", "--provider", "custom", "--base-url", "http://localhost:8787/v1", "--mode", "observe", "--key-stdin"], withKey(KEY), ); - expect(shadow.exitCode).toBe(0); - expect(readFile()).toMatchObject({ baseUrl: "http://localhost:8787/v1", mode: "shadow" }); + expect(observe.exitCode).toBe(0); + expect(readFile()).toMatchObject({ baseUrl: "http://localhost:8787/v1", mode: "observe" }); // And the mode cannot then be switched to enforce underneath it. const flip = await runJevCommand(["setup", "--mode", "enforce"], noTty); expect(flip.exitCode).toBe(1); - expect(readFile().mode).toBe("shadow"); + expect(readFile().mode).toBe("observe"); }); }); diff --git a/__tests__/hooks/jev-cli-review.test.ts b/__tests__/hooks/jev-cli-review.test.ts index a9bd2b02c..798d09e5d 100644 --- a/__tests__/hooks/jev-cli-review.test.ts +++ b/__tests__/hooks/jev-cli-review.test.ts @@ -72,7 +72,7 @@ describe("failproofai jev — review round", () => { expect(loadJevConfig()).toBeNull(); const before = readFileSync(jevConfigPath(), "utf8"); - const r = await runJevCommand(["setup", "--mode", "shadow"], noTty); + const r = await runJevCommand(["setup", "--mode", "observe"], noTty); expect(r.exitCode).toBe(1); expect(text(r)).toContain("open to other users (0664)"); expect(text(r)).toContain("https://attacker.example.com"); @@ -150,9 +150,9 @@ describe("failproofai jev — review round", () => { it("an owner-only file with a foreign endpoint keeps its key on a re-run, as before", async () => { await runJevCommand(["setup", "--provider", "typesafe", "--base-url", ATTACKER, "--key-stdin"], withKey(KEY)); - const r = await runJevCommand(["setup", "--mode", "shadow"], noTty); + const r = await runJevCommand(["setup", "--mode", "observe"], noTty); expect(r.exitCode).toBe(0); - expect(readFile()).toEqual({ provider: "typesafe", apiKey: KEY, baseUrl: ATTACKER, mode: "shadow" }); + expect(readFile()).toEqual({ provider: "typesafe", apiKey: KEY, baseUrl: ATTACKER, mode: "observe" }); }); it("status shows the endpoint a too-open file names, next to the chmod hint (human and --json)", async () => { diff --git a/__tests__/hooks/jev-cli-status-stats.test.ts b/__tests__/hooks/jev-cli-status-stats.test.ts index 46601bce2..2e4374e1f 100644 --- a/__tests__/hooks/jev-cli-status-stats.test.ts +++ b/__tests__/hooks/jev-cli-status-stats.test.ts @@ -128,23 +128,23 @@ describe("failproofai jev status — activity comes from jevStats()", () => { expectOneDefaultCall(); }); - it("shadow mode's would-be clears are printed, not reported as 'cleared nothing'", async () => { + it("observe mode's would-be clears are printed, not reported as 'cleared nothing'", async () => { // T8's jevStats() counts a clear that CHANGED an outcome in clearsByPolicy - // — which only enforce mode can do — and shadow mode's would-be clears in - // shadowClearsByPolicy. A renderer reading only the first tells a shadow - // user nothing was cleared, which is the one number shadow mode exists to + // — which only enforce mode can do — and observe mode's would-be clears in + // observeClearsByPolicy. A renderer reading only the first tells an observe + // user nothing was cleared, which is the one number observe mode exists to // show. The field is optional on the stats this branch builds against. jevStatsMock.mockResolvedValue({ ...STATS, clearsByPolicy: {}, - shadowClearsByPolicy: { "block-read-outside-cwd": 9, "protect-env-vars": 2 }, + observeClearsByPolicy: { "block-read-outside-cwd": 9, "protect-env-vars": 2 }, } as JevStats); - await runJevCommand(["setup", "--provider", "typesafe", "--mode", "shadow", "--key-stdin"], withKey(KEY)); + await runJevCommand(["setup", "--provider", "typesafe", "--mode", "observe", "--key-stdin"], withKey(KEY)); const human = await runJevCommand(["status"], RENDER); expect(human.exitCode).toBe(0); const out = text(human); - expect(out).toContain("would have cleared (shadow) block-read-outside-cwd ×9, protect-env-vars ×2"); + expect(out).toContain("would have cleared (observe) block-read-outside-cwd ×9, protect-env-vars ×2"); expect(out).toContain("cleared nothing"); }); diff --git a/__tests__/hooks/jev-cli-url-token.test.ts b/__tests__/hooks/jev-cli-url-token.test.ts index a16d87c42..8f1db5aab 100644 --- a/__tests__/hooks/jev-cli-url-token.test.ts +++ b/__tests__/hooks/jev-cli-url-token.test.ts @@ -268,15 +268,15 @@ describe("failproofai jev --url --token ", () => { expect(existsSync(jevConfigPath())).toBe(false); }); - it("refuses plain http to localhost in enforce mode, and takes it in shadow", async () => { + it("refuses plain http to localhost in enforce mode, and takes it in observe", async () => { const enforced = await runJevCommand(["--url", "http://127.0.0.1:8088/v1", "--token", TOKEN], RENDER); expect(enforced.exitCode).toBe(1); - expect(text(enforced)).toContain("accepted only with mode shadow"); + expect(text(enforced)).toContain("accepted only with mode observe"); expect(existsSync(jevConfigPath())).toBe(false); - const shadow = await runJevCommand(["--url", "http://127.0.0.1:8088/v1", "--mode", "shadow", "--token", TOKEN], RENDER); - expect(shadow.exitCode).toBe(0); - expect(readFile()).toMatchObject({ provider: "custom", baseUrl: "http://127.0.0.1:8088/v1", mode: "shadow" }); + const observe = await runJevCommand(["--url", "http://127.0.0.1:8088/v1", "--mode", "observe", "--token", TOKEN], RENDER); + expect(observe.exitCode).toBe(0); + expect(readFile()).toMatchObject({ provider: "custom", baseUrl: "http://127.0.0.1:8088/v1", mode: "observe" }); }); it("refuses a URL carrying credentials, in the loader's words", async () => { diff --git a/__tests__/hooks/jev-cli.test.ts b/__tests__/hooks/jev-cli.test.ts index 833664bb1..70bfd415f 100644 --- a/__tests__/hooks/jev-cli.test.ts +++ b/__tests__/hooks/jev-cli.test.ts @@ -72,20 +72,20 @@ describe("failproofai jev", () => { expect(text(off)).toContain("off — Jev is not asked at all"); expect(text(off)).not.toContain("jev test"); // Switched back on, the step is back. - const on = await runJevCommand(["setup", "--mode", "shadow"], noTty); + const on = await runJevCommand(["setup", "--mode", "observe"], noTty); expect(on.exitCode, text(on)).toBe(0); expect(text(on)).toContain("failproofai jev test"); }); it("writes every option it is given", async () => { const r = await runJevCommand( - ["setup", "--provider=cloudflare", "--account-id", ACCOUNT, "--model", "typesafe/jev", "--mode", "shadow", "--timeout-ms", "900", "--key-stdin"], + ["setup", "--provider=cloudflare", "--account-id", ACCOUNT, "--model", "typesafe/jev", "--mode", "observe", "--timeout-ms", "900", "--key-stdin"], withKey(KEY), ); expect(r.exitCode).toBe(0); - expect(readFile()).toEqual({ provider: "cloudflare", apiKey: KEY, accountId: ACCOUNT, model: "typesafe/jev", mode: "shadow", timeoutMs: 900 }); + expect(readFile()).toEqual({ provider: "cloudflare", apiKey: KEY, accountId: ACCOUNT, model: "typesafe/jev", mode: "observe", timeoutMs: 900 }); expect(text(r)).toContain(`accounts/${ACCOUNT}/ai/run`); - expect(text(r)).toContain("shadow"); + expect(text(r)).toContain("observe"); }); it("needs --provider the first time, and a real one", async () => { @@ -159,21 +159,41 @@ describe("failproofai jev", () => { it("re-running for the same provider keeps the key, so a mode switch is one flag", async () => { await runJevCommand(["setup", "--provider", "cloudflare", "--account-id", ACCOUNT, "--key-stdin"], withKey(KEY)); - const r = await runJevCommand(["setup", "--mode", "shadow"], noTty); + const r = await runJevCommand(["setup", "--mode", "observe"], noTty); expect(r.exitCode).toBe(0); expect(text(r)).toContain("kept from the existing config"); - expect(readFile()).toEqual({ provider: "cloudflare", apiKey: KEY, accountId: ACCOUNT, mode: "shadow" }); + expect(readFile()).toEqual({ provider: "cloudflare", apiKey: KEY, accountId: ACCOUNT, mode: "observe" }); + }); + + it("takes --mode shadow, the old name, and saves it as observe with a note", async () => { + await runJevCommand(["setup", "--provider", "cloudflare", "--account-id", ACCOUNT, "--key-stdin"], withKey(KEY)); + const r = await runJevCommand(["setup", "--mode", "shadow"], noTty); + expect(r.exitCode, text(r)).toBe(0); + expect(readFile()).toEqual({ provider: "cloudflare", apiKey: KEY, accountId: ACCOUNT, mode: "observe" }); + expect(text(r)).toContain('"shadow" is now called "observe"'); + // Said only for the old name. + const again = await runJevCommand(["setup", "--mode", "observe"], noTty); + expect(text(again)).not.toContain("now called"); + }); + + it("rewrites an older file's mode `shadow` as observe on the next save", async () => { + await runJevCommand(["setup", "--provider", "cloudflare", "--account-id", ACCOUNT, "--key-stdin"], withKey(KEY)); + writeFileSync(jevConfigPath(), JSON.stringify({ ...readFile(), mode: "shadow" }), { mode: 0o600 }); + expect(loadJevConfig()?.mode).toBe("observe"); + const r = await runJevCommand(["setup", "--timeout-ms", "900"], noTty); + expect(r.exitCode, text(r)).toBe(0); + expect(readFile()).toMatchObject({ mode: "observe", timeoutMs: 900 }); }); it("switching provider starts over: no key, model or URL carries across, mode does", async () => { - await runJevCommand(["setup", "--provider", "custom", "--base-url", "https://jev.example.com/v1", "--mode", "shadow", "--key-stdin"], withKey(KEY)); + await runJevCommand(["setup", "--provider", "custom", "--base-url", "https://jev.example.com/v1", "--mode", "observe", "--key-stdin"], withKey(KEY)); const noKey = await runJevCommand(["setup", "--provider", "vercel"], noTty); expect(noKey.exitCode).toBe(1); expect(readFile().provider).toBe("custom"); const r = await runJevCommand(["setup", "--provider", "vercel", "--key-stdin"], withKey(OTHER_KEY)); expect(r.exitCode).toBe(0); - expect(readFile()).toEqual({ provider: "vercel", apiKey: OTHER_KEY, mode: "shadow" }); + expect(readFile()).toEqual({ provider: "vercel", apiKey: OTHER_KEY, mode: "observe" }); }); it("`default` clears a model or base URL override", async () => { @@ -208,9 +228,9 @@ describe("failproofai jev", () => { expect(readFile()).toEqual({ provider: "typesafe" }); expect(loadJevConfig()?.apiKey).toBe(KEY); // A re-run for the same provider keeps it an environment-key config. - const again = await runJevCommand(["setup", "--mode", "shadow"], noTty); + const again = await runJevCommand(["setup", "--mode", "observe"], noTty); expect(again.exitCode).toBe(0); - expect(readFile()).toEqual({ provider: "typesafe", mode: "shadow" }); + expect(readFile()).toEqual({ provider: "typesafe", mode: "observe" }); // A variable that is set but malformed is refused, not stored around. process.env[JEV_API_KEY_ENV] = "two words"; expect((await runJevCommand(["setup", "--provider", "typesafe", "--key-from-env"], noTty)).exitCode).toBe(1); @@ -238,7 +258,7 @@ describe("failproofai jev", () => { }); it("shows provider, endpoint, model, mode, path and permissions — never the key", async () => { - await runJevCommand(["setup", "--provider", "openrouter", "--mode", "shadow", "--key-stdin"], withKey(KEY)); + await runJevCommand(["setup", "--provider", "openrouter", "--mode", "observe", "--key-stdin"], withKey(KEY)); const r = await runJevCommand(["status"], RENDER); expect(r.exitCode).toBe(0); const out = text(r); @@ -246,7 +266,7 @@ describe("failproofai jev", () => { expect(out).toContain("openrouter"); expect(out).toContain("https://openrouter.ai/api/v1/systemone"); expect(out).toContain("typesafe/jev-1.13 (provider default)"); - expect(out).toContain("shadow"); + expect(out).toContain("observe"); expect(out).toContain(jevConfigPath()); if (posix) expect(out).toContain("0600 (owner-only)"); expect(out).toContain("set in the config file"); @@ -323,6 +343,15 @@ describe("failproofai jev", () => { expect(lines).toContain("block-read-outside-cwd ×12, protect-env-vars ×3"); expect(jevStatsLines(null).join("\n")).toContain("could not be read"); }); + + it("prints observe-mode would-be clears, from the current key or the pre-rename one", () => { + const base = { windowMs: 24 * 3_600_000, total: 3, fallbackRate: 0, fallbackReasons: {}, latencyP50Ms: 40, latencyP95Ms: 90, clearsByPolicy: {} }; + const now = jevStatsLines({ ...base, observeClearsByPolicy: { "block-env-files": 2 } }, { cols: 100 }).join("\n"); + expect(now).toContain("would have cleared (observe)"); + const legacy = jevStatsLines({ ...base, shadowClearsByPolicy: { "block-env-files": 2 } } as never, { cols: 100 }).join("\n"); + expect(legacy).toContain("would have cleared (observe)"); + expect(legacy).toContain("block-env-files ×2"); + }); }); describe("test", () => { diff --git a/__tests__/hooks/jev-cloud-disconnect-race.test.ts b/__tests__/hooks/jev-cloud-disconnect-race.test.ts index ae9027c1f..eaebe1f66 100644 --- a/__tests__/hooks/jev-cloud-disconnect-race.test.ts +++ b/__tests__/hooks/jev-cloud-disconnect-race.test.ts @@ -75,7 +75,7 @@ const { removeCloudJevConfig } = await import("../../src/hooks/jev-cloud-connect const { jevConfigPath } = await import("../../src/hooks/semantic/jev-config"); const { runDisconnectCommand } = await import("../../src/hooks/cloud-enrollment-cli"); -const CLOUD = { provider: "failproofai", baseUrl: "https://app.befailproof.ai/enforcement/v1/jev", mode: "shadow" }; +const CLOUD = { provider: "failproofai", baseUrl: "https://app.befailproof.ai/enforcement/v1/jev", mode: "observe" }; // Built at runtime: this repo's own hooks refuse secret-shaped literals. const BYOK_KEY = ["ts", "byok", "0123456789abcdef"].join("-"); const BYOK = { provider: "typesafe", apiKey: BYOK_KEY, mode: "enforce" }; @@ -192,7 +192,7 @@ describe("disconnect never deletes a BYOK jev.json", () => { hook.after = step; hook.run = () => { const tmp = `${jevConfigPath()}.other.tmp`; - realFs.writeFileSync(tmp, JSON.stringify({ ...BYOK, mode: "shadow" }), { mode: 0o600 }); + realFs.writeFileSync(tmp, JSON.stringify({ ...BYOK, mode: "observe" }), { mode: 0o600 }); realFs.renameSync(tmp, jevConfigPath()); }; disconnect(); @@ -230,7 +230,7 @@ describe("putting a BYOK jev.json back where no hard link can be made", () => { it("…and the copy never lands over a file written meanwhile: both kept, the original set aside", () => { writeAtomically(BYOK); const original = readFileSync(jevConfigPath(), "utf8"); - const other = { ...BYOK, mode: "shadow" }; + const other = { ...BYOK, mode: "observe" }; inject.linkSync = { ...EPERM, before: () => writeAtomically(other) }; const r = removeCloudJevConfig(); expect(r.status).toBe("set-aside"); @@ -282,7 +282,7 @@ describe("putting a BYOK jev.json back where no hard link can be made", () => { it("…and a set-aside one too", () => { writeAtomically(BYOK); - inject.linkSync = { ...EPERM, before: () => writeAtomically({ ...BYOK, mode: "shadow" }) }; + inject.linkSync = { ...EPERM, before: () => writeAtomically({ ...BYOK, mode: "observe" }) }; const text = runDisconnectCommand().lines.join("\n"); expect(text).toContain("This machine is not connected to FailproofAI Cloud."); expect(text).toContain("another jev.json was written in its place"); diff --git a/__tests__/hooks/jev-env-key.test.ts b/__tests__/hooks/jev-env-key.test.ts index 0e083b329..c86e56fe1 100644 --- a/__tests__/hooks/jev-env-key.test.ts +++ b/__tests__/hooks/jev-env-key.test.ts @@ -101,14 +101,14 @@ describe("a config whose key comes from the environment, in a shell without it", }); it("status --json reports a configured machine, not an invalid one", async () => { - await setupFromEnv("--provider", "vercel", "--mode", "shadow"); + await setupFromEnv("--provider", "vercel", "--mode", "observe"); const r = await runJevCommand(["status", "--json"], RENDER); expect(r.exitCode).toBe(0); const j = JSON.parse(r.json as string) as Record; expect(j.status).toBe("key-missing"); expect(j.reason).toBe("no-env-key"); expect(j.provider).toBe("vercel"); - expect(j.mode).toBe("shadow"); + expect(j.mode).toBe("observe"); expect(j.keySource).toBe("env"); expect(j.keyEnvVar).toBe(JEV_API_KEY_ENV); expect(String(j.endpoint)).toContain("vercel"); diff --git a/__tests__/hooks/jev-field-shapes.test.ts b/__tests__/hooks/jev-field-shapes.test.ts index b69eb2454..4fa133e0f 100644 --- a/__tests__/hooks/jev-field-shapes.test.ts +++ b/__tests__/hooks/jev-field-shapes.test.ts @@ -135,9 +135,9 @@ describe("a clear of a reviewable policy whose name has spaces", () => { it("is counted by jev status, shown on the dashboard and sent to PostHog", () => { const name = REGISTERED_WITH_SPACES[0]; - const s = computeJevStats([cleared(name), cleared(name, { jevMode: "shadow" })], { now: 6_000, windowMs: 60_000 }); + const s = computeJevStats([cleared(name), cleared(name, { jevMode: "observe" })], { now: 6_000, windowMs: 60_000 }); expect(s.clearsByPolicy).toEqual({ [name]: 1 }); - expect(s.shadowClearsByPolicy).toEqual({ [name]: 1 }); + expect(s.observeClearsByPolicy).toEqual({ [name]: 1 }); expect(describeJevActivity(cleared(name))).toContain(`cleared ${name}`); const props = jevTelemetryProperties(cleared(name)); expect(props.jev_cleared).toEqual([name]); diff --git a/__tests__/hooks/jev-no-request.test.ts b/__tests__/hooks/jev-no-request.test.ts index 8e953f6fc..0ede79d98 100644 --- a/__tests__/hooks/jev-no-request.test.ts +++ b/__tests__/hooks/jev-no-request.test.ts @@ -56,7 +56,7 @@ function row(overrides: Partial = {}): HookActivityEntry { } /** Exactly what the two-tier path records for a TodoWrite call: nothing to ask, no request sent. */ -const noRequest = (mode: "shadow" | "enforce" = "enforce", ts = NOW - 1_000) => +const noRequest = (mode: "observe" | "enforce" = "enforce", ts = NOW - 1_000) => row({ timestamp: ts, toolName: "TodoWrite", evaluator: "jev", jevDecision: "allow", jevMode: mode }); const answered = (latency: number, extra: Partial = {}) => row({ evaluator: "jev", jevDecision: "allow", jevLatencyMs: latency, jevModel: "jev-1.13.0", jevMode: "enforce", ...extra }); @@ -66,7 +66,7 @@ const hardDeny = () => row({ decision: "deny", policyName: "block-sudo", evaluat describe("jevOutcome: a call Jev sent no request for", () => { it("classifies the recorded no-request row as no-request, not answered", () => { expect(jevOutcome(noRequest("enforce"))).toBe("no-request"); - expect(jevOutcome(noRequest("shadow"))).toBe("no-request"); + expect(jevOutcome(noRequest("observe"))).toBe("no-request"); expect(jevOutcome({ evaluator: "jev", jevDecision: "allow" })).toBe("no-request"); for (const r of JEV_NO_REQUEST_ROWS) expect(jevOutcome(r), r.toolName ?? "").toBe("no-request"); }); @@ -103,7 +103,7 @@ describe("jev status stats", () => { expect(s.notConsulted).toBe(0); expect(s.fallbackRate).toBeCloseTo(0.5); expect(s.decisions).toEqual({ allow: 1, instruct: 0, deny: 0 }); - expect(s.modes).toEqual({ shadow: 0, enforce: 2 }); + expect(s.modes).toEqual({ observe: 0, enforce: 2 }); }); it("prints the no-request calls on a line of their own", () => { @@ -124,7 +124,7 @@ describe("jev status stats", () => { }); it("reports no evaluations when Jev never had anything to ask", () => { - const s = computeJevStats([noRequest(), noRequest("shadow")], { now: NOW, windowMs: 3_600_000 }); + const s = computeJevStats([noRequest(), noRequest("observe")], { now: NOW, windowMs: 3_600_000 }); expect(s.total).toBe(0); expect(s.answered).toBe(0); expect(s.noRequest).toBe(2); @@ -139,7 +139,7 @@ describe("jev status stats", () => { describe("the dashboard summary", () => { it("says no request was sent, and claims no verdict", () => { - for (const mode of ["enforce", "shadow"] as const) { + for (const mode of ["enforce", "observe"] as const) { expect(describeJevActivity(noRequest(mode))).toEqual([JEV_NO_REQUEST_FACT]); } }); diff --git a/__tests__/hooks/jev-not-consulted.test.ts b/__tests__/hooks/jev-not-consulted.test.ts index 8217430c6..08b6bf763 100644 --- a/__tests__/hooks/jev-not-consulted.test.ts +++ b/__tests__/hooks/jev-not-consulted.test.ts @@ -53,7 +53,7 @@ function row(overrides: Partial = {}): HookActivityEntry { } /** Exactly what the combine rules record for a hard deny: Jev aborted, never read. */ -const notConsulted = (mode: "shadow" | "enforce" = "enforce", ts = NOW - 1_000) => +const notConsulted = (mode: "observe" | "enforce" = "enforce", ts = NOW - 1_000) => row({ timestamp: ts, decision: "deny", @@ -69,7 +69,7 @@ const timedOut = () => row({ evaluator: "jev-fallback", jevFallbackReason: "time describe("jevOutcome", () => { it("classifies the combine rules' not-consulted row as not consulted", () => { expect(jevOutcome(notConsulted("enforce"))).toBe("not-consulted"); - expect(jevOutcome(notConsulted("shadow"))).toBe("not-consulted"); + expect(jevOutcome(notConsulted("observe"))).toBe("not-consulted"); expect(jevOutcome({ evaluator: "jev" })).toBe("not-consulted"); for (const r of JEV_NOT_CONSULTED_ROWS) expect(jevOutcome(r)).toBe("not-consulted"); }); @@ -113,7 +113,7 @@ describe("jev status stats", () => { expect(s.fallbackRate).toBeCloseTo(0.5); expect(s.answered + s.fallbacks).toBe(s.total); expect(s.decisions.allow + s.decisions.instruct + s.decisions.deny).toBe(s.answered); - expect(s.modes).toEqual({ shadow: 0, enforce: 2 }); + expect(s.modes).toEqual({ observe: 0, enforce: 2 }); expect(s.latencyP50Ms).toBe(40); }); @@ -131,7 +131,7 @@ describe("jev status stats", () => { }); it("reports no evaluations when every Jev row was a hard deny", () => { - const s = computeJevStats([notConsulted(), notConsulted("shadow")], { now: NOW, windowMs: 3_600_000 }); + const s = computeJevStats([notConsulted(), notConsulted("observe")], { now: NOW, windowMs: 3_600_000 }); expect(s.total).toBe(0); expect(s.answered).toBe(0); expect(s.fallbackRate).toBe(0); @@ -144,7 +144,7 @@ describe("jev status stats", () => { describe("the dashboard summary", () => { it("says Jev was not consulted, rather than an empty summary", () => { - for (const mode of ["enforce", "shadow"] as const) { + for (const mode of ["enforce", "observe"] as const) { expect(describeJevActivity(notConsulted(mode))).toEqual([JEV_NOT_CONSULTED_FACT]); } }); diff --git a/__tests__/hooks/jev-policy-page-golden.test.ts b/__tests__/hooks/jev-policy-page-golden.test.ts index 3b29fe8a2..08a598ca3 100644 --- a/__tests__/hooks/jev-policy-page-golden.test.ts +++ b/__tests__/hooks/jev-policy-page-golden.test.ts @@ -1,7 +1,7 @@ // @vitest-environment node /** * The collector's golden rows for the policy page's Jev data (contract §5): - * a Jev-decided enforce row attributed `policySource: "jev"`, and shadow rows + * a Jev-decided enforce row attributed `policySource: "jev"`, and observe rows * whose `observed` list carries Jev's "would have". * * `crates/fpai-collect/tests/hooks_jev.rs` reads the golden file this test @@ -42,7 +42,7 @@ describe("the collector's policy-page golden rows", () => { for (const row of [wouldDeny, wouldWarn]) { expect(row.decision).toBe("allow"); expect(row.policySource).toBeUndefined(); - expect(row.jevMode).toBe("shadow"); + expect(row.jevMode).toBe("observe"); expect(row.observed).toHaveLength(1); const [o] = row.observed!; expect(o.policyId).toMatch(/^semantic\/[a-z0-9-]+$/); diff --git a/__tests__/hooks/jev-telemetry-privacy.test.ts b/__tests__/hooks/jev-telemetry-privacy.test.ts index c406aba81..cc4218234 100644 --- a/__tests__/hooks/jev-telemetry-privacy.test.ts +++ b/__tests__/hooks/jev-telemetry-privacy.test.ts @@ -56,7 +56,7 @@ const answering = }); /** Record an outcome the way the handler does — greedily (see the header). */ -function record(outcome: SemanticOutcome, mode: "shadow" | "enforce"): HookActivityEntry { +function record(outcome: SemanticOutcome, mode: "observe" | "enforce"): HookActivityEntry { const base: HookActivityEntry = { timestamp: Date.now(), eventType: "PreToolUse", @@ -161,7 +161,7 @@ describe("Jev telemetry privacy", () => { for (const [name, run] of cases) { it(`${name}: nothing identifying reaches the row, PostHog, the dashboard or the stats`, async () => { const outcome = await run(); - for (const mode of ["enforce", "shadow"] as const) { + for (const mode of ["enforce", "observe"] as const) { const entry = record(outcome, mode); persistHookActivity(entry); @@ -207,7 +207,7 @@ describe("Jev telemetry privacy", () => { `git push origin failproofai/zebra-archive && ${COMMAND}`, `mv pack/tangerine-ledger.csv cloud/marmalade review`, ]; - const poisonedRow = (mode: "shadow" | "enforce", overrides: Partial = {}): HookActivityEntry => ({ + const poisonedRow = (mode: "observe" | "enforce", overrides: Partial = {}): HookActivityEntry => ({ timestamp: Date.now(), eventType: "PreToolUse", integration: "claude", @@ -226,7 +226,7 @@ describe("Jev telemetry privacy", () => { jevMode: mode, ...overrides, }); - const rows: Array<[string, (mode: "shadow" | "enforce") => HookActivityEntry]> = [ + const rows: Array<[string, (mode: "observe" | "enforce") => HookActivityEntry]> = [ ["answered", (mode) => poisonedRow(mode)], [ "fell back", @@ -242,7 +242,7 @@ describe("Jev telemetry privacy", () => { for (const [name, make] of rows) { it(`${name}: none of it reaches the row, PostHog, the dashboard or the stats`, async () => { const entries: HookActivityEntry[] = []; - for (const mode of ["enforce", "shadow"] as const) { + for (const mode of ["enforce", "observe"] as const) { const entry = make(mode); entries.push(entry); persistHookActivity(entry); diff --git a/__tests__/hooks/semantic/combine-shadow-verdict.test.ts b/__tests__/hooks/semantic/combine-observe-verdict.test.ts similarity index 61% rename from __tests__/hooks/semantic/combine-shadow-verdict.test.ts rename to __tests__/hooks/semantic/combine-observe-verdict.test.ts index 63012747d..d33943d2a 100644 --- a/__tests__/hooks/semantic/combine-shadow-verdict.test.ts +++ b/__tests__/hooks/semantic/combine-observe-verdict.test.ts @@ -1,12 +1,12 @@ /** - * Shadow mode's "would have": Jev's own deny or instruct, recorded rather than + * Observe mode's "would have": Jev's own deny or instruct, recorded rather than * applied (contract §5B). * - * `combineTwoTier` returns it as `shadowVerdict`, and the handler files it in + * `combineTwoTier` returns it as `observeVerdict`, and the handler files it in * the row's `observed` list. What is pinned here is that it is the verdict * ENFORCE mode would have applied — same name, same decision, same reason — - * and that it appears in shadow mode only, for deny/instruct only, without - * changing anything shadow mode enforces. + * and that it appears in observe mode only, for deny/instruct only, without + * changing anything observe mode enforces. */ import { describe, expect, it } from "vitest"; import { combineTwoTier, regexOnly, type JevReview, type RegexVerdict } from "../../../src/hooks/semantic/combine"; @@ -31,64 +31,64 @@ function answered(over: Partial> = {}): }; } -describe("shadowVerdict", () => { - it.each(["deny", "instruct"] as const)("a Jev %s in shadow mode is recorded exactly as enforce mode would apply it", (decision) => { +describe("observeVerdict", () => { + it.each(["deny", "instruct"] as const)("a Jev %s in observe mode is recorded exactly as enforce mode would apply it", (decision) => { const review = answered({ decision }); - const shadow = combineTwoTier(allowAll, review, "shadow"); + const observe = combineTwoTier(allowAll, review, "observe"); const enforce = combineTwoTier(allowAll, review, "enforce"); - // Shadow enforces the regex result, unchanged. - expect(shadow.final).toEqual(regexOnly(allowAll)); - expect(shadow.decidedByJev).toBe(false); + // Observe enforces the regex result, unchanged. + expect(observe.final).toEqual(regexOnly(allowAll)); + expect(observe.decidedByJev).toBe(false); - // Enforce applied Jev's verdict; shadow records the same one. + // Enforce applied Jev's verdict; observe records the same one. expect(enforce.decidedByJev).toBe(true); expect(enforce.final.decision).toBe(decision); - expect(shadow.shadowVerdict).toEqual({ + expect(observe.observeVerdict).toEqual({ policyName: enforce.final.entries[0].policyName, decision, reason: enforce.final.entries[0].reason, version: "jev-1.13.0", }); - expect(enforce.shadowVerdict).toBeUndefined(); + expect(enforce.observeVerdict).toBeUndefined(); }); it("uses enforce mode's fixed template when Jev gave no reason", () => { - const shadow = combineTwoTier(allowAll, answered({ reason: null }), "shadow"); - expect(shadow.shadowVerdict?.reason).toBe("Flagged by semantic review (semantic/destructive-deletion)"); + const observe = combineTwoTier(allowAll, answered({ reason: null }), "observe"); + expect(observe.observeVerdict?.reason).toBe("Flagged by semantic review (semantic/destructive-deletion)"); }); it("records nothing when Jev allowed", () => { - expect(combineTwoTier(allowAll, answered({ decision: "allow" }), "shadow").shadowVerdict).toBeUndefined(); + expect(combineTwoTier(allowAll, answered({ decision: "allow" }), "observe").observeVerdict).toBeUndefined(); }); it("records nothing when Jev did not answer or was not consulted", () => { - expect(combineTwoTier(allowAll, { kind: "fallback", reason: "http-503", latencyMs: 40, model: null }, "shadow").shadowVerdict).toBeUndefined(); - expect(combineTwoTier(allowAll, { kind: "not-consulted" }, "shadow").shadowVerdict).toBeUndefined(); + expect(combineTwoTier(allowAll, { kind: "fallback", reason: "http-503", latencyMs: 40, model: null }, "observe").observeVerdict).toBeUndefined(); + expect(combineTwoTier(allowAll, { kind: "not-consulted" }, "observe").observeVerdict).toBeUndefined(); }); it("keeps Jev's own verdict on a cut or injected call, as enforce mode does (upward only)", () => { for (const over of [{ requestCut: true, truncated: true }, { injected: true }]) { const review = answered(over); expect(combineTwoTier(allowAll, review, "enforce").final.decision).toBe("deny"); - expect(combineTwoTier(allowAll, review, "shadow").shadowVerdict?.decision).toBe("deny"); + expect(combineTwoTier(allowAll, review, "observe").observeVerdict?.decision).toBe("deny"); } }); it("records it beside a regex deny too: it is Jev's verdict, not the row's", () => { const regexDeny: RegexVerdict[] = [{ policyName: "failproofai/block-sudo", decision: "deny", reason: "no sudo", authority: "hard", reviewedBy: [] }]; - const shadow = combineTwoTier(regexDeny, answered(), "shadow"); - expect(shadow.final.decision).toBe("deny"); - expect(shadow.final.entries[0].policyName).toBe("failproofai/block-sudo"); - expect(shadow.shadowVerdict?.policyName).toBe("semantic/destructive-deletion"); + const observe = combineTwoTier(regexDeny, answered(), "observe"); + expect(observe.final.decision).toBe("deny"); + expect(observe.final.entries[0].policyName).toBe("failproofai/block-sudo"); + expect(observe.observeVerdict?.policyName).toBe("semantic/destructive-deletion"); }); it("files the version as the model id, or `jev` when there is none this build would store", () => { - expect(combineTwoTier(allowAll, answered({ model: null }), "shadow").shadowVerdict?.version).toBe("jev"); - expect(combineTwoTier(allowAll, answered({ model: "typesafe/jev-1.13-20260917" }), "shadow").shadowVerdict?.version).toBe( + expect(combineTwoTier(allowAll, answered({ model: null }), "observe").observeVerdict?.version).toBe("jev"); + expect(combineTwoTier(allowAll, answered({ model: "typesafe/jev-1.13-20260917" }), "observe").observeVerdict?.version).toBe( "typesafe/jev-1.13-20260917", ); // A reported id carrying a space or a newline is not a model id. - expect(combineTwoTier(allowAll, answered({ model: "jev 1.13\nrm -rf" }), "shadow").shadowVerdict?.version).toBe("jev"); + expect(combineTwoTier(allowAll, answered({ model: "jev 1.13\nrm -rf" }), "observe").observeVerdict?.version).toBe("jev"); }); }); diff --git a/__tests__/hooks/semantic/combine.test.ts b/__tests__/hooks/semantic/combine.test.ts index eb8f009d2..20982dce5 100644 --- a/__tests__/hooks/semantic/combine.test.ts +++ b/__tests__/hooks/semantic/combine.test.ts @@ -1,12 +1,12 @@ /** - * The §4 combine table, exhaustively: every row × {shadow, enforce} × + * The §4 combine table, exhaustively: every row × {observe, enforce} × * {whole, request-cut}. Each row is driven from a SemanticOutcome — what * `evaluateSemantic` actually returns — through `toReview` (how the handler * reads it) and `combineTwoTier` (what it enforces), so the cut → fallback * step is covered by the same table rather than beside it. * * The expected result is written out for `enforce` + whole. The other columns - * follow from rules the table asserts on every row: shadow enforces the regex + * follow from rules the table asserts on every row: observe enforces the regex * result, and a call part of which was never shown to Jev — the tool input, a * computed fact, a redacted span — withdraws every clear, so every regex deny * counts (§4) and the call is recorded `jev-fallback` / `request-cut`. A cut @@ -380,7 +380,7 @@ const ROWS: Row[] = [ }, ]; -const MODES: JevMode[] = ["enforce", "shadow"]; +const MODES: JevMode[] = ["enforce", "observe"]; const CUTS: Cut[] = ["whole", "request-cut"]; /** allow < instruct < deny, for the "never more permissive" invariant. */ const SEVERITY: Record<"allow" | "instruct" | "deny", number> = { allow: 0, instruct: 1, deny: 2 }; @@ -390,7 +390,7 @@ function reviewFor(row: Row, cut: Cut): JevReview { return toReview(withCut(row.outcome, cut)); } -describe("combine table (§4) — every row × shadow/enforce × whole/request-cut", () => { +describe("combine table (§4) — every row × observe/enforce × whole/request-cut", () => { for (const row of ROWS) { for (const mode of MODES) { for (const cut of CUTS) { @@ -412,8 +412,8 @@ describe("combine table (§4) — every row × shadow/enforce × whole/request-c const wholeAnswer = !hardDecided && !degraded && !cutAnswer; // What is ENFORCED. - if (mode === "shadow" || hardDecided || degraded) { - // shadow, a degraded Jev and a hard deny all enforce exactly what + if (mode === "observe" || hardDecided || degraded) { + // observe, a degraded Jev and a hard deny all enforce exactly what // the regex engine says alone. expect(out.final).toEqual(legacy); expect(out.decidedByJev).toBe(false); @@ -464,7 +464,7 @@ describe("combine table (§4) — every row × shadow/enforce × whole/request-c expect(out.activity.evaluator).toBe("jev"); expect(out.activity.jevFallbackReason).toBeUndefined(); expect(out.activity.jevDecision).toBe(row.outcome!.status === "ok" ? row.outcome!.verdict.decision : undefined); - // Shadow records what enforce WOULD have cleared. + // Observe records what enforce WOULD have cleared. expect(out.cleared).toEqual(row.enforce.cleared); const recorded = row.enforce.recorded ?? row.enforce.cleared; expect(out.activity.jevCleared).toEqual(recorded.length > 0 ? recorded : undefined); @@ -721,9 +721,9 @@ describe("the clear rule, on hand-built reviews", () => { expect(out.decidedByJev).toBe(true); }); - it("shadow still enforces the regex result for a cut answer", () => { + it("observe still enforces the regex result for a cut answer", () => { const review = answered({ requestCut: true, truncated: true, decision: "deny", reason: "deletes the database", policyName: "semantic/destructive-deletion" }); - const out = combineTwoTier([], review, "shadow"); + const out = combineTwoTier([], review, "observe"); expect(out.final).toEqual(regexOnly([])); expect(out.decidedByJev).toBe(false); }); @@ -783,8 +783,8 @@ describe("the clear rule, on hand-built reviews", () => { expect(out.final.decision).toBe("instruct"); }); - it("shadow mode is unchanged, and still records the reason", () => { - const out = combineTwoTier([], answered({ requestCut: true, truncated: true }), "shadow"); + it("observe mode is unchanged, and still records the reason", () => { + const out = combineTwoTier([], answered({ requestCut: true, truncated: true }), "observe"); expect(out.final).toEqual(regexOnly([])); expect(out.activity.jevFallbackReason).toBe("request-cut"); }); diff --git a/__tests__/hooks/semantic/jev-client-hardening.test.ts b/__tests__/hooks/semantic/jev-client-hardening.test.ts index 17ee6d195..57e7d3443 100644 --- a/__tests__/hooks/semantic/jev-client-hardening.test.ts +++ b/__tests__/hooks/semantic/jev-client-hardening.test.ts @@ -4,7 +4,7 @@ // error message by any route (Cloudflare's 200 {success:false} envelope, a job // state, a reported model id, a network error), a configured Cloudflare model // reaches the wire, a custom endpoint must say which Jev answered, plain-http -// loopback is shadow-only, and the 64 KiB config cap holds. +// loopback is observe-only, and the 64 KiB config cap holds. import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; import { chmodSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; @@ -207,7 +207,7 @@ describe("config hardening", () => { chmodSync(jevConfigPath(), 0o600); }; - describe("plain http to loopback is accepted in shadow mode only", () => { + describe("plain http to loopback is accepted in observe mode only", () => { const problem = (obj: unknown) => { const r = validateJevConfig(obj); return r.ok ? null : r.problem; @@ -215,11 +215,11 @@ describe("config hardening", () => { it.each(["http://localhost:8787/v1", "http://127.0.0.1:8787", "http://[::1]:8787"])("%s", (baseUrl) => { // enforce, explicit or by default: refused, and the reason says what to do. - expect(problem({ provider: "custom", apiKey: KEY, baseUrl })).toMatch(/shadow/); + expect(problem({ provider: "custom", apiKey: KEY, baseUrl })).toMatch(/observe/); expect(problem({ provider: "custom", apiKey: KEY, baseUrl, mode: "enforce" })).toMatch(/https/); - expect(problem({ provider: "typesafe", apiKey: KEY, baseUrl })).toMatch(/shadow/); - // shadow: accepted; a forged answer there changes no decision. - expect(problem({ provider: "custom", apiKey: KEY, baseUrl, mode: "shadow" })).toBeNull(); + expect(problem({ provider: "typesafe", apiKey: KEY, baseUrl })).toMatch(/observe/); + // observe: accepted; a forged answer there changes no decision. + expect(problem({ provider: "custom", apiKey: KEY, baseUrl, mode: "observe" })).toBeNull(); }); it("https to loopback is fine in enforce mode: the agent cannot present a trusted certificate", () => { @@ -231,8 +231,8 @@ describe("config hardening", () => { expect(loadJevConfig()).toBeNull(); const r = inspectJevConfig(); expect(r.status === "refused" && r.reason).toBe("invalid"); - write(JSON.stringify({ provider: "custom", apiKey: KEY, baseUrl: "http://localhost:8787/v1", mode: "shadow" })); - expect(loadJevConfig()).toMatchObject({ provider: "custom", mode: "shadow", baseUrl: "http://localhost:8787/v1" }); + write(JSON.stringify({ provider: "custom", apiKey: KEY, baseUrl: "http://localhost:8787/v1", mode: "observe" })); + expect(loadJevConfig()).toMatchObject({ provider: "custom", mode: "observe", baseUrl: "http://localhost:8787/v1" }); }); }); diff --git a/__tests__/hooks/semantic/jev-client-redirect.test.ts b/__tests__/hooks/semantic/jev-client-redirect.test.ts index d94c03280..b0bc60b76 100644 --- a/__tests__/hooks/semantic/jev-client-redirect.test.ts +++ b/__tests__/hooks/semantic/jev-client-redirect.test.ts @@ -64,7 +64,7 @@ describe("the Jev client never follows a redirect", () => { b.close(); }); - const custom = (): JevConfig => ({ provider: "custom", apiKey: KEY, baseUrl: `http://localhost:${aPort}/v1`, mode: "shadow" }); + const custom = (): JevConfig => ({ provider: "custom", apiKey: KEY, baseUrl: `http://localhost:${aPort}/v1`, mode: "observe" }); it.each([307, 308, 302, 301, 303])("HTTP %s from the configured endpoint is an error, and the target is never contacted", async (s) => { status = s; @@ -76,7 +76,7 @@ describe("the Jev client never follows a redirect", () => { it("the endpoint that does answer directly still works", async () => { hitsB.length = 0; - const cfg: JevConfig = { provider: "custom", apiKey: KEY, baseUrl: `http://127.0.0.1:${bPort}/v1`, mode: "shadow" }; + const cfg: JevConfig = { provider: "custom", apiKey: KEY, baseUrl: `http://127.0.0.1:${bPort}/v1`, mode: "observe" }; const res = await transportForConfig(cfg).transport(request, AbortSignal.timeout(5_000)); expect(res.answers.a.noul).toBe(0.99); expect(hitsB).toHaveLength(1); diff --git a/__tests__/hooks/semantic/jev-cloud-config.test.ts b/__tests__/hooks/semantic/jev-cloud-config.test.ts index 156f5b215..d3c3886cb 100644 --- a/__tests__/hooks/semantic/jev-cloud-config.test.ts +++ b/__tests__/hooks/semantic/jev-cloud-config.test.ts @@ -89,7 +89,7 @@ describe("jev-config: the FailproofAI Cloud provider", () => { if (!current.ingest) writeCredentials({ ...current, ingest: { url: `${url}/v1/events`, key } }); return writeJevCloudCredential({ url, key }); }; - const cloudFile = (over: Record = {}) => ({ provider: "failproofai", baseUrl: BASE, mode: "shadow", ...over }); + const cloudFile = (over: Record = {}) => ({ provider: "failproofai", baseUrl: BASE, mode: "observe", ...over }); describe("the credentials.json slot", () => { it("round-trips at 0600, beside every other credential, and clears alone", () => { @@ -250,7 +250,7 @@ describe("jev-config: the FailproofAI Cloud provider", () => { const r = inspectJevConfig(); expect(r.status).toBe("key-lacks-jev"); if (r.status !== "key-lacks-jev") return; - expect(r.routing).toEqual({ provider: "failproofai", baseUrl: BASE, mode: "shadow", timeoutMs: 3000 }); + expect(r.routing).toEqual({ provider: "failproofai", baseUrl: BASE, mode: "observe", timeoutMs: 3000 }); expect(r.problem).toContain("no Jev key is stored"); expect(r.problem).not.toMatch(/not connected/); expect(JSON.stringify(r)).not.toContain(KEY); @@ -285,7 +285,7 @@ describe("jev-config: the FailproofAI Cloud provider", () => { expect(r.status).toBe("ok"); if (r.status !== "ok") return; expect(r.keySource).toBe("cloud"); - expect(r.config).toMatchObject({ provider: "failproofai", apiKey: KEY, baseUrl: BASE, mode: "shadow", timeoutMs: 3000 }); + expect(r.config).toMatchObject({ provider: "failproofai", apiKey: KEY, baseUrl: BASE, mode: "observe", timeoutMs: 3000 }); expect(loadJevConfig()?.apiKey).toBe(KEY); // The route the transport will POST to. expect(jevRoute(r.config).endpoint).toBe(`${BASE}/systemone`); @@ -363,7 +363,7 @@ describe("jev-config: the FailproofAI Cloud provider", () => { it("needs a baseUrl", () => { connect(); - writeJev({ provider: "failproofai", mode: "shadow" }); + writeJev({ provider: "failproofai", mode: "observe" }); expect(inspectJevConfig()).toMatchObject({ status: "refused", reason: "invalid" }); }); @@ -391,7 +391,7 @@ describe("jev-config: the FailproofAI Cloud provider", () => { const r = inspectJevConfig(); expect(r.status).toBe("not-connected"); if (r.status !== "not-connected") return; - expect(r.routing).toEqual({ provider: "failproofai", baseUrl: BASE, mode: "shadow", timeoutMs: 3000 }); + expect(r.routing).toEqual({ provider: "failproofai", baseUrl: BASE, mode: "observe", timeoutMs: 3000 }); expect(r.problem).toMatch(/not connected to FailproofAI Cloud/); expect(r.problem).toMatch(/config --token/); expect(loadJevConfig()).toBeNull(); @@ -431,7 +431,7 @@ describe("jev-config: the FailproofAI Cloud provider", () => { it("keeps plain http to localhost out of enforce mode, as for every provider", () => { const local = "http://localhost:8080"; connect(local); - writeJev(cloudFile({ baseUrl: `${local}/enforcement/v1/jev`, mode: "shadow" })); + writeJev(cloudFile({ baseUrl: `${local}/enforcement/v1/jev`, mode: "observe" })); expect(loadJevConfig()?.baseUrl).toBe(`${local}/enforcement/v1/jev`); writeJev(cloudFile({ baseUrl: `${local}/enforcement/v1/jev`, mode: "enforce" })); expect(inspectJevConfig()).toMatchObject({ status: "refused", reason: "invalid" }); @@ -477,8 +477,8 @@ describe("jev-config: the FailproofAI Cloud provider", () => { expect(inspectJevConfig()).toMatchObject({ status: "refused", reason: "invalid" }); }); - it("validates as a mode, alongside shadow and enforce, and nothing else", () => { - for (const mode of ["off", "shadow", "enforce"]) { + it("validates as a mode, alongside observe and enforce, and nothing else", () => { + for (const mode of ["off", "observe", "enforce"]) { expect(validateJevConfig({ provider: "typesafe", apiKey: KEY, mode }).ok).toBe(true); } for (const mode of ["disabled", "OFF", "", 0, false, null]) { diff --git a/__tests__/hooks/semantic/jev-cloud-transport.test.ts b/__tests__/hooks/semantic/jev-cloud-transport.test.ts index 183949adf..2b081af5d 100644 --- a/__tests__/hooks/semantic/jev-cloud-transport.test.ts +++ b/__tests__/hooks/semantic/jev-cloud-transport.test.ts @@ -119,12 +119,12 @@ describe("the FailproofAI Cloud route, over a real socket", () => { rmSync(home, { recursive: true, force: true }); }); - /** The config the loader produces for a connected machine (plain http to loopback: shadow only). */ + /** The config the loader produces for a connected machine (plain http to loopback: observe only). */ const cloud = (): JevConfig => ({ provider: "failproofai", apiKey: KEY, baseUrl: `http://127.0.0.1:${port}/enforcement/v1/jev`, - mode: "shadow", + mode: "observe", timeoutMs: 3000, // The loader records the origin of the credential it validated against. credentialOrigin: `http://127.0.0.1:${port}`, @@ -341,7 +341,7 @@ describe("the FailproofAI Cloud route, over a real socket", () => { }); it("a BYOK route's 429 is exactly what it was: no cool-down", async () => { - const byok: JevConfig = { provider: "custom", apiKey: KEY, baseUrl: `http://127.0.0.1:${port}/v1`, mode: "shadow", timeoutMs: 3000 }; + const byok: JevConfig = { provider: "custom", apiKey: KEY, baseUrl: `http://127.0.0.1:${port}/v1`, mode: "observe", timeoutMs: 3000 }; const sendByok = () => transportForConfig(byok).transport(request, AbortSignal.timeout(5_000)); reply = () => rateLimited("30"); expect((await failure(sendByok())).code).toBe("http-429"); diff --git a/__tests__/hooks/semantic/jev-config.test.ts b/__tests__/hooks/semantic/jev-config.test.ts index 5ae123a48..03cd400dd 100644 --- a/__tests__/hooks/semantic/jev-config.test.ts +++ b/__tests__/hooks/semantic/jev-config.test.ts @@ -7,12 +7,14 @@ import { DEFAULT_JEV_MODE, JEV_API_KEY_ENV, JEV_CONFIG_DEFAULT_TIMEOUT_MS, + LEGACY_OBSERVE_MODE, baseUrlWithoutQuery, inspectJevConfig, isCalibratedJevModel, jevConfigPath, jevModelVersion, loadJevConfig, + normalizeJevMode, validateBaseUrl, validateJevConfig, } from "../../../src/hooks/semantic/jev-config"; @@ -88,7 +90,7 @@ describe("semantic/jev-config", () => { }); it.each([ - ["typesafe", { provider: "typesafe", apiKey: KEY, model: "jev-1.13.0", mode: "shadow", timeoutMs: 900 }], + ["typesafe", { provider: "typesafe", apiKey: KEY, model: "jev-1.13.0", mode: "observe", timeoutMs: 900 }], ["openrouter", { provider: "openrouter", apiKey: KEY, model: "typesafe/jev-1.13" }], ["vercel", { provider: "vercel", apiKey: KEY }], ["cloudflare", { provider: "cloudflare", apiKey: KEY, accountId: ACCOUNT }], @@ -211,8 +213,26 @@ describe("semantic/jev-config", () => { expect(problem({ provider: "typesafe", apiKey: KEY, mode: "disabled" })).toMatch(/mode/); // `off` is a mode now: it keeps the file and runs no Jev (see jev-cloud-config.test.ts). expect(problem({ provider: "typesafe", apiKey: KEY, mode: "off" })).toBeNull(); - const r = validateJevConfig({ provider: "typesafe", apiKey: KEY, mode: "shadow", timeoutMs: 800 }); - expect(r.ok && [r.value.mode, r.value.timeoutMs]).toEqual(["shadow", 800]); + const r = validateJevConfig({ provider: "typesafe", apiKey: KEY, mode: "observe", timeoutMs: 800 }); + expect(r.ok && [r.value.mode, r.value.timeoutMs]).toEqual(["observe", 800]); + }); + + it("reads the old mode name `shadow` as `observe`", () => { + expect(normalizeJevMode("shadow")).toBe("observe"); + expect(normalizeJevMode(LEGACY_OBSERVE_MODE)).toBe("observe"); + for (const m of ["off", "observe", "enforce"]) expect(normalizeJevMode(m)).toBe(m); + for (const m of ["Shadow", "observing", "", null, undefined, 1]) expect(normalizeJevMode(m)).toBeNull(); + const r = validateJevConfig({ provider: "typesafe", apiKey: KEY, mode: "shadow" }); + expect(r.ok && r.value.mode).toBe("observe"); + // Plain http to loopback is accepted in observe mode under either name. + expect(problem({ provider: "custom", apiKey: KEY, baseUrl: "http://localhost:8787/v1", mode: "shadow" })).toBeNull(); + }); + + it("loads a jev.json an older build wrote with mode `shadow` as observe", () => { + write({ provider: "typesafe", apiKey: KEY, mode: "shadow" }); + expect(loadJevConfig()?.mode).toBe("observe"); + const inspected = inspectJevConfig(); + expect(inspected.status === "ok" && inspected.config.mode).toBe("observe"); }); it("refuses a model naming a Jev family the thresholds were not calibrated for", () => { diff --git a/__tests__/hooks/semantic/jev-review.test.ts b/__tests__/hooks/semantic/jev-review.test.ts index 1879c7829..998143430 100644 --- a/__tests__/hooks/semantic/jev-review.test.ts +++ b/__tests__/hooks/semantic/jev-review.test.ts @@ -385,12 +385,12 @@ describe("the local verdict log", () => { it("records what the handler did with each outcome", async () => { await startJevReview(CFG, bash("ls")).review; - await startJevReview({ ...CFG, mode: "shadow" }, bash("ls")).review; + await startJevReview({ ...CFG, mode: "observe" }, bash("ls")).review; respond = async () => { throw new JevError("http-429", "slow down"); }; await startJevReview(CFG, bash("ls")).review; - expect(rows().map((r) => r.applied)).toEqual(["two-tier", "shadow", "legacy-fallback"]); + expect(rows().map((r) => r.applied)).toEqual(["two-tier", "observe", "legacy-fallback"]); expect(rows()[2]).toMatchObject({ status: "degraded", reason: "http-429" }); }); @@ -406,7 +406,7 @@ describe("the local verdict log", () => { describe("mode", () => { it("defaults to enforce (D2)", () => { expect(resolveMode(CFG)).toBe("enforce"); - expect(resolveMode({ ...CFG, mode: "shadow" })).toBe("shadow"); + expect(resolveMode({ ...CFG, mode: "observe" })).toBe("observe"); expect(resolveMode({ ...CFG, mode: "loud" as never })).toBe("enforce"); }); }); @@ -570,7 +570,7 @@ describe("how long a call may wait for Jev", () => { // ── Round-2 review findings ────────────────────────────────────────────────── describe("the throttle's cache is scoped to where answers come from", () => { - const LOOPBACK_SHADOW: JevConfig = { provider: "custom", apiKey: "not-a-real-key", baseUrl: "http://127.0.0.1:9", mode: "shadow" }; + const LOOPBACK_OBSERVE: JevConfig = { provider: "custom", apiKey: "not-a-real-key", baseUrl: "http://127.0.0.1:9", mode: "observe" }; const TYPESAFE_ENFORCE: JevConfig = { provider: "typesafe", apiKey: "not-a-real-key", baseUrl: "https://jev.invalid", mode: "enforce" }; it("passes a scope naming the provider, endpoint, account and model", async () => { @@ -580,8 +580,8 @@ describe("the throttle's cache is scoped to where answers come from", () => { { ...CFG, model: "typesafe/jev" }, { provider: "typesafe", apiKey: "not-a-real-key" }, TYPESAFE_ENFORCE, - LOOPBACK_SHADOW, - { ...LOOPBACK_SHADOW, baseUrl: "http://127.0.0.1:10" }, + LOOPBACK_OBSERVE, + { ...LOOPBACK_OBSERVE, baseUrl: "http://127.0.0.1:10" }, { provider: "openrouter", apiKey: "not-a-real-key" }, ]; for (const cfg of configs) await startJevReview(cfg, bash("ls")).review; @@ -591,14 +591,14 @@ describe("the throttle's cache is scoped to where answers come from", () => { expect(new Set(scopes).size).toBe(configs.length); // The same config always gets the same scope (the cache still works), // whatever its key or mode — neither changes who answers. - expect(throttleScope({ ...CFG, apiKey: "another" , mode: "shadow" }, { via: "cloudflare", model: "jev-1.13.0" })).toBe(scopes[0]); + expect(throttleScope({ ...CFG, apiKey: "another" , mode: "observe" }, { via: "cloudflare", model: "jev-1.13.0" })).toBe(scopes[0]); }); it("an answer cached under one provider is never served under another", async () => { fakeCache.on = true; intent = { userSaid: ["show me my notes"], agentLastMessage: null }; // Both routes ask for the same model, so the requests are byte-identical. - const first = await startJevReview(LOOPBACK_SHADOW, bash("cat ~/other/notes.txt")).review; + const first = await startJevReview(LOOPBACK_OBSERVE, bash("cat ~/other/notes.txt")).review; expect(first).toMatchObject({ kind: "answered" }); expect(transportCalls).toHaveLength(1); diff --git a/__tests__/hooks/semantic/jev-stats-hardening.test.ts b/__tests__/hooks/semantic/jev-stats-hardening.test.ts index 847503f92..1264231f3 100644 --- a/__tests__/hooks/semantic/jev-stats-hardening.test.ts +++ b/__tests__/hooks/semantic/jev-stats-hardening.test.ts @@ -43,14 +43,14 @@ describe("computeJevStats: names that collide with Object.prototype", () => { [ answered({ jevCleared: PROTO_NAMES }), answered({ jevCleared: PROTO_NAMES }), - answered({ jevMode: "shadow", jevCleared: PROTO_NAMES }), + answered({ jevMode: "observe", jevCleared: PROTO_NAMES }), answered({ jevModel: "constructor" }), ], { now: NOW }, ); for (const name of PROTO_NAMES) { expect(Object.getOwnPropertyDescriptor(s.clearsByPolicy, name)?.value, name).toBe(2); - expect(Object.getOwnPropertyDescriptor(s.shadowClearsByPolicy, name)?.value, name).toBe(1); + expect(Object.getOwnPropertyDescriptor(s.observeClearsByPolicy, name)?.value, name).toBe(1); } expect(Object.entries(s.clearsByPolicy).sort()).toEqual(PROTO_NAMES.map((n) => [n, 2]).sort()); expect(Object.getOwnPropertyDescriptor(s.models, "constructor")?.value).toBe(1); diff --git a/__tests__/hooks/semantic/jev-stats.test.ts b/__tests__/hooks/semantic/jev-stats.test.ts index 116ed6ee5..2ae99c107 100644 --- a/__tests__/hooks/semantic/jev-stats.test.ts +++ b/__tests__/hooks/semantic/jev-stats.test.ts @@ -91,19 +91,44 @@ describe("computeJevStats", () => { expect(s.latencyP95Ms).toBe(190); }); - it("counts clears per policy, separating shadow-mode would-be clears", () => { + it("counts clears per policy, separating observe-mode would-be clears", () => { const s = computeJevStats( [ answered(30, { jevCleared: ["block-read-outside-cwd"] }), answered(30, { jevCleared: ["block-read-outside-cwd", "protect-env-vars"] }), answered(30, { jevCleared: [] }), - answered(30, { jevMode: "shadow", jevCleared: ["block-env-files"] }), + answered(30, { jevMode: "observe", jevCleared: ["block-env-files"] }), ], { now: NOW }, ); expect(s.clearsByPolicy).toEqual({ "block-read-outside-cwd": 2, "protect-env-vars": 1 }); - expect(s.shadowClearsByPolicy).toEqual({ "block-env-files": 1 }); - expect(s.modes).toEqual({ shadow: 1, enforce: 3 }); + expect(s.observeClearsByPolicy).toEqual({ "block-env-files": 1 }); + expect(s.modes).toEqual({ observe: 1, enforce: 3 }); + }); + + it("counts rows an older build wrote with jevMode `shadow` as observe", () => { + const s = computeJevStats( + [ + answered(30, { jevMode: "shadow" as never, jevCleared: ["block-env-files"] }), + answered(30, { jevMode: "observe", jevCleared: ["block-env-files"] }), + ], + { now: NOW }, + ); + expect(s.observeClearsByPolicy).toEqual({ "block-env-files": 2 }); + expect(s.clearsByPolicy).toEqual({}); + expect(s.modes).toEqual({ observe: 2, enforce: 0 }); + }); + + it("formats a stats object an older build returned under the pre-rename keys", () => { + const legacy = { + ...computeJevStats([answered(30), answered(30)], { now: NOW }), + modes: { shadow: 1, enforce: 1 }, + shadowClearsByPolicy: { "block-env-files": 1 }, + observeClearsByPolicy: undefined, + } as unknown as Parameters[0]; + const out = formatJevStats(legacy); + expect(out).toContain("Would clear: block-env-files 1 (observe mode)"); + expect(out).toContain("Modes: enforce 1, observe 1"); }); it("tallies Jev's own verdicts and the models that answered", () => { @@ -211,7 +236,7 @@ describe("formatJevStats", () => { answered(30, { jevCleared: ["block-read-outside-cwd"] }), answered(50, { jevDecision: "deny" }), fellBack("timeout"), - answered(40, { jevMode: "shadow", jevCleared: ["block-env-files"] }), + answered(40, { jevMode: "observe", jevCleared: ["block-env-files"] }), ], { now: NOW, windowMs: 6 * HOUR }, ); @@ -222,8 +247,8 @@ describe("formatJevStats", () => { " Fell back: 1 (25.0%) — timeout 1", " Latency: p50 40 ms, p95 50 ms", " Cleared: block-read-outside-cwd 1", - " Would clear: block-env-files 1 (shadow mode)", - " Modes: enforce 3, shadow 1", + " Would clear: block-env-files 1 (observe mode)", + " Modes: enforce 3, observe 1", ].join("\n"), ); }); diff --git a/__tests__/hooks/semantic/truncation-severity.test.ts b/__tests__/hooks/semantic/truncation-severity.test.ts index 6abfef4d9..eeb8dd9a5 100644 --- a/__tests__/hooks/semantic/truncation-severity.test.ts +++ b/__tests__/hooks/semantic/truncation-severity.test.ts @@ -125,9 +125,9 @@ describe("padding a command cannot take Jev's own deny away", () => { }); }); - it("shadow mode is unaffected: the regex result is enforced, cut or not", async () => { + it("observe mode is unaffected: the regex result is enforced, cut or not", async () => { const { review } = await judged(bash(`${DANGEROUS} ${PADDING}`)); - const out = combineTwoTier([], review, "shadow"); + const out = combineTwoTier([], review, "observe"); expect(out.final).toEqual(regexOnly([])); expect(out.decidedByJev).toBe(false); }); diff --git a/__tests__/hooks/two-tier-handler.test.ts b/__tests__/hooks/two-tier-handler.test.ts index 8000c0141..901ca422e 100644 --- a/__tests__/hooks/two-tier-handler.test.ts +++ b/__tests__/hooks/two-tier-handler.test.ts @@ -35,7 +35,7 @@ import type { JevConfig } from "../../src/hooks/semantic/jev-config"; let jevConfig: JevConfig | null = null; /** Overrides the build's DEFAULT_JEV_MODE (D2) for one test; undefined → the real one. */ -let defaultModeOverride: "shadow" | "enforce" | undefined; +let defaultModeOverride: "observe" | "enforce" | undefined; vi.mock("../../src/hooks/semantic/jev-config", async (importOriginal) => { const actual = await importOriginal(); return { @@ -523,8 +523,8 @@ describe("a reviewable deny", () => { expect(typeof row.jevLatencyMs).toBe("number"); }); - it("in shadow mode: the regex deny is enforced and the would-be clear recorded", async () => { - jevConfig = { ...CFG, mode: "shadow" }; + it("in observe mode: the regex deny is enforced and the would-be clear recorded", async () => { + jevConfig = { ...CFG, mode: "observe" }; const enforced = await outsideRead(); expect(enforced.outcome.evaluation?.decision).toBe("deny"); expect(enforced.outcome.evaluation?.policyName).toBe("failproofai/block-read-outside-cwd"); @@ -532,9 +532,9 @@ describe("a reviewable deny", () => { evaluator: "jev", jevDecision: "allow", jevCleared: ["failproofai/block-read-outside-cwd"], - jevMode: "shadow", + jevMode: "observe", }); - // …and what shadow enforces is byte-identical to the unconfigured answer. + // …and what observe enforces is byte-identical to the unconfigured answer. jevConfig = null; const plain = await outsideRead(); expect(enforced.outcome.stdout).toBe(plain.outcome.stdout); @@ -1042,24 +1042,24 @@ describe("a padded call cannot make Jev's own deny go away", () => { expect(row.jevFallbackReason).toBeUndefined(); }); - it("shadow mode still enforces the regex result for both spellings", async () => { - jevConfig = { ...CFG, mode: "shadow" }; + it("observe mode still enforces the regex result for both spellings", async () => { + jevConfig = { ...CFG, mode: "observe" }; respond = answers({ "destructive-deletion": 0.97 }); const field = await bash(padField()); expect(field.outcome.evaluation?.decision).toBe("allow"); - expect(field.row).toMatchObject({ evaluator: "jev-fallback", jevDecision: "deny", jevMode: "shadow" }); + expect(field.row).toMatchObject({ evaluator: "jev-fallback", jevDecision: "deny", jevMode: "observe" }); const budget = await run("PreToolUse", { tool_name: "Bash", tool_input: padBudget() }); expect(budget.outcome.evaluation?.decision).toBe("allow"); - expect(budget.row).toMatchObject({ evaluator: "jev-fallback", jevDecision: "deny", jevMode: "shadow" }); + expect(budget.row).toMatchObject({ evaluator: "jev-fallback", jevDecision: "deny", jevMode: "observe" }); - // And a call the envelope had to cut: shadow enforces the regex result + // And a call the envelope had to cut: observe enforces the regex result // either way. respond = seeingTransport as typeof respond; const hidden = await bash(`echo ${overflow("x")} ; ${DELETE} ; echo ${overflow("y")}`); expect(hidden.outcome.evaluation?.decision).toBe("allow"); - expect(hidden.row).toMatchObject({ evaluator: "jev-fallback", jevFallbackReason: "request-cut", jevMode: "shadow" }); + expect(hidden.row).toMatchObject({ evaluator: "jev-fallback", jevFallbackReason: "request-cut", jevMode: "observe" }); }); it("control: unpadded, the very same deny is a plain `jev` row", async () => { @@ -1331,21 +1331,21 @@ describe("a Jev review that cannot start", () => { it("follows DEFAULT_JEV_MODE rather than restating it", async () => { jevConfig = CFG; - defaultModeOverride = "shadow"; + defaultModeOverride = "observe"; vi.mocked(startJevReview).mockImplementationOnce(() => { throw new Error("module failed to initialise"); }); const { row } = await bash("ls -la"); - expect(row).toMatchObject({ evaluator: "jev-fallback", jevFallbackReason: "error", jevMode: "shadow" }); + expect(row).toMatchObject({ evaluator: "jev-fallback", jevFallbackReason: "error", jevMode: "observe" }); }); it("an explicit mode in the config wins", async () => { - jevConfig = { ...CFG, mode: "shadow" }; + jevConfig = { ...CFG, mode: "observe" }; vi.mocked(startJevReview).mockImplementationOnce(() => { throw new Error("module failed to initialise"); }); const { row } = await bash("ls -la"); - expect(row).toMatchObject({ evaluator: "jev-fallback", jevMode: "shadow" }); + expect(row).toMatchObject({ evaluator: "jev-fallback", jevMode: "observe" }); }); }); @@ -1397,20 +1397,20 @@ describe("captureIntent and a prompt a policy acted on", () => { describe("the throttle's cache across a jev.json change", () => { const outsideRead = () => readFile(join(home, "other", "notes.txt")); - // T1 accepts plain http to a loopback proxy only in shadow mode: its answers + // T1 accepts plain http to a loopback proxy only in observe mode: its answers // must never clear a deny. Both routes ask for the same model, so their // requests are byte-identical. - const LOOPBACK_SHADOW: JevConfig = { provider: "custom", apiKey: "not-a-real-key", baseUrl: "http://127.0.0.1:9", mode: "shadow" }; + const LOOPBACK_OBSERVE: JevConfig = { provider: "custom", apiKey: "not-a-real-key", baseUrl: "http://127.0.0.1:9", mode: "observe" }; const TYPESAFE_ENFORCE: JevConfig = { provider: "typesafe", apiKey: "not-a-real-key", baseUrl: "https://jev.invalid", mode: "enforce" }; it("never serves one provider's answer under another: switching providers asks the new one", async () => { const { JevError } = await import("../../src/hooks/semantic/jev-client"); fakeCache.on = true; - jevConfig = LOOPBACK_SHADOW; - const shadow = await outsideRead(); - expect(shadow.outcome.evaluation?.decision).toBe("deny"); - expect(shadow.row).toMatchObject({ evaluator: "jev", jevMode: "shadow", jevCleared: ["failproofai/block-read-outside-cwd"] }); + jevConfig = LOOPBACK_OBSERVE; + const observe = await outsideRead(); + expect(observe.outcome.evaluation?.decision).toBe("deny"); + expect(observe.row).toMatchObject({ evaluator: "jev", jevMode: "observe", jevCleared: ["failproofai/block-read-outside-cwd"] }); expect(jevCalls).toHaveLength(1); jevConfig = TYPESAFE_ENFORCE; @@ -1831,7 +1831,7 @@ describe("FAILPROOFAI_EVALUATOR=legacy (§4 row 1) under every configured mode: it.each<[string, JevConfig["mode"]]>([ ["no mode (the build's default)", undefined], - ["shadow", "shadow"], + ["observe", "observe"], ["enforce", "enforce"], ])("%s", async (_label, mode) => { respond = answers({ "destructive-deletion": 0.97 }); @@ -1949,19 +1949,19 @@ describe("what the policy page reads", () => { expect(row).toMatchObject({ policyName: "failproofai/block-sudo", policySource: "builtin" }); }); - it("B: in shadow mode, Jev's deny is a 'would have' in observed — and the regex result is enforced", async () => { - jevConfig = { ...CFG, mode: "shadow" }; + it("B: in observe mode, Jev's deny is a 'would have' in observed — and the regex result is enforced", async () => { + jevConfig = { ...CFG, mode: "observe" }; respond = answers({ "destructive-deletion": 0.97 }); - const shadow = await bash(DELETE_ALL); - expect(shadow.outcome.evaluation?.decision).toBe("allow"); - expect(shadow.row.policySource).toBeUndefined(); - expect(shadow.row).toMatchObject({ evaluator: "jev", jevDecision: "deny", jevMode: "shadow" }); + const observe = await bash(DELETE_ALL); + expect(observe.outcome.evaluation?.decision).toBe("allow"); + expect(observe.row.policySource).toBeUndefined(); + expect(observe.row).toMatchObject({ evaluator: "jev", jevDecision: "deny", jevMode: "observe" }); // The reason is the one enforce mode shows for the same answer. jevConfig = CFG; store._resetForTest(join(root, "activity-enforce")); const enforce = await bash(DELETE_ALL); - expect(shadow.row.observed).toEqual([ + expect(observe.row.observed).toEqual([ { policyId: "semantic/destructive-deletion", version: "jev-1.13.0", @@ -1971,8 +1971,8 @@ describe("what the policy page reads", () => { ]); }); - it("B: a shadow instruct is recorded as an instruct", async () => { - jevConfig = { ...CFG, mode: "shadow" }; + it("B: an observe instruct is recorded as an instruct", async () => { + jevConfig = { ...CFG, mode: "observe" }; // An instruct-mode check firing: a warning, not a block. respond = answers({ "system-modification": 0.9 }); const { row } = await bash("sysctl -w vm.swappiness=10"); @@ -1985,7 +1985,7 @@ describe("what the policy page reads", () => { }); it("B: nothing is recorded when Jev allowed, or fell back", async () => { - jevConfig = { ...CFG, mode: "shadow" }; + jevConfig = { ...CFG, mode: "observe" }; respond = answers(); expect((await bash("ls -la")).row.observed).toBeUndefined(); respond = async () => { diff --git a/app/actions/get-jev-config.ts b/app/actions/get-jev-config.ts index 03b67a666..48b68dd93 100644 --- a/app/actions/get-jev-config.ts +++ b/app/actions/get-jev-config.ts @@ -55,6 +55,7 @@ import { baseUrlWithoutQuery, inspectJevConfig, looksLikeCredential, + normalizeJevMode, readJevConfigForUpdate, type JevConfig, type JevProviderKind, @@ -307,7 +308,6 @@ function routingFromRaw(raw: Record | null): { const provider = (JEV_PROVIDER_KINDS as readonly string[]).includes(providerRaw) ? (providerRaw as JevProviderKind) : null; - const modeRaw = raw?.mode; return { provider, baseUrl: asString(raw?.baseUrl), @@ -315,7 +315,8 @@ function routingFromRaw(raw: Record | null): { // refused file may be a pasted key, and is not shown. accountId: CLOUDFLARE_ACCOUNT_ID_RE.test(asString(raw?.accountId)) ? asString(raw?.accountId) : "", model: asString(raw?.model), - mode: modeRaw === "off" || modeRaw === "shadow" || modeRaw === "enforce" ? modeRaw : DEFAULT_JEV_MODE, + // `shadow`, the old name, is shown as `observe`. + mode: normalizeJevMode(raw?.mode) ?? DEFAULT_JEV_MODE, }; } diff --git a/app/actions/update-jev-config.ts b/app/actions/update-jev-config.ts index 8be1285a1..2a390b61e 100644 --- a/app/actions/update-jev-config.ts +++ b/app/actions/update-jev-config.ts @@ -10,7 +10,7 @@ * `writeJsonAtomically` the CLI's `jev setup` calls, with the same * `{ mode: 0o600, dirMode: 0o700 }`, after the same `validateJevConfig` the * loader itself runs. Nothing here re-states a rule that lives in - * `jev-config.ts`: not the URL scheme, not "plain http only in shadow mode", + * `jev-config.ts`: not the URL scheme, not "plain http only in observe mode", * not "cloudflare needs an account id". A second copy of those rules is how the * dashboard ends up writing a file the hooks then refuse — the exact failure * mode this module exists to avoid. @@ -112,6 +112,7 @@ import { baseUrlWithoutQuery, endpointGivenAsBase, jevConfigPath, + normalizeJevMode, providerHostConflict, readJevConfigFileForUpdate, validateApiKey, @@ -134,7 +135,7 @@ export interface JevConfigInput { baseUrl: string; /** Cloudflare only. */ accountId: string; - /** "off" | "shadow" | "enforce". */ + /** "off" | "observe" | "enforce" — and "shadow", the old name, read as "observe". */ mode: string; /** "" keeps the key where it is: the stored one, or the environment's. */ token: string; @@ -303,8 +304,10 @@ export async function saveJevConfigAction(input: JevConfigInput): Promise { +export async function setJevModeAction(requested: string): Promise { const refusal = await crossOriginRefusal(); if (refusal) return { ok: false, problem: refusal }; - if (mode !== "off" && mode !== "shadow" && mode !== "enforce") { - return { ok: false, problem: 'mode must be "off", "shadow" or "enforce".' }; + const mode = normalizeJevMode(requested); + if (mode === null) { + return { ok: false, problem: 'mode must be "off", "observe" or "enforce".' }; } const existingFile = readJevConfigFileForUpdate(); diff --git a/app/components/jev-notices.tsx b/app/components/jev-notices.tsx index f9667cf59..e974d240b 100644 --- a/app/components/jev-notices.tsx +++ b/app/components/jev-notices.tsx @@ -7,7 +7,7 @@ * Only the rows where Jev changed or could have changed something get a pill: * a clear (a regex deny Jev overruled), a fallback (Jev's answer was not used — * unavailable, truncated or mismatched — and the regex policies decided alone), - * and a shadow-mode row where Jev disagreed with what was enforced. A fallback + * and an observe-mode row where Jev disagreed with what was enforced. A fallback * where Jev did answer (the call was truncated to fit the envelope) and its * unapplied verdict was stricter than what was enforced gets a louder fallback * pill; the collector ships that row on its own for the same reason. @@ -26,16 +26,16 @@ const SEVERITY: Record = { allow: 0, instruct: 1, deny: 2 }; /** Which pill a row gets, if any. Exported for tests. */ export function jevPillKind( item: JevRow, -): "cleared" | "would-clear" | "fallback" | "fallback-stricter" | "shadow-stricter" | null { +): "cleared" | "would-clear" | "fallback" | "fallback-stricter" | "observe-stricter" | null { const e = sanitizeJevActivity(item); const outcome = jevOutcome(e); if (outcome === null || outcome === "not-consulted" || outcome === "no-request") return null; const jevWasStricter = () => (SEVERITY[e.jevDecision ?? "allow"] ?? 0) > (SEVERITY[item.decision ?? "allow"] ?? 0); if (outcome === "fallback") return e.jevDecision !== undefined && jevWasStricter() ? "fallback-stricter" : "fallback"; const cleared = (e.jevCleared ?? []).length > 0; - if (e.jevMode === "shadow") { + if (e.jevMode === "observe") { if (cleared) return "would-clear"; - return jevWasStricter() ? "shadow-stricter" : null; + return jevWasStricter() ? "observe-stricter" : null; } return cleared ? "cleared" : null; } @@ -47,13 +47,13 @@ const PILLS = { className: "border-sky-500/40 bg-sky-500/10 text-sky-600 dark:text-sky-400", }, "would-clear": { - label: "jev shadow", - title: "Shadow mode: Jev would have cleared a block here; the regex result was enforced", + label: "jev observe", + title: "Observe mode: Jev would have cleared a block here; the regex result was enforced", className: "border-sky-500/30 bg-sky-500/5 text-sky-600/80 dark:text-sky-400/80", }, - "shadow-stricter": { - label: "jev shadow", - title: "Shadow mode: Jev would have been stricter here; the regex result was enforced", + "observe-stricter": { + label: "jev observe", + title: "Observe mode: Jev would have been stricter here; the regex result was enforced", className: "border-sky-500/30 bg-sky-500/5 text-sky-600/80 dark:text-sky-400/80", }, "fallback-stricter": { @@ -68,7 +68,7 @@ const PILLS = { }, } as const; -/** Marks a row where Jev cleared, fell back, or (in shadow mode) disagreed. */ +/** Marks a row where Jev cleared, fell back, or (in observe mode) disagreed. */ export function JevPill({ item }: { item: JevRow }) { const kind = jevPillKind(item); if (!kind) return null; diff --git a/app/settings/jev-panel.tsx b/app/settings/jev-panel.tsx index 1d78be36c..c51cb0543 100644 --- a/app/settings/jev-panel.tsx +++ b/app/settings/jev-panel.tsx @@ -45,7 +45,7 @@ * and a save leaves whatever is stored alone. * * Client-side validation here is a convenience only. Every rule — the URL - * scheme, plain http being refused outside shadow mode, cloudflare's account id + * scheme, plain http being refused outside observe mode, cloudflare's account id * — is enforced server-side by the same `validateJevConfig` the loader runs. * * ## FailproofAI Cloud @@ -53,7 +53,7 @@ * When `jev.json` names the FailproofAI Cloud provider, the endpoint and the * key are not this page's to edit: both come from the machine's connection * (`config --token`). So the form is replaced by the two controls that are - * the owner's — an on/off switch and shadow/enforce — both through + * the owner's — an on/off switch and observe/enforce — both through * `setJevModeAction`, which rewrites `mode` and nothing else. "Off" there keeps * the file (`mode: "off"`): deleting it would leave nothing on this page to * switch back on — the only way to get the file back would be re-running @@ -87,7 +87,7 @@ const PROVIDERS = [ const MODES = [ { value: "enforce", label: "enforce — jev's verdict counts" }, - { value: "shadow", label: "shadow — log only, regex decides" }, + { value: "observe", label: "observe — log only, regex decides" }, { value: "off", label: "off — keep this config, don't ask jev" }, ] as const; @@ -279,10 +279,10 @@ export default function JevPanel({ initial }: { initial: JevSettingsView | null }, [form, token]); /** - * The FailproofAI Cloud switch: `off`, `shadow` or `enforce`, and nothing + * The FailproofAI Cloud switch: `off`, `observe` or `enforce`, and nothing * else about the file changes (see `setJevModeAction`). */ - const onMode = useCallback(async (mode: "off" | "shadow" | "enforce") => { + const onMode = useCallback(async (mode: "off" | "observe" | "enforce") => { setBusy(true); setProblem(null); try { @@ -372,8 +372,8 @@ export default function JevPanel({ initial }: { initial: JevSettingsView | null {view?.on - ? view.mode === "shadow" - ? "shadow: jev's answers are logged, the regex result is what gets enforced." + ? view.mode === "observe" + ? "observe: jev's answers are logged, the regex result is what gets enforced." : "enforce: jev's answers can clear a reviewable deny." : cloudRoute ? "through FailproofAI Cloud, on your org's plan — no endpoint or token of your own." @@ -483,9 +483,9 @@ export default function JevPanel({ initial }: { initial: JevSettingsView | null