diff --git a/CHANGELOG.md b/CHANGELOG.md index 7bf5b6c5..78e1bad2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,11 @@ new version heading in the same commit. ## [Unreleased] +## [0.454.0] - 2026-10-05 +### Added +- **The bets board, on the goal it belongs to.** A goal room gains a **Bets** tab beside Tasks (`BetsBoard` in `web/src/App.tsx`): every attempt at that goal's number, in lanes β€” Proposed, Running, Judged, Closed. A card carries the hypothesis, the lever, "day N of M" against its window with the judge date, the lift so far measured on the bet's OWN assets against what it predicted, the verdict pill once the arithmetic has run, the lesson, and the assets themselves as links with their value (or "unmeasured" β€” never a zero somebody would read as a result, and an unindexed asset says so). Owner/admin get **Judge now** on a running bet and Keep / Expand / Kill on a judged one, each requiring the lesson the server also requires. `verdict` and `observedLift` are displayed and never offered as an input, on either lane. Lanes size themselves and an empty lane is not drawn (named in one line underneath instead): the goal room's main column is ~750px on a laptop, where a fixed four-lane grid gives 170px of truncated URLs. + **For users:** Open a goal and you can see each bet being made on it β€” what it predicted, how far into its window it is, which pages it shipped and what they have earned β€” and keep or kill a judged one with the reason. [Open Goals](#/goals) + ## [0.453.1] - 2026-10-05 ### Fixed - **πŸ”΄ `GET /api/bets` was readable with no session.** The route shipped in v0.453.0 was added beside the agent loopback routes, which sit BEFORE the `/api/*` auth gate, so anyone who could reach the port got every bet on the tenant β€” titles, hypotheses, baselines and asset URLs. The handler was correct; its position in the file was the bug. Moved below the gate beside the other goal routes, with a comment saying why it must stay there, and `scripts/bets-test.cjs` now asserts that an un-cookied read and an un-cookied write both come back 401 β€” a position bug needs a reachability test, not a handler test. Also adds the human write side the board needs (`PATCH /api/bets/:id`, `POST /api/bets/:id/judge`, owner/admin): a human may decide and record the lesson, and may force the arithmetic early, but cannot type a number over a measurement either. diff --git a/package-lock.json b/package-lock.json index 264d04c3..f4f24b7a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "agent-os", - "version": "0.453.1", + "version": "0.454.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "agent-os", - "version": "0.453.1", + "version": "0.454.0", "license": "MIT", "bin": { "agent-os": "bin/agent-os" diff --git a/package.json b/package.json index 72585e23..80556082 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "agent-os", - "version": "0.453.1", + "version": "0.454.0", "description": "A generic, governed operating system for running autonomous agents safely across brands. Ships with a local web console.", "license": "MIT", "type": "commonjs", diff --git a/web/src/App.tsx b/web/src/App.tsx index ab9d2927..ada7f590 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -15,7 +15,7 @@ import { Separator } from '@/components/ui/separator' import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from '@/components/ui/select' import { DropdownMenu, DropdownMenuTrigger, DropdownMenuContent, DropdownMenuItem, DropdownMenuSeparator } from '@/components/ui/dropdown-menu' import { api, isDraftTask, EFFORTS, PERMISSION_MODES, type PermissionMode, type StateResp, type HostMetrics, type RequestMetricsSnapshot, type AgentInfo, type Session, type Msg, type Member, type Role, type TeamResp, type AgentAccess, type MemberIdentity, type IdentityProvider, IDENTITY_PROVIDERS, type Automation, type Task, type TaskEvent, type TaskAttachment, type TaskChild, type TaskRun, type TaskPr, type TaskPrSummary, type TaskWorkers, type TaskTimelineEntry, type TaskDiscussionSummary, type TaskDiscussionDelivery, type TaskStatus, type AddTaskReq, type Goal, type GoalEvent, type GoalMetricStatus, type GoalReading, type GoalStatus, type GoalCounts, type GoalProgress, type AddGoalReq, type MemoryRecord, type MemoryHealth, type MemoryBackend, type MemorySettings, type MemorySettingsReq, type OllamaStatus, type KbPage, type KbRevision, type AgentRevision, type AgentStats, type AgentProposalTrust, type Recommendation, type DigestConfig, type DigestModel, type DreamingState, type Measurement, type Insights, type ImprovementTile, type MemoryCleanupPlan, type KbTidyPlan, type TaskReconcilePlan, type LibraryTidyPlan, type SessionTidyPlan, type StuckGoal, type TroubledAutomation, type PolicyDocument, type PolicyRule, type PolicyOutcome, type PolicyOp, type PolicyProposal, type PolicyRevision, type PolicyDrift, type AutomationProposal, type AgentUpdateProposal, type GoalUpdateProposal, type DirListing, type FileEntry, type FileContent, type Artifact, type AppInfo, type AppFile, type AppCapabilities, type SkillSummary, type SkillsResp, type CatalogSkill, type CatalogAgent, type SkillSource, type RemoteSkill, type SkillshHit, type SkillRequest, type SecretRequest, type IntegrationsResp, type SlackStatus, type DiscordStatus, type TelegramStatus, type AuditEvent, type Effort, type RuntimeTuning, type RuntimeTuningPatch, type OutputStylesResp, type OutputStyleAdoption, type Concurrency, type RuntimeAccount, type RuntimeAccountKind, type RuntimeAccountsResp, type RuntimePresence, type RuntimeLogin, type SecretMeta, type UpdateStatus, type UpdateApplyResult, type UpdateWatchConfig, type UpdateWatchMode, type ActivityEvent, type ActivitySummaryRow, type SystemMetrics, type DepsReport, type DepStatus, type DepsInstallResult, type ChatTurn, type ChatArtifactRef, type ChatKbRef, type ChatAppRef, type RouterPreviewResp, type RouterCard, type SessionChain, type ChainNode, type ChainPending, type SessionProgress, type WhatsNewEntry, type DriftMode } from '@/lib/api' -import { type Branding, type PublicBranding, type NotificationPrefs, DEFAULT_NOTIFICATION_PREFS, type PromptShortcut, type SessionMetrics, type Brief, type AutoApproval, type FeedItem, type FeedResponse, type FeedFilter, type TaskRunState, type GoalChatState } from '@/lib/api' +import { type Branding, type PublicBranding, type NotificationPrefs, DEFAULT_NOTIFICATION_PREFS, type PromptShortcut, type SessionMetrics, type Brief, type AutoApproval, type FeedItem, type FeedResponse, type FeedFilter, type TaskRunState, type GoalChatState, type Bet, type BetAsset, type BetState, type BetVerdict } from '@/lib/api' import { applyAccent, applyFavicon, faviconDataUri, readableOn } from '@/lib/branding' import { ENTITY_ID_SRC, entityHref, isEntityId } from '@/lib/entity-links' import { createGithubApp } from '@/lib/github-app' @@ -9632,7 +9632,7 @@ function GoalProgressBar({ p, className = '' }: { p?: GoalProgress; className?: ) } -type GoalTab = 'tasks' | 'description' | 'activity' | 'chat' +type GoalTab = 'tasks' | 'bets' | 'description' | 'activity' | 'chat' /** Why a linked task has no Run button, compressed to fit a row (the full sentence is the `title=`). * Codes, not prose matching, so the label can't drift from the server's refusal. */ @@ -9863,6 +9863,237 @@ const GOAL_VERDICT: Record = { + met: { label: 'met', role: 'ok', tip: 'the measured lift reached the prediction' }, + short: { label: 'short', role: 'partial', tip: 'the bet was tested and the lift fell short of what it predicted' }, + no_signal: { label: 'no signal', role: 'inactive', tip: 'untested, not failed β€” nothing shipped, nothing measured, or nothing indexed' }, +} + +/** The three closing decisions, in the words the board offers them in. Each one needs a lesson. */ +const BET_CLOSE: { state: BetState; label: string; icon: LucideIcon; variant: 'default' | 'outline' | 'destructive' }[] = [ + { state: 'kept', label: 'Keep', icon: Check, variant: 'default' }, + { state: 'expanded', label: 'Expand', icon: Rocket, variant: 'outline' }, + { state: 'killed', label: 'Kill', icon: X, variant: 'destructive' }, +] + +/** The BETS BOARD β€” every falsifiable attempt at this goal's number, in the four lanes a human acts on. + * + * Two things it deliberately does not do. Lift is summed over the bet's OWN assets, never over the goal's + * metric: that is what keeps two concurrent bets separable (docs/bets-plan.md Β§1). And `verdict` / + * `observedLift` are the server's arithmetic β€” displayed, never offered as an input. The half a human owns + * is the decision and the lesson, and a terminal state without a lesson is refused server-side, so the + * error that comes back is shown verbatim rather than guessed at here. */ +function BetsBoard({ goalId, onCount }: { goalId: string; onCount?: (n: number) => void }) { + const [bets, setBets] = useState<(Bet & { assets: BetAsset[] })[]>([]) + const [counts, setCounts] = useState>({}) + const [canEdit, setCanEdit] = useState(false) + const [loaded, setLoaded] = useState(false) + const [busy, setBusy] = useState('') + const [lesson, setLesson] = useState>({}) + const [hint, setHint] = useState>({}) + + const load = useCallback(async () => { + const r = await api.bets(goalId) + setBets(r.bets ?? []); setCounts(r.counts ?? {}); setCanEdit(!!r.canEdit); setLoaded(true) + onCount?.((r.bets ?? []).length) + }, [goalId, onCount]) + useEffect(() => { setLoaded(false); setLesson({}); setHint({}); void load() }, [load]) + + const say = (id: string, msg: string) => setHint((h) => ({ ...h, [id]: msg })) + + /** Force the arithmetic early on a running bet β€” the verdict only; the decision still comes after. */ + const judge = async (id: string) => { + setBusy(id); say(id, '') + const r = await api.judgeBet(id) + setBusy('') + if (!r.ok) return say(id, r.error || 'could not judge this bet') + await load() + } + /** Close a judged bet with the lesson it taught. The server is the one validator β€” surface its words. */ + const decide = async (id: string, state: BetState) => { + setBusy(id); say(id, '') + const r = await api.patchBet(id, { state, lesson: (lesson[id] ?? '').trim() || undefined }) + setBusy('') + if (!r.ok) return say(id, r.error || 'could not record the decision') + setLesson((m) => ({ ...m, [id]: '' })) + await load() + } + + const fmt = (v: number) => (Number.isInteger(v) ? v.toLocaleString() : v.toFixed(2)) + const signed = (v: number) => (v >= 0 ? '+' : '') + fmt(v) + const day = (ms: number) => new Date(ms).toLocaleDateString() + + /** lift = Ξ£(the assets' measured values) βˆ’ baseline. An asset with no `value` is UNMEASURED, not zero, + * so a bet with nothing measured has no lift to report at all rather than a misleading `βˆ’baseline`. */ + const liftOf = (b: Bet & { assets: BetAsset[] }) => { + const measured = b.assets.filter((a) => a.value != null) + if (measured.length === 0) return null + return measured.reduce((s, a) => s + (a.value as number), 0) - (b.baseline ?? 0) + } + + const card = (b: Bet & { assets: BetAsset[] }) => { + const lift = liftOf(b) + const v = b.verdict ? BET_VERDICT[b.verdict] : null + const elapsed = b.startedAt ? Math.max(1, Math.min(b.windowDays, Math.ceil((Date.now() - b.startedAt) / 86_400_000))) : null + const pct = elapsed == null ? 0 : Math.min(100, Math.round((elapsed / Math.max(1, b.windowDays)) * 100)) + const unindexed = b.assets.filter((a) => a.indexed === false).length + const pending = (lesson[b.id] ?? '').trim() || (b.lesson ?? '').trim() + return ( +
+
+ + {b.title} + {v && {v.label}} +
+ {b.lever && {b.lever}} + {b.hypothesis &&

{b.hypothesis}

} + + {/* The window. A running bet says which day of its own window it is on; a closed one says when the + call was made, so "judged three weeks ago" never reads as "still going". */} + {b.state === 'running' && elapsed != null ? ( +
+
+ day {elapsed} of {b.windowDays} + {b.judgeAt && judges {day(b.judgeAt)}} +
+
+
+
+
+ ) : ( +
+ {b.windowDays}d window + {b.judgedAt ? ` Β· judged ${day(b.judgedAt)}` : b.judgeAt ? ` Β· judges ${day(b.judgeAt)}` : ' Β· not started'} +
+ )} + + {/* Lift so far, on the bet's own assets, against what it predicted. */} +
+ + {lift != null ? {signed(lift)} : lift unmeasured} + {b.expectedLift != null && of {signed(b.expectedLift)} predicted} + {b.observedLift != null && Β· judged at {signed(b.observedLift)}} +
+ + {/* The assets are the evidence, so they are links, not a count β€” and an unmeasured or unindexed one + says so in words instead of showing a zero somebody would read as a result. */} + {b.assets.length > 0 && ( +
+ {b.assets.map((a) => ( +
+ + {a.url.replace(/^https?:\/\//, '')} + + {a.indexed === false && unindexed} + + {a.value != null ? fmt(a.value) : 'unmeasured'} + +
+ ))} + {unindexed === b.assets.length && ( +
Nothing indexed β€” a publishing problem, not evidence against the bet.
+ )} +
+ )} + + {b.lesson && ( +
+ {b.lesson} +
+ )} + + {canEdit && b.state === 'running' && ( + + )} + {canEdit && b.state === 'judging' && ( +
+