diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 85c79d9..5aa26f5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,7 +13,7 @@ jobs: steps: - name: Checkout code - uses: actions/checkout@v4 + uses: actions/checkout@v7 - name: Setup Node.js uses: actions/setup-node@v4 @@ -21,7 +21,7 @@ jobs: node-version: '22' - name: Setup pnpm - uses: pnpm/action-setup@v3 + uses: pnpm/action-setup@v6 with: version: 9 @@ -32,7 +32,7 @@ jobs: echo "STORE_PATH=$(pnpm store path)" >> $GITHUB_OUTPUT - name: Setup pnpm cache - uses: actions/cache@v4 + uses: actions/cache@v6 with: path: ${{ steps.pnpm-cache.outputs.STORE_PATH }} key: ${{ runner.os }}-pnpm-store-${{ hashFiles('**/pnpm-lock.yaml') }} @@ -58,7 +58,7 @@ jobs: steps: - name: Checkout code - uses: actions/checkout@v4 + uses: actions/checkout@v7 - name: Setup Node.js uses: actions/setup-node@v4 @@ -66,7 +66,7 @@ jobs: node-version: '22' - name: Setup pnpm - uses: pnpm/action-setup@v3 + uses: pnpm/action-setup@v6 with: version: 9 @@ -77,7 +77,7 @@ jobs: echo "STORE_PATH=$(pnpm store path)" >> $GITHUB_OUTPUT - name: Setup pnpm cache - uses: actions/cache@v4 + uses: actions/cache@v6 with: path: ${{ steps.pnpm-cache.outputs.STORE_PATH }} key: ${{ runner.os }}-pnpm-store-${{ hashFiles('**/pnpm-lock.yaml') }} @@ -92,7 +92,7 @@ jobs: run: pnpm test:ci - name: Upload coverage to artifacts - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v7 if: always() with: name: coverage-report @@ -106,7 +106,7 @@ jobs: steps: - name: Checkout code - uses: actions/checkout@v4 + uses: actions/checkout@v7 - name: Setup Node.js uses: actions/setup-node@v4 @@ -114,7 +114,7 @@ jobs: node-version: '22' - name: Setup pnpm - uses: pnpm/action-setup@v3 + uses: pnpm/action-setup@v6 with: version: 9 @@ -125,7 +125,7 @@ jobs: echo "STORE_PATH=$(pnpm store path)" >> $GITHUB_OUTPUT - name: Setup pnpm cache - uses: actions/cache@v4 + uses: actions/cache@v6 with: path: ${{ steps.pnpm-cache.outputs.STORE_PATH }} key: ${{ runner.os }}-pnpm-store-${{ hashFiles('**/pnpm-lock.yaml') }} @@ -141,7 +141,7 @@ jobs: NODE_ENV: production - name: Upload build artifacts - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v7 with: name: build-output path: .next/ diff --git a/.github/workflows/scorecard.yml b/.github/workflows/scorecard.yml index 3f557b4..7d5a27e 100644 --- a/.github/workflows/scorecard.yml +++ b/.github/workflows/scorecard.yml @@ -20,7 +20,7 @@ jobs: contents: read steps: - name: Checkout code - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: persist-credentials: false @@ -32,7 +32,7 @@ jobs: publish_results: false - name: Upload results - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: scorecard-results path: scorecard-results.json diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ef6fded..96349ac 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -583,7 +583,7 @@ We value all contributions and will: - **General Questions**: Open a [GitHub Discussion](https://github.com/csupenn/topflow/discussions) - **Bug Reports**: Open a [GitHub Issue](https://github.com/csupenn/topflow/issues) -- **Security Issues**: Email charlie@charliesu.com +- **Security Issues**: Report privately; see [SECURITY.md](SECURITY.md) - **Feature Proposals**: Open an issue with the "feature request" label --- diff --git a/README.md b/README.md index fd750ab..8acfcf5 100644 --- a/README.md +++ b/README.md @@ -30,6 +30,8 @@ database and explains what to upgrade. ([why](https://www.topflow.dev/blog/untrusted-reasoning-worker-llm-security)). - **Runs show sample results by default.** Turn on **Run a real scan** in the run dialog for live data. Public repos work without a key; a GitHub token raises GitHub's rate limit and allows private repos. +- **No AI spending unless you ask.** The report is built from the scan data. An LLM report (your own key and + quota) runs only when you switch on **Write the report with my AI key** for that run; saved keys never turn it on. **[Scan a repo →](https://www.topflow.dev/builder?template=github-security-scanner)** diff --git a/app/api/execute-workflow/__tests__/scanner-real-scan-and-cost.test.ts b/app/api/execute-workflow/__tests__/scanner-real-scan-and-cost.test.ts new file mode 100644 index 0000000..afcbfd5 --- /dev/null +++ b/app/api/execute-workflow/__tests__/scanner-real-scan-and-cost.test.ts @@ -0,0 +1,200 @@ +/** + * @jest-environment node + */ + +/** + * Real-scan design revision 2 (docs/architecture/osv-real-scan-design.md §15), acceptance criteria 1, 2, 5. + * + * Runs the REAL route + engine with the GitHub Scanner template and mocks only the edges: + * - `ai` (every AI-provider call goes through generateText / generateObject / embed) → counted, + * - `lib/osv/scanner` (no network) → observable in-process calls, + * - global fetch → serves api.github.com metadata, and the app's own routes the way local development + * would (so the old HTTP-to-self path can run far enough to reach the LLM and fail for the right reason). + */ + +import { GITHUB_SCANNER_NODES, GITHUB_SCANNER_EDGES } from "@/lib/templates/github-scanner" +import { getRepoAnalysis } from "@/lib/demo-data/github-repos" + +jest.mock("@/lib/security/upstash-rate-limit-store", () => ({ createUpstashStore: () => null })) +// Full scanner runs include the template's simulated step delays (~5 s each). +jest.setTimeout(30_000) + +jest.mock("ai", () => { + const actual = jest.requireActual("ai") + return { + ...actual, + generateText: jest.fn(async () => ({ text: "generated", files: [] })), + generateObject: jest.fn(async () => ({ + object: { prioritizedFindingIds: [], recommendations: [], summaryLabel: "MINOR_ISSUES" }, + })), + embed: jest.fn(async () => ({ embedding: [0] })), + } +}) + +const FAKE_SCAN = { + repository: "facebook/react", + stars: 1, + forks: 1, + language: "JavaScript", + lastAnalyzed: "2026-09-27T00:00:00.000Z", + scanMode: "real-osv", + securityScore: 72, + grade: "B", + vulnerabilities: { + critical: 0, + high: 1, + medium: 0, + low: 0, + details: [ + { + id: "CVE-2026-0001", + osvId: "GHSA-test-0001", + severity: "HIGH", + component: "lodash@4.17.21", + description: "test advisory", + fix: "Upgrade to 4.18.0", + effort: "30+ minutes", + }, + ], + }, + dependencyAudit: { + total: 10, + vulnerable: 1, + outdated: null, + licenses: [], + riskBreakdown: { high: 1, medium: 0, low: 0 }, + ecosystemsScanned: ["npm"], + }, + securityPractices: { + has_security_policy: true, + dependabot_enabled: false, + code_scanning: null, + secret_scanning: null, + branch_protection: null, + signed_commits: null, + two_factor_required: null, + }, + owaspCompliance: { A06_vulnerable_components: "FAIL" }, + _meta: { manifestSources: ["package-lock.json (10)"], dataSources: ["GitHub REST API", "OSV.dev"], byok: false }, +} + +jest.mock("@/lib/osv/scanner", () => ({ scanRepository: jest.fn() })) + +import { generateText, generateObject, embed } from "ai" +import { scanRepository } from "@/lib/osv/scanner" +import { POST } from "../route" + +const ownOrigin = /^https?:\/\/(localhost|127\.0\.0\.1)(:\d+)?\/|^\/api\// +let fetchCalls: string[] = [] +let ip = 0 + +function mockFetch() { + fetchCalls = [] + jest.spyOn(global, "fetch").mockImplementation(async (input: any) => { + const url = typeof input === "string" ? input : input.url + fetchCalls.push(url) + const json = (v: unknown) => new Response(JSON.stringify(v), { headers: { "Content-Type": "application/json" } }) + if (url.startsWith("https://api.github.com/repos/")) { + return json({ full_name: "facebook/react", stargazers_count: 1, forks_count: 1, language: "JavaScript" }) + } + // What the app's own routes would answer if the engine could reach them (local development). + if (/\/api\/demo\/github-scan\//.test(url)) return json(getRepoAnalysis("facebook/react")) + if (/\/api\/scan\/github\//.test(url)) return json(FAKE_SCAN) + throw new Error(`unexpected fetch in test: ${url}`) + }) +} + +async function runScanner(body: Record) { + const req = new Request("http://localhost:3000/api/execute-workflow", { + method: "POST", + headers: { "Content-Type": "application/json", "x-forwarded-for": `198.51.100.${++ip}` }, + body: JSON.stringify({ + workflowId: "github-security-scanner", + nodes: GITHUB_SCANNER_NODES, + edges: GITHUB_SCANNER_EDGES, + userInputs: { start: "https://github.com/facebook/react" }, + apiKeys: {}, + ...body, + }), + }) + const res = await POST(req) + const reader = res.body!.getReader() + const decoder = new TextDecoder() + let text = "" + for (;;) { + const { done, value } = await reader.read() + if (done) break + text += decoder.decode(value, { stream: true }) + } + return text + .split("\n") + .filter(Boolean) + .map((l) => JSON.parse(l)) +} + +const aiCalls = () => + (generateText as jest.Mock).mock.calls.length + + (generateObject as jest.Mock).mock.calls.length + + (embed as jest.Mock).mock.calls.length + +describe("Scanner revision 2: real scans in process, AI spending only on request", () => { + beforeEach(() => { + jest.spyOn(console, "log").mockImplementation(() => {}) + jest.spyOn(console, "warn").mockImplementation(() => {}) + jest.spyOn(console, "error").mockImplementation(() => {}) + ;(generateText as jest.Mock).mockClear() + ;(generateObject as jest.Mock).mockClear() + ;(embed as jest.Mock).mockClear() + ;(scanRepository as jest.Mock).mockReset().mockResolvedValue(FAKE_SCAN) + mockFetch() + }) + afterEach(() => jest.restoreAllMocks()) + + // Acceptance criterion 2 + test("a real scan calls the scanner in process and never fetches the app's own origin", async () => { + const updates = await runScanner({ scanMode: "real" }) + expect(fetchCalls.filter((u) => ownOrigin.test(u))).toEqual([]) + expect(scanRepository).toHaveBeenCalledWith("facebook", "react", { githubToken: undefined }) + expect(updates.filter((u) => u.type === "node_error" || u.type === "error")).toEqual([]) + // Score 72: the grade check's "input1.score >= 80" survives input sanitizing and evaluates to false. + // (Branch skipping itself is tracked separately: the base engine runs both branches, S14.) + const gradeCheck = updates.find((u) => u.type === "node_complete" && u.nodeId === "grade-check") + expect(gradeCheck?.output).toBe(false) + }) + + test("the visitor's GitHub token goes to the in-process scan, not over HTTP", async () => { + await runScanner({ scanMode: "real", githubToken: "ghp_visitor_token" }) + expect(scanRepository).toHaveBeenCalledWith("facebook", "react", { githubToken: "ghp_visitor_token" }) + expect(fetchCalls.filter((u) => ownOrigin.test(u))).toEqual([]) + }) + + // Acceptance criterion 1 + test("sample data + saved AI keys, AI report off → zero AI-provider calls", async () => { + await runScanner({ scanMode: "demo", apiKeys: { openai: "sk-test", google: "g-test" } }) + expect(aiCalls()).toBe(0) + }) + + test("real scan + saved AI keys, AI report off → zero AI-provider calls", async () => { + await runScanner({ scanMode: "real", apiKeys: { openai: "sk-test", google: "g-test" } }) + expect(aiCalls()).toBe(0) + }) + + test("AI report switched on → the URW-constrained model call runs", async () => { + await runScanner({ scanMode: "real", apiKeys: { openai: "sk-test" }, aiReport: true }) + expect((generateObject as jest.Mock).mock.calls.length).toBeGreaterThan(0) + }) + + // Acceptance criterion 5 + test("AI report on + Google key → the image step runs; without the switch it doesn't", async () => { + await runScanner({ scanMode: "real", apiKeys: { openai: "sk-test", google: "g-test" }, aiReport: true }) + expect((generateText as jest.Mock).mock.calls.length).toBeGreaterThan(0) + ;(generateText as jest.Mock).mockClear() + await runScanner({ scanMode: "real", apiKeys: { openai: "sk-test", google: "g-test" } }) + expect((generateText as jest.Mock).mock.calls.length).toBe(0) + }) + + test("AI report on without a Google key → no image-model call", async () => { + await runScanner({ scanMode: "real", apiKeys: { openai: "sk-test" }, aiReport: true }) + expect((generateText as jest.Mock).mock.calls.length).toBe(0) + }) +}) diff --git a/app/api/execute-workflow/route.ts b/app/api/execute-workflow/route.ts index 5bf6a54..d049431 100644 --- a/app/api/execute-workflow/route.ts +++ b/app/api/execute-workflow/route.ts @@ -84,6 +84,7 @@ export async function POST(req: Request) { userInputs, githubToken, scanMode, + aiReport, }: { nodes: Node[] edges: Edge[] @@ -92,6 +93,8 @@ export async function POST(req: Request) { userInputs?: Record githubToken?: string scanMode?: ScanMode + /** Scanner: the per-run "Write the report with my AI key" switch (default off). */ + aiReport?: boolean } = await req.json() // Privacy: log shape only (counts/flags), never user-supplied content — @@ -127,15 +130,13 @@ export async function POST(req: Request) { // Demo Mode Handling // ============================================================================ - // Two-axis BYOK resolution for the GitHub Scanner (see lib/demo-mode). - // GitHub token => real scan DATA; AI key => LLM report NARRATIVE. - // Real-scan per-axis behavior is strictly opt-in: with no token / no - // explicit scanMode, this collapses to the prior single-boolean demo gate. + // Two-axis resolution for the GitHub Scanner (design doc §15): the data axis follows the + // "Run a real scan" switch; the LLM report runs only when the user switched it on for this + // run (aiReport) and has an AI key. Saved keys alone never trigger AI spending. const isScanner = workflowId === "github-security-scanner" const scanAxes = isScanner - ? resolveScanModes({ apiKeys, githubToken, scanMode }) + ? resolveScanModes({ apiKeys, githubToken, scanMode, aiReport: aiReport === true }) : null - const realScan = Boolean(isScanner && scanAxes && scanAxes.dataMode === "real") const demoMode = isScanner ? scanAxes!.demoMode : shouldUseDemoMode(apiKeys, workflowId) // Check both legacy and new demo mode systems @@ -197,7 +198,8 @@ export async function POST(req: Request) { const sanitizedNodes = processedNodes.map((node) => ({ ...node, - data: sanitizeInput(node.data, ["code", "schema", "output"]), + // Conditions are read only by the safe parser (never rendered), so their < and > must survive. + data: sanitizeInput(node.data, ["code", "schema", "output", "condition"]), })) // ============================================================================ @@ -228,11 +230,11 @@ export async function POST(req: Request) { return } - // API key validation - only when an LLM will actually run. - // (A real scan with a templated, no-LLM report needs no AI key.) - const needsAiKeyValidation = isScanner - ? scanAxes!.narrativeMode === "llm" - : !demoMode + // API key validation - only when an LLM will actually run. Not for the scanner: its LLM report + // runs only when an AI key exists (resolveScanModes) and picks that key's provider (URW), and its + // image step runs only with a Google key, so checking every template node's hard-coded model + // would wrongly demand keys the run won't use. + const needsAiKeyValidation = isScanner ? false : !demoMode if (needsAiKeyValidation) { const apiKeyIssues = validateApiKeys(apiKeys, sanitizedNodes) const apiKeyErrors = apiKeyIssues.filter((issue) => issue.type === "error") @@ -251,10 +253,10 @@ export async function POST(req: Request) { // Execute Workflow using TopFlowExecutionEngine // ============================================================================ - // Pass per-axis options only for an opt-in real scan; otherwise keep the - // exact legacy construction so demo/live behavior is unchanged. + // The scanner always runs per-axis (no path where its report node runs as a plain + // text-model call); other workflows keep the legacy construction. const engine = new TopFlowExecutionEngine( - realScan + isScanner ? { demoMode, workflowId, diff --git a/app/api/scan/github/[...repo]/route.ts b/app/api/scan/github/[...repo]/route.ts index 5c36a02..0a154bc 100644 --- a/app/api/scan/github/[...repo]/route.ts +++ b/app/api/scan/github/[...repo]/route.ts @@ -10,7 +10,8 @@ * (`x-github-token`), forwarded by the browser from localStorage exactly like * the AI provider keys. It is never persisted server-side — consistent with * TopFlow's zero-storage model. (A server-side GITHUB_TOKEN env var is honored - * as a fallback only for self-hosted deployments.) + * as a fallback only for self-hosted deployments; the hosted service sets none.) + * Rate-limited per client (10/min), like the execution route. * * @route GET /api/scan/github/[owner]/[repo] * @header x-github-token: (optional but recommended: 60 -> 5,000 req/hr + private repos) @@ -18,6 +19,17 @@ */ import { NextRequest, NextResponse } from "next/server"; import { scanRepository } from "@/lib/osv/scanner"; +import { RateLimiter, rateLimitKey } from "@/lib/security/rate-limit"; +import { createUpstashStore } from "@/lib/security/upstash-rate-limit-store"; + +// Each scan costs ~15 GitHub requests, one OSV query per vulnerable package and up to 30 s of +// function time, so direct callers get the same budget as the execution route (design doc §15.5). +// Builder scans don't come through here: the engine calls scanRepository() in process. +const limiter = new RateLimiter({ + limit: 10, + windowMs: 60_000, + store: createUpstashStore() ?? undefined, +}); export const runtime = "nodejs"; export const maxDuration = 30; // matches the workflow engine's 30s budget @@ -26,6 +38,22 @@ export async function GET( request: NextRequest, { params }: { params: Promise<{ repo: string[] }> } ) { + const clientIp = (request.headers.get("x-forwarded-for") || "").split(",")[0].trim() || "anonymous"; + const rl = await limiter.check(`scan:${rateLimitKey(clientIp)}`); + if (!rl.allowed) { + return NextResponse.json( + { error: "Rate limit exceeded. Please try again shortly." }, + { + status: 429, + headers: { + "Retry-After": String(Math.max(1, Math.ceil(rl.resetMs / 1000))), + "X-RateLimit-Limit": String(rl.limit), + "X-RateLimit-Remaining": String(rl.remaining), + }, + } + ); + } + const { repo } = await params; const repoPath = (repo || []).join("/"); diff --git a/app/api/scan/github/__tests__/rate-limit.test.ts b/app/api/scan/github/__tests__/rate-limit.test.ts new file mode 100644 index 0000000..3240f94 --- /dev/null +++ b/app/api/scan/github/__tests__/rate-limit.test.ts @@ -0,0 +1,40 @@ +/** + * @jest-environment node + */ + +/** + * Real-scan design revision 2, acceptance criterion 3: the public real-scan route is rate-limited like the + * execution route (10 requests per minute per client), because each call costs ~15 GitHub requests, + * one OSV query per vulnerable package and up to 30 s of function time. + */ + +jest.mock("@/lib/security/upstash-rate-limit-store", () => ({ createUpstashStore: () => null })) +jest.mock("@/lib/osv/scanner", () => ({ scanRepository: jest.fn(async () => ({ repository: "facebook/react" })) })) + +import { GET } from "../[...repo]/route" + +const call = (ip: string) => + // The route only reads headers; a plain Request stands in for NextRequest in this environment. + GET(new Request("http://localhost:3000/api/scan/github/facebook/react", { headers: { "x-forwarded-for": ip } }) as any, { + params: Promise.resolve({ repo: ["facebook", "react"] }), + }) + +describe("GET /api/scan/github rate limit", () => { + test("the 11th request within a minute from one client gets 429; another client is unaffected", async () => { + const statuses: number[] = [] + for (let i = 0; i < 11; i++) statuses.push((await call("192.0.2.10")).status) + expect(statuses.slice(0, 10)).toEqual(Array(10).fill(200)) + expect(statuses[10]).toBe(429) + expect((await call("192.0.2.11")).status).toBe(200) + }) + + test("a limited response says when to retry and runs no scan", async () => { + const { scanRepository } = jest.requireMock("@/lib/osv/scanner") + for (let i = 0; i < 10; i++) await call("192.0.2.20") + scanRepository.mockClear() + const res = await call("192.0.2.20") + expect(res.status).toBe(429) + expect(Number(res.headers.get("Retry-After"))).toBeGreaterThan(0) + expect(scanRepository).not.toHaveBeenCalled() + }) +}) diff --git a/app/showcase/security-scanner/page.tsx b/app/showcase/security-scanner/page.tsx index 7e966e4..b3458e7 100644 --- a/app/showcase/security-scanner/page.tsx +++ b/app/showcase/security-scanner/page.tsx @@ -66,7 +66,7 @@ const STEPS = [ { name: "Fetch Repo Metadata", detail: "Stars, language and default branch from the GitHub API." }, { name: "Security Scan", detail: "Reads the manifests from GitHub, queries OSV.dev, and scores the result with the formula below." }, { name: "Calculate Score", detail: "Prepares the score, grade and breakdown for the report. No AI involved." }, - { name: "Write the report", detail: "A template, or an LLM with your own key, explains the findings." }, + { name: "Write the report", detail: "Built from the scan data; or, if you switch it on, an LLM with your own key explains the findings." }, ] const SCORE = [ @@ -219,7 +219,7 @@ export default function SecurityScannerShowcase() { {/* Modes and data */}
-
+

Sample results (default)

@@ -236,10 +236,20 @@ export default function SecurityScannerShowcase() {

-

What goes where

+

AI-written report (optional)

+

+ Off by default, even if you've saved AI keys. Turn on{" "} + Write the report with my AI key for a run to have an LLM, using + your own key and quota, explain the findings. It can't change them. With a Google key it also draws + an illustration, labeled as AI-generated. +

+
+
+

What goes where, and who pays

Our server receives the repo name and, if you add one, your GitHub token for that request. It calls the - GitHub API and OSV.dev, and stores neither the results nor your token. The optional AI report uses your own provider key. + GitHub API and OSV.dev (free), and stores neither the results nor your token. Nothing calls an AI + provider unless you switch the AI report on.

diff --git a/components/execution-panel.tsx b/components/execution-panel.tsx index 4d7f850..8b7f9bf 100644 --- a/components/execution-panel.tsx +++ b/components/execution-panel.tsx @@ -138,7 +138,7 @@ export function ExecutionPanel({ const handleInputSubmit = ( inputs: Record, - scanOptions?: { githubToken?: string; scanMode?: "demo" | "real" } + scanOptions?: { githubToken?: string; scanMode?: "demo" | "real"; aiReport?: boolean } ) => { setShowInputDialog(false) setPendingUserInputs(inputs) @@ -152,7 +152,7 @@ export function ExecutionPanel({ const executeWorkflow = async ( userInputs: Record | null, - scanOptions?: { githubToken?: string; scanMode?: "demo" | "real" } + scanOptions?: { githubToken?: string; scanMode?: "demo" | "real"; aiReport?: boolean } ) => { setIsExecuting(true) setExecutionLog([]) @@ -190,6 +190,7 @@ export function ExecutionPanel({ userInputs: userInputs, // Pass user inputs to API githubToken, // BYOK scan-data axis (scanner only; undefined => demo data) scanMode: scanOptions?.scanMode, // "real" | "demo" | undefined (auto) + aiReport: scanOptions?.aiReport === true, // scanner: LLM report only when switched on (design §15) }), }) diff --git a/components/github-scanner-results.tsx b/components/github-scanner-results.tsx index 10830a4..cf7fa7f 100644 --- a/components/github-scanner-results.tsx +++ b/components/github-scanner-results.tsx @@ -288,11 +288,14 @@ Try it yourself: https://www.topflow.dev/builder?template=github-security-scanne Security Dashboard +

+ AI-generated illustration, not scan data. The numbers above come from the scan. +

Security Dashboard diff --git a/components/ui/switch.tsx b/components/ui/switch.tsx index 3c4cfa3..3c123d9 100644 --- a/components/ui/switch.tsx +++ b/components/ui/switch.tsx @@ -13,7 +13,7 @@ function Switch({ diff --git a/components/workflow-input-dialog.tsx b/components/workflow-input-dialog.tsx index f3edc9f..dcbb3ba 100644 --- a/components/workflow-input-dialog.tsx +++ b/components/workflow-input-dialog.tsx @@ -11,7 +11,7 @@ import { Switch } from "@/components/ui/switch" import { encryptValue, decryptValue } from "@/lib/security/encryption" import type { StartNodeData } from "@/components/nodes/start-node" -type ScanOptions = { githubToken?: string; scanMode?: "demo" | "real" } +type ScanOptions = { githubToken?: string; scanMode?: "demo" | "real"; aiReport?: boolean } type WorkflowInputDialogProps = { open: boolean @@ -22,12 +22,14 @@ type WorkflowInputDialogProps = { } export function WorkflowInputDialog({ open, startNodes, onSubmit, onCancel, workflowId }: WorkflowInputDialogProps) { - // GitHub Scanner gets an extra control: a real/demo toggle + an optional BYOK - // GitHub token (scan-DATA axis). The AI key (report-NARRATIVE axis) is handled - // separately via the existing API Settings. + // GitHub Scanner gets two per-run switches (design doc §15): "Run a real scan" (scan-DATA axis, + // optional BYOK GitHub token) and "Write the report with my AI key" (report-NARRATIVE axis). Both are + // off by default; a saved AI key only makes the second one available, it never turns it on. const isScanner = workflowId === "github-security-scanner" const [githubToken, setGithubToken] = useState("") const [realScan, setRealScan] = useState(false) + const [aiReport, setAiReport] = useState(false) + const [hasAiKey, setHasAiKey] = useState(false) const [inputs, setInputs] = useState>(() => { const initialInputs: Record = {} startNodes.forEach((node) => { @@ -46,6 +48,17 @@ export function WorkflowInputDialog({ open, startNodes, onSubmit, onCancel, work }) setInputs(newInputs) setErrors({}) + setAiReport(false) + + // Scanner: the AI-report switch is only available when an AI provider key is saved. + if (isScanner && typeof window !== "undefined") { + try { + const keys = JSON.parse(localStorage.getItem("ai-agent-api-keys") || "{}") + setHasAiKey(["openai", "anthropic", "google", "groq"].some((p) => Boolean(keys?.[p]))) + } catch { + setHasAiKey(false) + } + } // Scanner: load + decrypt any saved GitHub token (legacy plaintext passes through). if (isScanner && typeof window !== "undefined") { @@ -129,7 +142,11 @@ export function WorkflowInputDialog({ open, startNodes, onSubmit, onCancel, work else localStorage.removeItem("ai-agent-github-token") } // realScan off => force demo; on => real (token optional: public repos scan tokenless). - scanOptions = { githubToken: realScan && token ? token : undefined, scanMode: realScan ? "real" : "demo" } + scanOptions = { + githubToken: realScan && token ? token : undefined, + scanMode: realScan ? "real" : "demo", + aiReport: aiReport && hasAiKey, + } } onSubmit(inputs, scanOptions) @@ -212,6 +229,18 @@ export function WorkflowInputDialog({ open, startNodes, onSubmit, onCancel, work

)} + +
+
+ +

+ {hasAiKey + ? "Uses your own AI provider key and quota. Off = a report built from the scan data, no AI calls. With a Google key, also draws an AI-generated illustration." + : "Add an AI provider key under API Keys to enable. Without it, the report is built from the scan data, with no AI calls."} +

+
+ +
)} diff --git a/docs/architecture/osv-real-scan-design.md b/docs/architecture/osv-real-scan-design.md index 243e1fe..883ebc5 100644 --- a/docs/architecture/osv-real-scan-design.md +++ b/docs/architecture/osv-real-scan-design.md @@ -2,7 +2,7 @@ | | | |---|---| -| **Status** | Proposed / proof-of-concept (initial backend + docs) | +| **Status** | Revision 1 implemented (real scan, two axes, templated fallback). **Revision 2 accepted 2026-09-27 (§15):** explicit opt-in for AI spending, in-process scanning, rate-limited public route; sample labeling (§15.4) follows in a later PR | | **Owner** | TopFlow | | **Related** | `docs/architecture/architecture-overview.md`, `lib/templates/github-scanner.ts`, `lib/demo-mode.ts`, `lib/topflow-execution-engine.ts`, `app/api/demo/github-scan/[...repo]/route.ts` | | **Scope** | Adds a real, opt-in scan path. Demo mode is unchanged. | @@ -140,6 +140,10 @@ scanner. They must be treated separately — neither implies the other. | **Report narrative** | **AI provider key** | the `AI Security Analysis` text-model node | `apiKeys` in the `/api/execute-workflow` body (never stored) | fall back to a **templated** (no-LLM) report rendered from `RepoAnalysis` | ### 6.1 Decoupled gating + +> **Superseded by §15.3 (Revision 2):** the narrative axis no longer follows "any AI key present"; it follows +> an explicit per-run switch. The matrix below is kept as the history of Revision 1. + Today `shouldUseDemoMode` returns one boolean for the whole workflow, keyed **only** on the AI key — so a GitHub token alone cannot trigger a real scan, and the data/narrative decisions are conflated. The target design splits the decision **per axis**: @@ -171,6 +175,9 @@ whichever provider the user actually has a key for (e.g. priority `anthropic → groq`, mapping to a sensible default model per provider), rather than a fixed model string. ### 6.3 Templated (no-LLM) report fallback + +> **Revision 2:** the templated report is the **default** for every scanner run, not a fallback (§15.3). + A real scan should not *require* an AI key. When scan data is real but no AI key is present, render the report deterministically from `RepoAnalysis` — exactly the approach `getGitHubScannerMockResponse` already uses to synthesize report markdown from counts/practices/recommendations. This is factored @@ -235,6 +242,10 @@ it; no downstream change is required either way. ## 10. Security & privacy +> **Revision 2 additions (§15):** the hosted service sets **no** server-side GitHub token (the `GITHUB_TOKEN` fallback is +> for self-hosted deployments only); the public scan route is rate-limited per client (HMAC-keyed IP); builder scans +> call `scanRepository()` in process. + - **Two BYOK keys, neither stored.** The **GitHub token** arrives in a request header (`x-github-token`) and the **AI key** in the `execute-workflow` body; both are used only for the duration of the request and never written to disk, logs, or a database — consistent with the @@ -257,6 +268,10 @@ it; no downstream change is required either way. ## 11. Error handling +> **Revision 2 additions:** the public `/api/scan/github` route answers `429` (with `Retry-After`) after 10 requests +> per minute per client (§15.5). + + | Case | Behavior | |---|---| | Repo not found / private without token | `502` with an actionable message | @@ -278,6 +293,10 @@ it; no downstream change is required either way. ## 13. Rollout +> **Revision 2:** in-process scanning and the opt-in narrative ship in the same release, because fixing the +> builder's real-scan path alone would also have made saved keys start spending (§15.1). + + Standard `feature → dev → main` flow. Backend + docs land first (this PR, into `dev`). A follow-up implements §6 (decoupled gating, provider-agnostic report model, templated fallback) and the UI: a **GitHub-token field** (scan-data axis) alongside the existing **AI-key settings** (narrative @@ -293,3 +312,89 @@ templated fallback); transitive coverage via lockfiles for all ecosystems; addit secret/code scanning via the GitHub security APIs); short-TTL caching keyed by repo + commit SHA; surface CVE links + CVSS scores in the results UI; broaden OWASP mapping; a GitHub Action / PR-comment bot reusing `lib/osv/scanner.ts`. + +--- + +## 15. Revision 2 — proposed (2026-09-27) + +> **Status: Accepted (2026-09-27)** — §15.2, §15.3, §15.5, §15.6 ship together in one release (branch +> `feat/scanner-real-scan-and-cost`); §15.4 (sample labeling, no substitution) follows in a later PR. Where §1–§14 +> disagree with this section, this section wins; the notes added to §6.1, §6.3, §10, §11 and §13 point here. + +### 15.1 Why + +Two owner constraints, plus what a review of the shipped behavior showed: + +- **The demo is intentional.** v1.4.0 launched the scanner with pre-loaded sample data so it runs instantly + with no keys. That stays; what changes is that sample data is always labeled as sample data. +- **No AI spending unless the user asks for it in that run.** Revision 1 (§6.1) sets the narrative axis to + "LLM when any AI provider key is present". A visitor who saved a key for other workflows therefore spends + their provider quota (and, with a Google key, an image-model call) just by running the scanner. +- **Real scans must not depend on the engine calling its own origin over HTTP.** The engine resolves the + scanner's relative app routes against `NEXT_PUBLIC_BASE_URL`, falling back to `http://localhost:3000`; on the + hosted service that variable isn't set, so builder-initiated real scans fail at the Security Scan node. + +### 15.2 Principles + +1. **Keys enable, switches decide.** A saved key makes an option *available*; it never turns one on. +2. **Every run's cost is visible before it starts.** The run dialog says which parts will call which service. +3. **Sample data is always labeled**, in the report header, the score badge and the exports. +4. **No server-side secrets on the hosted service for scanning.** Hosted scans run anonymously or with the + visitor's own GitHub token; `GITHUB_TOKEN` remains an option for self-hosted deployments only. + +### 15.3 Revised mode matrix (replaces the key-driven matrix in §6.1) + +Two explicit, per-run switches in the run dialog, both **off by default**: + +| "Run a real scan" | "Write the report with my AI key" | Scan data | Report | Dashboard image | Who pays | +|:---:|:---:|---|---|---|---| +| off | off | **sample** (labeled; sample repos only, §15.4) | templated | none | nobody | +| off | on | **sample** (labeled) | URW-constrained LLM over the labeled sample | optional (§15.6) | visitor's AI key | +| on | off | **real** (OSV + GitHub) | templated | none | GitHub quota (anonymous or visitor token); OSV is free | +| on | on | **real** | URW-constrained LLM | optional (§15.6) | visitor's AI key + the above | + +The AI switch is disabled (with a short explanation) when no AI provider key is saved. Every scanner run takes +the per-axis path; there is no longer a "live" path in which the report node runs as a plain text-model node. + +### 15.4 Unknown repositories in sample mode + +Sample data exists only for a few well-known repositories. For any other repository, sample mode no longer +substitutes another repository's data. The report shows: "*owner/repo* isn't in the sample set. Run a real +scan: it's free and needs no AI key." Exports and badges never pair a requested repository with a sample grade. + +### 15.5 Execution: in-process scanning and a rate-limited public route + +- The engine calls `scanRepository()` (real data) and the sample-data provider (sample mode) **in process** + for the Security Scan node, instead of fetching `/api/scan/github` or `/api/demo/github-scan` over HTTP. The + template keeps the node (so the workflow stays readable), and the SSRF provenance rule for engine-generated + routes becomes unnecessary for this node. +- `GET /api/scan/github/{owner}/{repo}` remains for direct/API use and applies the same limiter as the + execution route (10 requests per minute per client, HMAC-keyed IP), answering `429` when exceeded. + +### 15.6 The dashboard image + +The "Generate Dashboard" image-model step runs only when the AI-report switch is on **and** a Google key is +saved. The image is labeled "AI-generated illustration, not scan data"; the report's numbers always come from +the data, never from the image. + +### 15.7 Changes to other sections (applied when accepted) + +| Section | Change | +|---|---| +| §6.1 | Replace the key-driven matrix with §15.3 | +| §6.3 | Templated report is the default, not a fallback | +| §10 | Add: no server-side GitHub token on the hosted service; public route rate-limited | +| §11 | Add: `429` on the public route; unknown repo in sample mode is not an error (§15.4) | +| §13 | Both parts (in-process scanning and opt-in narrative) ship in the same release | + +### 15.8 Acceptance criteria + +1. With AI keys saved and the AI-report switch off, a scanner run makes **zero** AI-provider calls, with the + real-scan switch on or off (test mocks the provider SDK and asserts no call). +2. A real-scan run completes on the hosted service without `NEXT_PUBLIC_BASE_URL`, and the engine makes no + HTTP request to its own origin. +3. The 11th request within a minute to `/api/scan/github` from one client gets `429`. +4. A sample-mode run for a repository outside the sample set never shows that repository with a grade. +5. The image step runs only with the AI-report switch on and a Google key; the image carries its label. +6. Scanner page and README describe the switches and costs exactly as in §15.3. + diff --git a/docs/development/osv-scanner/00-roadmap.md b/docs/development/osv-scanner/00-roadmap.md index b5b6946..3644d3d 100644 --- a/docs/development/osv-scanner/00-roadmap.md +++ b/docs/development/osv-scanner/00-roadmap.md @@ -45,6 +45,7 @@ is ranked by **trust delivered per unit effort**, not by feature surface. Two tr | W3 | Make a second template real (PII detection; GDPR stretch) | **P1** | M | W2 Phase 1 pattern | `03-p1-second-template.md` | | W4 | Distribution loop (GitHub Action / PR bot + real README badges + SEO pages) | **P1** | M | real scan endpoint; CI enabled | `04-p1-distribution-loop.md` | | W5 | Observability / execution history (audit substrate) | P2 | M | W2 | (roadmap Phase 3) | +| W6 | Real scans on the hosted service, AI spending only on request, accurate results (Sept 2026) | **P0/P1** | M | real scan endpoint; design §15 | `07-w6-real-scans-and-accurate-results.md` | Effort: **S** < 1 day · **M** 1–3 days · **L** ≥ 1 week. diff --git a/docs/development/osv-scanner/05-implementation-status.md b/docs/development/osv-scanner/05-implementation-status.md index 411b78c..283b311 100644 --- a/docs/development/osv-scanner/05-implementation-status.md +++ b/docs/development/osv-scanner/05-implementation-status.md @@ -1,7 +1,7 @@ # OSV Scanner — Implementation Status -**Last updated:** 2026-09-26 -**Branch baseline:** `main` = `dev` @ `b180e80` (PR #31) +**Last updated:** 2026-09-27 +**Branch baseline:** `main` @ `e0d2dc5` (release v1.5.0); W6 work on `feat/scanner-real-scan-and-cost` This document is the ground truth between what the roadmap plans and what the code actually does. Update it as things land or get blocked — not after the fact. @@ -92,6 +92,13 @@ defects in shipped code (H4 privacy, H11 SSRF bypass). Each shipped with a red-b | **H6 Honest coverage** | Global 75% threshold was never met (~17%) and CI hid it (`continue-on-error`). Now per-file thresholds on the security core + execute route, a global ratchet floor, and a blocking `pnpm test:ci` | #28 | ✅ Shipped | | **H11 SSRF bypass** | IPv4-mapped IPv6 in the URL parser's hex form (`[::ffff:a9fe:a9fe]` = `169.254.169.254`) passed `checkOutboundUrl`. Hex-embedded IPv4 is now decoded and re-checked; tests go through the parser | #29 | ✅ Shipped, verified in production | | **H10 Validation panel parity** | `validation-engine.ts` (builder panel) had 0% coverage and a drifted copy of SSRF/cycle rules (reported "passed" for `[::1]`, CGNAT, `*.internal`, `file:`). Now delegates to `ssrf.ts` + `workflow-graph.ts`; 0% → 100% lines | #30 | ✅ Shipped | +| **P3 Rate-limit key privacy** | Rate-limit keys are HMAC-SHA256 of the client IP under `RATE_LIMIT_KEY_SECRET` (a plain hash of an IPv4 address is brute-forceable) | #36 | ✅ Shipped | +| **A3 Content Security Policy** | CSP in **report-only** mode from one source (`lib/security/security-headers.cjs`) + privacy-safe `/api/csp-report`; enforcement pending review of reports | #37 | ✅ Report-only | +| **H13 Condition tester + builder values** | Condition "Test" isolated from the page; visual-builder values emitted as JSON literals (code injection) | #38 | ✅ Shipped | +| **H17 User code containment** | JavaScript/Tool code ran via `new Function` with access to `process.env`/`fetch`. Hosted service now runs only built-in template code; conditions use a safe parser. Isolation design: `docs/architecture/js-node-isolation-design.md` | #42 (released #43) | ✅ Contained | +| **Honest scanner surfaces** | Scanner page, README and badge describe what the scanner does; badge carries no score and accepts no writes | #50–#52 | ✅ Shipped | +| **H12 Dependency patch** | Next.js 15.5.7 → 15.5.26, unused Auth.js removed; `pnpm audit --prod` 64 (6 critical) → 10 (0 critical); CI fails on new critical advisories (#61) | #53, #61 | ✅ Shipped | +| **Repository foundation** | SECURITY.md, Dependabot, OpenSSF Scorecard workflow, Code of Conduct, issue/PR templates; plain MIT license (#59); release v1.5.0 | #59, #61 | ✅ Shipped | ### ssrf.ts: IPv4-mapped IPv6 bypass via URL normalization (H11) @@ -108,6 +115,27 @@ IPv4 hosts, which the parser already normalizes safely). Tutorial 01 documents t --- +## M1.5 — Real scans on the hosted service; AI spending only on request (W6) + +Workstream doc: `07-w6-real-scans-and-accurate-results.md`. Design: `docs/architecture/osv-real-scan-design.md` §15. + +| Item | What | PR | Status | +|------|------|----|:------:| +| **In-process scanning** | Builder real scans called `/api/scan/github` over HTTP at `NEXT_PUBLIC_BASE_URL` (unset on the hosted service → `fetch failed`). The engine now calls `scanRepository()` in process | feat/scanner-real-scan-and-cost | ✅ in `dev` | +| **AI report on request** | The LLM report (and the Gemini image) ran whenever any AI key was saved. Now a per-run "Write the report with my AI key" switch, off by default; the scanner always takes the per-axis path (URW), never a plain text-model call | same | ✅ in `dev` | +| **Public scan route rate limit** | `/api/scan/github` limited to 10 requests/min per client (HMAC-keyed IP), like the execution route; no server GitHub token on the hosted service | same | ✅ in `dev` | +| **Conditions keep `<` / `>`** | Input sanitizing stripped `<>` from `condition` fields (`score >= 80` → `score = 80`); conditions are now exempt (read only by the safe parser) | same | ✅ in `dev` | +| **Scanner key validation** | The route demanded keys for every template node's hard-coded model (e.g. Google for the image step); skipped for the scanner, which only uses the keys a run needs | same | ✅ in `dev` | +| **Visible switches** | Unchecked switches were the same color as the page background; now have a visible track | same | ✅ in `dev` | +| Sample-data labeling | Label sample results everywhere; no substituted data for repos outside the sample set; no `85`/`B+` defaults | — | 🔲 Planned | +| Fix-version accuracy | Fix suggestions use the installed version's range (no downgrades or major jumps) | — | 🔲 Planned | +| Production vs development dependencies | Report both; score on production | — | 🔲 Planned | +| Score formula | Keep distinguishing repositories with many findings | — | 🔲 Planned | +| Accuracy CI gate | Recorded OSV/GitHub fixtures for known-vulnerable and known-clean lockfiles | — | 🔲 Planned | +| Conditional branches | The base engine runs both sides of every condition; skip the side that wasn't chosen | — | 🔲 Planned | + +--- + ## Open blockers ### B1 — `pnpm install --frozen-lockfile` prevents adding new dependencies in CI @@ -133,13 +161,15 @@ Evaluate in this order: (1) `quickjs-emscripten` — pure wasm, no native binari ## What's next (ordered) +0. **W6** — release the in-process/opt-in work, then the planned rows in "M1.5" above (`07-w6-…`) + 1. ~~**W2 Phase 1** — constrained-selector~~ ✅ shipped (`lib/security/urw.ts`) 2. ~~**T4 durable rate limiter**~~ ✅ shipped (`lib/security/upstash-rate-limit-store.ts`; PR #20) 3. ~~**T6 claims reconciliation**~~ ✅ shipped (`docs/architecture/architecture-overview.md`; PR #19) 4. ~~**T7 drop `ignoreBuildErrors`**~~ ✅ shipped (`next.config.mjs`; PR #19) 5. ~~**Post-M1 hardening** (H4, H6, H10, H11)~~ ✅ shipped (PRs #25–#31, Sept 2026) -6. **Rate-limit key privacy** — key Redis by a keyed hash of the client IP, not the raw IP (today: raw IP, ~65 s TTL) -7. **CSP header** — report-only first, then enforce +6. ~~**Rate-limit key privacy**~~ ✅ shipped (HMAC-SHA256 keys; PR #36) +7. **CSP header** — ✅ report-only shipped (PR #37); enforcement pending review of reports 8. **T3 JS-node isolation** — design: [`docs/architecture/js-node-isolation-design.md`](../../architecture/js-node-isolation-design.md) (QuickJS/WebAssembly inside a `worker_thread`; spike results included). Since Sept 2026 (H17) custom JS/Tool code is refused on the hosted service until this ships 9. **W2 Phase 2** — trifecta guard + human-gated sinks (co-develops with T3) 10. **W3 PII Detection** — M2, after URW Phase 1 establishes the pattern diff --git a/docs/development/osv-scanner/06-handoff.md b/docs/development/osv-scanner/06-handoff.md index a7f92d2..06a2875 100644 --- a/docs/development/osv-scanner/06-handoff.md +++ b/docs/development/osv-scanner/06-handoff.md @@ -1,5 +1,8 @@ # Handoff — OSV Scanner Security Program +> **Historical (June 2026).** Kept as a record of the June pause. For current status see +> `05-implementation-status.md`; for the September work see its "Post-M1 hardening" and "M1.5" sections. + **Written:** 2026-06-14 **Branch at pause:** `feature/t4-durable-rate-limiter` (current working branch) **Program tracker:** `05-implementation-status.md` diff --git a/docs/development/osv-scanner/07-w6-real-scans-and-accurate-results.md b/docs/development/osv-scanner/07-w6-real-scans-and-accurate-results.md new file mode 100644 index 0000000..65a0374 --- /dev/null +++ b/docs/development/osv-scanner/07-w6-real-scans-and-accurate-results.md @@ -0,0 +1,72 @@ +# W6 — Real Scans on the Hosted Service, AI Spending on Request, Accurate Results (P0/P1) + +**Priority:** P0 (release blockers first), then P1 · **Effort:** M overall, in small PRs +**Design:** `docs/architecture/osv-real-scan-design.md` §15 (Revision 2) +**Tracker rows:** `05-implementation-status.md` → "M1.5" + +--- + +## Goal + +The GitHub Security Scanner should do what its page says, for any visitor, without surprise costs: + +1. **Keep the zero-setup demo** (sample data for a few well-known repos), labeled as sample data. +2. **Real scans work on the hosted service** for public repos, anonymously or with the visitor's own GitHub token. +3. **No AI-provider spending unless the visitor asks for it in that run.** Saved keys make the AI report + *available*; a per-run switch decides. +4. **Results are accurate:** fix suggestions, dependency scope and the score hold up when someone checks them. + +## Owner constraints + +- The v1.4.0 demo used sample data on purpose, to launch quickly; it stays. +- No LLM or image-model calls by default. +- No server-side GitHub token on the hosted service (the `GITHUB_TOKEN` fallback is for self-hosting). + +## Target behavior + +Two per-run switches in the run dialog, both off by default (design §15.3): + +| "Run a real scan" | "Write the report with my AI key" | Data | Report | Who pays | +|:---:|:---:|---|---|---| +| off | off | sample (labeled) | built from the data | nobody | +| off | on | sample (labeled) | URW-constrained LLM | visitor's AI key | +| on | off | real (OSV + GitHub) | built from the data | GitHub quota (anonymous or visitor token); OSV is free | +| on | on | real | URW-constrained LLM | visitor's AI key + the above | + +With the AI report on and a Google key saved, the image step also runs; the image is labeled "AI-generated +illustration, not scan data". + +## Tasks + +| # | Task | Status | +|---|---|:--:| +| 1 | Engine calls `scanRepository()` in process for the Security Scan node (no HTTP to its own origin) | ✅ in `dev` | +| 2 | Narrative axis follows the per-run switch (`aiReport`); scanner always takes the per-axis path (URW); image only with the switch + Google key, labeled | ✅ in `dev` | +| 3 | Rate-limit `GET /api/scan/github` (10/min per client, HMAC-keyed IP) | ✅ in `dev` | +| 4 | Keep `<` / `>` in `condition` fields through input sanitizing | ✅ in `dev` | +| 5 | Scanner skips template-wide API-key validation (uses only the keys a run needs) | ✅ in `dev` | +| 6 | Run dialog: AI-report switch (disabled without a saved AI key); visible switch styling | ✅ in `dev` | +| 7 | Sample-data labeling in report header, badge and exports; no substituted data for repos outside the sample set; remove `85`/`B+` defaults | 🔲 | +| 8 | Fix-version suggestions from the affected range that contains the installed version (no downgrades, no needless major or pre-release jumps) | 🔲 | +| 9 | Production vs development dependencies from lockfiles; report both, score on production | 🔲 | +| 10 | Score formula that keeps distinguishing repositories with many findings; scanner page formula table in sync | 🔲 | +| 11 | Accuracy CI gate: recorded OSV/GitHub responses for a known-vulnerable and a known-clean lockfile (no network in CI) | 🔲 | +| 12 | Skip the unchosen side of conditional branches in the engine (the base engine runs both) | 🔲 | +| 13 | Later: real cached badge (W4 task 1); decide whether real scans become the default | 🔲 | + +## Acceptance criteria + +Design §15.8, plus: + +- Tasks 1–6: `app/api/execute-workflow/__tests__/scanner-real-scan-and-cost.test.ts` (real route + engine, AI SDK + and scanner mocked) and `app/api/scan/github/__tests__/rate-limit.test.ts`. Each was checked to fail on the + previous code for the intended reason (AI calls > 0; request to the app's own origin; 11th request → 200; + `grade-check` error from the stripped `>=`). +- Task 8: fixtures built from real OSV records for packages with more than one affected release line. +- A stranger scans their own public repo on the hosted service without help, and every number shown can be + traced to code. + +## Rollout + +Tasks 1–6 ship in one release: fixing the builder's real-scan path alone would have let saved AI keys start +spending on the hosted service. Tasks 7–12 ship as separate small PRs, each with its own failing-first test. diff --git a/docs/development/osv-scanner/README.md b/docs/development/osv-scanner/README.md index 9b7c132..9d62fde 100644 --- a/docs/development/osv-scanner/README.md +++ b/docs/development/osv-scanner/README.md @@ -8,7 +8,7 @@ This folder is the working documentation hub for the GitHub Security Scanner fea The OSV scanner turns TopFlow into a **real** security tool: given a GitHub repo, it fetches dependency manifests, queries the [OSV.dev](https://osv.dev) API, and returns structured vulnerability findings. The LLM is used only as a constrained selector over those findings — it cannot invent CVEs or modify severities. -The work here is tracked across two P0 workstreams (W1 security hardening, W2 URW trust boundary) and two P1 workstreams (W3 second template, W4 distribution loop). +The work here is tracked across two P0 workstreams (W1 security hardening, W2 URW trust boundary), two P1 workstreams (W3 second template, W4 distribution loop), and W6 (real scans on the hosted service, AI spending on request, accurate results). --- @@ -23,6 +23,8 @@ The work here is tracked across two P0 workstreams (W1 security hardening, W2 UR | `03-p1-second-template.md` | W3: making PII Detection real (the second URW-compliant template) | | `04-p1-distribution-loop.md` | W4: real badge API, GitHub Action/PR bot, shareable report cards | | `05-implementation-status.md` | Live tracking: what's shipped, what's in progress, what's blocked and why | +| `06-handoff.md` | Historical (June 2026 pause); superseded by `05-implementation-status.md` | +| `07-w6-real-scans-and-accurate-results.md` | W6: real scans on the hosted service, AI spending only on request, accurate results (design §15) | ## Tutorial series diff --git a/lib/__tests__/scanner-axes.test.ts b/lib/__tests__/scanner-axes.test.ts index ffa847e..009a19f 100644 --- a/lib/__tests__/scanner-axes.test.ts +++ b/lib/__tests__/scanner-axes.test.ts @@ -48,6 +48,29 @@ describe("resolveScanModes (two-axis BYOK)", () => { }) }) +describe("resolveScanModes with the scanner's AI-report switch (design §15)", () => { + test("AI key saved but switch off -> templated report, no LLM (demo and real data)", () => { + expect(resolveScanModes({ apiKeys: { openai: "sk-x" }, aiReport: false }).narrativeMode).toBe("templated") + expect(resolveScanModes({ apiKeys: { openai: "sk-x" }, scanMode: "real", aiReport: false }).narrativeMode).toBe("templated") + }) + + test("switch on with an AI key -> LLM report", () => { + expect(resolveScanModes({ apiKeys: { anthropic: "sk-ant-x" }, aiReport: true }).narrativeMode).toBe("llm") + }) + + test("switch on without any AI key -> still templated", () => { + expect(resolveScanModes({ aiReport: true }).narrativeMode).toBe("templated") + }) + + test("switch off + sample data -> full demo, even with keys saved", () => { + expect(resolveScanModes({ apiKeys: { openai: "sk-x", google: "g" }, scanMode: "demo", aiReport: false }).demoMode).toBe(true) + }) + + test("other workflows (no aiReport given) keep the key-driven rule", () => { + expect(resolveScanModes({ apiKeys: { openai: "sk-x" } }).narrativeMode).toBe("llm") + }) +}) + describe("resolveReportModel (provider-agnostic)", () => { test("prefers anthropic, then openai, then google, then groq", () => { expect(resolveReportModel({ anthropic: "x", openai: "y" })).toBe("anthropic/claude-3-5-sonnet-20241022") diff --git a/lib/demo-mode.ts b/lib/demo-mode.ts index 5e20cf8..51bd4e6 100644 --- a/lib/demo-mode.ts +++ b/lib/demo-mode.ts @@ -73,7 +73,10 @@ export interface ScanAxes { /** * Resolve the two scan axes from the available keys + explicit preferences. * - * - narrative: explicit demo/live preference wins; otherwise LLM iff an AI key exists. + * - narrative: explicit demo/live preference wins. When `aiReport` is given (the GitHub Scanner always + * passes it), the LLM runs only if the user switched the AI report on for this run AND an + * AI key exists: saved keys never trigger spending on their own (design §15). Without + * `aiReport` (other workflows via shouldUseDemoMode): LLM iff an AI key exists. * - data: explicit scanMode wins; a "demo" preference forces mock data; otherwise * real iff a GitHub token is present. (Public repos can scan tokenless at * the lower rate limit, but we only flip to "real" on an explicit signal so @@ -84,6 +87,8 @@ export function resolveScanModes(opts: { githubToken?: string scanMode?: ScanMode userPreference?: DemoModePreference + /** Scanner only: the per-run "Write the report with my AI key" switch. */ + aiReport?: boolean }): ScanAxes { const apiKeys = opts.apiKeys || {} const hasAiKey = Boolean( @@ -96,6 +101,7 @@ export function resolveScanModes(opts: { let narrativeMode: "llm" | "templated" if (pref === "demo") narrativeMode = "templated" else if (pref === "live") narrativeMode = "llm" + else if (opts.aiReport !== undefined) narrativeMode = opts.aiReport && hasAiKey ? "llm" : "templated" else narrativeMode = hasAiKey ? "llm" : "templated" let dataMode: "real" | "demo" diff --git a/lib/topflow-execution-engine.ts b/lib/topflow-execution-engine.ts index 0401e6b..bd25446 100644 --- a/lib/topflow-execution-engine.ts +++ b/lib/topflow-execution-engine.ts @@ -16,6 +16,7 @@ import { resolveReportModel } from './demo-mode' import { assertSafeOutboundUrl } from './security/ssrf' +import { scanRepository } from './osv/scanner' import { isTrustedCode, UNTRUSTED_CODE_MESSAGE } from './security/trusted-code' import { evaluateCondition } from './conditions/safe-evaluate' import { @@ -131,7 +132,14 @@ export class TopFlowExecutionEngine extends ExecutionEngine { // Data axis: mock when demo, real otherwise. if (dataLogicNodes.includes(id)) { - return this.dataMode === 'demo' ? { handled: true, result: await mock() } : { handled: false } + if (this.dataMode === 'demo') return { handled: true, result: await mock() } + // Real scan: call the scanner in process instead of fetching our own /api/scan/github over + // HTTP (the hosted service has no base URL for that, and HTTP-to-self would share one client + // IP for every visitor's scan). Design doc §15.5. + if (id === 'fetch-security') { + return { handled: true, result: await this.scanInProcess(inputs) } + } + return { handled: false } } // Narrative axis: URW-constrained LLM when "llm"; templated render otherwise. @@ -166,6 +174,16 @@ export class TopFlowExecutionEngine extends ExecutionEngine { return { handled: false } } + /** + * Real scan of the repository chosen upstream (extract-repo → { fullName: "owner/repo" }). + */ + private async scanInProcess(inputs: Record): Promise { + const fullName = String(inputs.input1?.fullName ?? '') + const match = fullName.match(/^([A-Za-z0-9_.-]+)\/([A-Za-z0-9_.-]+)$/) + if (!match) throw new Error('Expected a GitHub repository in the form owner/repo') + return scanRepository(match[1], match[2], { githubToken: this.githubToken }) + } + /** * Real execution for TopFlow-specific node types (the original switch). */ @@ -224,20 +242,16 @@ export class TopFlowExecutionEngine extends ExecutionEngine { private async executeHttpRequestNode(node: Node, inputs: Record): Promise { const data = node.data as any - let url = data.url || '' + const url = data.url || '' const method = (data.method || 'GET').toUpperCase() // Templates may store headers as a JSON string; only spread real objects. const headers: Record = (data.headers && typeof data.headers === 'object') ? { ...data.headers } : {} const body = data.body || '' - // Real-scan (per-axis) overrides for the GitHub Scanner data nodes: - // - point the security node at the real OSV endpoint - // - attach the user's GitHub token (BYOK): private repos + 5,000 req/hr + // Real scan: attach the visitor's GitHub token (BYOK) to the metadata request + // (private repos, 5,000 req/hr). The security node itself runs in process (scanInProcess). if (this.perAxis && this.dataMode === 'real' && this.workflowId === 'github-security-scanner') { - if (node.id === 'fetch-security') { - url = '/api/scan/github/$input1.fullName' - if (this.githubToken) headers['x-github-token'] = this.githubToken - } else if (node.id === 'fetch-metadata' && this.githubToken) { + if (node.id === 'fetch-metadata' && this.githubToken) { headers['Authorization'] = `Bearer ${this.githubToken}` } }