From fdf9b0607b2d11e258da700e726d9f7195af8ae1 Mon Sep 17 00:00:00 2001 From: crypt0rr <57799908+crypt0rr@users.noreply.github.com> Date: Wed, 7 Oct 2026 17:33:19 +0200 Subject: [PATCH] fix: record shutdown interruptions and keep scan cancellation requests - A scan that stops because EdgeWatch stopped is recorded as canceled with "scan interrupted because EdgeWatch stopped" and a scan-interrupted activity event that is never delivered, instead of a scan-canceled alert to every destination. Resumable attempts keep their saved progress and say so. - Active scans report cancel_requested once someone asks to cancel, so the job page and Overview keep showing Cancellation requested while the scanner reports its last progress and the result is saved. - A cancel request after finalization has begun is refused with 409 scan_finalizing, because it could no longer change the outcome. - model.Fingerprint sorts a copy of the CPEs instead of the caller's slice. Fixes #1227, #1228, #1237 --- .../content/docs/user-guide/notifications.md | 7 +- docs/src/content/docs/user-guide/scanning.md | 6 + internal/app/app.go | 45 +++- internal/app/coverage_more_test.go | 6 +- internal/app/resumable.go | 8 + internal/app/scan_interruption_test.go | 199 ++++++++++++++++++ internal/app/units.go | 8 + internal/engine/coverage_extra_test.go | 5 +- internal/engine/engine.go | 10 + internal/model/model.go | 102 +++++---- internal/model/model_test.go | 14 ++ internal/store/incident_actions.go | 2 +- internal/store/runtime.go | 3 + internal/web/scan_handlers.go | 4 + src/main.test.tsx | 4 + src/main.tsx | 1 + src/pages/Activity.test.tsx | 7 +- src/pages/Activity.tsx | 1 + src/pages/Dashboard.test.tsx | 7 + src/pages/Dashboard.tsx | 2 +- src/pages/JobDetail.actions.test.tsx | 10 + src/pages/JobDetail.tsx | 4 +- src/types.ts | 2 +- 23 files changed, 400 insertions(+), 57 deletions(-) create mode 100644 internal/app/scan_interruption_test.go diff --git a/docs/src/content/docs/user-guide/notifications.md b/docs/src/content/docs/user-guide/notifications.md index e8574b3a..ba770db2 100644 --- a/docs/src/content/docs/user-guide/notifications.md +++ b/docs/src/content/docs/user-guide/notifications.md @@ -67,7 +67,12 @@ pause uses none of their retries and is not reported as a delivery failure. ## Delivery retries and health Scan changes, scan failures, cancellations, timeouts, stalled cycles, and -recovery events can all generate notifications. Definitive provider failures +recovery events can all generate notifications. A scan that stops because +EdgeWatch stopped, for example during an upgrade or restart, is recorded as +canceled with the reason "scan interrupted because EdgeWatch stopped" and +appears in Activity as **Scan interrupted**, but sends no notification. A +daemon that keeps stopping is still reported by the job's silence alert. +Definitive provider failures are retried durably for up to 15 attempts over roughly 77 hours; the delay doubles from two minutes and caps at 12 hours. A restart preserves each delivery's retry schedule. Terminal failures are visible in the console diff --git a/docs/src/content/docs/user-guide/scanning.md b/docs/src/content/docs/user-guide/scanning.md index 064b9649..0ccb667c 100644 --- a/docs/src/content/docs/user-guide/scanning.md +++ b/docs/src/content/docs/user-guide/scanning.md @@ -73,6 +73,12 @@ next cycle. Accepting an incident, approving or resetting the baseline, or changing the monitored scope discards paused progress, and the next run starts a fresh cycle. +**Cancel scan** stops the scanner, and the job page and Overview show +**Cancellation requested** until the scan ends. Once EdgeWatch is saving a +scan's result, the scan can no longer be canceled. A scan that stops because +EdgeWatch itself stopped is recorded as interrupted rather than canceled; see +[notifications](/user-guide/notifications/#delivery-retries-and-health). + ## Runtime capabilities The default Compose configuration grants `NET_RAW`. Naabu SYN additionally diff --git a/internal/app/app.go b/internal/app/app.go index 938ba1f3..7f73ac43 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -84,8 +84,30 @@ type activeRun struct { cancel context.CancelFunc // tenant is the ID of the run's tenant, set by registerRun. tenant string + // finalizing is set once the result is being saved. A cancellation can + // no longer change the outcome then. + finalizing bool } +// interruptedByShutdown reports whether scanCtx ended because the run +// context ctx ended, as it does when the daemon stops or loses its lease, +// rather than because someone asked to cancel this scan. +func (r *activeRun) interruptedByShutdown(ctx, scanCtx context.Context) bool { + if !errors.Is(scanCtx.Err(), context.Canceled) || ctx.Err() == nil { + return false + } + if r == nil { + return true + } + r.mu.RLock() + defer r.mu.RUnlock() + return !r.scan.CancelRequested +} + +// ScanInterruptedMessage is the error of a scan that stopped because +// EdgeWatch stopped while it ran. +const ScanInterruptedMessage = "scan interrupted because EdgeWatch stopped" + type cronSlogLogger struct{ logger *slog.Logger } func (l cronSlogLogger) Info(msg string, keysAndValues ...interface{}) { @@ -108,6 +130,10 @@ func (l cronSlogLogger) Error(err error, msg string, keysAndValues ...interface{ // accepted because the daemon is stopping. var ErrShuttingDown = errors.New("application is shutting down") +// ErrScanFinalizing is returned by CancelScan once a scan is saving its +// result, when cancelling it would no longer change the outcome. +var ErrScanFinalizing = errors.New("scan is saving its result and can no longer be canceled") + // ErrScanWorkBudget is returned before a lease is acquired when a job's // estimated probe count exceeds the deployment guard and the job has not // explicitly opted into high-cost work. @@ -966,6 +992,10 @@ func (a *App) runJobWithQueueMarker(ctx context.Context, scope store.TenantScope if errors.Is(scanCtx.Err(), context.Canceled) { scan.Status = "canceled" scan.Error = "scan canceled" + if run.interruptedByShutdown(ctx, scanCtx) { + scan.Error = ScanInterruptedMessage + scan.Interrupted = true + } } else if errors.Is(scanCtx.Err(), context.DeadlineExceeded) || errors.Is(scanErr, context.DeadlineExceeded) { scan.Status = "timed_out" scan.Error = "scan timed out" @@ -983,7 +1013,7 @@ func (a *App) runJobWithQueueMarker(ctx context.Context, scope store.TenantScope // target scopes and will not advance a baseline from this scan. engine.MarkIncompleteScan(&scan) } - a.updateActivePhase(scan.ID, "finalizing") + a.beginActiveFinalization(scan.ID) persistTimeout := scanPersistenceTimeout(len(scan.Snapshot.Hosts)) if a.persistenceBudget != nil { persistTimeout = a.persistenceBudget(len(scan.Snapshot.Hosts)) @@ -1153,8 +1183,10 @@ func (a *App) CancelScan(scope store.TenantScope, id string) error { if run.cancel == nil { return store.ErrNotFound } - run.cancel() - run.scan.Phase = "cancelling" + if run.finalizing { + return ErrScanFinalizing + } + run.requestCancelLocked() return nil } @@ -1232,7 +1264,9 @@ func (a *App) updateActiveProgress(id string, progress scanner.Progress) { } } -func (a *App) updateActivePhase(id, phase string) { +// beginActiveFinalization reports the finalizing phase and refuses later +// cancellation requests, which could no longer change the result. +func (a *App) beginActiveFinalization(id string) { value, ok := a.running.Load(id) if !ok { return @@ -1242,8 +1276,9 @@ func (a *App) updateActivePhase(id, phase string) { return } run.mu.Lock() - run.scan.Phase = phase + run.scan.Phase = "finalizing" run.scan.ProcessAlive = false + run.finalizing = true run.mu.Unlock() } diff --git a/internal/app/coverage_more_test.go b/internal/app/coverage_more_test.go index 64c0c80e..2232dace 100644 --- a/internal/app/coverage_more_test.go +++ b/internal/app/coverage_more_test.go @@ -68,10 +68,10 @@ func TestProgressPercentAndActiveRunUpdates(t *testing.T) { t.Fatalf("active progress = %#v", got) } a.updateActiveProgress("scan", scanner.Progress{CompletedProbes: 30, CompletedInvocations: 2, ProcessAlive: false}) - a.updateActivePhase("missing", "ignored") - a.updateActivePhase("scan", "finalizing") + a.beginActiveFinalization("missing") + a.beginActiveFinalization("scan") got = run.snapshot() - if got.CompletedProbes != 30 || got.CompletedInvocations != 2 || got.ProcessAlive || got.Phase != "finalizing" { + if got.CompletedProbes != 30 || got.CompletedInvocations != 2 || got.ProcessAlive || got.Phase != "finalizing" || !run.finalizing { t.Fatalf("active phase update = %#v", got) } if (&activeRun{}).snapshot().ElapsedSeconds != 0 { diff --git a/internal/app/resumable.go b/internal/app/resumable.go index bad2e2dc..c6d2845a 100644 --- a/internal/app/resumable.go +++ b/internal/app/resumable.go @@ -136,6 +136,10 @@ func (a *App) runResumableAttempt(ctx, scanCtx context.Context, ts *store.Tenant if errors.Is(scanCtx.Err(), context.Canceled) { scan.Status = "canceled" scan.Error = "scan canceled while creating the scan plan" + if run.interruptedByShutdown(ctx, scanCtx) { + scan.Error = ScanInterruptedMessage + " while creating the scan plan" + scan.Interrupted = true + } } else if errors.Is(scanCtx.Err(), context.DeadlineExceeded) || errors.Is(planErr, context.DeadlineExceeded) { scan.Status = "timed_out" scan.Error = "scan timed out while creating the scan plan" @@ -441,6 +445,10 @@ func (a *App) runResumableAttempt(ctx, scanCtx context.Context, ts *store.Tenant } else if canceled { scan.Status = "canceled" scan.Error = "scan canceled; progress was saved" + if run.interruptedByShutdown(ctx, scanCtx) { + scan.Error = ScanInterruptedMessage + "; progress was saved" + scan.Interrupted = true + } } else { scan.Status = "timed_out" scan.Error = "scan timed out; progress was saved for the next trigger" diff --git a/internal/app/scan_interruption_test.go b/internal/app/scan_interruption_test.go new file mode 100644 index 00000000..430567de --- /dev/null +++ b/internal/app/scan_interruption_test.go @@ -0,0 +1,199 @@ +package app + +import ( + "context" + "encoding/json" + "errors" + "io" + "log/slog" + "testing" + "time" + + "github.com/crypt0rr/edgewatch/internal/config" + "github.com/crypt0rr/edgewatch/internal/model" + "github.com/crypt0rr/edgewatch/internal/scanner" + "github.com/crypt0rr/edgewatch/internal/store" + "github.com/crypt0rr/edgewatch/internal/store/storetest" +) + +func interruptionTestApp(t *testing.T) (*App, *store.Store) { + t.Helper() + s, err := store.Open(storetest.FreshPath(t)) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = s.Close() }) + cfg := &config.Config{ + Version: 1, Database: "test", Retention: config.Duration(24 * time.Hour), + Scheduler: config.Scheduler{MaxConcurrent: 1}, + Web: config.Web{Listen: "127.0.0.1:8080"}, + Notifications: config.Notifications{URLs: []string{"generic://localhost/edgewatch?disabletls=yes&template=json"}}, + } + a, err := New(cfg, s, "missing", slog.New(slog.NewTextHandler(io.Discard, nil))) + if err != nil { + t.Fatal(err) + } + return a, s +} + +func outboxCount(t *testing.T, s *store.Store) int { + t.Helper() + var count int + if err := s.DB.QueryRow(`SELECT COUNT(*) FROM outbox`).Scan(&count); err != nil { + t.Fatal(err) + } + return count +} + +func TestShutdownInterruptedScanIsRecordedWithoutNotification(t *testing.T) { + t.Parallel() + a, s := interruptionTestApp(t) + record, err := defaultTenant(s).CreateJob(context.Background(), config.NormalizeJob(config.Job{ + Name: "interrupted", Schedule: "0 * * * *", Timezone: "UTC", Targets: []string{"127.0.0.1"}, + TCP: &config.Protocol{Ports: "1", Mode: "connect"}, Timing: "balanced", Timeout: config.Duration(time.Minute), + })) + if err != nil { + t.Fatal(err) + } + blocking := &blockingScanner{started: make(chan struct{}), release: make(chan struct{})} + a.Scanner = blocking + runCtx, stop := context.WithCancel(context.Background()) + done := make(chan struct{}) + var scan model.Scan + var events []model.Event + var runErr error + go func() { + scan, events, runErr = a.RunJobRecord(runCtx, record) + close(done) + }() + select { + case <-blocking.started: + case <-time.After(2 * time.Second): + t.Fatal("scan did not start") + } + // The daemon stopping cancels the run context, not this scan. + stop() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("interrupted scan did not finish") + } + if !errors.Is(runErr, context.Canceled) || scan.Status != "canceled" || scan.Error != ScanInterruptedMessage { + t.Fatalf("interrupted scan = status %q error %q err %v", scan.Status, scan.Error, runErr) + } + if len(events) != 1 || events[0].Type != model.EventScanInterrupted || events[0].Message != "Scan interrupted because EdgeWatch stopped" { + t.Fatalf("interrupted scan events = %#v", events) + } + if count := outboxCount(t, s); count != 0 { + t.Fatalf("an interrupted scan queued %d notifications", count) + } + var raw []byte + if err := s.DB.QueryRow(`SELECT payload_json FROM events ORDER BY id DESC LIMIT 1`).Scan(&raw); err != nil { + t.Fatal(err) + } + var stored model.Event + if err := json.Unmarshal(raw, &stored); err != nil || stored.Type != model.EventScanInterrupted { + t.Fatalf("activity history = %s (%v), want the interruption", raw, err) + } +} + +func TestShutdownInterruptedResumableScanKeepsProgressWithoutNotification(t *testing.T) { + t.Parallel() + a, s := interruptionTestApp(t) + probe := &resumableTestScanner{} + a.Scanner = probe + record, err := defaultTenant(s).CreateJob(context.Background(), config.NormalizeJob(config.Job{ + Name: "broad", Schedule: "0 * * * *", Timezone: "UTC", Targets: []string{"192.0.2.1"}, + TCP: &config.Protocol{Ports: "1-2", Mode: "syn"}, Timeout: config.Duration(time.Minute), ResumeWindow: config.Duration(time.Hour), + })) + if err != nil { + t.Fatal(err) + } + runCtx, stop := context.WithCancel(context.Background()) + done := make(chan struct{}) + var scan model.Scan + var events []model.Event + go func() { + scan, events, _ = a.RunJobRecord(runCtx, record) + close(done) + }() + // The second unit blocks until its context ends; stop once it runs. + deadline := time.Now().Add(5 * time.Second) + for { + probe.mu.Lock() + started := probe.calls[1] > 0 + probe.mu.Unlock() + if started { + break + } + if time.Now().After(deadline) { + t.Fatal("second work unit did not start") + } + time.Sleep(5 * time.Millisecond) + } + stop() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("interrupted resumable scan did not finish") + } + if scan.Status != "canceled" || scan.Error != ScanInterruptedMessage+"; progress was saved" || scan.CycleStatus != "paused" { + t.Fatalf("interrupted resumable scan = status %q error %q cycle %q", scan.Status, scan.Error, scan.CycleStatus) + } + if len(events) != 1 || events[0].Type != model.EventScanInterrupted { + t.Fatalf("interrupted resumable scan events = %#v", events) + } + if count := outboxCount(t, s); count != 0 { + t.Fatalf("an interrupted resumable scan queued %d notifications", count) + } +} + +func TestCancelRequestOutlivesProgressAndEndsAtFinalization(t *testing.T) { + t.Parallel() + a := &App{} + scope := store.DefaultTenantScope() + canceled := false + run := &activeRun{scan: model.ActiveScan{ID: "scan", Phase: "scanning"}, cancel: func() { canceled = true }} + a.registerRun(scope.ID(), "scan", run) + if err := a.CancelScan(scope, "scan"); err != nil || !canceled { + t.Fatalf("CancelScan = %v, canceled %v", err, canceled) + } + // Nmap and Naabu report one last progress update when their process is + // stopped. The phase follows it; the cancellation request stays visible. + a.updateActiveProgress("scan", scanner.Progress{Phase: "scanning"}) + if got := run.snapshot(); !got.CancelRequested || got.Phase != "scanning" { + t.Fatalf("after progress = cancel_requested %v phase %q", got.CancelRequested, got.Phase) + } + a.beginActiveFinalization("scan") + if got := run.snapshot(); !got.CancelRequested || got.Phase != "finalizing" { + t.Fatalf("after finalization began = cancel_requested %v phase %q", got.CancelRequested, got.Phase) + } + if err := a.CancelScan(scope, "scan"); !errors.Is(err, ErrScanFinalizing) { + t.Fatalf("CancelScan during finalization = %v, want ErrScanFinalizing", err) + } +} + +func TestInterruptedByShutdownNeedsAStoppedRunAndNoCancelRequest(t *testing.T) { + t.Parallel() + live := context.Background() + stopped := canceledContext() + if (&activeRun{}).interruptedByShutdown(live, stopped) { + t.Fatal("a scan canceled while the run context is live was read as a shutdown") + } + if !(&activeRun{}).interruptedByShutdown(stopped, stopped) { + t.Fatal("a scan stopped with its run context was not read as a shutdown") + } + if (&activeRun{scan: model.ActiveScan{CancelRequested: true}}).interruptedByShutdown(stopped, stopped) { + t.Fatal("a requested cancellation during shutdown was read as a shutdown") + } + var missing *activeRun + if !missing.interruptedByShutdown(stopped, stopped) { + t.Fatal("a scan without an active run was not read as a shutdown") + } + timedOut, cancel := context.WithTimeout(live, 0) + defer cancel() + <-timedOut.Done() + if (&activeRun{}).interruptedByShutdown(stopped, timedOut) { + t.Fatal("a timed-out scan was read as a shutdown") + } +} diff --git a/internal/app/units.go b/internal/app/units.go index 71f7b251..855f8aab 100644 --- a/internal/app/units.go +++ b/internal/app/units.go @@ -166,9 +166,17 @@ func (a *App) registerRun(tenantID, id string, run *activeRun) { func (run *activeRun) requestCancel() { run.mu.Lock() defer run.mu.Unlock() + run.requestCancelLocked() +} + +// requestCancelLocked cancels the run and records that the cancellation was +// requested, so the scan is reported as canceled rather than interrupted. +// The caller holds run.mu. +func (run *activeRun) requestCancelLocked() { if run.cancel != nil { run.cancel() run.scan.Phase = "cancelling" + run.scan.CancelRequested = true } } diff --git a/internal/engine/coverage_extra_test.go b/internal/engine/coverage_extra_test.go index 6e934dff..f1b48893 100644 --- a/internal/engine/coverage_extra_test.go +++ b/internal/engine/coverage_extra_test.go @@ -19,6 +19,7 @@ func TestProcessFailureAndOutcomeMessageVariants(t *testing.T) { status string resumable bool cycleStatus string + interrupted bool wantEvent string wantMessage string wantFailures int @@ -26,6 +27,8 @@ func TestProcessFailureAndOutcomeMessageVariants(t *testing.T) { {name: "paused", status: "failed", resumable: true, cycleStatus: "paused", wantEvent: "scan-paused", wantMessage: "Scan failed: nmap stopped", wantFailures: 0}, {name: "paused canceled", status: "canceled", resumable: true, cycleStatus: "paused", wantEvent: "scan-canceled", wantMessage: "Scan canceled: detail", wantFailures: 0}, {name: "canceled", status: "canceled", wantEvent: "scan-canceled", wantMessage: "Scan canceled: detail", wantFailures: 0}, + {name: "interrupted", status: "canceled", interrupted: true, wantEvent: model.EventScanInterrupted, wantMessage: "Detail", wantFailures: 0}, + {name: "interrupted paused", status: "canceled", resumable: true, cycleStatus: "paused", interrupted: true, wantEvent: model.EventScanInterrupted, wantMessage: "Detail", wantFailures: 0}, {name: "empty status", wantEvent: "scan-failure", wantMessage: "Scan failed: detail", wantFailures: 1}, {name: "timed out", status: "timeout", wantEvent: "scan-failure", wantMessage: "Scan timed out: detail", wantFailures: 1}, {name: "custom status", status: "provider_error", wantEvent: "scan-failure", wantMessage: "Scan provider error: detail", wantFailures: 1}, @@ -33,7 +36,7 @@ func TestProcessFailureAndOutcomeMessageVariants(t *testing.T) { for _, test := range cases { t.Run(test.name, func(t *testing.T) { state := model.JobState{} - scan := model.Scan{Status: test.status, Resumable: test.resumable, CycleStatus: test.cycleStatus, Error: "detail"} + scan := model.Scan{Status: test.status, Resumable: test.resumable, CycleStatus: test.cycleStatus, Error: "detail", Interrupted: test.interrupted} if test.name == "paused" { scan.Error = "nmap stopped" } diff --git a/internal/engine/engine.go b/internal/engine/engine.go index 3b08bb34..9c418776 100644 --- a/internal/engine/engine.go +++ b/internal/engine/engine.go @@ -1314,6 +1314,16 @@ func (e *Engine) FailureForJobWithDestinations(ctx context.Context, jobID, job s func processFailure(state *model.JobState, job string, scan model.Scan) ([]model.Event, error) { clearUnconfirmedTotalLoss(state) + if scan.Interrupted { + // A restart is not an operator's cancellation. Keep it in the + // activity history without notifying anyone; the job-silence alert + // still reports a daemon that keeps stopping. + message := "Scan interrupted" + if reason := strings.TrimSpace(scan.Error); reason != "" { + message = sanitizeNotificationText(strings.ToUpper(reason[:1]) + reason[1:]) + } + return []model.Event{{Type: model.EventScanInterrupted, Job: job, ScanID: scan.ID, Message: message, CreatedAt: scan.FinishedAt}}, nil + } if scan.Resumable && scan.CycleStatus == "paused" { if scan.Status == "canceled" { return []model.Event{{Type: "scan-canceled", Job: job, ScanID: scan.ID, Message: scanOutcomeMessage(scan), CreatedAt: scan.FinishedAt}}, nil diff --git a/internal/model/model.go b/internal/model/model.go index 8c76f7c9..ce7ed72a 100644 --- a/internal/model/model.go +++ b/internal/model/model.go @@ -190,6 +190,10 @@ type Scan struct { BaselineConfigHash string `json:"baseline_config_hash,omitempty"` Changes []Change `json:"changes,omitempty"` Snapshot Snapshot `json:"snapshot"` + // Interrupted marks a canceled scan that stopped because the daemon + // stopped, not because someone canceled it. It decides the scan's event + // and is not stored. + Interrupted bool `json:"-"` } // ScanSummary is the metadata needed for history and dashboard lists. Full @@ -234,45 +238,48 @@ type ScanSummary struct { // executing. It intentionally contains metadata only; the result is not // persisted until the scanner reaches a terminal state. type ActiveScan struct { - ID string `json:"id"` - JobID string `json:"job_id,omitempty"` - Job string `json:"job"` - JobRevision int64 `json:"job_revision,omitempty"` - StartedAt time.Time `json:"started_at"` - EstimatedProbes int64 `json:"estimated_probes,omitempty"` - NmapInvocations int64 `json:"nmap_invocations,omitempty"` - EstimatedSeconds int64 `json:"estimated_seconds,omitempty"` - CompletedProbes int64 `json:"completed_probes,omitempty"` - TotalProbes int64 `json:"total_probes,omitempty"` - CompletedInvocations int64 `json:"completed_invocations,omitempty"` - TotalInvocations int64 `json:"total_invocations,omitempty"` - ProgressPercent int `json:"progress_percent"` - Phase string `json:"phase,omitempty"` - Protocol string `json:"protocol,omitempty"` - Scanner string `json:"scanner,omitempty"` - ScannerProfileID string `json:"scanner_profile_id,omitempty"` - ScannerProfileRevision int64 `json:"scanner_profile_revision,omitempty"` - DiscoveryPortsFound int `json:"discovery_ports_found,omitempty"` - DiscoveryAddresses int `json:"discovery_addresses,omitempty"` - DiscoveryDurationMS int64 `json:"discovery_duration_ms,omitempty"` - EnrichmentDurationMS int64 `json:"enrichment_duration_ms,omitempty"` - CurrentInvocation int64 `json:"current_invocation,omitempty"` - TotalBatches int64 `json:"total_batches,omitempty"` - ProcessProgressPercent int `json:"process_progress_percent,omitempty"` - ElapsedSeconds int64 `json:"elapsed_seconds,omitempty"` - LastOutput string `json:"last_output,omitempty"` - ProcessAlive bool `json:"process_alive"` - CycleID string `json:"cycle_id,omitempty"` - CycleAttempt int `json:"cycle_attempt,omitempty"` - CycleStatus string `json:"cycle_status,omitempty"` - CycleCompletedProbes int64 `json:"cycle_completed_probes,omitempty"` - CycleTotalProbes int64 `json:"cycle_total_probes,omitempty"` - CycleCompletedUnits int `json:"cycle_completed_units,omitempty"` - CycleTotalUnits int `json:"cycle_total_units,omitempty"` - CycleNoProgressAttempts int `json:"cycle_no_progress_attempts,omitempty"` - CurrentUnit int64 `json:"current_unit,omitempty"` - CurrentUnitPorts string `json:"current_unit_ports,omitempty"` - CurrentUnitAddresses int `json:"current_unit_addresses,omitempty"` + ID string `json:"id"` + JobID string `json:"job_id,omitempty"` + Job string `json:"job"` + JobRevision int64 `json:"job_revision,omitempty"` + StartedAt time.Time `json:"started_at"` + EstimatedProbes int64 `json:"estimated_probes,omitempty"` + NmapInvocations int64 `json:"nmap_invocations,omitempty"` + EstimatedSeconds int64 `json:"estimated_seconds,omitempty"` + CompletedProbes int64 `json:"completed_probes,omitempty"` + TotalProbes int64 `json:"total_probes,omitempty"` + CompletedInvocations int64 `json:"completed_invocations,omitempty"` + TotalInvocations int64 `json:"total_invocations,omitempty"` + ProgressPercent int `json:"progress_percent"` + Phase string `json:"phase,omitempty"` + // CancelRequested stays true once someone asked to cancel the scan, while + // Phase keeps reporting the scanner's progress until the scan ends. + CancelRequested bool `json:"cancel_requested,omitempty"` + Protocol string `json:"protocol,omitempty"` + Scanner string `json:"scanner,omitempty"` + ScannerProfileID string `json:"scanner_profile_id,omitempty"` + ScannerProfileRevision int64 `json:"scanner_profile_revision,omitempty"` + DiscoveryPortsFound int `json:"discovery_ports_found,omitempty"` + DiscoveryAddresses int `json:"discovery_addresses,omitempty"` + DiscoveryDurationMS int64 `json:"discovery_duration_ms,omitempty"` + EnrichmentDurationMS int64 `json:"enrichment_duration_ms,omitempty"` + CurrentInvocation int64 `json:"current_invocation,omitempty"` + TotalBatches int64 `json:"total_batches,omitempty"` + ProcessProgressPercent int `json:"process_progress_percent,omitempty"` + ElapsedSeconds int64 `json:"elapsed_seconds,omitempty"` + LastOutput string `json:"last_output,omitempty"` + ProcessAlive bool `json:"process_alive"` + CycleID string `json:"cycle_id,omitempty"` + CycleAttempt int `json:"cycle_attempt,omitempty"` + CycleStatus string `json:"cycle_status,omitempty"` + CycleCompletedProbes int64 `json:"cycle_completed_probes,omitempty"` + CycleTotalProbes int64 `json:"cycle_total_probes,omitempty"` + CycleCompletedUnits int `json:"cycle_completed_units,omitempty"` + CycleTotalUnits int `json:"cycle_total_units,omitempty"` + CycleNoProgressAttempts int `json:"cycle_no_progress_attempts,omitempty"` + CurrentUnit int64 `json:"current_unit,omitempty"` + CurrentUnitPorts string `json:"current_unit_ports,omitempty"` + CurrentUnitAddresses int `json:"current_unit_addresses,omitempty"` } type Change struct { @@ -394,6 +401,17 @@ type Event struct { TenantID string `json:"-"` } +// EventScanInterrupted records a scan that stopped because the daemon +// stopped. It appears in the activity history but is never delivered: a +// restart is not an operator's cancellation. +const EventScanInterrupted = "scan-interrupted" + +// EventDelivered reports whether events of the given type are queued for the +// job's notification destinations. +func EventDelivered(eventType string) bool { + return eventType != EventScanInterrupted +} + // EventPayloadLimit is the maximum serialized size of a durable event or // notification outbox payload. Change details remain available on the scan // history endpoint; oversized alert events carry a bounded summary instead. @@ -608,8 +626,10 @@ func (s Snapshot) Hash() string { func Fingerprint(name, product, version, extra string, cpes []string) string { parts := []string{name, product, version, extra} - sort.Strings(cpes) - parts = append(parts, cpes...) + // Sort a copy: the caller's slice is often a stored observation's CPEs. + sorted := append([]string(nil), cpes...) + sort.Strings(sorted) + parts = append(parts, sorted...) for i := range parts { parts[i] = strings.TrimSpace(parts[i]) } diff --git a/internal/model/model_test.go b/internal/model/model_test.go index 97bae8ee..66f6ef32 100644 --- a/internal/model/model_test.go +++ b/internal/model/model_test.go @@ -203,11 +203,25 @@ func TestSnapshotNormalizeCanonicalizesAndSortsHostStates(t *testing.T) { } } +func TestEventDeliveredSkipsOnlyScanInterruptions(t *testing.T) { + for _, eventType := range []string{"scan-canceled", "scan-failure", "changes-detected", "job-silent"} { + if !EventDelivered(eventType) { + t.Errorf("EventDelivered(%q) = false", eventType) + } + } + if EventDelivered(EventScanInterrupted) { + t.Error("a scan interruption would be delivered") + } +} + func TestFingerprintAndChangeSummary(t *testing.T) { cpes := []string{"cpe:/b", "cpe:/a"} if got := Fingerprint(" ssh ", " OpenSSH", " 9", " Linux ", cpes); got != "ssh | OpenSSH | 9 | Linux | cpe:/a | cpe:/b" { t.Fatalf("fingerprint = %q", got) } + if cpes[0] != "cpe:/b" || cpes[1] != "cpe:/a" { + t.Fatalf("Fingerprint reordered the caller's CPEs to %v", cpes) + } if got := ChangeSummary(Change{Kind: "dns-added", Target: "router", New: "192.0.2.1"}); got != "router dns-added: 192.0.2.1" { t.Fatalf("DNS summary = %q", got) } diff --git a/internal/store/incident_actions.go b/internal/store/incident_actions.go index 82ece02e..e388864e 100644 --- a/internal/store/incident_actions.go +++ b/internal/store/incident_actions.go @@ -505,7 +505,7 @@ func acceptedServiceForEvidence(snapshot *model.Snapshot, protocol string, port // and cause an unchanged next scan to look like a service change. fingerprint = strings.TrimSpace(service.Product) case "probed": - fingerprint = strings.TrimSpace(model.Fingerprint(service.Name, service.Product, service.Version, service.ExtraInfo, append([]string(nil), service.CPEs...))) + fingerprint = strings.TrimSpace(model.Fingerprint(service.Name, service.Product, service.Version, service.ExtraInfo, service.CPEs)) default: // Nmap's table-only service guesses are descriptive metadata; the // scanner does not include them in the baseline fingerprint. diff --git a/internal/store/runtime.go b/internal/store/runtime.go index 3076def3..1014dc34 100644 --- a/internal/store/runtime.go +++ b/internal/store/runtime.go @@ -988,6 +988,9 @@ func queueEventsTx(ctx context.Context, tx *sql.Tx, events []model.Event, destin resolved := make(map[intent]managedIntent, len(destinations)) var discarded managedIntentDiscards for _, event := range events { + if !model.EventDelivered(event.Type) { + continue + } bounded, payload, err := model.MarshalBoundedEvent(event, model.EventPayloadLimit) if err != nil { return err diff --git a/internal/web/scan_handlers.go b/internal/web/scan_handlers.go index f35b6fab..78dce12c 100644 --- a/internal/web/scan_handlers.go +++ b/internal/web/scan_handlers.go @@ -63,6 +63,10 @@ func (s *Server) cancelScan(w http.ResponseWriter, r *http.Request, session stor writeError(w, http.StatusConflict, "scan_not_active", "scan is no longer active", nil) return } + if errors.Is(err, app.ErrScanFinalizing) { + writeError(w, http.StatusConflict, "scan_finalizing", "The scan is saving its result and can no longer be canceled.", nil) + return + } s.writeInternalError(w, r, "cancel_failed", err) return } diff --git a/src/main.test.tsx b/src/main.test.tsx index 81c4d7d1..5d05c7f6 100644 --- a/src/main.test.tsx +++ b/src/main.test.tsx @@ -171,6 +171,10 @@ describe('application shell', () => { expect(invalidate).toHaveBeenCalledWith({ queryKey: ['active-scans'] }) expect(invalidate).toHaveBeenCalledWith({ queryKey: ['jobs'] }) expect(invalidate).toHaveBeenCalledWith({ queryKey: ['job', 'job-9'] }) + invalidate.mockClear() + act(() => stream.emit('scan-interrupted', 'job-9')) + expect(invalidate).toHaveBeenCalledWith({ queryKey: ['activity-events'] }) + expect(invalidate).not.toHaveBeenCalledWith() act(() => stream.emit('scan.skipped', 'job-9', 'paused')) expect(skipped).toHaveBeenCalledWith(expect.objectContaining({ detail: { job_id: 'job-9', reason: 'paused' } })) expect(invalidate).toHaveBeenCalledWith({ queryKey: ['active-scans'] }) diff --git a/src/main.tsx b/src/main.tsx index 97e946ad..187f62ce 100644 --- a/src/main.tsx +++ b/src/main.tsx @@ -142,6 +142,7 @@ export function Shell({ displayName, role, permissions, onLogout, unit }: { disp case 'scan-failure': case 'scan-incomplete': case 'scan-canceled': + case 'scan-interrupted': case 'scan-anomaly': void client.invalidateQueries({ queryKey: ['jobs'] }) void client.invalidateQueries({ queryKey: ['active-scans'] }) diff --git a/src/pages/Activity.test.tsx b/src/pages/Activity.test.tsx index 3c9a557a..683ad92e 100644 --- a/src/pages/Activity.test.tsx +++ b/src/pages/Activity.test.tsx @@ -65,14 +65,17 @@ describe('activity history', () => { { type: 'scan-failure', job_id: 'job-1', job: 'Production', message: 'Scan failed', created_at: '2026-09-20T12:00:00Z' }, { type: 'scan-anomaly', job_id: 'job-1', job: 'Production', message: 'Coverage was incomplete', created_at: '2026-09-20T11:00:00Z' }, { type: 'scan-canceled', job_id: 'job-1', job: 'Production', message: 'Scan canceled', created_at: '2026-09-20T10:00:00Z' }, - ], pagination: { ...page, total: 3 } }) + { type: 'scan-interrupted', job_id: 'job-1', job: 'Production', message: 'Scan interrupted because EdgeWatch stopped', created_at: '2026-09-20T09:00:00Z' }, + ], pagination: { ...page, total: 4 } }) renderWithProviders() - await waitFor(() => expect(document.querySelectorAll('.activity-event')).toHaveLength(3)) + await waitFor(() => expect(document.querySelectorAll('.activity-event')).toHaveLength(4)) const events = Array.from(document.querySelectorAll('.activity-event')) expect(events[0]).toHaveClass('failure') expect(events[1]).toHaveClass('warning') expect(events[2].querySelector('.activity-event-heading strong')).toHaveTextContent('Scan canceled') + expect(events[3].querySelector('.activity-event-heading strong')).toHaveTextContent('Scan interrupted') + expect(events[3]).not.toHaveClass('failure') expect(screen.queryByText('Scan cancelled', { exact: true })).not.toBeInTheDocument() }) diff --git a/src/pages/Activity.tsx b/src/pages/Activity.tsx index 1cd220c8..82fa4235 100644 --- a/src/pages/Activity.tsx +++ b/src/pages/Activity.tsx @@ -20,6 +20,7 @@ const eventLabels: Record = { 'scan-incomplete': 'Scan incomplete', 'scan-failure': 'Scan failed', 'scan-canceled': 'Scan canceled', + 'scan-interrupted': 'Scan interrupted', 'scan-anomaly': 'Scan anomaly', 'application-update-available': 'Update available', 'application-updated': 'Application updated', diff --git a/src/pages/Dashboard.test.tsx b/src/pages/Dashboard.test.tsx index 2b6b9b30..8a28f30e 100644 --- a/src/pages/Dashboard.test.tsx +++ b/src/pages/Dashboard.test.tsx @@ -190,6 +190,13 @@ describe('dashboard', () => { expect(container.textContent).toContain('1 destination tested') }) + it('confirms a requested cancellation while the scanner keeps reporting progress', async () => { + vi.mocked(activeScans).mockResolvedValue({ scans: [{ ...activeScan, phase: 'scanning', cancel_requested: true }] }) + await renderDashboard() + await vi.waitFor(() => expect(container.querySelector('.active-scan-row .pill')?.textContent).toBe('Cancellation requested'), { timeout: 1000 }) + expect(Array.from(container.querySelectorAll('button')).some(button => button.textContent?.includes('Cancel scan'))).toBe(false) + }) + it('shows queued runs reported by the server without offering a cancel action', async () => { vi.mocked(activeScans).mockResolvedValue({ scans: [], queued_runs: [{ job_id: 'job-1', job: 'demo', queued_at: '2026-10-06T08:00:00Z', trigger: 'manual' }] }) await renderDashboard() diff --git a/src/pages/Dashboard.tsx b/src/pages/Dashboard.tsx index 60609a97..bd9983cf 100644 --- a/src/pages/Dashboard.tsx +++ b/src/pages/Dashboard.tsx @@ -118,7 +118,7 @@ function ActiveScanRow({ scan, cancelBusy, onCancel }: { scan: ActiveScan; cance const liveness = scan.process_alive ? ` · ${scanner} process active` : '' const cycle = scan.cycle_id ? `Cycle ${scan.cycle_completed_units ?? 0}/${scan.cycle_total_units ?? '?'} units · ${scan.cycle_status ?? 'running'}` : '' const discovery = scan.scanner === 'naabu_nmap' ? ` · ${scan.discovery_ports_found ?? 0} ports found across ${scan.discovery_addresses ?? 0} hosts` : '' - return
{scan.job}{scanner} · {phase}{protocol} · {completed} of {total} probes · elapsed {elapsed}{batch}{discovery}{liveness}{cycle && {cycle}{scan.current_unit_ports ? ` · current scope ${scan.current_unit_ports} across ${scan.current_unit_addresses ?? 0} host${scan.current_unit_addresses === 1 ? '' : 's'}` : ''}}{scan.process_alive && scan.process_progress_percent !== undefined ? Current {scanner} process: {scan.process_progress_percent}% : null}{scan.last_output ? Latest scanner output: {scan.last_output} : null}
Show scan details
Scanner
{scanner}{scan.scanner_profile_revision ? ` · profile r${scan.scanner_profile_revision}` : ''}
Phase
{phase}{protocol}
Elapsed
{elapsed}
Progress
{completed} of {total} probes ({scan.progress_percent ?? 0}%)
{scan.scanner === 'naabu_nmap' && <>
Discovery
{(scan.discovery_ports_found ?? 0).toLocaleString()} ports found across {(scan.discovery_addresses ?? 0).toLocaleString()} hosts{scan.discovery_duration_ms ? ` · ${formatDurationMS(scan.discovery_duration_ms)}` : ''}
Enrichment
{scan.enrichment_duration_ms ? formatDurationMS(scan.enrichment_duration_ms) : 'In progress'}
}{cycle &&
Resumable cycle
{cycle}
}{scan.current_unit_ports ?
Current work unit
{scan.current_unit_ports} · {scan.current_unit_addresses ?? 0} host{scan.current_unit_addresses === 1 ? '' : 's'}
: null}{scan.current_invocation && scan.total_batches ?
Batch
{scan.current_invocation} of {scan.total_batches}
: null}
Process
{scan.process_alive ? `${scanner} process active${scan.process_progress_percent !== undefined ? ` · ${scan.process_progress_percent}%` : ''}` : 'Waiting for process update'}
{scan.last_output ?
Latest output
{scan.last_output}
: null}
+ return
{scan.job}{scanner} · {phase}{protocol} · {completed} of {total} probes · elapsed {elapsed}{batch}{discovery}{liveness}{cycle && {cycle}{scan.current_unit_ports ? ` · current scope ${scan.current_unit_ports} across ${scan.current_unit_addresses ?? 0} host${scan.current_unit_addresses === 1 ? '' : 's'}` : ''}}{scan.process_alive && scan.process_progress_percent !== undefined ? Current {scanner} process: {scan.process_progress_percent}% : null}{scan.last_output ? Latest scanner output: {scan.last_output} : null}
Show scan details
Scanner
{scanner}{scan.scanner_profile_revision ? ` · profile r${scan.scanner_profile_revision}` : ''}
Phase
{phase}{protocol}
Elapsed
{elapsed}
Progress
{completed} of {total} probes ({scan.progress_percent ?? 0}%)
{scan.scanner === 'naabu_nmap' && <>
Discovery
{(scan.discovery_ports_found ?? 0).toLocaleString()} ports found across {(scan.discovery_addresses ?? 0).toLocaleString()} hosts{scan.discovery_duration_ms ? ` · ${formatDurationMS(scan.discovery_duration_ms)}` : ''}
Enrichment
{scan.enrichment_duration_ms ? formatDurationMS(scan.enrichment_duration_ms) : 'In progress'}
}{cycle &&
Resumable cycle
{cycle}
}{scan.current_unit_ports ?
Current work unit
{scan.current_unit_ports} · {scan.current_unit_addresses ?? 0} host{scan.current_unit_addresses === 1 ? '' : 's'}
: null}{scan.current_invocation && scan.total_batches ?
Batch
{scan.current_invocation} of {scan.total_batches}
: null}
Process
{scan.process_alive ? `${scanner} process active${scan.process_progress_percent !== undefined ? ` · ${scan.process_progress_percent}%` : ''}` : 'Waiting for process update'}
{scan.last_output ?
Latest output
{scan.last_output}
: null}
{scan.cancel_requested || scan.phase === 'cancelling' ? Cancellation requested : }
} function QueuedRunRow({ run }: { run: QueuedRun }) { diff --git a/src/pages/JobDetail.actions.test.tsx b/src/pages/JobDetail.actions.test.tsx index b8243766..b7ec565f 100644 --- a/src/pages/JobDetail.actions.test.tsx +++ b/src/pages/JobDetail.actions.test.tsx @@ -169,6 +169,16 @@ describe('job detail actions', () => { expect(await screen.findByText('Cancellation requested')).toBeInTheDocument() }) + it('keeps a requested cancellation visible after the scanner reports progress again', async () => { + vi.mocked(activeScans).mockResolvedValue({ scans: [{ ...activeScan, phase: 'finalizing', cancel_requested: true }] } as never) + renderPage() + + expect(await screen.findByRole('heading', { name: 'Scan in progress' })).toBeInTheDocument() + expect(screen.getByText(/· finalizing/)).toBeInTheDocument() + expect(screen.getByText('Cancellation requested')).toBeInTheDocument() + expect(screen.queryByRole('button', { name: 'Cancel scan' })).not.toBeInTheDocument() + }) + it('clears the accepted state after the completed scan appears in history', async () => { const { client } = renderPage() await screen.findByRole('button', { name: /scan-1/i }) diff --git a/src/pages/JobDetail.tsx b/src/pages/JobDetail.tsx index 98379c32..c09b3a91 100644 --- a/src/pages/JobDetail.tsx +++ b/src/pages/JobDetail.tsx @@ -679,7 +679,9 @@ function JobScanStatus({ scan, queuedRun, cancelBusy, onCancel }: { const total = scan.total_probes || scan.estimated_probes || 0 const completed = scan.completed_probes ?? 0 const progress = Math.max(0, Math.min(100, scan.progress_percent ?? 0)) - const cancelling = scan.phase === 'cancelling' + // The phase keeps following the scanner after a cancel request; the flag + // stays set until the scan ends. Older servers only report the phase. + const cancelling = scan.cancel_requested || scan.phase === 'cancelling' return
diff --git a/src/types.ts b/src/types.ts index 09e74a57..9ffd3e36 100644 --- a/src/types.ts +++ b/src/types.ts @@ -16,7 +16,7 @@ export type ActiveScan = { id: string; job_id?: string; job: string; job_revision?: number; started_at: string estimated_probes?: number; nmap_invocations?: number; estimated_seconds?: number completed_probes?: number; total_probes?: number; completed_invocations?: number - total_invocations?: number; progress_percent: number; phase?: string; protocol?: string + total_invocations?: number; progress_percent: number; phase?: string; cancel_requested?: boolean; protocol?: string current_invocation?: number; total_batches?: number; process_progress_percent?: number elapsed_seconds?: number; last_output?: string; process_alive?: boolean cycle_id?: string; cycle_attempt?: number; cycle_status?: string; cycle_completed_probes?: number; cycle_total_probes?: number