Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 3 additions & 2 deletions docs/src/content/docs/user-guide/notifications.md
Original file line number Diff line number Diff line change
Expand Up @@ -66,8 +66,9 @@ pause uses none of their retries and is not reported as a delivery failure.

## Delivery retries and health

Scan changes, scan failures, cancellations, timeouts, stalled cycles, and
recovery events can all generate notifications. A scan that stops because
Scan changes, scan failures, cancellations, timeouts, stalled cycles,
scheduled runs skipped because they exceed the probe budget, and recovery
events can all generate notifications. A scan that stops because
EdgeWatch stopped, for example during an upgrade or restart, is recorded as
canceled with the reason "scan interrupted because EdgeWatch stopped" and
appears in Activity as **Scan interrupted**, but sends no notification. A
Expand Down
11 changes: 10 additions & 1 deletion docs/src/content/docs/user-guide/scanning.md
Original file line number Diff line number Diff line change
Expand Up @@ -62,7 +62,16 @@ executables.

Full-range scans are deliberately bounded by scheduler probe budgets. A scan
that runs as a single invocation resolves DNS again when it starts, and the
budget is checked against that resolution before any scanner runs. A broad
budget is checked against that resolution before any scanner runs.

A job whose estimated work exceeds its unit's budget needs an administrator's
high-cost approval. **Scan now** on such a job is refused with an error. A
scheduled run is skipped before it starts and is reported in Activity as
**Scheduled scan skipped**, with a notification to the job's destinations. It
is reported once for each scope and budget, and again after a scan of the job
has run. The job page shows when a job exceeds its budget and whether an
approval would let it run. An operator's change to a job's targets, ports, or
scanner clears its high-cost approval; the scope-change confirmation says so. A broad
scan may be split into resumable address, discovery, enrichment, and UDP work
units. A timeout or restart preserves completed work for the configured resume
window; partial work cannot change a baseline. The dashboard shows scanner
Expand Down
46 changes: 46 additions & 0 deletions internal/app/app.go
Original file line number Diff line number Diff line change
Expand Up @@ -1769,6 +1769,11 @@ func (a *App) startManagedScheduled(ctx context.Context, scope store.TenantScope
a.Logger.Info("scheduled run skipped because the job's tenant is not active", "job", record.Job.Name)
return
}
var budgetErr *ScanWorkBudgetError
if scan.ID == "" && errors.As(runErr, &budgetErr) {
a.recordScheduledBudgetSkip(ctx, scope, record, budgetErr)
return
}
if runErr != nil {
a.Logger.Error("scan failed", "job", record.Job.Name, "scan_id", scan.ID, "error", runErr)
return
Expand All @@ -1777,6 +1782,47 @@ func (a *App) startManagedScheduled(ctx context.Context, scope store.TenantScope
})
}

// recordScheduledBudgetSkip reports a scheduled run that its probe budget
// stopped before a scan record existed. A manual run reports the refusal to
// the person who started it; nobody sees a scheduled one, so it becomes an
// activity event delivered to the job's destinations. It is reported once for
// each scope and budget, until a scan of the job runs again.
func (a *App) recordScheduledBudgetSkip(ctx context.Context, scope store.TenantScope, record store.JobRecord, budgetErr *ScanWorkBudgetError) {
a.Logger.Warn("scheduled run skipped because it exceeds the probe budget", "job", record.Job.Name, "estimated_probes", budgetErr.Estimate.Probes, "budget", budgetErr.Budget)
var destinations []string
if a.Notifier != nil {
var err error
destinations, err = a.Notifier.Tenant(a.Store.Tenant(scope)).QueueDestinationsForJob(ctx, record.Job)
if err != nil {
// Like the silence watchdog, record nothing rather than an alert
// without its deliveries; the next scheduled run tries again.
a.Logger.Warn("budget skip notification destinations unavailable", "job", record.Job.Name, "error", err)
return
}
}
hash := record.Job.SecurityHash()
key := fmt.Sprintf("%s|%d", hash, budgetErr.Budget)
message := fmt.Sprintf("Scheduled scan skipped: about %d probes exceed the probe budget of %d. An administrator must approve high-cost scans for this job, or its scope must be reduced.", budgetErr.Estimate.Probes, budgetErr.Budget)
if record.Job.AllowHighCost {
message = fmt.Sprintf("Scheduled scan skipped: about %d probes exceed the high-cost limit of %d. Reduce the job's scope.", budgetErr.Estimate.Probes, budgetErr.Budget)
}
events, err := a.Store.System().UpdateRuntimeForScanWithOutbox(ctx, record.ID, hash, destinations, func(state *model.JobState) ([]model.Event, error) {
if state.BudgetSkipAlertKey == key {
return nil, nil
}
state.BudgetSkipAlertKey = key
return []model.Event{{Type: model.EventScanBudgetExceeded, JobID: record.ID, Job: record.Job.Name, Message: message, CreatedAt: time.Now().UTC()}}, nil
})
if err != nil {
a.Logger.Warn("budget skip could not be recorded", "job", record.Job.Name, "error", err)
return
}
if len(events) > 0 {
a.emitTenantEvents(scope, events)
a.wakeDelivery()
}
}

func (a *App) startTracked(fn func()) bool {
a.runMu.Lock()
if !a.runAccepting {
Expand Down
114 changes: 114 additions & 0 deletions internal/app/budget_skip_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
package app

import (
"context"
"encoding/json"
"io"
"log/slog"
"testing"
"time"

"github.com/crypt0rr/edgewatch/internal/config"
"github.com/crypt0rr/edgewatch/internal/model"
"github.com/crypt0rr/edgewatch/internal/store"
"github.com/crypt0rr/edgewatch/internal/store/storetest"
)

func TestScheduledBudgetSkipIsRecordedOnceUntilAScanRuns(t *testing.T) {
t.Parallel()
ctx := context.Background()
s, err := store.Open(storetest.FreshPath(t))
if err != nil {
t.Fatal(err)
}
defer s.Close()
cfg := &config.Config{
Version: 1, Database: "test", Retention: config.Duration(24 * time.Hour),
Scheduler: config.Scheduler{MaxConcurrent: 1, MaxProbeCount: 100},
Web: config.Web{Listen: "127.0.0.1:8080"},
Notifications: config.Notifications{URLs: []string{"generic://localhost/edgewatch?disabletls=yes&template=json"}},
}
a, err := New(cfg, s, "missing", slog.New(slog.NewTextHandler(io.Discard, nil)))
if err != nil {
t.Fatal(err)
}
a.Scanner = schedulerFake{}
record, err := defaultTenant(s).CreateJob(ctx, config.NormalizeJob(config.Job{
Name: "over-budget", Schedule: "0 * * * *", Timezone: "UTC", Targets: []string{"192.0.2.1"},
TCP: &config.Protocol{Ports: "1-1000", Mode: "connect", Engine: config.EngineNmap}, Timing: "balanced", Timeout: config.Duration(time.Minute),
}))
if err != nil {
t.Fatal(err)
}
live := make(chan model.Event, 32)
a.SetEventHandler(func(event model.Event) { live <- event })
scheduled := func() {
t.Helper()
a.BeginRun(ctx)
a.startManagedScheduled(ctx, store.DefaultTenantScope(), record.ID)
a.StopRun()
}
budgetEvents := func() (events []model.Event, outbox int) {
t.Helper()
rows, err := s.DB.QueryContext(ctx, `SELECT payload_json FROM events ORDER BY id`)
if err != nil {
t.Fatal(err)
}
defer rows.Close()
for rows.Next() {
var raw []byte
if err := rows.Scan(&raw); err != nil {
t.Fatal(err)
}
var event model.Event
if err := json.Unmarshal(raw, &event); err != nil {
t.Fatal(err)
}
if event.Type == model.EventScanBudgetExceeded {
events = append(events, event)
}
}
if err := rows.Err(); err != nil {
t.Fatal(err)
}
if err := s.DB.QueryRowContext(ctx, `SELECT COUNT(*) FROM outbox WHERE CAST(payload_json AS TEXT) LIKE '%scan-budget-exceeded%'`).Scan(&outbox); err != nil {
t.Fatal(err)
}
return events, outbox
}

scheduled()
events, outbox := budgetEvents()
if len(events) != 1 || outbox != 1 || events[0].JobID != record.ID {
t.Fatalf("first skip = %+v (outbox %d), want one delivered event", events, outbox)
}
if want := "Scheduled scan skipped: about 1000 probes exceed the probe budget of 100. An administrator must approve high-cost scans for this job, or its scope must be reduced."; events[0].Message != want {
t.Fatalf("skip message = %q, want %q", events[0].Message, want)
}
var sawLive bool
for len(live) > 0 {
if event := <-live; event.Type == model.EventScanBudgetExceeded {
sawLive = true
}
}
if !sawLive {
t.Fatal("the budget skip was not published as a live update")
}

// The same scope and budget are not reported again.
scheduled()
if events, outbox := budgetEvents(); len(events) != 1 || outbox != 1 {
t.Fatalf("repeated skip = %d events, %d deliveries, want 1 and 1", len(events), outbox)
}

// Once a scan runs, a later skip is reported again.
a.Config.Scheduler.MaxProbeCount = 0
if _, _, err := a.RunJobRecord(ctx, record); err != nil {
t.Fatal(err)
}
a.Config.Scheduler.MaxProbeCount = 100
scheduled()
if events, outbox := budgetEvents(); len(events) != 2 || outbox != 2 {
t.Fatalf("skip after a scan ran = %d events, %d deliveries, want 2 and 2", len(events), outbox)
}
}
3 changes: 3 additions & 0 deletions internal/engine/engine.go
Original file line number Diff line number Diff line change
Expand Up @@ -57,6 +57,9 @@ func (e *Engine) FinalizeManagedScanWithOptions(ctx context.Context, jobID strin
return nil, fmt.Errorf("scan is required")
}
return e.Store.System().FinalizeManagedScanWithOptions(ctx, scan, jobID, scan.ConfigHash, destinations, options, func(state *model.JobState, current *model.Scan, reminderSettings store.IncidentReminderSettings) ([]model.Event, error) {
// The scan passed its probe budget when it started, so a later
// budget skip is reported again.
state.BudgetSkipAlertKey = ""
if current.Status == "success" {
MarkIncompleteScan(current)
}
Expand Down
8 changes: 8 additions & 0 deletions internal/model/model.go
Original file line number Diff line number Diff line change
Expand Up @@ -364,6 +364,10 @@ type JobState struct {
// emitted for this job. It lives in runtime state so cadence survives
// restarts without affecting scan snapshots or baseline hashes.
LastIncidentReminderAt *time.Time `json:"last_incident_reminder_at,omitempty"`
// BudgetSkipAlertKey names the scope and probe budget of the last
// scheduled run that the budget stopped before it started. A skip with
// the same key is not reported again; any scan that runs clears it.
BudgetSkipAlertKey string `json:"budget_skip_alert_key,omitempty"`
}

// QueuedRun describes an accepted manual or scheduled run that has not yet
Expand Down Expand Up @@ -412,6 +416,10 @@ func EventDelivered(eventType string) bool {
return eventType != EventScanInterrupted
}

// EventScanBudgetExceeded records a scheduled run that its probe budget
// stopped before it started.
const EventScanBudgetExceeded = "scan-budget-exceeded"

// EventPayloadLimit is the maximum serialized size of a durable event or
// notification outbox payload. Change details remain available on the scan
// history endpoint; oversized alert events carry a bounded summary instead.
Expand Down
109 changes: 109 additions & 0 deletions internal/web/high_cost_visibility_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,109 @@
package web

import (
"context"
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"testing"

"github.com/crypt0rr/edgewatch/internal/config"
"github.com/crypt0rr/edgewatch/internal/store"
)

func TestOperatorScopeEditShowsTheClearedHighCostApproval(t *testing.T) {
t.Parallel()
ctx := context.Background()
server, db, admin := newUsersTestServer(t)
server.App.Config.Scheduler.MaxProbeCount = 100
operator := admin
operator.Role = store.RoleOperator
job := config.NormalizeJob(config.Job{
Name: "approved-broad", Schedule: "0 * * * *", Timezone: "UTC", Targets: []string{"192.0.2.1"},
TCP: &config.Protocol{Ports: "1-1000", Mode: "connect", Engine: config.EngineNmap}, AllowHighCost: true,
})
record, err := defaultTenant(db).CreateJob(ctx, job)
if err != nil {
t.Fatal(err)
}
serve := func(session store.Session, method, body string) *httptest.ResponseRecorder {
req := httptest.NewRequest(method, "/api/v1/jobs/"+record.ID, strings.NewReader(body))
if body != "" {
req.Header.Set("Content-Type", "application/json")
}
rec := httptest.NewRecorder()
server.jobRoute(rec, req, session, defaultTenantStore(server), record.ID)
return rec
}
type response struct {
Job struct {
AllowHighCost bool `json:"allow_high_cost"`
} `json:"job"`
ScanBudget *struct {
Exceeded bool `json:"exceeded"`
EstimatedProbes int64 `json:"estimated_probes"`
Limit int64 `json:"limit"`
ApprovalWouldFit bool `json:"approval_would_fit"`
} `json:"scan_budget"`
Cleared bool `json:"high_cost_approval_cleared"`
}
decode := func(rec *httptest.ResponseRecorder) response {
t.Helper()
var value response
if err := json.Unmarshal(rec.Body.Bytes(), &value); err != nil {
t.Fatalf("decode %s: %v", rec.Body.String(), err)
}
return value
}

approved := serve(admin, http.MethodGet, "")
if got := decode(approved); approved.Code != http.StatusOK || got.ScanBudget == nil || got.ScanBudget.Exceeded {
t.Fatalf("approved job = %d %s, want a budget that fits", approved.Code, approved.Body.String())
}

edit := fromConfig(record.Job)
edit.Revision = record.Revision
edit.TCP.Ports = "1-999"
body, err := json.Marshal(edit)
if err != nil {
t.Fatal(err)
}
prompt := serve(operator, http.MethodPut, string(body))
if prompt.Code != http.StatusConflict || !strings.Contains(prompt.Body.String(), "high-cost approval: cleared; an administrator must approve the new scope again (about 999 probes exceed the budget of 100") {
t.Fatalf("operator scope edit = %d %s, want the cleared approval in the confirmation", prompt.Code, prompt.Body.String())
}

edit.ConfirmRebaseline = true
body, err = json.Marshal(edit)
if err != nil {
t.Fatal(err)
}
saved := serve(operator, http.MethodPut, string(body))
got := decode(saved)
if saved.Code != http.StatusOK || !got.Cleared || got.Job.AllowHighCost {
t.Fatalf("confirmed operator edit = %d %s, want the approval reported as cleared", saved.Code, saved.Body.String())
}
if got.ScanBudget == nil || !got.ScanBudget.Exceeded || got.ScanBudget.EstimatedProbes != 999 || got.ScanBudget.Limit != 100 || !got.ScanBudget.ApprovalWouldFit {
t.Fatalf("budget after the edit = %+v", got.ScanBudget)
}

// A routine edit that keeps the scope keeps the approval and says nothing.
reloaded, err := defaultTenant(db).GetJob(ctx, record.ID)
if err != nil {
t.Fatal(err)
}
if reloaded.Job.AllowHighCost {
t.Fatal("the approval survived the operator's scope change")
}
routine := fromConfig(reloaded.Job)
routine.Revision = reloaded.Revision
routine.Schedule = "30 * * * *"
body, err = json.Marshal(routine)
if err != nil {
t.Fatal(err)
}
if rec := serve(operator, http.MethodPut, string(body)); rec.Code != http.StatusOK || decode(rec).Cleared {
t.Fatalf("routine edit = %d %s", rec.Code, rec.Body.String())
}
}
33 changes: 32 additions & 1 deletion internal/web/job_handlers.go
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ import (
"strings"
"time"

"github.com/crypt0rr/edgewatch/internal/app"
"github.com/crypt0rr/edgewatch/internal/auth"
"github.com/crypt0rr/edgewatch/internal/config"
"github.com/crypt0rr/edgewatch/internal/model"
Expand Down Expand Up @@ -170,7 +171,37 @@ func jobJSONFromStateSummary(record store.JobRecord, summary store.RuntimeStateS

func (s *Server) jobJSONWithCycle(ctx context.Context, ts *store.TenantStore, record store.JobRecord, state model.JobState) map[string]any {
value := s.addNotificationRouting(ctx, s.tenantNotifier(ts), jobJSON(record, state))
return s.addJobCycleAndProfile(ctx, ts, record, value)
value = s.addJobCycleAndProfile(ctx, ts, record, value)
if budget, ok := s.jobScanBudget(ctx, ts, record.Job); ok {
value["scan_budget"] = budget
}
return value
}

// jobScanBudget reports whether the job's estimated work fits the probe
// budget of its unit, so the job page can explain why its scheduled runs are
// skipped. approval_would_fit says whether an administrator's high-cost
// approval would let it run. It is left out when the budget cannot be read.
func (s *Server) jobScanBudget(ctx context.Context, ts *store.TenantStore, job config.Job) (map[string]any, bool) {
if s.App == nil {
return nil, false
}
_, err := s.App.CheckScanWorkBudget(ctx, ts, job)
var budgetErr *app.ScanWorkBudgetError
switch {
case err == nil:
return map[string]any{"exceeded": false}, true
case !errors.As(err, &budgetErr):
return nil, false
}
budget := map[string]any{"exceeded": true, "estimated_probes": budgetErr.Estimate.Probes, "limit": budgetErr.Budget, "approval_would_fit": false}
if !job.AllowHighCost {
approved := job
approved.AllowHighCost = true
_, approvedErr := s.App.CheckScanWorkBudget(ctx, ts, approved)
budget["approval_would_fit"] = approvedErr == nil
}
return budget, true
}

type pendingChangeView struct {
Expand Down
Loading
Loading