-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathvalidate.go
More file actions
102 lines (95 loc) · 2.73 KB
/
Copy pathvalidate.go
File metadata and controls
102 lines (95 loc) · 2.73 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
package main
import (
"context"
"sync/atomic"
"github.com/betterleaks/betterleaks/detect"
"github.com/betterleaks/betterleaks/sources"
"github.com/git-pkgs/history"
)
// blobSource adapts history.WalkBlobs to betterleaks' sources.Source so the
// detector's Run pipeline (which owns the async validation pool) can drive
// the scan.
type blobSource struct {
repo *history.Repo
pf prefilter
paths map[string][]string
skip sources.SkipFunc
scanned, skipped *atomic.Int64
total *int64
}
const attrBlobOID = "git-pkgs.blob-oid"
func (s *blobSource) Fragments(ctx context.Context, yield sources.FragmentsFunc) error {
total, err := s.repo.WalkBlobs(history.BlobOptions{
Workers: workers,
Limit: func(string) int64 { return maxBlobSize },
Skip: func(string) { s.skipped.Add(1) },
}, func(b history.Blob) error {
if err := ctx.Err(); err != nil {
return err
}
data, binaryContent := prepareBlob(b.Data)
contentMatched := !binaryContent && s.pf.match(data)
paths := s.paths[b.OID]
if s.paths == nil {
paths = []string{""}
}
scanned := false
for _, path := range paths {
if err := ctx.Err(); err != nil {
return err
}
attributes := map[string]string{attrBlobOID: b.OID}
if path != "" {
attributes[sources.AttrPath] = path
}
if s.skip != nil && s.skip(attributes) {
continue
}
pathMatched := s.pf.pathMatch(path)
if binaryContent && !pathMatched {
continue
}
if !binaryContent && !contentMatched && !pathMatched {
continue
}
if !binaryContent && !scanned {
s.scanned.Add(1)
scanned = true
}
if err := yield(sources.Fragment{Raw: string(data), Attributes: attributes}, nil); err != nil {
return err
}
}
if binaryContent || !scanned {
s.skipped.Add(1)
}
return nil
})
*s.total = int64(total)
return err
}
// detectBlobsRun uses detector.Run so validation is applied.
func detectBlobsRun(ctx context.Context, r *history.Repo, d *detect.Detector, pf prefilter, paths map[string][]string, stats *scanStats) (detectionHits, error) {
var scanned, skipped atomic.Int64
src := &blobSource{
repo: r, pf: pf, paths: paths, skip: d.SkipFunc(),
scanned: &scanned, skipped: &skipped, total: &stats.total,
}
hits := make(detectionHits)
for result := range d.Run(ctx, src) {
if result.Err != nil {
stats.scanned, stats.skipped = scanned.Load(), skipped.Load()
return hits, result.Err
}
key := blobPath{
blob: result.Finding.Attr(attrBlobOID),
path: result.Finding.Attr(sources.AttrPath),
}
hits[key] = append(hits[key], result.Finding)
}
stats.scanned, stats.skipped = scanned.Load(), skipped.Load()
if err := ctx.Err(); err != nil {
return hits, err
}
return hits, nil
}