Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -185,6 +185,7 @@ require (
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect
github.com/bahlo/generic-list-go v0.2.0 // indirect
github.com/beorn7/perks v1.0.1 // indirect
github.com/betterleaks/go-re2 v1.11.0-betterleaks.3 // indirect
github.com/blang/semver v3.5.1+incompatible // indirect
github.com/bmatcuk/doublestar v1.3.4 // indirect
github.com/bmatcuk/doublestar/v4 v4.8.1 // indirect
Expand Down Expand Up @@ -427,6 +428,7 @@ require (
github.com/transparency-dev/merkle v0.0.2 // indirect
github.com/ulikunitz/xz v0.5.15 // indirect
github.com/valyala/fastjson v1.6.10 // indirect
github.com/wasilibs/wazero-helpers v0.0.0-20250123031827-cd30c44769bb // indirect
github.com/x448/float16 v0.8.4 // indirect
github.com/xanzy/ssh-agent v0.3.3 // indirect
github.com/xeipuuv/gojsonpointer v0.0.0-20190905194746-02993c407bfb // indirect
Expand Down
4 changes: 4 additions & 0 deletions go.sum
Original file line number Diff line number Diff line change
Expand Up @@ -220,6 +220,8 @@ github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM=
github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw=
github.com/betterleaks/betterleaks v1.8.1 h1:tTSQAdQe+S4MF69o6VtpRDlml1b73F1xNUUA3PcrHiI=
github.com/betterleaks/betterleaks v1.8.1/go.mod h1:M8pR9QvWFx+iWcRH2HG6Qe1wP9DLXhHt75iMNyZ7Y+s=
github.com/betterleaks/go-re2 v1.11.0-betterleaks.3 h1:89efHsGJOOCl1KH6GxBtixZNBkNzYIlKPtmQJizCBXU=
github.com/betterleaks/go-re2 v1.11.0-betterleaks.3/go.mod h1:UzEofSJPdMG4B9tgB7ExxybS/jEV4nZxVdd2T5PcsEY=
github.com/bgentry/speakeasy v0.1.0/go.mod h1:+zsyZBPWlz7T6j88CTgSN5bM796AkVf0kBD4zp0CCIs=
github.com/blang/semver v3.5.1+incompatible h1:cQNTCjp13qL8KC3Nbxr/y2Bqb63oX6wdnnjpJbkM4JQ=
github.com/blang/semver v3.5.1+incompatible/go.mod h1:kRBLl5iJ+tD4TcOOxsy/0fnwebNt5EWlYSAyrTnjyyk=
Expand Down Expand Up @@ -1284,6 +1286,8 @@ github.com/valyala/fastjson v1.6.10 h1:/yjJg8jaVQdYR3arGxPE2X5z89xrlhS0eGXdv+ADT
github.com/valyala/fastjson v1.6.10/go.mod h1:e6FubmQouUNP73jtMLmcbxS6ydWIpOfhz34TSfO3JaE=
github.com/vektah/gqlparser/v2 v2.5.37 h1:jbb1Ilv+xBklV6653tKb4oVUupPNTLb5LmrnBKVI12Y=
github.com/vektah/gqlparser/v2 v2.5.37/go.mod h1:9O4Ox6Ngd3Y12bMD3w6i3CRQXh8W1oC1q0m6olCymDM=
github.com/wasilibs/wazero-helpers v0.0.0-20250123031827-cd30c44769bb h1:gQ+ZV4wJke/EBKYciZ2MshEouEHFuinB85dY3f5s1q8=
github.com/wasilibs/wazero-helpers v0.0.0-20250123031827-cd30c44769bb/go.mod h1:jMeV4Vpbi8osrE/pKUxRZkVaA0EX7NZN0A9/oRzgpgY=
github.com/x448/float16 v0.8.4 h1:qLwI1I70+NjRFUR3zs1JPUCgaCXSh3SW62uAKT1mSBM=
github.com/x448/float16 v0.8.4/go.mod h1:14CWIYCyZA/cWjXOioeEpHeN/83MdbZDRQHoFcYsOfg=
github.com/xanzy/ssh-agent v0.3.3 h1:+/15pJfg/RsTxqYcX6fHqOXZwwMP+2VyYWJeWM2qQFM=
Expand Down
163 changes: 146 additions & 17 deletions internal/redaction/betterleaks.go
Original file line number Diff line number Diff line change
Expand Up @@ -18,10 +18,38 @@ package redaction
import (
"context"
"fmt"
"runtime"
"slices"
"strconv"
"strings"
"sync"

"github.com/betterleaks/betterleaks/detect"
betterleaksregexp "github.com/betterleaks/betterleaks/regexp"
"github.com/betterleaks/betterleaks/regexp/re2"
"github.com/betterleaks/betterleaks/sources"
"golang.org/x/sync/errgroup"
)

const (
// defaultChunkSize is how much new text each fragment handed to the detector
// carries. A rule only runs on a fragment that contains one of its keywords,
// and a whole transcript contains nearly all of them, so scanning it as a
// single fragment runs nearly every rule over every byte. Small fragments let
// the keyword prefilter skip most rules. It matches the chunk size betterleaks
// itself reads files with.
defaultChunkSize = 100_000

// chunkOverlapLines is how many trailing lines of a chunk the next chunk
// scans again. Composite rules only fire when a component is found within a
// window of lines around the primary match, so the overlap has to cover the
// widest window in the ruleset or a pair straddling a boundary is missed.
// TestChunkOverlapCoversComponentWindows keeps it in step with the ruleset.
chunkOverlapLines = 64

// attrChunk tags each fragment, and therefore each finding, with the index of
// its chunk, so that findings can be put back in document order.
attrChunk = "chainloop.redaction.chunk"
)

// betterleaksScanner detects secrets with the betterleaks default ruleset.
Expand All @@ -30,21 +58,40 @@ type betterleaksScanner struct {
// (ValidationCounts is cleared at the start of every Run), so concurrent
// scans on a shared detector would race. Redaction is not on a hot
// concurrent path, so serialising is cheaper than owning a detector per
// caller: constructing one compiles the whole ruleset.
// caller: constructing one compiles the whole ruleset. A single scan still
// uses every core, by scanning its chunks concurrently.
mu sync.Mutex
detector *detect.Detector
// chunkSize is the amount of new text per fragment; see defaultChunkSize.
chunkSize int
}

var defaultScanner = sync.OnceValues(newBetterleaksScanner)
var (
defaultScanner = sync.OnceValues(newBetterleaksScanner)
// useRE2 selects the regex engine. It is process-wide and only affects
// patterns compiled afterwards, so it has to run before the ruleset loads.
useRE2 = sync.OnceFunc(func() { betterleaksregexp.SetEngine(re2.RE2{}) })
)

// DefaultScanner returns the process-wide betterleaks-backed scanner.
// Constructing a detector compiles several hundred regexes and builds a keyword
// trie, so it is built once, lazily, and only for attestations that need it.
func DefaultScanner() (Scanner, error) {
return defaultScanner()
s, err := defaultScanner()
if err != nil {
// Returned bare so that a failure is a nil interface, not a nil pointer
// wrapped in one.
return nil, err
}
return s, nil
}

func newBetterleaksScanner() (Scanner, error) {
func newBetterleaksScanner() (*betterleaksScanner, error) {
// The library defaults to the standard library engine, which is several
// times slower on this ruleset; the betterleaks CLI defaults to RE2 for the
// same reason.
useRE2()

// Validation stays off, which is what the default constructor gives us:
// validating would reach out to third-party APIs to check whether a candidate
// credential is live, and crafting a material must not do that.
Expand Down Expand Up @@ -73,7 +120,7 @@ func newBetterleaksScanner() (Scanner, error) {
// bound. Results are read from the Run iterator instead.
d.SkipFindingAppend = true

return &betterleaksScanner{detector: d}, nil
return &betterleaksScanner{detector: d, chunkSize: defaultChunkSize}, nil
}

func (s *betterleaksScanner) Scan(ctx context.Context, text string) ([]Finding, error) {
Expand All @@ -84,13 +131,27 @@ func (s *betterleaksScanner) Scan(ctx context.Context, text string) ([]Finding,
s.mu.Lock()
defer s.mu.Unlock()

var findings []Finding
for result := range s.detector.Run(ctx, stringSource{text: text}) {
source := chunkedSource{text: text, chunks: splitLines(text, s.chunkSize, chunkOverlapLines)}

// Chunks are scanned concurrently, so findings arrive interleaved. They are
// collected per chunk and joined in chunk order, which keeps the result
// deterministic: a single goroutine scans each chunk, so a chunk's own
// findings arrive in the detector's order. That matters because when two
// rules report the same secret the first finding names its placeholder, and
// the placeholder is part of the redacted document's digest. Duplicates from
// the chunk overlap are left for the caller, which deduplicates by secret.
perChunk := make([][]Finding, len(source.chunks))
for result := range s.detector.Run(ctx, source) {
if result.Err != nil {
return nil, fmt.Errorf("scanning for secrets: %w", result.Err)
}

findings = appendSecret(findings, result.Finding.RuleID, result.Finding.Secret)
chunk, err := strconv.Atoi(result.Finding.Attr(attrChunk))
if err != nil || chunk < 0 || chunk >= len(perChunk) {
return nil, fmt.Errorf("scanning for secrets: finding with an invalid chunk index %q", result.Finding.Attr(attrChunk))
}

perChunk[chunk] = appendSecret(perChunk[chunk], result.Finding.RuleID, result.Finding.Secret)

// Composite rules report only their primary match. An AWS access key id,
// for instance, only matches when a secret access key is found near it,
Expand All @@ -103,7 +164,7 @@ func (s *betterleaksScanner) Scan(ctx context.Context, text string) ([]Finding,
if component == nil {
continue
}
findings = appendSecret(findings, component.RuleID, component.Secret)
perChunk[chunk] = appendSecret(perChunk[chunk], component.RuleID, component.Secret)
}
}
}
Expand All @@ -114,7 +175,7 @@ func (s *betterleaksScanner) Scan(ctx context.Context, text string) ([]Finding,
return nil, err
}

return findings, nil
return slices.Concat(perChunk...), nil
}

// appendSecret records a locatable secret. A finding without one cannot be
Expand All @@ -126,16 +187,84 @@ func appendSecret(dst []Finding, ruleID, secret string) []Finding {
return append(dst, Finding{RuleID: ruleID, Secret: secret})
}

// stringSource adapts an in-memory document to the source interface the scanner
// consumes. Run is the only scanning entry point that is neither deprecated nor
// span is a half-open byte range [start, end) of a text.
type span struct {
start, end int
}

// splitLines cuts text into chunks that each carry at least size bytes of new
// text, ending at a line end. Each chunk after the first also begins with up to
// overlap trailing lines of its predecessor. Only the last of those may be
// longer than size bytes; the others must fit within size bytes together, so
// that a run of very long lines is not scanned again and again.
//
// A line is never split, even when it is longer than size. That is what keeps a
// secret whole, and it relies on how Redactor renders the document: as indented
// JSON, where a line holds at most one string leaf and newlines inside strings
// are escaped, so no secret it can locate spans a line break.
func splitLines(text string, size, overlap int) []span {
if len(text) <= size {
return []span{{0, len(text)}}
}

var chunks []span
for start, pos := 0, 0; pos < len(text); {
for fresh := pos; pos < len(text) && pos-fresh < size; {
if nl := strings.IndexByte(text[pos:], '\n'); nl >= 0 {
pos += nl + 1
} else {
pos = len(text)
}
}
chunks = append(chunks, span{start, pos})

// Walk back over the trailing lines the next chunk re-reads. The last
// line always is, however long: a composite match on it may have its
// component at the start of the next chunk.
start = pos
for range overlap {
if start == 0 {
break
}
lineStart := strings.LastIndexByte(text[:start-1], '\n') + 1
if pos-lineStart > size && start != pos {
break
}
start = lineStart
}
}
return chunks
}

// chunkedSource adapts an in-memory document to the source interface the
// scanner consumes, yielding its chunks from several goroutines at once. Run
// supports that: betterleaks' own file source scans files concurrently the same
// way. Run is the only scanning entry point that is neither deprecated nor
// context-blind, and it takes a source rather than a string.
type stringSource struct {
text string
type chunkedSource struct {
text string
chunks []span
}

func (s stringSource) Fragments(ctx context.Context, yield sources.FragmentsFunc) error {
if err := ctx.Err(); err != nil {
func (s chunkedSource) Fragments(ctx context.Context, yield sources.FragmentsFunc) error {
// The group's context is cancelled once Wait returns, so the caller's is the
// one checked at the end.
g, gctx := errgroup.WithContext(ctx)
g.SetLimit(runtime.GOMAXPROCS(0))

for i, c := range s.chunks {
if gctx.Err() != nil {
break
}
g.Go(func() error {
fragment := sources.Fragment{Raw: s.text[c.start:c.end]}
fragment.SetAttr(attrChunk, strconv.Itoa(i))
return yield(fragment, nil)
})
}

if err := g.Wait(); err != nil {
return err
}
return yield(sources.Fragment{Raw: s.text}, nil)
return ctx.Err()
}
Loading
Loading