Files
trufflehog/pkg/engine/engine_test.go
Chethas DileepandGitHub 75add79b92 fix(postgres): honor ignore tags for default port URLs (#4968)
* fix(postgres): honor ignore tags for default port URLs

* test(postgres): add raw vs primary secret test and document design decision
2026-06-05 18:17:44 +05:00

2002 lines
61 KiB
Go

package engine
import (
aCtx "context"
"fmt"
"math/rand"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"sync/atomic"
"testing"
"time"
"github.com/google/go-cmp/cmp"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"google.golang.org/protobuf/testing/protocmp"
"github.com/trufflesecurity/trufflehog/v3/pkg/config"
"github.com/trufflesecurity/trufflehog/v3/pkg/context"
"github.com/trufflesecurity/trufflehog/v3/pkg/custom_detectors"
"github.com/trufflesecurity/trufflehog/v3/pkg/decoders"
"github.com/trufflesecurity/trufflehog/v3/pkg/detectors"
"github.com/trufflesecurity/trufflehog/v3/pkg/detectors/gitlab/v2"
"github.com/trufflesecurity/trufflehog/v3/pkg/engine/ahocorasick"
"github.com/trufflesecurity/trufflehog/v3/pkg/engine/defaults"
"github.com/trufflesecurity/trufflehog/v3/pkg/pb/custom_detectorspb"
"github.com/trufflesecurity/trufflehog/v3/pkg/pb/detector_typepb"
"github.com/trufflesecurity/trufflehog/v3/pkg/pb/detectorspb"
"github.com/trufflesecurity/trufflehog/v3/pkg/pb/source_metadatapb"
"github.com/trufflesecurity/trufflehog/v3/pkg/pb/sourcespb"
"github.com/trufflesecurity/trufflehog/v3/pkg/sources"
"github.com/trufflesecurity/trufflehog/v3/pkg/verificationcache"
)
const fakeDetectorKeyword = "fakedetector"
type fakeDetectorV1 struct{}
type fakeDetectorV2 struct{}
var _ detectors.Detector = (*fakeDetectorV1)(nil)
var _ detectors.Versioner = (*fakeDetectorV1)(nil)
var _ detectors.Detector = (*fakeDetectorV2)(nil)
var _ detectors.Versioner = (*fakeDetectorV2)(nil)
func (f fakeDetectorV1) FromData(_ aCtx.Context, _ bool, _ []byte) ([]detectors.Result, error) {
return []detectors.Result{
{
DetectorType: detector_typepb.DetectorType(-1),
Verified: true,
Raw: []byte("fake secret v1"),
},
}, nil
}
func (f fakeDetectorV1) Keywords() []string { return []string{fakeDetectorKeyword} }
func (f fakeDetectorV1) Type() detector_typepb.DetectorType { return detector_typepb.DetectorType(-1) }
func (f fakeDetectorV1) Version() int { return 1 }
func (f fakeDetectorV1) Description() string { return "fake detector v1" }
func (f fakeDetectorV2) FromData(_ aCtx.Context, _ bool, _ []byte) ([]detectors.Result, error) {
return []detectors.Result{
{
DetectorType: detector_typepb.DetectorType(-1),
Verified: true,
Raw: []byte("fake secret v2"),
},
}, nil
}
func (f fakeDetectorV2) Keywords() []string { return []string{fakeDetectorKeyword} }
func (f fakeDetectorV2) Type() detector_typepb.DetectorType { return detector_typepb.DetectorType(-1) }
func (f fakeDetectorV2) Version() int { return 2 }
func (f fakeDetectorV2) Description() string { return "fake detector v2" }
func TestFragmentLineOffset(t *testing.T) {
tests := []struct {
name string
chunk *sources.Chunk
result *detectors.Result
expectedLine int64
ignore bool
}{
{
name: "ignore found on same line",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\nsecret here trufflehog:ignore\nline4"),
},
result: &detectors.Result{
Raw: []byte("secret here"),
},
expectedLine: 2,
ignore: true,
},
{
name: "no ignore",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\nsecret here\nline4"),
},
result: &detectors.Result{
Raw: []byte("secret here"),
},
expectedLine: 2,
ignore: false,
},
{
name: "ignore on different line",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\ntrufflehog:ignore\nline4\nsecret here\nline6"),
},
result: &detectors.Result{
Raw: []byte("secret here"),
},
expectedLine: 4,
ignore: false,
},
{
name: "match on consecutive lines",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\ntrufflehog:ignore\nline4\nsecret\nhere\nline6"),
},
result: &detectors.Result{
Raw: []byte("secret\nhere"),
},
expectedLine: 4,
ignore: false,
},
{
name: "ignore on last consecutive lines",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\nline3\nsecret\nhere // trufflehog:ignore\nline5"),
},
result: &detectors.Result{
Raw: []byte("secret\nhere"),
},
expectedLine: 3,
ignore: true,
},
{
name: "ignore on last line",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\nline3\nsecret here // trufflehog:ignore"),
},
result: &detectors.Result{
Raw: []byte("secret here"),
},
expectedLine: 3,
ignore: true,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
lineOffset, isIgnored := FragmentLineOffset(tt.chunk, tt.result)
if lineOffset != tt.expectedLine {
t.Errorf("Expected line offset to be %d, got %d", tt.expectedLine, lineOffset)
}
if isIgnored != tt.ignore {
t.Errorf("Expected isIgnored to be %v, got %v", tt.ignore, isIgnored)
}
})
}
}
func TestFragmentLineOffsetWithPrimarySecret(t *testing.T) {
primarySecretResult1 := &detectors.Result{
Raw: []byte("id heresecret here"), // RAW has two secrets merged
}
primarySecretResult1.SetPrimarySecretValue("secret here") // set `secret here` as primary secret value for line number calculation
primarySecretResult2 := &detectors.Result{
Raw: []byte("idsecret"), // RAW has two secrets merged
}
tests := []struct {
name string
chunk *sources.Chunk
result *detectors.Result
expectedLine int64
ignore bool
}{
{
name: "primary secret line number - correct line number",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\nid here\nsecret here\nline5"),
},
result: primarySecretResult1,
expectedLine: 3,
ignore: false,
},
{
name: "no primary secret set - wrong line number",
chunk: &sources.Chunk{
Data: []byte("line1\nline2\nid\nsecret\nline5"),
},
result: primarySecretResult2,
expectedLine: 0,
ignore: false,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
lineOffset, isIgnored := FragmentLineOffset(tt.chunk, tt.result)
if lineOffset != tt.expectedLine {
t.Errorf("Expected line offset to be %d, got %d", tt.expectedLine, lineOffset)
}
if isIgnored != tt.ignore {
t.Errorf("Expected isIgnored to be %v, got %v", tt.ignore, isIgnored)
}
})
}
}
func TestFragmentLineOffsetWithPrimarySecretMultiline(t *testing.T) {
result := &detectors.Result{
Raw: []byte("secret here"),
}
result.SetPrimarySecretValue("secret:\nsecret here")
chunk := &sources.Chunk{
Data: []byte("line1\nline2\nsecret:\nsecret here\nline5"),
}
lineOffset, isIgnored := FragmentLineOffset(chunk, result)
assert.False(t, isIgnored)
// offset 2 means line 3
assert.Equal(t, int64(2), lineOffset)
}
// TestFragmentLineOffset_DuplicateSecrets verifies that when the same secret
// appears on multiple lines within a chunk, each result receives the correct
// line number rather than always reporting the first occurrence's line.
// Regression test for https://github.com/trufflesecurity/trufflehog/issues/2502
func TestFragmentLineOffset_DuplicateSecrets(t *testing.T) {
secret := []byte("AKIA1234567890ABCDEF")
chunk := &sources.Chunk{
Data: []byte("line1\n" + // line 0
"line2\n" + // line 1
"AKIA1234567890ABCDEF\n" + // line 2 (first occurrence)
"line4\n" + // line 3
"AKIA1234567890ABCDEF\n" + // line 4 (second occurrence)
"line6\n" + // line 5
"AKIA1234567890ABCDEF\n"), // line 6 (third occurrence)
}
results := []detectors.Result{
{Raw: secret},
{Raw: secret},
{Raw: secret},
}
expectedLines := []int64{2, 4, 6}
AssignDuplicateLineOffsets(chunk, results)
seen := make(map[int64]bool)
for i, res := range results {
lineOffset, _ := FragmentLineOffset(chunk, &res)
assert.Equal(t, expectedLines[i], lineOffset,
"result[%d]: expected line %d but got %d", i, expectedLines[i], lineOffset)
assert.False(t, seen[lineOffset],
"result[%d]: line %d was already reported by a previous result (duplicate line number)", i, lineOffset)
seen[lineOffset] = true
}
}
func TestAssignDuplicateLineOffsets(t *testing.T) {
chunk := &sources.Chunk{
Data: []byte("aaa\nbbb\naaa\nccc\naaa\n"),
}
results := []detectors.Result{
{Raw: []byte("aaa")},
{Raw: []byte("aaa")},
{Raw: []byte("aaa")},
{Raw: []byte("bbb")},
}
AssignDuplicateLineOffsets(chunk, results)
// Duplicates get offsets assigned.
assert.True(t, results[0].HasChunkOffset())
assert.Equal(t, int64(0), results[0].ChunkOffset())
assert.True(t, results[1].HasChunkOffset())
assert.Equal(t, int64(8), results[1].ChunkOffset()) // "aaa\nbbb\n" = 8 bytes
assert.True(t, results[2].HasChunkOffset())
assert.Equal(t, int64(16), results[2].ChunkOffset()) // "aaa\nbbb\naaa\nccc\n" = 16 bytes
// Unique secret does not get an offset.
assert.False(t, results[3].HasChunkOffset())
}
func TestFragmentLineOffset_DuplicateSecretsWithIgnoreTag(t *testing.T) {
secret := []byte("mysecret")
chunk := &sources.Chunk{
Data: []byte("mysecret\nfoo\nmysecret trufflehog:ignore\nbar\nmysecret\n"),
}
results := []detectors.Result{
{Raw: secret},
{Raw: secret},
{Raw: secret},
}
AssignDuplicateLineOffsets(chunk, results)
line0, ignored0 := FragmentLineOffset(chunk, &results[0])
assert.Equal(t, int64(0), line0)
assert.False(t, ignored0)
line1, ignored1 := FragmentLineOffset(chunk, &results[1])
assert.Equal(t, int64(2), line1)
assert.True(t, ignored1)
line2, ignored2 := FragmentLineOffset(chunk, &results[2])
assert.Equal(t, int64(4), line2)
assert.False(t, ignored2)
}
func setupFragmentLineOffsetBench(totalLines, needleLine int) (*sources.Chunk, *detectors.Result) {
data := make([]byte, 0, 4096)
needle := []byte("needle")
for i := 0; i < totalLines; i++ {
if i != needleLine {
data = append(data, []byte(fmt.Sprintf("line%d\n", i))...)
continue
}
data = append(data, needle...)
data = append(data, '\n')
}
chunk := &sources.Chunk{Data: data}
result := &detectors.Result{Raw: needle}
return chunk, result
}
func BenchmarkFragmentLineOffsetStart(b *testing.B) {
chunk, result := setupFragmentLineOffsetBench(512, 2)
for i := 0; i < b.N; i++ {
_, _ = FragmentLineOffset(chunk, result)
}
}
func BenchmarkFragmentLineOffsetMiddle(b *testing.B) {
chunk, result := setupFragmentLineOffsetBench(512, 256)
for i := 0; i < b.N; i++ {
_, _ = FragmentLineOffset(chunk, result)
}
}
func BenchmarkFragmentLineOffsetEnd(b *testing.B) {
chunk, result := setupFragmentLineOffsetBench(512, 510)
for i := 0; i < b.N; i++ {
_, _ = FragmentLineOffset(chunk, result)
}
}
// Test to make sure that DefaultDecoders always returns the UTF8 decoder first.
// Technically a decoder test but we want this to run and fail in CI
func TestDefaultDecoders(t *testing.T) {
ds := decoders.DefaultDecoders()
if _, ok := ds[0].(*decoders.UTF8); !ok {
t.Errorf("DefaultDecoders() = %v, expected UTF8 decoder to be first", ds)
}
}
func TestSupportsLineNumbers(t *testing.T) {
tests := []struct {
name string
sourceType sourcespb.SourceType
expectedValue bool
}{
{"Git source", sourcespb.SourceType_SOURCE_TYPE_GIT, true},
{"Github source", sourcespb.SourceType_SOURCE_TYPE_GITHUB, true},
{"Gitlab source", sourcespb.SourceType_SOURCE_TYPE_GITLAB, true},
{"Bitbucket source", sourcespb.SourceType_SOURCE_TYPE_BITBUCKET, true},
{"Gerrit source", sourcespb.SourceType_SOURCE_TYPE_GERRIT, true},
{"Github unauthenticated org source", sourcespb.SourceType_SOURCE_TYPE_GITHUB_UNAUTHENTICATED_ORG, true},
{"Public Git source", sourcespb.SourceType_SOURCE_TYPE_PUBLIC_GIT, true},
{"Filesystem source", sourcespb.SourceType_SOURCE_TYPE_FILESYSTEM, true},
{"Azure Repos source", sourcespb.SourceType_SOURCE_TYPE_AZURE_REPOS, true},
{"Unsupported type", sourcespb.SourceType_SOURCE_TYPE_BUILDKITE, false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
result := SupportsLineNumbers(tt.sourceType)
assert.Equal(t, tt.expectedValue, result)
})
}
}
func BenchmarkSupportsLineNumbersLoop(b *testing.B) {
sourceType := sourcespb.SourceType_SOURCE_TYPE_GITHUB
for i := 0; i < b.N; i++ {
_ = SupportsLineNumbers(sourceType)
}
}
// TestEngine_DuplicateSecrets is a test that detects ALL duplicate secrets with the same decoder.
func TestEngine_DuplicateSecrets(t *testing.T) {
ctx := context.Background()
absPath, err := filepath.Abs("./testdata/secrets.txt")
assert.Nil(t, err)
ctx, cancel := context.WithTimeout(ctx, 10*time.Second)
defer cancel()
const defaultOutputBufferSize = 64
opts := []func(*sources.SourceManager){
sources.WithSourceUnits(),
sources.WithBufferedOutput(defaultOutputBufferSize),
}
sourceManager := sources.NewManager(opts...)
conf := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: defaults.DefaultDetectors(),
Verify: false,
SourceManager: sourceManager,
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
}
e, err := NewEngine(ctx, &conf)
assert.NoError(t, err)
e.Start(ctx)
cfg := sources.FilesystemConfig{Paths: []string{absPath}}
if _, err := e.ScanFileSystem(ctx, cfg); err != nil {
return
}
// Wait for all the chunks to be processed.
assert.Nil(t, e.Finish(ctx))
want := uint64(2)
assert.Equal(t, want, e.GetMetrics().UnverifiedSecretsFound)
}
// lineCaptureDispatcher is a test dispatcher that captures the line number
// of detected secrets. It implements the Dispatcher interface and is used
// to verify that the Engine correctly identifies and reports the line numbers
// where secrets are found in the source code.
type lineCaptureDispatcher struct{ line int64 }
func (d *lineCaptureDispatcher) Dispatch(_ context.Context, result detectors.ResultWithMetadata) error {
d.line = result.SourceMetadata.GetFilesystem().GetLine()
return nil
}
func TestEngineLineVariations(t *testing.T) {
tests := []struct {
name string
content string
expectedLine int64
}{
{
name: "secret on first line",
content: `AKIA2OGYBAH6STMMNXNN
aws_secret_access_key = 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`,
expectedLine: 1,
},
{
name: "secret after multiple newlines",
content: `
AKIA2OGYBAH6STMMNXNN
aws_secret_access_key = 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`,
expectedLine: 4,
},
{
name: "secret with mixed whitespace before",
content: `first line
AKIA2OGYBAH6STMMNXNN
aws_secret_access_key = 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`,
expectedLine: 4,
},
{
name: "secret with content after",
content: `[default]
region = us-east-1
AKIA2OGYBAH6STMMNXNN
aws_secret_access_key = 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
more content
even more`,
expectedLine: 3,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel()
tmpFile, err := os.CreateTemp("", "test_aws_credentials")
assert.NoError(t, err)
defer func() { _ = os.Remove(tmpFile.Name()) }()
err = os.WriteFile(tmpFile.Name(), []byte(tt.content), os.ModeAppend)
assert.NoError(t, err)
const defaultOutputBufferSize = 64
opts := []func(*sources.SourceManager){
sources.WithSourceUnits(),
sources.WithBufferedOutput(defaultOutputBufferSize),
}
sourceManager := sources.NewManager(opts...)
lineCapturer := new(lineCaptureDispatcher)
conf := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: defaults.DefaultDetectors(),
Verify: false,
SourceManager: sourceManager,
Dispatcher: lineCapturer,
}
eng, err := NewEngine(ctx, &conf)
assert.NoError(t, err)
eng.Start(ctx)
cfg := sources.FilesystemConfig{Paths: []string{tmpFile.Name()}}
_, err = eng.ScanFileSystem(ctx, cfg)
assert.NoError(t, err)
assert.NoError(t, eng.Finish(ctx))
want := uint64(1)
assert.Equal(t, want, eng.GetMetrics().UnverifiedSecretsFound)
assert.Equal(t, tt.expectedLine, lineCapturer.line)
})
}
}
// TestEngine_VersionedDetectorsVerifiedSecrets is a test that detects ALL verified secrets across
// versioned detectors.
func TestEngine_VersionedDetectorsVerifiedSecrets(t *testing.T) {
ctx, cancel := context.WithTimeout(context.Background(), time.Second*10)
defer cancel()
tmpFile, err := os.CreateTemp("", "testfile")
assert.Nil(t, err)
defer func() { _ = tmpFile.Close() }()
defer func() { _ = os.Remove(tmpFile.Name()) }()
_, err = fmt.Fprintf(tmpFile, "test data using keyword %s", fakeDetectorKeyword)
assert.NoError(t, err)
const defaultOutputBufferSize = 64
opts := []func(*sources.SourceManager){
sources.WithSourceUnits(),
sources.WithBufferedOutput(defaultOutputBufferSize),
}
sourceManager := sources.NewManager(opts...)
conf := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: []detectors.Detector{new(fakeDetectorV1), new(fakeDetectorV2)},
Verify: true,
SourceManager: sourceManager,
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
}
e, err := NewEngine(ctx, &conf)
assert.NoError(t, err)
e.Start(ctx)
cfg := sources.FilesystemConfig{Paths: []string{tmpFile.Name()}}
if _, err := e.ScanFileSystem(ctx, cfg); err != nil {
return
}
assert.NoError(t, e.Finish(ctx))
want := uint64(2)
assert.Equal(t, want, e.GetMetrics().VerifiedSecretsFound)
}
// TestEngine_CustomDetectorsDetectorsVerifiedSecrets is a test that covers an edge case where there are
// multiple detectors with the same type, keywords and regex that match the same secret.
// This ensures that those secrets get verified.
func TestEngine_CustomDetectorsDetectorsVerifiedSecrets(t *testing.T) {
tmpFile, err := os.CreateTemp("", "testfile")
assert.Nil(t, err)
defer func() { _ = tmpFile.Close() }()
defer func() { _ = os.Remove(tmpFile.Name()) }()
_, err = tmpFile.WriteString("test stuff")
assert.Nil(t, err)
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusOK)
}))
defer ts.Close()
customDetector1, err := custom_detectors.NewWebhookCustomRegex(&custom_detectorspb.CustomRegex{
Name: "custom detector 1",
Keywords: []string{"test"},
Regex: map[string]string{"test": "\\w+"},
Verify: []*custom_detectorspb.VerifierConfig{{Endpoint: ts.URL, Unsafe: true, SuccessRanges: []string{"200"}}},
})
assert.Nil(t, err)
customDetector2, err := custom_detectors.NewWebhookCustomRegex(&custom_detectorspb.CustomRegex{
Name: "custom detector 2",
Keywords: []string{"test"},
Regex: map[string]string{"test": "\\w+"},
Verify: []*custom_detectorspb.VerifierConfig{{Endpoint: ts.URL, Unsafe: true, SuccessRanges: []string{"200"}}},
})
assert.Nil(t, err)
allDetectors := []detectors.Detector{customDetector1, customDetector2}
ctx, cancel := context.WithTimeout(context.Background(), time.Second*5)
defer cancel()
const defaultOutputBufferSize = 64
opts := []func(*sources.SourceManager){
sources.WithSourceUnits(),
sources.WithBufferedOutput(defaultOutputBufferSize),
}
sourceManager := sources.NewManager(opts...)
conf := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: allDetectors,
Verify: true,
SourceManager: sourceManager,
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
}
e, err := NewEngine(ctx, &conf)
assert.NoError(t, err)
e.Start(ctx)
cfg := sources.FilesystemConfig{Paths: []string{tmpFile.Name()}}
if _, err := e.ScanFileSystem(ctx, cfg); err != nil {
return
}
assert.Nil(t, e.Finish(ctx))
// We should have 4 verified secrets, 2 for each custom detector.
want := uint64(4)
assert.Equal(t, want, e.GetMetrics().VerifiedSecretsFound)
}
func TestProcessResult_SourceSupportsLineNumbers_LinkUpdated(t *testing.T) {
// Arrange: Create an engine
e := Engine{results: make(chan detectors.ResultWithMetadata, 1)}
// Arrange: Create a Chunk
chunk := sources.Chunk{
Data: []byte("abcde\nswordfish"),
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Github{
Github: &source_metadatapb.Github{
Line: 1,
Link: "https://github.com/org/repo/blob/abcdef/file.txt#L1",
},
},
},
SourceType: sourcespb.SourceType_SOURCE_TYPE_GIT,
}
// Arrange: Create a Result
result := detectors.Result{
Raw: []byte("swordfish"),
Verified: true,
}
// Act
e.processResult(context.AddLogger(t.Context()), result, chunk, 0, "", nil)
// Assert that the link has been correctly updated
require.Len(t, e.results, 1)
r := <-e.results
assert.Equal(t, "https://github.com/org/repo/blob/abcdef/file.txt#L2", r.SourceMetadata.GetGithub().GetLink())
}
func TestProcessResult_IgnoreLinePresent_NothingGenerated(t *testing.T) {
// Arrange: Create an engine
e := Engine{results: make(chan detectors.ResultWithMetadata, 1)}
// Arrange: Create a Chunk
chunk := sources.Chunk{
Data: []byte("swordfish trufflehog:ignore"),
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Git{
Git: &source_metadatapb.Git{
Line: 1,
},
},
},
SourceType: sourcespb.SourceType_SOURCE_TYPE_GIT,
}
// Arrange: Create a Result
result := detectors.Result{
Raw: []byte("swordfish"),
Verified: true,
}
// Act
e.processResult(context.AddLogger(t.Context()), result, chunk, 0, "", nil)
// Assert that no results were generated
assert.Empty(t, e.results)
}
func TestProcessResult_AllFieldsCopied(t *testing.T) {
// Arrange: Create an engine
e := Engine{results: make(chan detectors.ResultWithMetadata, 1)}
// Arrange: Create a Chunk
chunk := sources.Chunk{
SourceName: "test source",
SourceID: 1,
JobID: 2,
SecretID: 3,
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Docker{
Docker: &source_metadatapb.Docker{
File: "file",
Image: "image",
Layer: "layer",
Tag: "tag",
},
},
},
SourceType: sourcespb.SourceType_SOURCE_TYPE_DOCKER,
}
// Arrange: Create a Result
result := detectors.Result{
DetectorType: detector_typepb.DetectorType(-1),
ExtraData: map[string]string{"key": "value"},
Raw: []byte("something"),
RawV2: []byte("something:else"),
Redacted: "someth***",
Verified: true,
}
// Act
e.processResult(context.AddLogger(t.Context()), result, chunk, detectorspb.DecoderType_PLAIN, "a detector that detects", nil)
// Assert that the single generated result has the correct fields
require.Len(t, e.results, 1)
r := <-e.results
if diff := cmp.Diff(chunk.SourceMetadata, r.SourceMetadata, protocmp.Transform()); diff != "" {
t.Errorf("metadata mismatch (-want +got):\n%s", diff)
}
assert.Equal(t, map[string]string{"key": "value"}, r.ExtraData)
assert.Equal(t, []byte("something"), r.Raw)
assert.Equal(t, []byte("something:else"), r.RawV2)
assert.Equal(t, "someth***", r.Redacted)
assert.True(t, r.Verified)
assert.Equal(t, detector_typepb.DetectorType(-1), r.DetectorType)
assert.Equal(t, sources.SourceID(1), r.SourceID)
assert.Equal(t, sources.JobID(2), r.JobID)
assert.Equal(t, int64(3), r.SecretID)
assert.Equal(t, "test source", r.SourceName)
assert.Equal(t, sourcespb.SourceType_SOURCE_TYPE_DOCKER, r.SourceType)
assert.Equal(t, detectorspb.DecoderType_PLAIN, r.DecoderType)
assert.Equal(t, "a detector that detects", r.DetectorDescription)
}
func TestProcessResult_FalsePositiveFlagSetCorrectly(t *testing.T) {
testcases := []struct {
name string
verified bool
isFalsePositive bool
wantIsFalsePositive bool
}{
{
name: "unverified/false positive",
verified: false,
isFalsePositive: true,
wantIsFalsePositive: true,
},
{
name: "unverified/not false positive",
verified: false,
isFalsePositive: false,
wantIsFalsePositive: false,
},
{
name: "verified/false positive",
verified: true,
isFalsePositive: true,
wantIsFalsePositive: false, // The false positive check should not be run for verified secrets
},
{
name: "verified/not false positive",
verified: true,
isFalsePositive: false,
wantIsFalsePositive: false,
},
}
for _, tt := range testcases {
t.Run(tt.name, func(t *testing.T) {
// Arrange: Create an Engine
e := Engine{results: make(chan detectors.ResultWithMetadata, 1)}
// Arrange: Create a Result
res := detectors.Result{
Raw: []byte("something not nil"), // The false positive check is not run when Raw is nil
Verified: tt.verified,
}
// Arrange: Create the false positive check
isFalsePositive := func(_ detectors.Result) (bool, string) { return tt.isFalsePositive, "" }
// Act
e.processResult(context.AddLogger(t.Context()), res, sources.Chunk{}, 0, "", isFalsePositive)
// Assert that the single generated result has the correct false positive flag
require.Len(t, e.results, 1)
assert.Equal(t, tt.wantIsFalsePositive, (<-e.results).IsWordlistFalsePositive)
})
}
}
func TestVerificationOverlapChunk(t *testing.T) {
ctx := context.Background()
absPath, err := filepath.Abs("./testdata/verificationoverlap_secrets.txt")
assert.Nil(t, err)
ctx, cancel := context.WithTimeout(ctx, 10*time.Second)
defer cancel()
confPath, err := filepath.Abs("./testdata/verificationoverlap_detectors.yaml")
assert.Nil(t, err)
conf, err := config.Read(confPath)
assert.Nil(t, err)
const defaultOutputBufferSize = 64
opts := []func(*sources.SourceManager){
sources.WithSourceUnits(),
sources.WithBufferedOutput(defaultOutputBufferSize),
}
sourceManager := sources.NewManager(opts...)
c := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: conf.Detectors,
IncludeDetectors: "904", // isolate this test to only the custom detectors provided
Verify: false,
SourceManager: sourceManager,
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
}
e, err := NewEngine(ctx, &c)
assert.NoError(t, err)
e.verificationOverlapTracker = &atomic.Int32{}
e.Start(ctx)
cfg := sources.FilesystemConfig{Paths: []string{absPath}}
if _, err := e.ScanFileSystem(ctx, cfg); err != nil {
return
}
// Wait for all the chunks to be processed.
assert.Nil(t, e.Finish(ctx))
// We want TWO secrets that match both the custom regexes.
want := uint64(2)
assert.Equal(t, want, e.GetMetrics().UnverifiedSecretsFound)
// We want 0 because these are custom detectors and verification should still occur.
wantDupe := int32(0)
assert.Equal(t, wantDupe, e.verificationOverlapTracker.Load())
}
func TestEngine_FalsePositivesRetainedCorrectly(t *testing.T) {
// Arrange: Generate the absolute path of the file to scan
secretsPath, err := filepath.Abs("./testdata/verificationoverlap_secrets_fp.txt")
require.NoError(t, err)
testCases := []struct {
name string
detectors []detectors.Detector
retainFalsePositives bool
wantUnverifiedSecretCount uint64
}{
{
name: "no overlap, retain false positives",
detectors: []detectors.Detector{
passthroughDetector{detectorType: detector_typepb.DetectorType(-1), keywords: []string{"sample"}},
},
retainFalsePositives: true,
wantUnverifiedSecretCount: 1,
},
{
name: "no overlap, do not retain false positives",
detectors: []detectors.Detector{
passthroughDetector{detectorType: detector_typepb.DetectorType(-1), keywords: []string{"sample"}},
},
retainFalsePositives: false,
wantUnverifiedSecretCount: 0,
},
{
name: "overlap, retain false positives",
detectors: []detectors.Detector{
passthroughDetector{detectorType: detector_typepb.DetectorType(-1), keywords: []string{"sample"}},
passthroughDetector{detectorType: detector_typepb.DetectorType(-2), keywords: []string{"ample"}},
},
retainFalsePositives: true,
wantUnverifiedSecretCount: 2,
},
{
name: "overlap, do not retain false positives",
detectors: []detectors.Detector{
passthroughDetector{detectorType: detector_typepb.DetectorType(-1), keywords: []string{"sample"}},
passthroughDetector{detectorType: detector_typepb.DetectorType(-2), keywords: []string{"ample"}},
},
retainFalsePositives: false,
wantUnverifiedSecretCount: 0,
},
}
for _, tt := range testCases {
t.Run(tt.name, func(t *testing.T) {
ctx := context.AddLogger(t.Context())
// Arrange: Generate a base engine config
engineConfig := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: tt.detectors,
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
Results: map[string]struct{}{"verified": {}, "unverified": {}, "unknown": {}},
SourceManager: sources.NewManager(sources.WithSourceUnits()),
Verify: false,
}
// Arrange: Set the appropriate false positive flag
if tt.retainFalsePositives {
engineConfig.Results["filtered_unverified"] = struct{}{}
}
// Arrange: Create and start an engine
e, err := NewEngine(ctx, &engineConfig)
require.NoError(t, err)
e.Start(ctx)
// Act: Scan the file
cfg := sources.FilesystemConfig{Paths: []string{secretsPath}}
_, err = e.ScanFileSystem(ctx, cfg)
require.NoError(t, err)
require.NoError(t, e.Finish(ctx))
// Assert that the unverified secret count was expected
assert.Equal(t, tt.wantUnverifiedSecretCount, e.GetMetrics().UnverifiedSecretsFound)
})
}
}
func TestFragmentFirstLineAndLink(t *testing.T) {
tests := []struct {
name string
chunk *sources.Chunk
expectedLine int64
expectedLink string
}{
{
name: "Test Git Metadata",
chunk: &sources.Chunk{
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Git{
Git: &source_metadatapb.Git{
Line: 10,
},
},
},
},
expectedLine: 10,
expectedLink: "", // Git doesn't support links
},
{
name: "Test Github Metadata",
chunk: &sources.Chunk{
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Github{
Github: &source_metadatapb.Github{
Line: 5,
Link: "https://example.github.com",
},
},
},
},
expectedLine: 5,
expectedLink: "https://example.github.com",
},
{
name: "Test Azure Repos Metadata",
chunk: &sources.Chunk{
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_AzureRepos{
AzureRepos: &source_metadatapb.AzureRepos{
Line: 5,
Link: "https://example.azure.com",
},
},
},
},
expectedLine: 5,
expectedLink: "https://example.azure.com",
},
{
name: "Line number not set",
chunk: &sources.Chunk{
SourceMetadata: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Github{
Github: &source_metadatapb.Github{
Link: "https://example.github.com",
},
},
},
},
expectedLine: 1,
expectedLink: "https://example.github.com",
},
{
name: "Unsupported Type",
chunk: &sources.Chunk{},
expectedLine: 0,
expectedLink: "",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
line, linePtr, link := FragmentFirstLineAndLink(tt.chunk)
assert.Equal(t, tt.expectedLink, link, "Mismatch in link")
assert.Equal(t, tt.expectedLine, line, "Mismatch in line")
if linePtr != nil {
assert.Equal(t, tt.expectedLine, *linePtr, "Mismatch in linePtr value")
}
})
}
}
func TestSetLink(t *testing.T) {
tests := []struct {
name string
input *source_metadatapb.MetaData
link string
line int64
wantLink string
wantErr bool
}{
{
name: "Github link set",
input: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Github{
Github: &source_metadatapb.Github{},
},
},
link: "https://github.com/example",
line: 42,
wantLink: "https://github.com/example#L42",
},
{
name: "Gitlab link set",
input: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Gitlab{
Gitlab: &source_metadatapb.Gitlab{},
},
},
link: "https://gitlab.com/example",
line: 10,
wantLink: "https://gitlab.com/example#L10",
},
{
name: "Bitbucket link set",
input: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Bitbucket{
Bitbucket: &source_metadatapb.Bitbucket{},
},
},
link: "https://bitbucket.com/example",
line: 8,
wantLink: "https://bitbucket.com/example#L8",
},
{
name: "Filesystem link set",
input: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Filesystem{
Filesystem: &source_metadatapb.Filesystem{},
},
},
link: "file:///path/to/example",
line: 3,
wantLink: "file:///path/to/example#L3",
},
{
name: "Azure Repos link set",
input: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_AzureRepos{
AzureRepos: &source_metadatapb.AzureRepos{},
},
},
link: "https://dev.azure.com/example",
line: 3,
wantLink: "https://dev.azure.com/example?line=3&lineEnd=4&lineStartColumn=1",
},
{
name: "Unsupported metadata type",
input: &source_metadatapb.MetaData{
Data: &source_metadatapb.MetaData_Git{
Git: &source_metadatapb.Git{},
},
},
link: "https://git.example.com/link",
line: 5,
wantErr: true,
},
{
name: "Metadata nil",
input: nil,
link: "https://some.link",
line: 1,
wantErr: true,
},
}
ctx := context.Background()
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
err := UpdateLink(ctx, tt.input, tt.link, tt.line)
if err != nil && !tt.wantErr {
t.Errorf("Unexpected error: %v", err)
}
if tt.wantErr {
return
}
switch data := tt.input.GetData().(type) {
case *source_metadatapb.MetaData_Github:
assert.Equal(t, tt.wantLink, data.Github.Link, "Github link mismatch")
case *source_metadatapb.MetaData_Gitlab:
assert.Equal(t, tt.wantLink, data.Gitlab.Link, "Gitlab link mismatch")
case *source_metadatapb.MetaData_Bitbucket:
assert.Equal(t, tt.wantLink, data.Bitbucket.Link, "Bitbucket link mismatch")
case *source_metadatapb.MetaData_Filesystem:
assert.Equal(t, tt.wantLink, data.Filesystem.Link, "Filesystem link mismatch")
case *source_metadatapb.MetaData_AzureRepos:
assert.Equal(t, tt.wantLink, data.AzureRepos.Link, "Azure Repos link mismatch")
}
})
}
}
func TestLikelyDuplicate(t *testing.T) {
// Initialize detectors
// (not actually calling detector FromData or anything, just using detector struct for key creation)
detectorA := ahocorasick.DetectorMatch{
Key: ahocorasick.CreateDetectorKey(defaults.DefaultDetectors()[0]),
Detector: defaults.DefaultDetectors()[0],
}
detectorB := ahocorasick.DetectorMatch{
Key: ahocorasick.CreateDetectorKey(defaults.DefaultDetectors()[1]),
Detector: defaults.DefaultDetectors()[1],
}
// Define test cases
tests := []struct {
name string
val chunkSecretKey
dupes map[chunkSecretKey]struct{}
expected bool
}{
{
name: "exact duplicate different detector",
val: chunkSecretKey{"PMAK-qnwfsLyRSyfCwfpHaQP1UzDhrgpWvHjbYzjpRCMshjt417zWcrzyHUArs7r", detectorA.Key},
dupes: map[chunkSecretKey]struct{}{
{"PMAK-qnwfsLyRSyfCwfpHaQP1UzDhrgpWvHjbYzjpRCMshjt417zWcrzyHUArs7r", detectorB.Key}: {},
},
expected: true,
},
{
name: "non-duplicate length outside range",
val: chunkSecretKey{"short", detectorA.Key},
dupes: map[chunkSecretKey]struct{}{
{"muchlongerthanthevalstring", detectorB.Key}: {},
},
expected: false,
},
{
name: "similar within threshold",
val: chunkSecretKey{"PMAK-qnwfsLyRSyfCwfpHaQP1UzDhrgpWvHjbYzjpRCMshjt417zWcrzyHUArs7r", detectorA.Key},
dupes: map[chunkSecretKey]struct{}{
{"qnwfsLyRSyfCwfpHaQP1UzDhrgpWvHjbYzjpRCMshjt417zWcrzyHUArs7r", detectorB.Key}: {},
},
expected: true,
},
{
name: "similar outside threshold",
val: chunkSecretKey{"anotherkey", detectorA.Key},
dupes: map[chunkSecretKey]struct{}{
{"completelydifferent", detectorB.Key}: {},
},
expected: false,
},
{
name: "empty strings",
val: chunkSecretKey{"", detectorA.Key},
dupes: map[chunkSecretKey]struct{}{{"", detectorB.Key}: {}},
expected: true,
},
{
name: "similar within threshold same detector",
val: chunkSecretKey{"PMAK-qnwfsLyRSyfCwfpHaQP1UzDhrgpWvHjbYzjpRCMshjt417zWcrzyHUArs7r", detectorA.Key},
dupes: map[chunkSecretKey]struct{}{
{"qnwfsLyRSyfCwfpHaQP1UzDhrgpWvHjbYzjpRCMshjt417zWcrzyHUArs7r", detectorA.Key}: {},
},
expected: false,
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
ctx := context.Background()
result := likelyDuplicate(ctx, tc.val, tc.dupes)
if result != tc.expected {
t.Errorf("expected %v, got %v", tc.expected, result)
}
})
}
}
type customCleaner struct {
ignoreConfig bool
}
var _ detectors.CustomResultsCleaner = (*customCleaner)(nil)
var _ detectors.Detector = (*customCleaner)(nil)
func (c customCleaner) FromData(aCtx.Context, bool, []byte) ([]detectors.Result, error) {
return []detectors.Result{}, nil
}
func (c customCleaner) Keywords() []string { return []string{} }
func (c customCleaner) Type() detector_typepb.DetectorType { return detector_typepb.DetectorType(-1) }
func (customCleaner) Description() string { return "" }
func (c customCleaner) CleanResults([]detectors.Result) []detectors.Result {
return []detectors.Result{}
}
func (c customCleaner) ShouldCleanResultsIrrespectiveOfConfiguration() bool { return c.ignoreConfig }
func TestFilterResults_CustomCleaner(t *testing.T) {
testCases := []struct {
name string
cleaningConfigured bool
ignoreConfig bool
resultsToClean []detectors.Result
wantResults []detectors.Result
}{
{
name: "respect config to clean",
cleaningConfigured: true,
ignoreConfig: false,
resultsToClean: []detectors.Result{{}},
wantResults: []detectors.Result{},
},
{
name: "respect config to not clean",
cleaningConfigured: false,
ignoreConfig: false,
resultsToClean: []detectors.Result{{}},
wantResults: []detectors.Result{{}},
},
{
name: "clean irrespective of config",
cleaningConfigured: false,
ignoreConfig: true,
resultsToClean: []detectors.Result{{}},
wantResults: []detectors.Result{},
},
}
for _, tt := range testCases {
t.Run(tt.name, func(t *testing.T) {
match := ahocorasick.DetectorMatch{
Detector: customCleaner{
ignoreConfig: tt.ignoreConfig,
},
}
engine := Engine{
filterUnverified: tt.cleaningConfigured,
retainFalsePositives: true,
}
cleaned := engine.filterResults(context.Background(), &match, tt.resultsToClean)
assert.ElementsMatch(t, tt.wantResults, cleaned)
})
}
}
func BenchmarkPopulateMatchingDetectors(b *testing.B) {
allDetectors := defaults.DefaultDetectors()
ac := ahocorasick.NewAhoCorasickCore(allDetectors)
// Generate sample data with keywords from detectors.
dataSize := 1 << 20 // 1 MB
sampleData := generateRandomDataWithKeywords(dataSize, allDetectors)
smallChunk := 1 << 10 // 1 KB
mediumChunk := 1 << 12 // 4 KB
current := sources.TotalChunkSize
largeChunk := 1 << 14 // 16 KB
xlChunk := 1 << 15 // 32 KB
xxlChunk := 1 << 16 // 64 KB
xxxlChunk := 1 << 18 // 256 KB
chunkSizes := []int{smallChunk, mediumChunk, current, largeChunk, xlChunk, xxlChunk, xxxlChunk}
for _, chunkSize := range chunkSizes {
b.Run(fmt.Sprintf("ChunkSize_%d", chunkSize), func(b *testing.B) {
b.ReportAllocs()
b.SetBytes(int64(chunkSize))
// Create a single chunk of the desired size.
chunk := sampleData[:chunkSize]
b.ResetTimer()
for i := 0; i < b.N; i++ {
ac.FindDetectorMatches([]byte(chunk)) // Match against the single chunk
}
})
}
}
func generateRandomDataWithKeywords(size int, detectors []detectors.Detector) string {
const charset = "abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789"
data := make([]byte, size)
r := rand.New(rand.NewSource(42)) // Seed for reproducibility
for i := range data {
data[i] = charset[r.Intn(len(charset))]
}
totalKeywords := 0
for _, d := range detectors {
totalKeywords += len(d.Keywords())
}
// Target keyword density (keywords per character)
// This ensures that the generated data has a reasonable number of keywords and is consistent
// across different data sizes.
keywordDensity := 0.01
targetKeywords := int(float64(size) * keywordDensity)
for i := 0; i < targetKeywords; i++ {
detectorIndex := r.Intn(len(detectors))
keywordIndex := r.Intn(len(detectors[detectorIndex].Keywords()))
keyword := detectors[detectorIndex].Keywords()[keywordIndex]
insertPosition := r.Intn(size - len(keyword))
copy(data[insertPosition:], keyword)
}
return string(data)
}
func TestEngine_ShouldVerifyChunk(t *testing.T) {
tests := []struct {
name string
detector detectors.Detector
overrideKey config.DetectorID
want func(sourceVerify, detectorVerify bool) bool
}{
{
name: "detector override by exact version",
detector: &gitlab.Scanner{},
overrideKey: config.DetectorID{ID: detector_typepb.DetectorType_Gitlab, Version: 2},
want: func(sourceVerify, detectorVerify bool) bool { return detectorVerify },
},
{
name: "detector override by versionless config",
detector: &gitlab.Scanner{},
overrideKey: config.DetectorID{ID: detector_typepb.DetectorType_Gitlab, Version: 0},
want: func(sourceVerify, detectorVerify bool) bool { return detectorVerify },
},
{
name: "no detector override because of detector type mismatch",
detector: &gitlab.Scanner{},
overrideKey: config.DetectorID{ID: detector_typepb.DetectorType_NpmToken, Version: 2},
want: func(sourceVerify, detectorVerify bool) bool { return sourceVerify },
},
{
name: "no detector override because of detector version mismatch",
detector: &gitlab.Scanner{},
overrideKey: config.DetectorID{ID: detector_typepb.DetectorType_Gitlab, Version: 1},
want: func(sourceVerify, detectorVerify bool) bool { return sourceVerify },
},
}
booleanChoices := [2]bool{true, false}
engine := &Engine{verify: true}
for _, tt := range tests {
for _, sourceVerify := range booleanChoices {
for _, detectorVerify := range booleanChoices {
t.Run(fmt.Sprintf("%s (source verify = %v, detector verify = %v)", tt.name, sourceVerify, detectorVerify), func(t *testing.T) {
overrides := map[config.DetectorID]bool{
tt.overrideKey: detectorVerify,
}
want := tt.want(sourceVerify, detectorVerify)
got := engine.shouldVerifyChunk(sourceVerify, tt.detector, overrides)
assert.Equal(t, want, got)
})
}
}
}
}
func TestEngineInitializesCloudProviderDetectors(t *testing.T) {
ctx := context.Background()
conf := Config{
Concurrency: 1,
Detectors: defaults.DefaultDetectors(),
Verify: false,
SourceManager: sources.NewManager(),
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
}
e, err := NewEngine(ctx, &conf)
assert.NoError(t, err)
var count int
noCloudEndpointDetectors := map[detector_typepb.DetectorType]struct{}{
detector_typepb.DetectorType_ArtifactoryAccessToken: {},
detector_typepb.DetectorType_ArtifactoryReferenceToken: {},
detector_typepb.DetectorType_TableauPersonalAccessToken: {},
detector_typepb.DetectorType_HashiCorpVaultAuth: {},
detector_typepb.DetectorType_JiraDataCenterPAT: {},
detector_typepb.DetectorType_ConfluenceDataCenter: {},
detector_typepb.DetectorType_BitbucketDataCenter: {},
// these do not have any cloud endpoint
}
for _, det := range e.detectors {
if endpoints, ok := det.(interface{ Endpoints(...string) []string }); ok {
id := config.GetDetectorID(det)
if len(endpoints.Endpoints()) == 0 {
if _, ok := noCloudEndpointDetectors[det.Type()]; !ok {
t.Fatalf("detector %q Endpoints() is empty", id.String())
}
}
count++
}
}
if count == 0 {
t.Fatal("no detectors found implementing Endpoints(), did EndpointSetter change?")
}
}
func TestEngineignoreLine(t *testing.T) {
tests := []struct {
name string
content string
expectedFindings int
}{
{
name: "ignore at end of line",
content: `
# tests/example_false_positive.py
def test_something():
connection_string = "who-cares"
# Ignoring this does not work
assert connection_string == "postgres://master_user:master_password@hostname:1234/main" # trufflehog:ignore`,
expectedFindings: 0,
},
{
name: "ignore postgres url without explicit port",
content: `
# tests/example_false_positive.py
def test_something():
connection_string = "who-cares"
# The detector normalizes this URL to include :5432, but the ignore tag should still be honored.
assert connection_string == "postgres://master_user:master_password@hostname/main" # trufflehog:ignore`,
expectedFindings: 0,
},
{
name: "ignore not on secret line",
content: `
# tests/example_false_positive.py
def test_something():
connection_string = "who-cares"
# Ignoring this does not work
assert some_other_stuff == "blah" # trufflehog:ignore
assert connection_string == "postgres://master_user:master_password@hostname:1234/main"`,
expectedFindings: 1,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel()
tmpFile, err := os.CreateTemp("", "test_creds")
assert.NoError(t, err)
defer func() { _ = os.Remove(tmpFile.Name()) }()
err = os.WriteFile(tmpFile.Name(), []byte(tt.content), os.ModeAppend)
assert.NoError(t, err)
const defaultOutputBufferSize = 64
opts := []func(*sources.SourceManager){
sources.WithSourceUnits(),
sources.WithBufferedOutput(defaultOutputBufferSize),
}
sourceManager := sources.NewManager(opts...)
conf := Config{
Concurrency: 1,
Decoders: decoders.DefaultDecoders(),
Detectors: defaults.DefaultDetectors(),
Verify: false,
SourceManager: sourceManager,
Dispatcher: NewPrinterDispatcher(new(discardPrinter)),
}
eng, err := NewEngine(ctx, &conf)
assert.NoError(t, err)
eng.Start(ctx)
cfg := sources.FilesystemConfig{Paths: []string{tmpFile.Name()}}
_, err = eng.ScanFileSystem(ctx, cfg)
assert.NoError(t, err)
assert.NoError(t, eng.Finish(ctx))
assert.Equal(t, tt.expectedFindings, int(eng.GetMetrics().UnverifiedSecretsFound))
})
}
}
type passthroughDetector struct {
detectorType detector_typepb.DetectorType
keywords []string
secret string
}
func (p passthroughDetector) FromData(_ aCtx.Context, verify bool, data []byte) ([]detectors.Result, error) {
raw := data
if p.secret != "" {
raw = []byte(p.secret)
}
return []detectors.Result{
{
Raw: raw,
Verified: verify,
},
}, nil
}
func (p passthroughDetector) Keywords() []string { return p.keywords }
func (p passthroughDetector) Type() detector_typepb.DetectorType { return p.detectorType }
func (p passthroughDetector) Description() string { return "fake detector for testing" }
type passthroughDecoder struct{}
func (p passthroughDecoder) FromChunk(chunk *sources.Chunk) *decoders.DecodableChunk {
return &decoders.DecodableChunk{
Chunk: chunk,
DecoderType: detectorspb.DecoderType(-1),
}
}
func (p passthroughDecoder) Type() detectorspb.DecoderType {
return detectorspb.DecoderType(-1)
}
// TestEngine_DetectChunk_UsesVerifyFlag validates that detectChunk correctly forwards detectableChunk.verify to
// detectors.
func TestEngine_DetectChunk_UsesVerifyFlag(t *testing.T) {
ctx := context.Background()
testCases := []struct {
name string
verify bool
}{
{name: "verify=true", verify: true},
{name: "verify=false", verify: false},
}
for _, tc := range testCases {
t.Run(tc.name, func(t *testing.T) {
// Arrange: Create a minimal engine.
e := &Engine{
results: make(chan detectors.ResultWithMetadata, 1),
verificationCache: verificationcache.New(nil, &verificationcache.InMemoryMetrics{}),
}
// Arrange: Create a detector match. We can't create one directly, so we have to use a minimal A-H core.
ahcore := ahocorasick.NewAhoCorasickCore([]detectors.Detector{passthroughDetector{keywords: []string{"keyword"}}})
detectorMatches := ahcore.FindDetectorMatches([]byte("keyword"))
require.Len(t, detectorMatches, 1)
// Arrange: Create a chunk to detect.
chunk := detectableChunk{
detector: detectorMatches[0],
verify: tc.verify,
wgDoneFn: func() {},
}
// Act
e.detectChunk(ctx, chunk)
close(e.results)
// Assert: Confirm that a result was generated and that it has the expected verify flag.
select {
case result := <-e.results:
assert.Equal(t, tc.verify, result.Verified)
default:
t.Errorf("expected a result but did not get one")
}
})
}
}
// TestEngine_ScannerWorker_DetectableChunkHasCorrectVerifyFlag validates that scannerWorker generates detectableChunk
// structs that have the correct verify flag set. It also validates that the original chunks' SourceVerify flags are
// unchanged.
func TestEngine_ScannerWorker_DetectableChunkHasCorrectVerifyFlag(t *testing.T) {
ctx := context.Background()
testCases := []struct {
name string
engineVerify bool
sourceVerify bool
wantVerify bool
}{
{name: "engineVerify=false,sourceVerify=false", engineVerify: false, sourceVerify: false, wantVerify: false},
{name: "engineVerify=false,sourceVerify=true", engineVerify: false, sourceVerify: true, wantVerify: false},
{name: "engineVerify=true,sourceVerify=false", engineVerify: true, sourceVerify: false, wantVerify: false},
{name: "engineVerify=true,sourceVerify=true", engineVerify: true, sourceVerify: true, wantVerify: true},
}
for _, tc := range testCases {
t.Run(tc.name, func(t *testing.T) {
// Arrange: Create a minimal engine.
detector := &passthroughDetector{keywords: []string{"keyword"}}
e := &Engine{
AhoCorasickCore: ahocorasick.NewAhoCorasickCore([]detectors.Detector{detector}),
decoders: []decoders.Decoder{passthroughDecoder{}},
detectableChunksChan: make(chan detectableChunk, 1),
sourceManager: sources.NewManager(),
verify: tc.engineVerify,
maxDecodeDepth: 1,
}
// Arrange: Create a chunk to scan.
chunk := sources.Chunk{
Data: []byte("keyword"),
SourceVerify: tc.sourceVerify,
}
// Arrange: Enqueue a chunk to be scanned.
e.sourceManager.ScanChunk(&chunk)
// Act
go e.scannerWorker(ctx)
// Assert: Confirm that a chunk was generated, that its SourceVerify flag is unchanged, and that its verify
// flag is correctly set.
select {
case chunk := <-e.detectableChunksChan:
assert.Equal(t, tc.sourceVerify, chunk.chunk.SourceVerify)
assert.Equal(t, tc.wantVerify, chunk.verify)
case <-time.After(1 * time.Second):
t.Errorf("expected a detectableChunk but did not get one")
}
})
}
}
// TestEngine_VerificationOverlapWorker_DetectableChunkHasCorrectVerifyFlag validates that the results directly
// generated by verificationOverlapWorker all came from detector invocations with the verify flag cleared (because these
// results were generated from verification overlaps). It also validates that detectableChunk structs generated by
// verificationOverlapWorker have their verify flags correctly set, and that these structs' original chunks'
// SourceVerify flags are unchanged.
func TestEngine_VerificationOverlapWorker_DetectableChunkHasCorrectVerifyFlag(t *testing.T) {
ctx := context.Background()
t.Run("overlap", func(t *testing.T) {
// Arrange: Create a minimal engine
e := &Engine{
detectableChunksChan: make(chan detectableChunk, 2),
results: make(chan detectors.ResultWithMetadata, 2),
retainFalsePositives: true,
verificationOverlapChunksChan: make(chan verificationOverlapChunk, 2),
verify: true,
}
// Arrange: Set up a fake detectableChunk processor so that any chunks (incorrectly) sent to
// e.detectableChunksChan don't block the test.
processedDetectableChunks := make(chan detectableChunk, 2)
go func() {
for chunk := range e.detectableChunksChan {
chunk.wgDoneFn()
processedDetectableChunks <- chunk
}
}()
// Arrange: Create a chunk to "scan."
chunk := sources.Chunk{
Data: []byte("keyword ;oahpow8heg;blaisd"),
SourceVerify: true,
}
// Arrange: Create overlapping detector matches. We can't create them directly, so we have to use a minimal A-H
// core.
ahcore := ahocorasick.NewAhoCorasickCore([]detectors.Detector{
passthroughDetector{detectorType: detector_typepb.DetectorType(-1), keywords: []string{"keyw"}},
passthroughDetector{detectorType: detector_typepb.DetectorType(-2), keywords: []string{"keyword"}},
})
detectorMatches := ahcore.FindDetectorMatches(chunk.Data)
require.Len(t, detectorMatches, 2)
// Arrange: Enqueue a verification overlap chunk
e.verificationOverlapChunksChan <- verificationOverlapChunk{
chunk: chunk,
detectors: detectorMatches,
verificationOverlapWgDoneFn: func() { close(e.verificationOverlapChunksChan) },
}
// Act
e.verificationOverlapWorker(ctx)
close(e.results)
close(e.detectableChunksChan)
close(processedDetectableChunks)
// Assert: Confirm that every generated result is unverified (because overlap detection precluded it).
for result := range e.results {
assert.False(t, result.Verified)
}
// Assert: Confirm that every generated detectable chunk's Chunk.SourceVerify flag is unchanged and that its
// verify flag is correctly set.
// CMR: There should be not be any of these chunks. However, due to what I believe is an unrelated bug, there
// are. This test ensures that even in that erroneous case, their contents are correct.
for detectableChunk := range processedDetectableChunks {
assert.True(t, detectableChunk.verify)
assert.True(t, detectableChunk.chunk.SourceVerify)
}
})
t.Run("no overlap", func(t *testing.T) {
// Arrange: Create a minimal engine
e := &Engine{
detectableChunksChan: make(chan detectableChunk, 2),
retainFalsePositives: true,
verificationOverlapChunksChan: make(chan verificationOverlapChunk, 2),
verify: true,
}
// Arrange: Set up a fake detectableChunk processor so that any chunks sent to e.detectableChunksChan don't
// block the test.
processedDetectableChunks := make(chan detectableChunk, 2)
go func() {
for chunk := range e.detectableChunksChan {
chunk.wgDoneFn()
processedDetectableChunks <- chunk
}
}()
// Arrange: Create a chunk to "scan."
chunk := sources.Chunk{
Data: []byte("keyword ;oahpow8heg;blaisd"),
SourceVerify: true,
}
// Arrange: Create non-overlapping detector matches. We can't create them directly, so we have to use a minimal
// A-H core.
ahcore := ahocorasick.NewAhoCorasickCore([]detectors.Detector{
passthroughDetector{detectorType: detector_typepb.DetectorType(-1), keywords: []string{"keyw"}, secret: "oahpow"},
passthroughDetector{detectorType: detector_typepb.DetectorType(-2), keywords: []string{"keyword"}, secret: "blaisd"},
})
detectorMatches := ahcore.FindDetectorMatches(chunk.Data)
require.Len(t, detectorMatches, 2)
// Arrange: Enqueue a verification overlap chunk
e.verificationOverlapChunksChan <- verificationOverlapChunk{
chunk: chunk,
detectors: detectorMatches,
verificationOverlapWgDoneFn: func() { close(e.verificationOverlapChunksChan) },
}
// Act
e.verificationOverlapWorker(ctx)
close(e.detectableChunksChan)
close(processedDetectableChunks)
// Assert: Confirm that SourceVerify flags are unchanged, and verify flags are correctly set.
for detectableChunk := range processedDetectableChunks {
assert.True(t, detectableChunk.chunk.SourceVerify)
assert.True(t, detectableChunk.verify)
}
})
}
func TestEngine_IterativeDecoding(t *testing.T) {
t.Parallel()
const (
// base64(base64("my-secret-key-test-value"))
doubleEncoded = "YlhrdGMyVmpjbVYwTFd0bGVTMTBaWE4wTFhaaGJIVmw="
detectorKeyword = "my-secret"
// escapedUnicode("my-secret-key-test-value")
escapedUnicode = `\u006d\u0079\u002d\u0073\u0065\u0063\u0072\u0065\u0074\u002d\u006b\u0065\u0079\u002d\u0074\u0065\u0073\u0074\u002d\u0076\u0061\u006c\u0075\u0065`
// escapedUnicode(escapedUnicode("my-secret-key-test-value"))
doubleEscapedUnicode = `\u005c\u0075\u0030\u0030\u0036\u0064\u005c\u0075\u0030\u0030\u0037\u0039\u005c\u0075\u0030\u0030\u0032\u0064\u005c\u0075\u0030\u0030\u0037\u0033\u005c\u0075\u0030\u0030\u0036\u0035\u005c\u0075\u0030\u0030\u0036\u0033\u005c\u0075\u0030\u0030\u0037\u0032\u005c\u0075\u0030\u0030\u0036\u0035\u005c\u0075\u0030\u0030\u0037\u0034\u005c\u0075\u0030\u0030\u0032\u0064\u005c\u0075\u0030\u0030\u0036\u0062\u005c\u0075\u0030\u0030\u0036\u0035\u005c\u0075\u0030\u0030\u0037\u0039\u005c\u0075\u0030\u0030\u0032\u0064\u005c\u0075\u0030\u0030\u0037\u0034\u005c\u0075\u0030\u0030\u0036\u0035\u005c\u0075\u0030\u0030\u0037\u0033\u005c\u0075\u0030\u0030\u0037\u0034\u005c\u0075\u0030\u0030\u0032\u0064\u005c\u0075\u0030\u0030\u0037\u0036\u005c\u0075\u0030\u0030\u0036\u0031\u005c\u0075\u0030\u0030\u0036\u0063\u005c\u0075\u0030\u0030\u0037\u0035\u005c\u0075\u0030\u0030\u0036\u0035`
// base64(escapedUnicode("my-secret-key-test-value"))
b64EscapedUnicode = "XHUwMDZkXHUwMDc5XHUwMDJkXHUwMDczXHUwMDY1XHUwMDYzXHUwMDcyXHUwMDY1XHUwMDc0XHUwMDJkXHUwMDZiXHUwMDY1XHUwMDc5XHUwMDJkXHUwMDc0XHUwMDY1XHUwMDczXHUwMDc0XHUwMDJkXHUwMDc2XHUwMDYxXHUwMDZjXHUwMDc1XHUwMDY1"
// escapedUnicode(base64("my-secret-key-test-value"))
escapedUnicodeB64 = `\u0062\u0058\u006b\u0074\u0063\u0032\u0056\u006a\u0063\u006d\u0056\u0030\u004c\u0057\u0074\u006c\u0065\u0053\u0031\u0030\u005a\u0058\u004e\u0030\u004c\u0058\u005a\u0068\u0062\u0048\u0056\u006c`
)
// "token: bXktc2VjcmV0LWtleS10ZXN0LXZhbHVl end" as UTF-16LE
utf16ContainingBase64 := []byte{
116, 0, 111, 0, 107, 0, 101, 0, 110, 0, 58, 0, 32, 0,
98, 0, 88, 0, 107, 0, 116, 0, 99, 0, 50, 0, 86, 0, 106, 0,
99, 0, 109, 0, 86, 0, 48, 0, 76, 0, 87, 0, 116, 0, 108, 0,
101, 0, 83, 0, 49, 0, 48, 0, 90, 0, 88, 0, 78, 0, 48, 0,
76, 0, 88, 0, 90, 0, 104, 0, 98, 0, 72, 0, 86, 0, 108, 0,
32, 0, 101, 0, 110, 0, 100, 0,
}
tests := []struct {
name string
input []byte
depth int
wantKeyword bool
}{
{
name: "double base64, depth=1, miss",
input: []byte("token: " + doubleEncoded),
depth: 1,
wantKeyword: false,
},
{
name: "double base64, depth=2, found",
input: []byte("token: " + doubleEncoded),
depth: 2,
wantKeyword: true,
},
{
name: "utf16+base64, depth=1, miss",
input: utf16ContainingBase64,
depth: 1,
wantKeyword: false,
},
{
name: "utf16+base64, depth=2, found",
input: utf16ContainingBase64,
depth: 2,
wantKeyword: true,
},
{
name: "escaped unicode, depth=1, found",
input: []byte("token: " + escapedUnicode),
depth: 1,
wantKeyword: true,
},
{
name: "double escaped unicode, depth=1, miss",
input: []byte("token: " + doubleEscapedUnicode),
depth: 1,
wantKeyword: false,
},
{
name: "double escaped unicode, depth=2, found",
input: []byte("token: " + doubleEscapedUnicode),
depth: 2,
wantKeyword: true,
},
{
name: "base64+escaped unicode, depth=1, miss",
input: []byte("token: " + b64EscapedUnicode),
depth: 1,
wantKeyword: false,
},
{
name: "base64+escaped unicode, depth=2, found",
input: []byte("token: " + b64EscapedUnicode),
depth: 2,
wantKeyword: true,
},
{
name: "escaped unicode+base64, depth=1, miss",
input: []byte("token: " + escapedUnicodeB64),
depth: 1,
wantKeyword: false,
},
{
name: "escaped unicode+base64, depth=2, found",
input: []byte("token: " + escapedUnicodeB64),
depth: 2,
wantKeyword: true,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
ctx := context.Background()
detector := &passthroughDetector{
keywords: []string{detectorKeyword},
detectorType: detector_typepb.DetectorType(9999),
}
e := &Engine{
AhoCorasickCore: ahocorasick.NewAhoCorasickCore([]detectors.Detector{detector}),
decoders: decoders.DefaultDecoders(),
detectableChunksChan: make(chan detectableChunk, 64),
sourceManager: sources.NewManager(),
maxDecodeDepth: tt.depth,
}
e.sourceManager.ScanChunk(&sources.Chunk{Data: tt.input})
go e.scannerWorker(ctx)
var found bool
timeout := time.After(2 * time.Second)
Loop:
for {
select {
case dc := <-e.detectableChunksChan:
dc.wgDoneFn()
found = true
for {
select {
case dc2 := <-e.detectableChunksChan:
dc2.wgDoneFn()
case <-time.After(200 * time.Millisecond):
break Loop
}
}
case <-timeout:
break Loop
}
}
if tt.wantKeyword {
assert.True(t, found, "expected detector match")
} else {
assert.False(t, found, "unexpected detector match")
}
})
}
}