[INS-242] Add more validations to Custom Detector config (#4642)

This commit is contained in:
Mustansir
2026-01-06 16:31:45 +05:00
committed by GitHub
parent a633174c3b
commit 50aa4695be
3 changed files with 177 additions and 0 deletions
+9
View File
@@ -47,6 +47,15 @@ func NewWebhookCustomRegex(pb *custom_detectorspb.CustomRegex) (*CustomRegexWebh
if err := ValidateRegex(pb.Regex); err != nil {
return nil, err
}
if err := ValidateRegexSlice(pb.ExcludeRegexesCapture); err != nil {
return nil, err
}
if err := ValidateRegexSlice(pb.ExcludeRegexesMatch); err != nil {
return nil, err
}
if err := ValidatePrimaryRegexName(pb.PrimaryRegexName, pb.Regex); err != nil {
return nil, err
}
for _, verify := range pb.Verify {
if err := ValidateVerifyEndpoint(verify.Endpoint, verify.Unsafe); err != nil {
@@ -2,6 +2,7 @@ package custom_detectors
import (
"context"
"strings"
"testing"
"github.com/google/go-cmp/cmp"
@@ -559,6 +560,153 @@ func TestDetectorValidations(t *testing.T) {
}
}
func TestNewWebhookCustomRegex_Validation(t *testing.T) {
t.Parallel()
// A known-good baseline; each test case mutates exactly one thing to trigger a specific validator.
base := func() *custom_detectorspb.CustomRegex {
return &custom_detectorspb.CustomRegex{
Name: "ok",
Keywords: []string{"kw"},
Regex: map[string]string{
"main": `\btoken_[a-z]+\b`,
},
PrimaryRegexName: "main",
ExcludeRegexesCapture: []string{
`^skip_.*$`,
},
ExcludeRegexesMatch: []string{
`^ignore_.*$`,
},
Verify: []*custom_detectorspb.VerifierConfig{
{
Endpoint: "https://example.com/verify",
Unsafe: false,
Headers: []string{"Authorization: Bearer x"},
},
},
}
}
tests := []struct {
name string
mutate func(*custom_detectorspb.CustomRegex)
wantErr bool
wantErrSubstr string // substring expected in error
}{
{
name: "Validate everything ok",
mutate: func(pb *custom_detectorspb.CustomRegex) {},
},
{
name: "ValidateKeywords: no keywords",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Keywords = nil
},
wantErr: true,
wantErrSubstr: "no keywords",
},
{
name: "ValidateKeywords: empty keyword",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Keywords = []string{""}
},
wantErr: true,
wantErrSubstr: "empty keyword",
},
{
name: "ValidateRegex: no regex",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Regex = nil
},
wantErr: true,
wantErrSubstr: "no regex",
},
{
name: "ValidateRegex: invalid regex in map",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Regex = map[string]string{"main": "("} // invalid
},
wantErr: true,
wantErrSubstr: "regex 'main':",
},
{
name: "ValidateRegexSlice: invalid exclude_regexes_capture",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.ExcludeRegexesCapture = []string{"("} // invalid
},
wantErr: true,
wantErrSubstr: "regex '1':",
},
{
name: "ValidateRegexSlice: invalid exclude_regexes_match",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.ExcludeRegexesMatch = []string{"("} // invalid
},
wantErr: true,
wantErrSubstr: "regex '1':",
},
{
name: "ValidatePrimaryRegexName: unknown primary regex name",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.PrimaryRegexName = "does-not-exist"
},
wantErr: true,
wantErrSubstr: `unknown primary regex name: "does-not-exist"`,
},
{
name: "ValidateVerifyEndpoint: empty endpoint",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Verify = []*custom_detectorspb.VerifierConfig{
{Endpoint: "", Unsafe: false, Headers: []string{"A: b"}},
}
},
wantErr: true,
wantErrSubstr: "no endpoint",
},
{
name: "ValidateVerifyEndpoint: http endpoint without unsafe=true",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Verify = []*custom_detectorspb.VerifierConfig{
{Endpoint: "http://example.com/verify", Unsafe: false, Headers: []string{"A: b"}},
}
},
wantErr: true,
wantErrSubstr: "http endpoint must have unsafe=true",
},
{
name: "ValidateVerifyHeaders: header missing colon",
mutate: func(pb *custom_detectorspb.CustomRegex) {
pb.Verify = []*custom_detectorspb.VerifierConfig{
{Endpoint: "https://example.com/verify", Unsafe: false, Headers: []string{"Authorization Bearer x"}},
}
},
wantErr: true,
wantErrSubstr: `must contain a colon`,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
pb := base()
tt.mutate(pb)
got, err := NewWebhookCustomRegex(pb)
if (err != nil) != tt.wantErr {
t.Fatalf("expected error=%v, got error=%v (result=%#v)", tt.wantErr, err != nil, got)
}
if tt.wantErr && got != nil {
t.Fatalf("expected nil result on error, got=%#v", got)
}
if tt.wantErr && !strings.Contains(err.Error(), tt.wantErrSubstr) {
t.Fatalf("error mismatch:\n got: %q\n want substring: %q", err.Error(), tt.wantErrSubstr)
}
})
}
}
func BenchmarkProductIndices(b *testing.B) {
for i := 0; i < b.N; i++ {
_ = productIndices(3, 2, 6)
+20
View File
@@ -32,6 +32,26 @@ func ValidateRegex(regex map[string]string) error {
return nil
}
func ValidateRegexSlice(regex []string) error {
for i, reg := range regex {
if _, err := regexp.Compile(reg); err != nil {
return fmt.Errorf("regex '%d': %w", i+1, err)
}
}
return nil
}
// validates if a provided non-empty primary regex name exists in the map of regexes
func ValidatePrimaryRegexName(primaryRegexName string, regexes map[string]string) error {
if primaryRegexName == "" {
return nil
}
if _, ok := regexes[primaryRegexName]; !ok {
return fmt.Errorf("unknown primary regex name: %q", primaryRegexName)
}
return nil
}
func ValidateVerifyEndpoint(endpoint string, unsafe bool) error {
if len(endpoint) == 0 {
return fmt.Errorf("no endpoint")