diff --git a/pkg/decoders/utf16.go b/pkg/decoders/utf16.go index 5949196e4..f3da9c370 100644 --- a/pkg/decoders/utf16.go +++ b/pkg/decoders/utf16.go @@ -37,12 +37,12 @@ func utf16ToUTF8(b []byte) ([]byte, error) { var bufBE, bufLE bytes.Buffer for i := 0; i < len(b)-1; i += 2 { if r := rune(binary.BigEndian.Uint16(b[i:])); b[i] == 0 && utf8.ValidRune(r) { - if isValidByte(byte(r)) { + if isPrintableByte(byte(r)) { bufBE.WriteRune(r) } } if r := rune(binary.LittleEndian.Uint16(b[i:])); b[i+1] == 0 && utf8.ValidRune(r) { - if isValidByte(byte(r)) { + if isPrintableByte(byte(r)) { bufLE.WriteRune(r) } } diff --git a/pkg/decoders/utf8.go b/pkg/decoders/utf8.go index 49cfea451..f8dd847a0 100644 --- a/pkg/decoders/utf8.go +++ b/pkg/decoders/utf8.go @@ -1,7 +1,6 @@ package decoders import ( - "bytes" "unicode/utf8" "github.com/trufflesecurity/trufflehog/v3/pkg/pb/detectorspb" @@ -29,35 +28,58 @@ func (d *UTF8) FromChunk(chunk *sources.Chunk) *DecodableChunk { return decodableChunk } -// extractSubstrings performs similarly to the strings binutil, -// extacting contigous portions of printable characters that we care -// about from some bytes -func extractSubstrings(b []byte) []byte { +// utf8ReplacementBytes holds the UTF-8 encoded form of the Unicode replacement character (U+FFFD). +// This is pre-computed since it's used frequently when replacing invalid UTF-8 sequences +// and control characters. +var utf8ReplacementBytes = []byte(string(utf8.RuneError)) - field := make([]byte, len(b)) - fieldLen := 0 - buf := &bytes.Buffer{} - for i, c := range b { - if isValidByte(c) { - field[fieldLen] = c - fieldLen++ - } else { - if fieldLen > 5 { - buf.Write(field[:fieldLen]) +// extractSubstrings sanitizes byte sequences to ensure consistent handling of malformed input +// while maintaining readable content. It handles ASCII and UTF-8 data as follows: +// +// For ASCII range (0-127): preserves printable characters (32-126) while replacing +// control characters with the UTF-8 replacement character. +// https://cs.opensource.google/go/go/+/refs/tags/go1.23.3:src/unicode/utf8/utf8.go;l=16 +// +// For multi-byte sequences: preserves valid UTF-8 as-is, while invalid sequences +// are replaced with a single UTF-8 replacement character. +func extractSubstrings(b []byte) []byte { + dataLen := len(b) + buf := make([]byte, 0, dataLen) + for idx := 0; idx < dataLen; { + // If it's ASCII, handle separately. + // This is faster than decoding for common cases. + if b[idx] < utf8.RuneSelf { + if isPrintableByte(b[idx]) { + buf = append(buf, b[idx]) + } else { + buf = append(buf, utf8ReplacementBytes...) } - fieldLen = 0 + idx++ + continue } - if i == len(b)-1 && fieldLen > 5 { - buf.Write(field[:fieldLen]) + r, size := utf8.DecodeRune(b[idx:]) + if r == utf8.RuneError { + // Collapse any malformed sequence into a single replacement character + // rather than replacing each byte individually. + buf = append(buf, utf8ReplacementBytes...) + idx++ + } else { + // Keep valid multi-byte UTF-8 sequences intact to preserve unicode characters. + buf = append(buf, b[idx:idx+size]...) + idx += size } } - return buf.Bytes() + return buf } -func isValidByte(c byte) bool { - // https://www.rapidtables.com/code/text/ascii-table.html - // split on anything that is not ascii space through tilde - return c > 31 && c < 127 -} +// isPrintableByte reports whether a byte represents a printable ASCII character +// using a fast byte-range check. This avoids the overhead of utf8.DecodeRune +// for the common case of ASCII characters (0-127), since we know any byte < 128 +// represents a complete ASCII character and doesn't need UTF-8 decoding. +// This includes letters, digits, punctuation, and symbols, but excludes control characters. +// The upper bound is 127 (not 128) because 127 is the DEL control character. +// +// https://www.rapidtables.com/code/text/ascii-table.html +func isPrintableByte(c byte) bool { return c > 31 && c < 127 } diff --git a/pkg/decoders/utf8_test.go b/pkg/decoders/utf8_test.go index 18205d327..7d1c0c86a 100644 --- a/pkg/decoders/utf8_test.go +++ b/pkg/decoders/utf8_test.go @@ -1,6 +1,7 @@ package decoders import ( + "strings" "testing" "github.com/kylelemons/godebug/pretty" @@ -8,7 +9,7 @@ import ( "github.com/trufflesecurity/trufflehog/v3/pkg/sources" ) -func TestUTF8_FromChunk(t *testing.T) { +func TestUTF8_FromChunk_ValidUTF8(t *testing.T) { type args struct { chunk *sources.Chunk } @@ -29,16 +30,7 @@ func TestUTF8_FromChunk(t *testing.T) { wantErr: false, }, { - name: "successful binary decode", - d: &UTF8{}, - args: args{ - chunk: &sources.Chunk{Data: []byte("\xf0\x28\x8c\x28 not-entirely utf8 chunk that should decode successfully")}, - }, - want: &sources.Chunk{Data: []byte("( not-entirely utf8 chunk that should decode successfully")}, - wantErr: false, - }, - { - name: "unsuccessful decode", + name: "empty chunk", d: &UTF8{}, args: args{ chunk: nil, @@ -46,18 +38,319 @@ func TestUTF8_FromChunk(t *testing.T) { want: nil, wantErr: false, }, + { + name: "valid UTF8 with control characters", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("FIRST_KEY_123456\x00SECOND_KEY_789012")}, + }, + want: &sources.Chunk{Data: []byte("FIRST_KEY_123456\x00SECOND_KEY_789012")}, + wantErr: false, + }, + { + name: "valid UTF8 with all ASCII control characters", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 'S', 'T', 'A', 'R', 'T', + 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, + 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0x10, 0x11, 0x12, 0x13, 0x14, + 0x15, 0x16, 0x17, 0x18, 0x19, 0x1A, 0x1B, 0x1C, 0x1D, 0x1E, 0x1F, + 'E', 'N', 'D', + }}, + }, + want: &sources.Chunk{Data: []byte("START\x01\x02\x03\x04\x05\x06\x07\x08\x09\x0A\x0B\x0C\x0D\x0E\x0F\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1A\x1B\x1C\x1D\x1E\x1FEND")}, + wantErr: false, + }, + { + name: "aws key in binary data - valid utf8", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("AWS_ACCESS_KEY_ID\x00\x00\x00AKIAEXAMPLEKEY123\x00")}, + }, + want: &sources.Chunk{Data: []byte("AWS_ACCESS_KEY_ID\x00\x00\x00AKIAEXAMPLEKEY123\x00")}, + wantErr: false, + }, } + for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { d := &UTF8{} got := d.FromChunk(tt.args.chunk) if got != nil && tt.want != nil { if diff := pretty.Compare(string(got.Data), string(tt.want.Data)); diff != "" { - t.Errorf("%s: Plain.FromChunk() diff: (-got +want)\n%s", tt.name, diff) + t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff) } } else { if diff := pretty.Compare(got, tt.want); diff != "" { - t.Errorf("%s: Plain.FromChunk() diff: (-got +want)\n%s", tt.name, diff) + t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff) + } + } + }) + } +} + +func TestUTF8_FromChunk_InvalidUTF8(t *testing.T) { + type args struct { + chunk *sources.Chunk + } + tests := []struct { + name string + d *UTF8 + args args + want *sources.Chunk + wantErr bool + }{ + { + name: "basic invalid utf8", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("\xF0\x28\x8C\x28")}, + }, + want: &sources.Chunk{Data: []byte("�(�(")}, + wantErr: false, + }, + { + name: "invalid utf8 between words", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("START\xF0\x28\x8C\x28MIDDLE\xC0\x80END")}, + }, + want: &sources.Chunk{Data: []byte("START�(�(MIDDLE��END")}, + wantErr: false, + }, + { + name: "binary data with embedded text", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0xF0, 'S', 'E', 'C', 'R', 'E', 'T', // Invalid UTF-8 before text + 0xC0, 0x80, // Invalid UTF-8 sequence + 'V', 'A', 'L', 'U', 'E', + 0xFF, 0x8C, // More invalid UTF-8 + }}, + }, + want: &sources.Chunk{Data: []byte("�SECRET��VALUE��")}, + wantErr: false, + }, + { + name: "binary protocol with length fields", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0x02, // frame type + 0x00, 0x00, 0x00, 0x0A, // length field + 'P', 'A', 'S', 'S', 'W', 'O', 'R', 'D', '1', '2', + 0xFE, 0xFF, // checksum + }}, + }, + want: &sources.Chunk{Data: []byte("�����PASSWORD12��")}, + wantErr: false, + }, + { + name: "truncated utf8 sequence", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("PREFIX\xF0\x28SUFFIX")}, + }, + want: &sources.Chunk{Data: []byte("PREFIX�(SUFFIX")}, + wantErr: false, + }, + { + name: "multiple invalid sequences", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0xF0, 'A', // Invalid + ASCII + 0xC0, 0x80, // Invalid sequence + 'B', + 0xFF, // Single invalid byte + 'C', + 0xF0, 0x28, 0x8C, 0x28, // Invalid sequence + 'D', + }}, + }, + want: &sources.Chunk{Data: []byte("�A��B�C�(�(D")}, + wantErr: false, + }, + { + name: "invalid utf8 header with embedded secret", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0xF0, 0x28, 0x8C, // Invalid UTF-8 sequence + 'S', 'E', 'C', 'R', 'E', 'T', '=', + 0xC0, 0x80, // Another invalid UTF-8 sequence + 'A', 'K', 'I', 'A', '1', '2', '3', '4', '5', '6', + 0xF8, 0x88, // More invalid UTF-8 + }}, + }, + want: &sources.Chunk{Data: []byte("�(�SECRET=��AKIA123456��")}, + wantErr: false, + }, + { + name: "key value pairs with length prefixes", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0x00, 0x01, // header + 'A', 'P', 'I', '_', 'K', 'E', 'Y', '=', + 0x00, 0x00, 0x00, 0x05, // length + 'A', 'K', 'I', 'A', '5', + 0xFF, // separator + 'S', 'E', 'C', 'R', 'E', 'T', '=', + 0x00, 0x00, 0x00, 0x06, + 'S', 'E', 'C', 'R', 'E', 'T', + }}, + }, + want: &sources.Chunk{Data: []byte("��API_KEY=����AKIA5�SECRET=����SECRET")}, + wantErr: false, + }, + { + name: "mixed binary and invalid utf8", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0x00, 0x01, // valid binary + 0xF0, 0x28, // invalid UTF-8 + 'K', 'E', 'Y', '=', + 0xC0, 0x80, // more invalid UTF-8 + 'V', 'A', 'L', 'U', 'E', + }}, + }, + want: &sources.Chunk{Data: []byte("���(KEY=��VALUE")}, + wantErr: false, + }, + { + name: "very large utf8 sequence", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte(strings.Repeat("世界", 1000))}, + }, + want: &sources.Chunk{Data: []byte(strings.Repeat("世界", 1000))}, + wantErr: false, + }, + { + name: "single byte chunk", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{0x41}}, // Single 'A' + }, + want: &sources.Chunk{Data: []byte("A")}, + wantErr: false, + }, + { + name: "chunk with zero bytes between valid utf8", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("hello\x00world\x00!")}, + }, + want: &sources.Chunk{Data: []byte("hello\x00world\x00!")}, + wantErr: false, + }, + { + name: "multi-byte unicode characters", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("🌍🌎🌏")}, + }, + want: &sources.Chunk{Data: []byte("🌍🌎🌏")}, + wantErr: false, + }, + { + name: "mixed ascii and multi-byte unicode with invalid sequences", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("Hello 世界\xF0\x28\x8C\x28Testing🌍")}, + }, + want: &sources.Chunk{Data: []byte("Hello 世界�(�(Testing🌍")}, + wantErr: false, + }, + { + name: "chunk ending with partial utf8 sequence", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("Hello\xE2\x80")}, // Incomplete UTF-8 sequence + }, + want: &sources.Chunk{Data: []byte("Hello��")}, + wantErr: false, + }, + { + name: "chunk with all printable ascii chars", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte(" !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~")}, + }, + want: &sources.Chunk{Data: []byte(" !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~")}, + wantErr: false, + }, + { + name: "alternating valid and invalid utf8", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte("A\xF0B\xF0C\xF0D")}, + }, + want: &sources.Chunk{Data: []byte("A�B�C�D")}, + wantErr: false, + }, + { + name: "overlong utf8 encoding", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{0xF0, 0x82, 0x82, 0xAC}}, // Overlong encoding of € + }, + want: &sources.Chunk{Data: []byte("����")}, + wantErr: false, + }, + { + name: "utf8 boundary conditions", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{ + 0xFF, // Invalid single byte -> � + 0xC2, 0x80, // Minimum valid 2-byte UTF-8 sequence (U+0080) -> \u0080 + 0xDF, 0xBF, // Maximum valid 2-byte UTF-8 sequence (U+07FF) -> ߿ + 0xE0, 0x80, 0x80, // Invalid 3-byte (overlong encoding) -> � + 0xEF, 0xBF, 0xBF, // Valid 3-byte sequence for U+FFFF -> \uffff + 0xF0, 0x28, 0x8C, 0x28, // Invalid UTF-8 mixed with ASCII -> �(�( + 0xF4, 0x8F, 0xBF, 0xBF, // Valid 4-byte sequence for U+10FFFF -> \U0010ffff + }}, + }, + want: &sources.Chunk{Data: []byte("�\u0080߿���\uffff�(�(\U0010ffff")}, + wantErr: false, + }, + { + name: "chunk with byte order mark (BOM)", + d: &UTF8{}, + args: args{ + chunk: &sources.Chunk{Data: []byte{0xEF, 0xBB, 0xBF, 'h', 'e', 'l', 'l', 'o'}}, + }, + want: &sources.Chunk{Data: []byte("\uFEFFhello")}, + wantErr: false, + }, + { + name: "chunk with surrogate pairs", + d: &UTF8{}, + args: args{ + // Invalid UTF-8 encoding of surrogate pairs + chunk: &sources.Chunk{Data: []byte{0xED, 0xA0, 0x80, 0xED, 0xB0, 0x80}}, + }, + want: &sources.Chunk{Data: []byte("������")}, + wantErr: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + d := &UTF8{} + got := d.FromChunk(tt.args.chunk) + if got != nil && tt.want != nil { + if diff := pretty.Compare(string(got.Data), string(tt.want.Data)); diff != "" { + t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff) + } + } else { + if diff := pretty.Compare(got, tt.want); diff != "" { + t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff) } } })