[fix] - Improve UTF8 decoder's handling of non-printable characters (#3588)
* Avoid removing non-printable characters when decoding * use byte slice * remove new line --------- Co-authored-by: Miccah <[email protected]>
This commit is contained in:
@@ -37,12 +37,12 @@ func utf16ToUTF8(b []byte) ([]byte, error) {
|
||||
var bufBE, bufLE bytes.Buffer
|
||||
for i := 0; i < len(b)-1; i += 2 {
|
||||
if r := rune(binary.BigEndian.Uint16(b[i:])); b[i] == 0 && utf8.ValidRune(r) {
|
||||
if isValidByte(byte(r)) {
|
||||
if isPrintableByte(byte(r)) {
|
||||
bufBE.WriteRune(r)
|
||||
}
|
||||
}
|
||||
if r := rune(binary.LittleEndian.Uint16(b[i:])); b[i+1] == 0 && utf8.ValidRune(r) {
|
||||
if isValidByte(byte(r)) {
|
||||
if isPrintableByte(byte(r)) {
|
||||
bufLE.WriteRune(r)
|
||||
}
|
||||
}
|
||||
|
||||
+46
-24
@@ -1,7 +1,6 @@
|
||||
package decoders
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/trufflesecurity/trufflehog/v3/pkg/pb/detectorspb"
|
||||
@@ -29,35 +28,58 @@ func (d *UTF8) FromChunk(chunk *sources.Chunk) *DecodableChunk {
|
||||
return decodableChunk
|
||||
}
|
||||
|
||||
// extractSubstrings performs similarly to the strings binutil,
|
||||
// extacting contigous portions of printable characters that we care
|
||||
// about from some bytes
|
||||
func extractSubstrings(b []byte) []byte {
|
||||
// utf8ReplacementBytes holds the UTF-8 encoded form of the Unicode replacement character (U+FFFD).
|
||||
// This is pre-computed since it's used frequently when replacing invalid UTF-8 sequences
|
||||
// and control characters.
|
||||
var utf8ReplacementBytes = []byte(string(utf8.RuneError))
|
||||
|
||||
field := make([]byte, len(b))
|
||||
fieldLen := 0
|
||||
buf := &bytes.Buffer{}
|
||||
for i, c := range b {
|
||||
if isValidByte(c) {
|
||||
field[fieldLen] = c
|
||||
fieldLen++
|
||||
} else {
|
||||
if fieldLen > 5 {
|
||||
buf.Write(field[:fieldLen])
|
||||
// extractSubstrings sanitizes byte sequences to ensure consistent handling of malformed input
|
||||
// while maintaining readable content. It handles ASCII and UTF-8 data as follows:
|
||||
//
|
||||
// For ASCII range (0-127): preserves printable characters (32-126) while replacing
|
||||
// control characters with the UTF-8 replacement character.
|
||||
// https://cs.opensource.google/go/go/+/refs/tags/go1.23.3:src/unicode/utf8/utf8.go;l=16
|
||||
//
|
||||
// For multi-byte sequences: preserves valid UTF-8 as-is, while invalid sequences
|
||||
// are replaced with a single UTF-8 replacement character.
|
||||
func extractSubstrings(b []byte) []byte {
|
||||
dataLen := len(b)
|
||||
buf := make([]byte, 0, dataLen)
|
||||
for idx := 0; idx < dataLen; {
|
||||
// If it's ASCII, handle separately.
|
||||
// This is faster than decoding for common cases.
|
||||
if b[idx] < utf8.RuneSelf {
|
||||
if isPrintableByte(b[idx]) {
|
||||
buf = append(buf, b[idx])
|
||||
} else {
|
||||
buf = append(buf, utf8ReplacementBytes...)
|
||||
}
|
||||
fieldLen = 0
|
||||
idx++
|
||||
continue
|
||||
}
|
||||
|
||||
if i == len(b)-1 && fieldLen > 5 {
|
||||
buf.Write(field[:fieldLen])
|
||||
r, size := utf8.DecodeRune(b[idx:])
|
||||
if r == utf8.RuneError {
|
||||
// Collapse any malformed sequence into a single replacement character
|
||||
// rather than replacing each byte individually.
|
||||
buf = append(buf, utf8ReplacementBytes...)
|
||||
idx++
|
||||
} else {
|
||||
// Keep valid multi-byte UTF-8 sequences intact to preserve unicode characters.
|
||||
buf = append(buf, b[idx:idx+size]...)
|
||||
idx += size
|
||||
}
|
||||
}
|
||||
|
||||
return buf.Bytes()
|
||||
return buf
|
||||
}
|
||||
|
||||
func isValidByte(c byte) bool {
|
||||
// https://www.rapidtables.com/code/text/ascii-table.html
|
||||
// split on anything that is not ascii space through tilde
|
||||
return c > 31 && c < 127
|
||||
}
|
||||
// isPrintableByte reports whether a byte represents a printable ASCII character
|
||||
// using a fast byte-range check. This avoids the overhead of utf8.DecodeRune
|
||||
// for the common case of ASCII characters (0-127), since we know any byte < 128
|
||||
// represents a complete ASCII character and doesn't need UTF-8 decoding.
|
||||
// This includes letters, digits, punctuation, and symbols, but excludes control characters.
|
||||
// The upper bound is 127 (not 128) because 127 is the DEL control character.
|
||||
//
|
||||
// https://www.rapidtables.com/code/text/ascii-table.html
|
||||
func isPrintableByte(c byte) bool { return c > 31 && c < 127 }
|
||||
|
||||
+306
-13
@@ -1,6 +1,7 @@
|
||||
package decoders
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kylelemons/godebug/pretty"
|
||||
@@ -8,7 +9,7 @@ import (
|
||||
"github.com/trufflesecurity/trufflehog/v3/pkg/sources"
|
||||
)
|
||||
|
||||
func TestUTF8_FromChunk(t *testing.T) {
|
||||
func TestUTF8_FromChunk_ValidUTF8(t *testing.T) {
|
||||
type args struct {
|
||||
chunk *sources.Chunk
|
||||
}
|
||||
@@ -29,16 +30,7 @@ func TestUTF8_FromChunk(t *testing.T) {
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "successful binary decode",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("\xf0\x28\x8c\x28 not-entirely utf8 chunk that should decode successfully")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("( not-entirely utf8 chunk that should decode successfully")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "unsuccessful decode",
|
||||
name: "empty chunk",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: nil,
|
||||
@@ -46,18 +38,319 @@ func TestUTF8_FromChunk(t *testing.T) {
|
||||
want: nil,
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid UTF8 with control characters",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("FIRST_KEY_123456\x00SECOND_KEY_789012")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("FIRST_KEY_123456\x00SECOND_KEY_789012")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid UTF8 with all ASCII control characters",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
'S', 'T', 'A', 'R', 'T',
|
||||
0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A,
|
||||
0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0x10, 0x11, 0x12, 0x13, 0x14,
|
||||
0x15, 0x16, 0x17, 0x18, 0x19, 0x1A, 0x1B, 0x1C, 0x1D, 0x1E, 0x1F,
|
||||
'E', 'N', 'D',
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("START\x01\x02\x03\x04\x05\x06\x07\x08\x09\x0A\x0B\x0C\x0D\x0E\x0F\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1A\x1B\x1C\x1D\x1E\x1FEND")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "aws key in binary data - valid utf8",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("AWS_ACCESS_KEY_ID\x00\x00\x00AKIAEXAMPLEKEY123\x00")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("AWS_ACCESS_KEY_ID\x00\x00\x00AKIAEXAMPLEKEY123\x00")},
|
||||
wantErr: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
d := &UTF8{}
|
||||
got := d.FromChunk(tt.args.chunk)
|
||||
if got != nil && tt.want != nil {
|
||||
if diff := pretty.Compare(string(got.Data), string(tt.want.Data)); diff != "" {
|
||||
t.Errorf("%s: Plain.FromChunk() diff: (-got +want)\n%s", tt.name, diff)
|
||||
t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff)
|
||||
}
|
||||
} else {
|
||||
if diff := pretty.Compare(got, tt.want); diff != "" {
|
||||
t.Errorf("%s: Plain.FromChunk() diff: (-got +want)\n%s", tt.name, diff)
|
||||
t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestUTF8_FromChunk_InvalidUTF8(t *testing.T) {
|
||||
type args struct {
|
||||
chunk *sources.Chunk
|
||||
}
|
||||
tests := []struct {
|
||||
name string
|
||||
d *UTF8
|
||||
args args
|
||||
want *sources.Chunk
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "basic invalid utf8",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("\xF0\x28\x8C\x28")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("�(�(")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "invalid utf8 between words",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("START\xF0\x28\x8C\x28MIDDLE\xC0\x80END")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("START�(�(MIDDLE��END")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "binary data with embedded text",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0xF0, 'S', 'E', 'C', 'R', 'E', 'T', // Invalid UTF-8 before text
|
||||
0xC0, 0x80, // Invalid UTF-8 sequence
|
||||
'V', 'A', 'L', 'U', 'E',
|
||||
0xFF, 0x8C, // More invalid UTF-8
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("�SECRET��VALUE��")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "binary protocol with length fields",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0x02, // frame type
|
||||
0x00, 0x00, 0x00, 0x0A, // length field
|
||||
'P', 'A', 'S', 'S', 'W', 'O', 'R', 'D', '1', '2',
|
||||
0xFE, 0xFF, // checksum
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("�����PASSWORD12��")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "truncated utf8 sequence",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("PREFIX\xF0\x28SUFFIX")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("PREFIX�(SUFFIX")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "multiple invalid sequences",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0xF0, 'A', // Invalid + ASCII
|
||||
0xC0, 0x80, // Invalid sequence
|
||||
'B',
|
||||
0xFF, // Single invalid byte
|
||||
'C',
|
||||
0xF0, 0x28, 0x8C, 0x28, // Invalid sequence
|
||||
'D',
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("�A��B�C�(�(D")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "invalid utf8 header with embedded secret",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0xF0, 0x28, 0x8C, // Invalid UTF-8 sequence
|
||||
'S', 'E', 'C', 'R', 'E', 'T', '=',
|
||||
0xC0, 0x80, // Another invalid UTF-8 sequence
|
||||
'A', 'K', 'I', 'A', '1', '2', '3', '4', '5', '6',
|
||||
0xF8, 0x88, // More invalid UTF-8
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("�(�SECRET=��AKIA123456��")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "key value pairs with length prefixes",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0x00, 0x01, // header
|
||||
'A', 'P', 'I', '_', 'K', 'E', 'Y', '=',
|
||||
0x00, 0x00, 0x00, 0x05, // length
|
||||
'A', 'K', 'I', 'A', '5',
|
||||
0xFF, // separator
|
||||
'S', 'E', 'C', 'R', 'E', 'T', '=',
|
||||
0x00, 0x00, 0x00, 0x06,
|
||||
'S', 'E', 'C', 'R', 'E', 'T',
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("��API_KEY=����AKIA5�SECRET=����SECRET")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "mixed binary and invalid utf8",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0x00, 0x01, // valid binary
|
||||
0xF0, 0x28, // invalid UTF-8
|
||||
'K', 'E', 'Y', '=',
|
||||
0xC0, 0x80, // more invalid UTF-8
|
||||
'V', 'A', 'L', 'U', 'E',
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("���(KEY=��VALUE")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "very large utf8 sequence",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte(strings.Repeat("世界", 1000))},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte(strings.Repeat("世界", 1000))},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "single byte chunk",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{0x41}}, // Single 'A'
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("A")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "chunk with zero bytes between valid utf8",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("hello\x00world\x00!")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("hello\x00world\x00!")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "multi-byte unicode characters",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("🌍🌎🌏")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("🌍🌎🌏")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "mixed ascii and multi-byte unicode with invalid sequences",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("Hello 世界\xF0\x28\x8C\x28Testing🌍")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("Hello 世界�(�(Testing🌍")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "chunk ending with partial utf8 sequence",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("Hello\xE2\x80")}, // Incomplete UTF-8 sequence
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("Hello��")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "chunk with all printable ascii chars",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte(" !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte(" !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "alternating valid and invalid utf8",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte("A\xF0B\xF0C\xF0D")},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("A�B�C�D")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "overlong utf8 encoding",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{0xF0, 0x82, 0x82, 0xAC}}, // Overlong encoding of €
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("����")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "utf8 boundary conditions",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{
|
||||
0xFF, // Invalid single byte -> �
|
||||
0xC2, 0x80, // Minimum valid 2-byte UTF-8 sequence (U+0080) -> \u0080
|
||||
0xDF, 0xBF, // Maximum valid 2-byte UTF-8 sequence (U+07FF) -> ߿
|
||||
0xE0, 0x80, 0x80, // Invalid 3-byte (overlong encoding) -> �
|
||||
0xEF, 0xBF, 0xBF, // Valid 3-byte sequence for U+FFFF -> \uffff
|
||||
0xF0, 0x28, 0x8C, 0x28, // Invalid UTF-8 mixed with ASCII -> �(�(
|
||||
0xF4, 0x8F, 0xBF, 0xBF, // Valid 4-byte sequence for U+10FFFF -> \U0010ffff
|
||||
}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("�\u0080߿���\uffff�(�(\U0010ffff")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "chunk with byte order mark (BOM)",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
chunk: &sources.Chunk{Data: []byte{0xEF, 0xBB, 0xBF, 'h', 'e', 'l', 'l', 'o'}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("\uFEFFhello")},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "chunk with surrogate pairs",
|
||||
d: &UTF8{},
|
||||
args: args{
|
||||
// Invalid UTF-8 encoding of surrogate pairs
|
||||
chunk: &sources.Chunk{Data: []byte{0xED, 0xA0, 0x80, 0xED, 0xB0, 0x80}},
|
||||
},
|
||||
want: &sources.Chunk{Data: []byte("������")},
|
||||
wantErr: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
d := &UTF8{}
|
||||
got := d.FromChunk(tt.args.chunk)
|
||||
if got != nil && tt.want != nil {
|
||||
if diff := pretty.Compare(string(got.Data), string(tt.want.Data)); diff != "" {
|
||||
t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff)
|
||||
}
|
||||
} else {
|
||||
if diff := pretty.Compare(got, tt.want); diff != "" {
|
||||
t.Errorf("%s: UTF8.FromChunk() diff: (-got +want)\n%s", tt.name, diff)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user