Files
trufflehog/pkg/handlers/default.go
Gleb Haranin f3eff52825 fix: report accurate line numbers for chunked file scanning (#1876) (#4615)
Fixes #1876

Problem
Files are processed in ~13KB chunks by default (10KB of actual data plus a 3KB peek into the next chunk). To infer the line the engine would take the line from chunk's metadata SourceMetadata.Line and would consider it absolute, that is it's relative to the entire file. The problem is it was never set anywhere. Since every chunk had line equals to 1 in metadata, all secret scan results appeared at line 1 + FragmentLineOffset() which resulted in line being relative to the chunk not the file itself.

Solution
Track cumulative line numbers in handleNonArchiveContent() by counting newlines as chunks are processed. Each DataOrErr now carries the correct starting line, which flows through to SourceMetadata.Filesystem.Line, giving FragmentFirstLineAndLink() the correct fragStart for the final calculation.

Affected Sources
Directly fixed:

Filesystem - now reports accurate line numbers. I have tested it using the test file provided in the issue wget https://gist.githubusercontent.com/det/1526b4c16d0e07ac023d75c912a68658/raw/c3061c14a811205a65cbdcf0065bd3c11d88bfcb/test.txt

Not affected (no Line field in proto):

S3, GCS, Jenkins, stdin - use handlers but their metadata protos lack a Line field. I think it makes sense for S3 and GCS to report line numbers so it could be a good future change.

Partially affected (own line tracking):

Git, GitHub, GitLab - regular text diffs use git-based scanning with built-in line tracking (unaffected). However, binary archives (tar.gz, zip) go through HandleBinary() → handlers.HandleFile() and benefit from this fix.
others like BitBucket use their own implementations so not affected
2026-01-15 13:21:29 -05:00

159 lines
5.8 KiB
Go

package handlers
import (
"bytes"
"context"
"errors"
"fmt"
"io"
"time"
"github.com/trufflesecurity/trufflehog/v3/pkg/common"
logContext "github.com/trufflesecurity/trufflehog/v3/pkg/context"
"github.com/trufflesecurity/trufflehog/v3/pkg/sources"
)
// defaultHandler is a handler for non-archive files.
// It is embedded in other specialized handlers to provide a consistent way of handling non-archive content
// once it has been extracted or decompressed by the specific handler.
// This allows the specialized handlers to focus on their specific archive formats while leveraging
// the common functionality provided by the defaultHandler for processing the extracted content.
type defaultHandler struct {
chunkReader sources.ChunkReader
metrics *metrics
}
// defaultHandlerOption is a functional option for configuring a defaultHandler.
type defaultHandlerOption func(*defaultHandler)
// withChunkReader sets a custom ChunkReader for the handler.
// This is primarily used for testing to inject mock chunk readers.
func withChunkReader(cr sources.ChunkReader) defaultHandlerOption {
return func(h *defaultHandler) { h.chunkReader = cr }
}
// newDefaultHandler creates a defaultHandler with metrics configured based on the provided handlerType.
// The handlerType parameter is used to initialize the metrics instance with the appropriate handler type,
// ensuring that the metrics recorded within the defaultHandler methods are correctly attributed to the
// specific handler that invoked them.
func newDefaultHandler(handlerType handlerType, opts ...defaultHandlerOption) *defaultHandler {
h := &defaultHandler{
chunkReader: sources.NewChunkReader(),
metrics: newHandlerMetrics(handlerType),
}
for _, opt := range opts {
opt(h)
}
return h
}
// HandleFile processes non-archive files.
//
// Fatal errors that will terminate processing include:
// - Context cancellation
// - Context deadline exceeded
// - Errors writing to the data channel
//
// Non-fatal errors that will be logged but allow processing to continue include:
// - Errors reading individual chunks from the input (wrapped as ErrProcessingWarning)
func (h *defaultHandler) HandleFile(ctx logContext.Context, input fileReader) chan DataOrErr {
// Shared channel for both archive and non-archive content.
dataOrErrChan := make(chan DataOrErr, defaultBufferSize)
go func() {
defer close(dataOrErrChan)
start := time.Now()
err := h.handleNonArchiveContent(ctx, newMimeTypeReaderFromFileReader(input), dataOrErrChan)
if err == nil {
h.metrics.incFilesProcessed()
}
// Update the metrics for the file processing and handle errors.
h.measureLatencyAndHandleErrors(ctx, start, err, dataOrErrChan)
}()
return dataOrErrChan
}
// measureLatencyAndHandleErrors measures the latency of the file processing and updates the metrics accordingly.
// It also records errors and timeouts in the metrics.
func (h *defaultHandler) measureLatencyAndHandleErrors(
ctx logContext.Context,
start time.Time,
err error,
dataErrChan chan<- DataOrErr,
) {
if err == nil {
h.metrics.observeHandleFileLatency(time.Since(start).Milliseconds())
return
}
dataOrErr := DataOrErr{}
h.metrics.incErrors()
if errors.Is(err, context.DeadlineExceeded) {
h.metrics.incFileProcessingTimeouts()
dataOrErr.Err = fmt.Errorf("%w: error processing chunk", err)
if err := common.CancellableWrite(ctx, dataErrChan, dataOrErr); err != nil {
ctx.Logger().Error(err, "error writing to data channel")
}
return
}
dataOrErr.Err = err
if err := common.CancellableWrite(ctx, dataErrChan, dataOrErr); err != nil {
ctx.Logger().Error(err, "error writing to data channel")
}
}
// handleNonArchiveContent processes files that do not contain nested archives, serving as the final stage in the
// extraction/decompression process. It reads the content to detect its MIME type and decides whether to skip based
// on the type, particularly for binary files. It manages reading file chunks and writing them to the archive channel,
// effectively collecting the final bytes for further processing. This function is a key component in ensuring that all
// file content, regardless of being an archive or not, is handled appropriately.
// It also tracks line numbers for each chunk to enable accurate line number reporting for secrets.
func (h *defaultHandler) handleNonArchiveContent(
ctx logContext.Context,
reader mimeTypeReader,
dataOrErrChan chan DataOrErr,
) error {
mimeExt := reader.mimeExt
if common.SkipFile(mimeExt) || common.IsBinary(mimeExt) {
ctx.Logger().V(4).Info("skipping file: extension is ignored", "ext", mimeExt)
h.metrics.incFilesSkipped()
// Make sure we consume the reader to avoid potentially blocking indefinitely.
_, _ = io.Copy(io.Discard, reader)
return nil
}
// Track the current line number (1-indexed) across chunks.
// This allows accurate line number reporting for secrets found in later chunks.
currentLine := int64(1)
for chunkResult := range h.chunkReader(ctx, reader) {
dataOrErr := DataOrErr{}
if err := chunkResult.Error(); err != nil {
h.metrics.incErrors()
dataOrErr.Err = fmt.Errorf("%w: error reading chunk: %v", ErrProcessingWarning, err)
if writeErr := common.CancellableWrite(ctx, dataOrErrChan, dataOrErr); writeErr != nil {
return fmt.Errorf("%w: error writing to data channel: %v", ErrProcessingFatal, writeErr)
}
continue
}
chunkBytes := chunkResult.Bytes()
dataOrErr.Data = chunkBytes
dataOrErr.LineNumber = currentLine
if err := common.CancellableWrite(ctx, dataOrErrChan, dataOrErr); err != nil {
return err
}
h.metrics.incBytesProcessed(len(chunkBytes))
// Count newlines in the content portion only (excluding peek chunkResult) to avoid
// double-counting newlines that will appear at the start of the next chunk.
contentSize := chunkResult.ContentSize()
currentLine += int64(bytes.Count(chunkBytes[:contentSize], []byte{'\n'}))
}
return nil
}