mirror of
https://github.com/pgsty/minio.git
synced 2026-07-20 12:40:24 +03:00
fix: CVE-2026-39414 harden S3 Select oversized record handling
Enforce the 1 MiB maxCharsPerRecord limit while splitting CSV and line-delimited JSON input so oversized records are rejected before they can be buffered and parsed. Return OverMaxRecordSize for these failures instead of collapsing them into InternalError, and preserve splitter errors in the JSON worker so oversized-record failures are not lost after successful partial decode.
This commit is contained in:
@@ -17,6 +17,10 @@
|
||||
|
||||
package json
|
||||
|
||||
import "errors"
|
||||
|
||||
var errLineTooLong = errors.New("line exceeds maximum allowed length")
|
||||
|
||||
type s3Error struct {
|
||||
code string
|
||||
message string
|
||||
@@ -61,3 +65,12 @@ func errJSONParsingError(err error) *s3Error {
|
||||
cause: err,
|
||||
}
|
||||
}
|
||||
|
||||
func errOverMaxRecordSize(err error) *s3Error {
|
||||
return &s3Error{
|
||||
code: "OverMaxRecordSize",
|
||||
message: "The length of a record in the input or result is greater than maxCharsPerRecord of 1 MB.",
|
||||
statusCode: 400,
|
||||
cause: err,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -118,10 +118,39 @@ func (r *PReader) nextSplit(skip int, dst []byte) ([]byte, error) {
|
||||
return dst, io.EOF
|
||||
}
|
||||
}
|
||||
// Read until next line.
|
||||
in, err := r.buf.ReadBytes('\n')
|
||||
dst = append(dst, in...)
|
||||
return dst, err
|
||||
|
||||
tailLen := len(dst)
|
||||
if i := bytes.LastIndexByte(dst, '\n'); i >= 0 {
|
||||
tailLen = len(dst) - i - 1
|
||||
}
|
||||
tailStart := len(dst) - tailLen
|
||||
if tailLen > maxCharsPerRecord {
|
||||
return dst[:tailStart], errOverMaxRecordSize(errLineTooLong)
|
||||
}
|
||||
|
||||
for {
|
||||
in, err := r.buf.ReadSlice('\n')
|
||||
switch err {
|
||||
case nil:
|
||||
if tailLen+len(in)-1 > maxCharsPerRecord {
|
||||
return dst[:tailStart], errOverMaxRecordSize(errLineTooLong)
|
||||
}
|
||||
dst = append(dst, in...)
|
||||
return dst, nil
|
||||
case bufio.ErrBufferFull:
|
||||
if tailLen+len(in) > maxCharsPerRecord {
|
||||
return dst[:tailStart], errOverMaxRecordSize(errLineTooLong)
|
||||
}
|
||||
dst = append(dst, in...)
|
||||
tailLen += len(in)
|
||||
default:
|
||||
if tailLen+len(in) > maxCharsPerRecord {
|
||||
return dst[:tailStart], errOverMaxRecordSize(errLineTooLong)
|
||||
}
|
||||
dst = append(dst, in...)
|
||||
return dst, err
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// jsonSplitSize is the size of each block.
|
||||
@@ -129,6 +158,9 @@ func (r *PReader) nextSplit(skip int, dst []byte) ([]byte, error) {
|
||||
// 128KB appears to be a very reasonable default.
|
||||
const jsonSplitSize = 128 << 10
|
||||
|
||||
// S3 Select allows up to 1 MiB per input record.
|
||||
const maxCharsPerRecord = 1 << 20
|
||||
|
||||
// startReaders will read the header if needed and spin up a parser
|
||||
// and a number of workers based on GOMAXPROCS.
|
||||
// If an error is returned no goroutines have been started and r.err will have been set.
|
||||
@@ -177,6 +209,10 @@ func (r *PReader) startReaders() {
|
||||
go func() {
|
||||
for in := range r.input {
|
||||
if len(in.input) == 0 {
|
||||
if in.input != nil {
|
||||
r.bufferPool.Put(in.input[:0])
|
||||
in.input = nil
|
||||
}
|
||||
in.dst <- nil
|
||||
continue
|
||||
}
|
||||
@@ -207,7 +243,9 @@ func (r *PReader) startReaders() {
|
||||
//nolint:staticcheck // SA6002 Using pointer would allocate more since we would have to copy slice header before taking a pointer.
|
||||
r.bufferPool.Put(in.input)
|
||||
in.input = nil
|
||||
in.err = d.Err()
|
||||
if err := d.Err(); err != nil {
|
||||
in.err = err
|
||||
}
|
||||
in.dst <- all
|
||||
}
|
||||
}()
|
||||
|
||||
Reference in New Issue
Block a user