@@ -15,7 +15,6 @@ const kContentType = Symbol('kContentType')
1515const kContentLength = Symbol ( 'kContentLength' )
1616const kUsed = Symbol ( 'kUsed' )
1717const kBytesRead = Symbol ( 'kBytesRead' )
18- const kPreservedBuffer = Symbol ( 'kPreservedBuffer' )
1918
2019const noop = ( ) => { }
2120
@@ -326,36 +325,14 @@ class BodyReadable extends Readable {
326325 */
327326 setEncoding ( encoding ) {
328327 if ( Buffer . isEncoding ( encoding ) ) {
329- // Preserve raw Buffer chunks for the consume path (body.text(),
330- // body.json(), etc.) before super.setEncoding() replaces them
331- // with decoded strings. Without this, the consume path would
332- // lose access to the original bytes — some of which may be held
333- // by the decoder for incomplete multi-byte sequences, and the
334- // rest converted to strings that can't be safely concatenated
335- // byte-wise.
336- const state = this . _readableState
337- const buffer = state . buffer
338- if ( buffer && state . length > 0 ) {
339- const bufferIndex = state . bufferIndex ?? 0
340- const preserved = [ ]
341- const source = typeof buffer . slice === 'function'
342- ? buffer . slice ( bufferIndex )
343- : buffer
344- for ( const data of source ) {
345- if ( Buffer . isBuffer ( data ) ) {
346- preserved . push ( data )
347- }
348- }
349- if ( preserved . length > 0 ) {
350- this [ kPreservedBuffer ] = ( this [ kPreservedBuffer ] || [ ] ) . concat ( preserved )
351- }
352- }
353-
354328 // Delegate to Node.js Readable.setEncoding() which initializes a
355329 // StringDecoder and re-encodes already-buffered chunks. This properly
356330 // handles multi-byte sequences split at chunk boundaries for the
357331 // for-await / on('data') paths. Without this, Node.js uses
358332 // buf.toString(encoding) on each chunk, producing U+FFFD for split chars.
333+ //
334+ // The consume path (body.text(), body.json(), ...) copes with the
335+ // decoded strings this leaves in state.buffer, see consumeStart().
359336 super . setEncoding ( encoding )
360337 }
361338 return this
@@ -464,17 +441,7 @@ function consumeStart (consume) {
464441
465442 const { _readableState : state } = consume . stream
466443
467- // If setEncoding() was called, state.buffer may contain decoded strings
468- // (which would break Buffer.concat in chunksDecode). Use the preserved
469- // raw Buffers (saved before super.setEncoding() in setEncoding()) for
470- // byte-level accurate consumption. Otherwise read from state.buffer.
471- const preserved = consume . stream [ kPreservedBuffer ]
472- if ( preserved && preserved . length > 0 ) {
473- for ( const chunk of preserved ) {
474- consumePush ( consume , chunk )
475- }
476- consume . stream [ kPreservedBuffer ] = null
477- } else if ( state . bufferIndex ) {
444+ if ( state . bufferIndex ) {
478445 const start = state . bufferIndex
479446 const end = state . buffer . length
480447 for ( let n = start ; n < end ; n ++ ) {
@@ -486,14 +453,29 @@ function consumeStart (consume) {
486453 }
487454 }
488455
456+ // If setEncoding() was called, state.buffer holds decoded strings, which
457+ // consumePush() turns back into bytes. The trailing bytes of a multi-byte
458+ // sequence split across a chunk boundary are not part of any of those
459+ // strings, they are held inside the decoder until the rest arrives, so
460+ // take them from there.
461+ const decoder = state . decoder
462+ if ( decoder != null && decoder . lastNeed > 0 ) {
463+ consumePush ( consume , Buffer . from ( decoder . lastChar . subarray ( 0 , decoder . lastTotal - decoder . lastNeed ) ) )
464+ }
465+
489466 if ( state . endEmitted ) {
490- consumeEnd ( this [ kConsume ] , this . _readableState . encoding )
491- } else {
492- consume . stream . on ( 'end' , function ( ) {
493- consumeEnd ( this [ kConsume ] , this . _readableState . encoding )
494- } )
467+ // No `this` to read the consume off here: consumeStart is a free function, called from
468+ // the queueMicrotask above. The callback below does have one, because the emitter passes
469+ // the stream as its receiver. Returning matters too - consumeEnd() clears consume.stream,
470+ // which the resume() below would then dereference.
471+ consumeEnd ( consume , state . encoding )
472+ return
495473 }
496474
475+ consume . stream . on ( 'end' , function ( ) {
476+ consumeEnd ( this [ kConsume ] , this . _readableState . encoding )
477+ } )
478+
497479 consume . stream . resume ( )
498480
499481 while ( consume . stream . read ( ) != null ) {
@@ -583,14 +565,22 @@ function consumeEnd (consume, encoding) {
583565
584566/**
585567 * @param {Consume } consume
586- * @param {Buffer } chunk
568+ * @param {Buffer|string } chunk
587569 * @returns {void }
588570 */
589571function consumePush ( consume , chunk ) {
590572 if ( consume . body === null ) {
591573 return
592574 }
593575
576+ if ( typeof chunk === 'string' ) {
577+ // Buffered before the consume started, while an encoding was set.
578+ // consume.length has to stay a byte count and chunksDecode()/chunksConcat()
579+ // only work on bytes, so re-encode. A string's own length is in UTF-16 code
580+ // units and Uint8Array.prototype.set() ignores a string argument entirely.
581+ chunk = Buffer . from ( chunk , consume . stream . _readableState . encoding )
582+ }
583+
594584 consume . length += chunk . length
595585 consume . body . push ( chunk )
596586}
0 commit comments