From 737c7c6b1e65676ba26dda1f8ff4c921f5afe47a Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 01:33:51 +0000 Subject: [PATCH 01/58] stream: keep long-lived stream/iter objects in fast mode Object literals with `__proto__: null` are created in V8 dictionary mode. Create the objects that live as long as a stream and are used for every chunk with ObjectSetPrototypeOf() instead, as was done for the share and broadcast consumer state, so that they keep fast properties: - the iterators returned by push(), pull(), share(), shareSync() and broadcast() consumers, and the pull() consumer-cleanup wrapper, - the iterators and the cancellation context used by from() normalization, - the async wrapper share() uses for sync sources. Objects created per call or per chunk (iterator results, options bags, promise resolver records, single-use iterables) keep the literal form: for those, setting the prototype after creation costs more than it saves, about 2x slower in a create-and-read microbenchmark. The fromWritable() writer is also unchanged, since V8 keeps object literals with accessors in dictionary mode regardless. With 200,000 16-byte chunks, pipeTo() is about 3.5% faster and pull() with a transform or a signal about 1-1.5% faster. In benchmark/streams/iter-throughput-share*.js, share() and shareSync() improve by 1-4.5%; no benchmark regressed significantly. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/broadcast.js | 5 ++--- lib/internal/streams/iter/from.js | 16 +++++++--------- lib/internal/streams/iter/pull.js | 16 +++++++--------- lib/internal/streams/iter/push.js | 6 +++--- lib/internal/streams/iter/share.js | 15 ++++++--------- 5 files changed, 25 insertions(+), 33 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index cdf25021bbc..e862f8d2a7a 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -225,8 +225,7 @@ class BroadcastImpl { return { __proto__: null, [SymbolAsyncIterator]() { - return { - __proto__: null, + return ObjectSetPrototypeOf({ next() { if (state.detached) { if (state.error !== kNoBroadcastError) { @@ -284,7 +283,7 @@ class BroadcastImpl { detach(); return kDone; }, - }; + }, null); }, }; } diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index e434ff9c253..9f39e0f614a 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -15,6 +15,7 @@ const { DataViewPrototypeGetByteLength, DataViewPrototypeGetByteOffset, FunctionPrototypeCall, + ObjectSetPrototypeOf, PromisePrototypeThen, PromiseResolve, PromiseWithResolvers, @@ -68,13 +69,12 @@ const kNormalizationCancelled = Symbol('kNormalizationCancelled'); const kFlushBatch = Symbol('kFlushBatch'); function createNormalizationContext() { - return { - __proto__: null, + return ObjectSetPrototypeOf({ cancelled: false, reason: undefined, resolve: null, suppressCleanup: false, - }; + }, null); } function cancelNormalization(context, reason, suppressCleanup = false) { @@ -112,8 +112,7 @@ async function waitForNormalization(value, context) { function createNormalizationIterator(createIterator) { const context = createNormalizationContext(); const iterator = createIterator(context); - return { - __proto__: null, + return ObjectSetPrototypeOf({ next(value) { return FunctionPrototypeCall(iterator.next, iterator, value); }, @@ -129,7 +128,7 @@ function createNormalizationIterator(createIterator) { [SymbolAsyncIterator]() { return this; }, - }; + }, null); } function createNormalizationSource(createIterator) { @@ -386,8 +385,7 @@ function yieldNormalizationAbortable(source, context) { } } - return { - __proto__: null, + return ObjectSetPrototypeOf({ async next() { if (completed) { return { __proto__: null, done: true, value: undefined }; @@ -433,7 +431,7 @@ function yieldNormalizationAbortable(source, context) { [SymbolAsyncIterator]() { return this; }, - }; + }, null); }, }; } diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 176d8dbfc83..6dff97e42be 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -11,6 +11,7 @@ const { ArrayPrototypePush, ArrayPrototypeSlice, FunctionPrototypeCall, + ObjectSetPrototypeOf, PromisePrototypeThen, PromiseReject, PromiseResolve, @@ -861,8 +862,7 @@ function pull(source, ...args) { yield* createAsyncPipeline(normalized, transforms, controller.signal); } const iterator = pipeline(); - return { - __proto__: null, + return ObjectSetPrototypeOf({ next(value) { return iterator.next(value); }, @@ -877,7 +877,7 @@ function pull(source, ...args) { [SymbolAsyncIterator]() { return this; }, - }; + }, null); } return createAbortablePullIterator(normalized, transforms, signal); }, @@ -907,8 +907,7 @@ function createAbortablePullIterator(source, transforms, signal) { throw error; } - return { - __proto__: null, + return ObjectSetPrototypeOf({ next(value) { if (aborted) return PromiseReject(signal.reason); return PromisePrototypeThen(iterator.next(value), undefined, onRejected); @@ -929,7 +928,7 @@ function createAbortablePullIterator(source, transforms, signal) { [SymbolAsyncIterator]() { return this; }, - }; + }, null); } // Keep ownership of a bonded consumer outside the transform pipeline so it can @@ -986,8 +985,7 @@ function pullWithConsumerCleanup(source, transforms, signal) { __proto__: null, [SymbolAsyncIterator]() { const iterator = pipeline[SymbolAsyncIterator](); - return { - __proto__: null, + return ObjectSetPrototypeOf({ next(value) { return PromisePrototypeThen( iterator.next(value), @@ -1011,7 +1009,7 @@ function pullWithConsumerCleanup(source, transforms, signal) { [SymbolAsyncIterator]() { return this; }, - }; + }, null); }, }; } diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index 764a3afac92..039e6b9f39d 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -7,6 +7,7 @@ const { ArrayPrototypePush, + ObjectSetPrototypeOf, PromisePrototypeThen, PromiseReject, PromiseResolve, @@ -733,8 +734,7 @@ function createReadable(queue) { return { __proto__: null, [SymbolAsyncIterator]() { - return { - __proto__: null, + return ObjectSetPrototypeOf({ async next() { return queue.read(); }, @@ -746,7 +746,7 @@ function createReadable(queue) { queue.consumerThrow(error); throw error; }, - }; + }, null); }, }; } diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index 6e9881b682a..a61d4627ad5 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -235,8 +235,7 @@ class ShareImpl { } }; - return { - __proto__: null, + return ObjectSetPrototypeOf({ next() { const next = PromisePrototypeThen( state.pendingNext, @@ -266,7 +265,7 @@ class ShareImpl { } return { __proto__: null, done: true, value: undefined }; }, - }; + }, null); }, }; } @@ -401,8 +400,7 @@ class ShareImpl { } else if (isSyncIterable(this.#source)) { const syncIterator = this.#source[SymbolIterator](); - this.#sourceIterator = { - __proto__: null, + this.#sourceIterator = ObjectSetPrototypeOf({ async next() { return syncIterator.next(); }, @@ -410,7 +408,7 @@ class ShareImpl { return syncIterator.return?.() ?? { __proto__: null, done: true, value: undefined }; }, - }; + }, null); } else { throw new ERR_INVALID_ARG_TYPE( 'source', ['AsyncIterable', 'Iterable'], this.#source); @@ -592,8 +590,7 @@ class SyncShareImpl { return { __proto__: null, [SymbolIterator]() { - return { - __proto__: null, + return ObjectSetPrototypeOf({ next() { if (state.detached) { if (state.error !== kNoShareError) throw state.error; @@ -710,7 +707,7 @@ class SyncShareImpl { } return { __proto__: null, done: true, value: undefined }; }, - }; + }, null); }, }; } From 1bdd59b93cbe2c5d39a6b287e771c6e094a39f1e Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 01:34:53 +0000 Subject: [PATCH 02/58] stream: make endSync() optional in pipeToSync() pipeToSync() threw ERR_INVALID_ARG_TYPE before writing anything when the writer had no endSync() method and preventClose was not set. endSync() is optional: the spec (pipeToSync() step 7) only calls it if the writer has it, and pipeTo() already treats it that way. A writer without endSync() now receives the data and is not closed. pipeToSync() still never falls back to the async end(). This also fixes the from-sync-writev case of benchmark/streams/iter-from-batching.js, whose writer has no endSync(). testPipeToSyncNoEndSync asserted the previous rejection and now checks that the data is written and end() is not called. The documentation of the writer requirements is corrected as well: only writeSync() is required. Assisted-by: OpenCode Signed-off-by: James M Snell --- doc/api/stream_iter.md | 9 ++++++--- lib/internal/streams/iter/pull.js | 9 +++------ test/parallel/test-stream-iter-pipeto-edge.js | 20 +++++++++---------- 3 files changed, 18 insertions(+), 20 deletions(-) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 49605b15c0d..0976f94854c 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -714,7 +714,7 @@ added: * `source` {Iterable} The sync data source. * `...transforms` {Function|Object} Zero or more sync transforms. -* `writer` {Object} Destination with `write(chunk)` method. +* `writer` {Object} Destination with a `writeSync(chunk)` method. * `options` {Object} * `failOnIncompleteClose` {boolean} If `true`, call `writer.fail()` when `writer.endSync()` cannot close the writer synchronously. Ignored when @@ -727,8 +727,11 @@ added: Synchronous version of [`pipeTo()`][]. The `source`, all transforms, and the `writer` must be synchronous. Cannot accept async iterables or promises. -The `writer` must have the `*Sync` methods (`writeSync`, `writevSync`, -`endSync`) and `fail()` for this to work. +The `writer` must have a `writeSync()` method. The other methods are +optional: `writevSync()` is used for batches of more than one chunk if it is +present, `endSync()` is called to close the writer (unless `preventClose` is +`true`), and `fail()` is called if the pipe fails (unless `preventFail` is +`true`). A writer without `endSync()` is not closed. `pipeToSync()` never falls back to the asynchronous writer methods. If `writer.endSync()` returns `-1` because the writer cannot close synchronously diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 6dff97e42be..70e89a99423 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -1032,12 +1032,9 @@ function pipeToSync(source, ...args) { context: 'options', }); const hasWritevSync = typeof writer.writevSync === 'function'; + // endSync() is optional: a writer without it is not closed. const endSync = writer.endSync; - - if (!options.preventClose && typeof endSync !== 'function') { - throw new ERR_INVALID_ARG_TYPE( - 'writer.endSync', 'Function', endSync); - } + const hasEndSync = typeof endSync === 'function'; // Normalize source and create pipeline const normalized = fromSync(source); @@ -1074,7 +1071,7 @@ function pipeToSync(source, ...args) { } } - if (!options.preventClose) { + if (!options.preventClose && hasEndSync) { closedSync = FunctionPrototypeCall(endSync, writer) >= 0; } } catch (error) { diff --git a/test/parallel/test-stream-iter-pipeto-edge.js b/test/parallel/test-stream-iter-pipeto-edge.js index 0c5448db5a8..082871a9e4c 100644 --- a/test/parallel/test-stream-iter-pipeto-edge.js +++ b/test/parallel/test-stream-iter-pipeto-edge.js @@ -47,20 +47,18 @@ async function testPipeToSyncEndSyncFailureDoesNotFailWriter() { assert.strictEqual(await result, 'abcdef'); } -// pipeToSync requires endSync() when closing is enabled. +// pipeToSync does not require endSync(). async function testPipeToSyncNoEndSync() { - let writeCalled = false; - let endCalled = false; + // endSync() is optional. Without it the data is still written and the + // writer is not closed; pipeToSync() never falls back to end(). + const written = []; const writer = { - writeSync() { writeCalled = true; return true; }, - end() { endCalled = true; }, + writeSync(chunk) { written.push(chunk); return true; }, + end: common.mustNotCall(), + fail: common.mustNotCall(), }; - assert.throws( - () => pipeToSync(fromSync('data'), writer), - { code: 'ERR_INVALID_ARG_TYPE' }, - ); - assert.strictEqual(writeCalled, false); - assert.strictEqual(endCalled, false); + assert.strictEqual(pipeToSync(fromSync('data'), writer), 4); + assert.deepStrictEqual(written, [new TextEncoder().encode('data')]); } // pipeToSync with preventFail: true — source error does NOT call fail() From c72ebe0e22d596704b2df3b56f7a3982200d6c9f Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 01:47:21 +0000 Subject: [PATCH 03/58] stream: create stream/iter iterator results with a constructor The iterators of push(), share(), shareSync(), broadcast() consumers and from() normalization created a `{ __proto__: null, done, value }` literal for every result. V8 creates such literals in dictionary mode, which makes them several times more expensive to create and read than ordinary objects. Create them with an IterResult constructor whose prototype is a single frozen, null-prototype object instead. Results have fast properties and a single shape, and still have no %Object.prototype% in their prototype chain, so a polluted Object.prototype.then still cannot turn a result into a thenable. Creating and reading a result is about 5x faster in a microbenchmark. benchmark/streams/iter-throughput-share-sync.js improves by 7% to 42% (more with more consumers) and iter-throughput-share.js by 4-6%; pipeTo() with small chunks is about 2.5% faster. This is observable: results are no longer null-prototype objects, so deepStrictEqual() comparisons against `{ __proto__: null, ... }` no longer match, and util.inspect() prints them as `IterResult { done, value }` (the prototype has a non-enumerable `constructor` for that purpose). The five tests that compared results that way now compare their own properties, and a new test covers the result contract and the prototype pollution case. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/broadcast.js | 17 ++-- lib/internal/streams/iter/from.js | 9 ++- lib/internal/streams/iter/pull.js | 3 +- lib/internal/streams/iter/push.js | 19 ++--- lib/internal/streams/iter/share.js | 37 ++++----- lib/internal/streams/iter/utils.js | 27 +++++++ .../test-stream-iter-broadcast-basic.js | 3 +- .../test-stream-iter-broadcast-from.js | 3 +- .../test-stream-iter-iterator-result.js | 79 +++++++++++++++++++ test/parallel/test-stream-iter-pull-async.js | 4 +- .../test-stream-iter-reason-propagation.js | 3 +- .../test-stream-iter-share-coverage.js | 3 +- 12 files changed, 157 insertions(+), 50 deletions(-) create mode 100644 test/parallel/test-stream-iter-iterator-result.js diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index e862f8d2a7a..df2a701f20c 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -57,6 +57,7 @@ const { } = require('internal/streams/iter/pull'); const { + IterResult, kMultiConsumerDefaultBudget, kResolvedPromise, convertChunks, @@ -207,13 +208,13 @@ class BroadcastImpl { const self = this; const kDone = PromiseResolve( - { __proto__: null, done: true, value: undefined }); + new IterResult(true, undefined)); function detach() { state.detached = true; self.#waiters.delete(state); if (state.resolve) { - state.resolve({ __proto__: null, done: true, value: undefined }); + state.resolve(new IterResult(true, undefined)); } self.#resolvePendingDone(state); if (self.#deleteConsumer(state)) { @@ -245,7 +246,7 @@ class BroadcastImpl { self.#tryTrimBuffer(); } return PromiseResolve( - { __proto__: null, done: false, value: chunk }); + new IterResult(false, chunk)); } if (self.#errored) { @@ -307,7 +308,7 @@ class BroadcastImpl { if (hasReason) { consumer.reject?.(reason); } else { - consumer.resolve({ __proto__: null, done: true, value: undefined }); + consumer.resolve(new IterResult(true, undefined)); } consumer.resolve = null; consumer.reject = null; @@ -397,9 +398,9 @@ class BroadcastImpl { --this.#cachedMinCursorConsumers === 0) { this.#tryTrimBuffer(); } - consumer.resolve({ __proto__: null, done: false, value: chunk }); + consumer.resolve(new IterResult(false, chunk)); } else { - consumer.resolve({ __proto__: null, done: true, value: undefined }); + consumer.resolve(new IterResult(true, undefined)); this.#resolvePendingDone(consumer); consumer.detached = true; } @@ -529,7 +530,7 @@ class BroadcastImpl { const resolve = consumer.resolve; consumer.resolve = null; consumer.reject = null; - resolve({ __proto__: null, done: false, value: chunk }); + resolve(new IterResult(false, chunk)); if (consumer.detached && this.#deleteConsumer(consumer)) { this.#tryTrimBuffer(); } else if (this.#promotePending(consumer)) { @@ -573,7 +574,7 @@ class BroadcastImpl { } while (consumer.pending.length > 0) { ArrayPrototypeShift(consumer.pending).resolve( - { __proto__: null, done: true, value: undefined }); + new IterResult(true, undefined)); } } diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 9f39e0f614a..78b61b267de 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -53,6 +53,7 @@ const { } = require('internal/streams/iter/types'); const { + IterResult, getProtocolMethod, toUint8Array, } = require('internal/streams/iter/utils'); @@ -388,7 +389,7 @@ function yieldNormalizationAbortable(source, context) { return ObjectSetPrototypeOf({ async next() { if (completed) { - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } throwIfNormalizationCancelled(context); reading = true; @@ -406,12 +407,12 @@ function yieldNormalizationAbortable(source, context) { throwIfNormalizationCancelled(context); completed = true; closed = true; - return { __proto__: null, done: true, value: result.value }; + return new IterResult(true, result.value); } const value = result.value; reading = false; throwIfNormalizationCancelled(context); - return { __proto__: null, done: false, value }; + return new IterResult(false, value); } catch (error) { if (context.cancelled) await closeSource(true); reading = false; @@ -421,7 +422,7 @@ function yieldNormalizationAbortable(source, context) { async return(value) { await closeSource( context.suppressCleanup || (context.cancelled && reading)); - return { __proto__: null, done: true, value }; + return new IterResult(true, value); }, async throw(error) { await closeSource( diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 70e89a99423..9a52a5fdfed 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -50,6 +50,7 @@ const { } = require('internal/streams/iter/from'); const { + IterResult, createBatchEntry, isTransformObject, parsePullArgs, @@ -914,7 +915,7 @@ function createAbortablePullIterator(source, transforms, signal) { }, return(value) { if (aborted) { - return PromiseResolve({ __proto__: null, done: true, value }); + return PromiseResolve(new IterResult(true, value)); } controller.abort(lazyDOMException('Aborted', 'AbortError')); return iterator.return(value); diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index 039e6b9f39d..d15f5752fa6 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -29,6 +29,7 @@ const { } = require('internal/streams/iter/types'); const { + IterResult, kPushDefaultBudget, kResolvedPromise, createBatchEntry, @@ -438,7 +439,7 @@ class PushQueue { async read() { if (this.#consumerState === 'returned') { - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } if (this.#consumerState === 'thrown') { throw this.#consumerError; @@ -448,17 +449,17 @@ class PushQueue { if (this.#slots.length > 0) { const result = this.#drain(); this.#resolvePendingWrites(); - return { __proto__: null, done: false, value: result }; + return new IterResult(false, result); } // Buffer empty and writer closing = drain complete if (this.#writerState === 'closing') { this.endDrained(); - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } if (this.#writerState === 'closed') { - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } if (this.#writerState === 'errored') { @@ -540,7 +541,7 @@ class PushQueue { while (this.#pendingReads.length > 0) { if (this.#consumerState === 'returned') { const pending = this.#pendingReads.shift(); - pending.resolve({ __proto__: null, done: true, value: undefined }); + pending.resolve(new IterResult(true, undefined)); } else if (this.#consumerState === 'thrown') { const pending = this.#pendingReads.shift(); pending.reject(this.#consumerError); @@ -549,7 +550,7 @@ class PushQueue { try { const result = this.#drain(); this.#resolvePendingWrites(); - pending.resolve({ __proto__: null, done: false, value: result }); + pending.resolve(new IterResult(false, result)); } catch (error) { pending.reject(error); } @@ -558,10 +559,10 @@ class PushQueue { this.#pendingWrites.length === 0) { this.endDrained(); const pending = this.#pendingReads.shift(); - pending.resolve({ __proto__: null, done: true, value: undefined }); + pending.resolve(new IterResult(true, undefined)); } else if (this.#writerState === 'closed') { const pending = this.#pendingReads.shift(); - pending.resolve({ __proto__: null, done: true, value: undefined }); + pending.resolve(new IterResult(true, undefined)); } else if (this.#writerState === 'errored') { const pending = this.#pendingReads.shift(); pending.reject(this.#writerError); @@ -740,7 +741,7 @@ function createReadable(queue) { }, async return() { queue.consumerReturn(); - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); }, async throw(error) { queue.consumerThrow(error); diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index a61d4627ad5..937528b3929 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -38,6 +38,7 @@ const { } = require('internal/streams/iter/pull'); const { + IterResult, kMultiConsumerDefaultBudget, createBatchEntry, getProtocolMethod, @@ -175,7 +176,7 @@ class ShareImpl { for (;;) { if (state.detached) { if (state.error !== kNoShareError) throw state.error; - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } if (self.#cancelled) { @@ -183,7 +184,7 @@ class ShareImpl { state.error = self.#cancelError; self.#deleteConsumer(state); if (state.error !== kNoShareError) throw state.error; - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } // Check if data is available in buffer @@ -196,7 +197,7 @@ class ShareImpl { --self.#cachedMinCursorConsumers === 0) { self.#tryTrimBuffer(); } - return { __proto__: null, done: false, value: chunk }; + return new IterResult(false, chunk); } if (self.#sourceExhausted) { @@ -206,7 +207,7 @@ class ShareImpl { state.error = self.#sourceError; throw state.error; } - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } // Need to pull from source - check buffer limit @@ -225,7 +226,7 @@ class ShareImpl { state.error = self.#cancelError; self.#deleteConsumer(state); if (state.error !== kNoShareError) throw state.error; - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } await self.#pullFromSource(!shouldBuffer); @@ -253,7 +254,7 @@ class ShareImpl { if (self.#deleteConsumer(state)) { self.#tryTrimBuffer(); } - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); }, async throw() { @@ -263,7 +264,7 @@ class ShareImpl { if (self.#deleteConsumer(state)) { self.#tryTrimBuffer(); } - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); }, }, null); }, @@ -302,7 +303,7 @@ class ShareImpl { if (hasReason) { consumer.reject?.(reason); } else { - consumer.resolve({ __proto__: null, done: true, value: undefined }); + consumer.resolve(new IterResult(true, undefined)); } consumer.resolve = null; consumer.reject = null; @@ -406,7 +407,7 @@ class ShareImpl { }, async return() { return syncIterator.return?.() ?? - { __proto__: null, done: true, value: undefined }; + new IterResult(true, undefined); }, }, null); } else { @@ -594,7 +595,7 @@ class SyncShareImpl { next() { if (state.detached) { if (state.error !== kNoShareError) throw state.error; - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } if (self.#sourceError !== kNoShareError) { state.detached = true; @@ -605,7 +606,7 @@ class SyncShareImpl { if (self.#cancelled) { state.detached = true; self.#deleteConsumer(state); - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } const bufferIndex = state.cursor - self.#bufferStart; @@ -617,13 +618,13 @@ class SyncShareImpl { --self.#cachedMinCursorConsumers === 0) { self.#tryTrimBuffer(); } - return { __proto__: null, done: false, value: chunk }; + return new IterResult(false, chunk); } if (self.#sourceExhausted) { state.detached = true; self.#deleteConsumer(state); - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } // Check buffer limit. 'unbounded' and 'drop-newest' are rejected @@ -680,16 +681,16 @@ class SyncShareImpl { --self.#cachedMinCursorConsumers === 0) { self.#tryTrimBuffer(); } - return { __proto__: null, done: false, value: chunk }; + return new IterResult(false, chunk); } if (self.#sourceExhausted) { state.detached = true; self.#deleteConsumer(state); - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); } - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); }, return() { @@ -697,7 +698,7 @@ class SyncShareImpl { if (self.#deleteConsumer(state)) { self.#tryTrimBuffer(); } - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); }, throw() { @@ -705,7 +706,7 @@ class SyncShareImpl { if (self.#deleteConsumer(state)) { self.#tryTrimBuffer(); } - return { __proto__: null, done: true, value: undefined }; + return new IterResult(true, undefined); }, }, null); }, diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 2f72a78c0d2..0c6e1daaa32 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -7,6 +7,8 @@ const { ArrayBufferPrototypeGetResizable, ArrayPrototypePush, ArrayPrototypeSlice, + ObjectDefineProperty, + ObjectFreeze, PromiseResolve, PromiseWithResolvers, SafePromisePrototypeFinally, @@ -61,6 +63,30 @@ const kPushDefaultBudget = 16384; /** Default byte budget for broadcast and share streams (multi-consumer). */ const kMultiConsumerDefaultBudget = 65536; +/** + * Iterator result object (`{ done, value }`) for the iterators returned by + * this module. A result is created for every chunk, so it must be cheap. + * + * `new IterResult(done, value)` literals are created in V8 dictionary + * mode, which costs several times more than an ordinary object. Instances + * of this constructor have fast properties, always with the same shape + * (`done` before `value`), and still have no %Object.prototype% in their + * prototype chain, so a polluted `Object.prototype.then` cannot turn a + * result into a thenable when an async `next()` resolves with it. The + * prototype is a single empty, frozen, null-prototype object (V8 gives + * objects fast properties once they are used as a prototype). + * @param {boolean} done + * @param {any} value + */ +function IterResult(done, value) { + this.done = done; + this.value = value; +} +// The non-enumerable `constructor` lets util.inspect() print results as +// `IterResult { done, value }`. +IterResult.prototype = ObjectFreeze(ObjectDefineProperty( + { __proto__: null }, 'constructor', { __proto__: null, value: IterResult })); + /** * Register a handler for an AbortSignal, handling the already-aborted case. * If the signal is already aborted, calls handler immediately. @@ -539,6 +565,7 @@ function validateBackpressure(value) { } module.exports = { + IterResult, kMultiConsumerDefaultBudget, kPushDefaultBudget, kResolvedPromise, diff --git a/test/parallel/test-stream-iter-broadcast-basic.js b/test/parallel/test-stream-iter-broadcast-basic.js index 232439cdf7d..c59689c2d12 100644 --- a/test/parallel/test-stream-iter-broadcast-basic.js +++ b/test/parallel/test-stream-iter-broadcast-basic.js @@ -385,8 +385,7 @@ async function testOverlappingNextKeepsEarlierRead() { assert.strictEqual(Buffer.concat(result.value).toString(), 'x'); writer.endSync(); - assert.deepStrictEqual(await second, { - __proto__: null, + assert.deepStrictEqual({ ...await second }, { done: true, value: undefined, }); diff --git a/test/parallel/test-stream-iter-broadcast-from.js b/test/parallel/test-stream-iter-broadcast-from.js index b76a2b31b7f..8bc2abde2e1 100644 --- a/test/parallel/test-stream-iter-broadcast-from.js +++ b/test/parallel/test-stream-iter-broadcast-from.js @@ -144,8 +144,7 @@ async function testBroadcastFromCancelWhileBlocked() { let writesAfterCancel = 0; writer.writevSync = () => { writesAfterCancel++; return true; }; bc.cancel(); - assert.deepStrictEqual(await pendingRead, { - __proto__: null, + assert.deepStrictEqual({ ...await pendingRead }, { done: true, value: undefined, }); diff --git a/test/parallel/test-stream-iter-iterator-result.js b/test/parallel/test-stream-iter-iterator-result.js new file mode 100644 index 00000000000..4f29bf05534 --- /dev/null +++ b/test/parallel/test-stream-iter-iterator-result.js @@ -0,0 +1,79 @@ +// Flags: --experimental-stream-iter +'use strict'; + +// Iterator results created by the stream/iter iterators themselves do not +// inherit from Object.prototype, so prototype pollution cannot affect them. +// (Iterators implemented as async generators, such as the one returned by +// pull(), return ordinary iterator results created by the engine.) + +const common = require('../common'); +const assert = require('assert'); +const { inspect } = require('util'); +const { + broadcast, + from, + push, + share, + shareSync, +} = require('stream/iter'); + +function assertResult(result, done) { + assert.strictEqual(result instanceof Object, false); + assert.deepStrictEqual(Object.keys(result), ['done', 'value']); + assert.strictEqual(result.done, done); + assert.match(inspect(result), /^IterResult \{ done: (true|false), value: /); +} + +async function testAsyncIterators() { + const sources = { + 'push()': () => { + const { writer, readable } = push(); + writer.writeSync('a'); + writer.endSync(); + return readable; + }, + 'share()': () => share(from('a')).pull(), + 'broadcast()': () => { + const { writer, broadcast: bc } = broadcast(); + const consumer = bc.push(); + writer.writeSync('a'); + writer.endSync(); + return consumer; + }, + }; + for (const create of Object.values(sources)) { + const iterator = create()[Symbol.asyncIterator](); + assertResult(await iterator.next(), false); + assertResult(await iterator.next(), true); + } +} + +function testShareSync() { + const iterator = shareSync(['a']).pull()[Symbol.iterator](); + assertResult(iterator.next(), false); + assertResult(iterator.next(), true); +} + +async function testPollutedThen() { + // Resolving an async next() with an object looks up `then`. Results must + // not pick it up from a polluted Object.prototype. + const { writer, readable } = push(); + writer.writeSync('ab'); + writer.endSync(); + const iterator = readable[Symbol.asyncIterator](); + Object.prototype.then = common.mustNotCall('Object.prototype.then'); + try { + const first = await iterator.next(); + assert.strictEqual(first.done, false); + assert.strictEqual(new TextDecoder().decode(first.value[0]), 'ab'); + assert.strictEqual((await iterator.next()).done, true); + } finally { + delete Object.prototype.then; + } +} + +(async () => { + await testAsyncIterators(); + testShareSync(); + await testPollutedThen(); +})().then(common.mustCall()); diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index 632306fa589..71fc27426ea 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -79,8 +79,8 @@ async function testPullWithAbortSignal() { await assert.rejects(iterator.next(), (error) => error === signal.reason); await assert.rejects(iterator.next(), (error) => error === signal.reason); assert.strictEqual(started, false); - assert.deepStrictEqual(await iterator.return(), - { __proto__: null, done: true, value: undefined }); + assert.deepStrictEqual({ ...await iterator.return() }, + { done: true, value: undefined }); await assert.rejects(text(pull(gen(), (chunks) => chunks, { signal })), (error) => error === signal.reason); diff --git a/test/parallel/test-stream-iter-reason-propagation.js b/test/parallel/test-stream-iter-reason-propagation.js index 7511dbb2c41..58ae7d4e5f4 100644 --- a/test/parallel/test-stream-iter-reason-propagation.js +++ b/test/parallel/test-stream-iter-reason-propagation.js @@ -249,8 +249,7 @@ async function testCompletedBroadcastConsumerStaysCompleted() { await iterator.return(); writer.fail(undefined); - assert.deepStrictEqual(await iterator.next(), { - __proto__: null, + assert.deepStrictEqual({ ...await iterator.next() }, { done: true, value: undefined, }); diff --git a/test/parallel/test-stream-iter-share-coverage.js b/test/parallel/test-stream-iter-share-coverage.js index fee28ae1568..73827f2d40f 100644 --- a/test/parallel/test-stream-iter-share-coverage.js +++ b/test/parallel/test-stream-iter-share-coverage.js @@ -113,8 +113,7 @@ async function testCompletedSyncConsumerStaysCompleted() { assert.strictEqual(error, reason); } assert.strictEqual(caught, true); - assert.deepStrictEqual(completed.next(), { - __proto__: null, + assert.deepStrictEqual({ ...completed.next() }, { done: true, value: undefined, }); From 57dbd6fc1dff2eae5f69b55bf49a69e706c02b53 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 02:13:10 +0000 Subject: [PATCH 04/58] stream: avoid per-write allocations in stream/iter writers Every write() and writeSync() allocated a `{ __proto__: null, context }` options object for the WebIDL chunk conversion, and for a Uint8Array chunk a second, six-property one with [AllowShared] and [AllowResizable]. Both are created in V8 dictionary mode. Return Uint8Array chunks directly from the WriterChunk converter: with [AllowShared] and [AllowResizable], the Uint8Array conversion cannot reject a value isUint8Array() accepts and returns the same object. Use shared, frozen conversion contexts for chunks, chunk sequences and write options; the converters only read them to build error messages. Writing 1e6 16-byte Uint8Array chunks into a push() stream and reading them back is about 27% faster with writeSync() and with writevSync() (4 chunks per call). String chunks are unaffected. Behavior and error messages are unchanged. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/utils.js | 23 +++++++++++------------ lib/internal/streams/iter/webidl.js | 20 +++++--------------- 2 files changed, 16 insertions(+), 27 deletions(-) diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 0c6e1daaa32..d50ec639ab7 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -412,6 +412,14 @@ function concatBytes(chunks) { return concatenated; } +// Conversion contexts for the per-write paths. The converters only read +// them (to build error messages), so they are shared rather than allocated +// on every write. +const kChunkContext = ObjectFreeze({ __proto__: null, context: 'chunk' }); +const kChunksContext = ObjectFreeze({ __proto__: null, context: 'chunks' }); +const kWriteOptionsContext = + ObjectFreeze({ __proto__: null, context: 'options' }); + /** * Convert an array of chunks (strings or Uint8Arrays) to a Uint8Array[]. * Always returns a fresh copy of the array. @@ -419,10 +427,7 @@ function concatBytes(chunks) { * @returns {Uint8Array[]} */ function convertChunks(chunks) { - chunks = converters.WriterChunkSequence(chunks, { - __proto__: null, - context: 'chunks', - }); + chunks = converters.WriterChunkSequence(chunks, kChunksContext); const len = chunks.length; const result = new Array(len); for (let i = 0; i < len; i++) { @@ -437,17 +442,11 @@ function convertChunks(chunks) { * @returns {AbortSignal|undefined} */ function getWriterSignal(options) { - return converters.WriteOptions(options, { - __proto__: null, - context: 'options', - }).signal; + return converters.WriteOptions(options, kWriteOptionsContext).signal; } function toWriterUint8Array(chunk) { - return toUint8Array(converters.WriterChunk(chunk, { - __proto__: null, - context: 'chunk', - })); + return toUint8Array(converters.WriterChunk(chunk, kChunkContext)); } /** diff --git a/lib/internal/streams/iter/webidl.js b/lib/internal/streams/iter/webidl.js index ef61abb7794..6ee1afd622a 100644 --- a/lib/internal/streams/iter/webidl.js +++ b/lib/internal/streams/iter/webidl.js @@ -27,17 +27,6 @@ function enforceRangeUnsignedLongLong(value, options = { __proto__: null }) { }); } -function allowStreamBufferOptions(options) { - return { - __proto__: null, - prefix: options.prefix, - context: options.context, - code: options.code, - allowShared: true, - allowResizable: true, - }; -} - converters.AbortSignal = baseConverters.AbortSignal; converters.BackpressurePolicy = createEnumConverter('BackpressurePolicy', [ 'strict', @@ -48,10 +37,11 @@ converters.BackpressurePolicy = createEnumConverter('BackpressurePolicy', [ converters.unsignedLongLong = unsignedLongLong; converters.enforceRangeUnsignedLongLong = enforceRangeUnsignedLongLong; converters.WriterChunk = (value, options = { __proto__: null }) => { - if (isUint8Array(value)) { - return baseConverters.Uint8Array( - value, allowStreamBufferOptions(options)); - } + // The chunk is a Uint8Array with [AllowShared] and [AllowResizable]. The + // Uint8Array conversion cannot reject a value isUint8Array() accepts when + // both are allowed, and returns the same object, so skip it: writes are + // hot, and the conversion would need its own options object. + if (isUint8Array(value)) return value; return baseConverters.USVString(value, options); }; converters.WriterChunkSequence = createSequenceConverter( From bc6aa2292bec6e9b85e6c8c6e20ff0eeea191b88 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 02:21:50 +0000 Subject: [PATCH 05/58] stream: construct stream/iter byte view snapshots and batch entries createBatchEntry() and recordChunk() snapshot every chunk in a seven-property `{ __proto__: null, ... }` literal, and every batch gets a `{ __proto__: null, views, byteLength }` record. V8 creates such literals in dictionary mode, which is expensive for objects created for every chunk. Create them with constructors whose prototype is an empty, frozen, null-prototype object instead, as for IterResult. They have fast properties and a single shape, and are only used internally. Writing 1e6 16-byte chunks into a push() stream and reading them back is about 6x faster with writeSync(), 3.7x faster with writevSync() (4 chunks per call) and 1.7x faster with string chunks. benchmark/streams/iter-throughput-share-sync.js improves by 67-93%, iter-throughput-share.js by 4-8%, and iter-throughput-broadcast.js with 4 consumers by 8%. pipeTo() with small chunks is about 1.5x faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/utils.js | 45 +++++++++++++++++++----------- 1 file changed, 28 insertions(+), 17 deletions(-) diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index d50ec639ab7..915e5340ceb 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -217,23 +217,35 @@ function toUint8Array(chunk) { return chunk; } +// Byte view snapshots and batch entries are created for every chunk and +// every batch, so like IterResult they are constructed rather than created +// as `{ __proto__: null, ... }` literals (which are dictionary-mode objects), +// and their prototype is an empty null-prototype object. +function ByteViewSnapshot(value, buffer, sharedBufferView) { + this.value = value; + this.buffer = buffer; + this.bufferByteLength = sharedBufferView === undefined ? + ArrayBufferPrototypeGetByteLength(buffer) : + TypedArrayPrototypeGetByteLength(sharedBufferView); + this.byteLength = TypedArrayPrototypeGetByteLength(value); + this.byteOffset = TypedArrayPrototypeGetByteOffset(value); + this.detached = sharedBufferView === undefined && + ArrayBufferPrototypeGetDetached(buffer); + this.sharedBufferView = sharedBufferView; +} +ByteViewSnapshot.prototype = ObjectFreeze({ __proto__: null }); + +function BatchEntry(views, byteLength) { + this.views = views; + this.byteLength = byteLength; +} +BatchEntry.prototype = ObjectFreeze({ __proto__: null }); + function snapshotByteView(value) { const buffer = TypedArrayPrototypeGetBuffer(value); const sharedBufferView = isSharedArrayBuffer(buffer) ? new Uint8Array(buffer) : undefined; - return { - __proto__: null, - value, - buffer, - bufferByteLength: sharedBufferView === undefined ? - ArrayBufferPrototypeGetByteLength(buffer) : - TypedArrayPrototypeGetByteLength(sharedBufferView), - byteLength: TypedArrayPrototypeGetByteLength(value), - byteOffset: TypedArrayPrototypeGetByteOffset(value), - detached: sharedBufferView === undefined && - ArrayBufferPrototypeGetDetached(buffer), - sharedBufferView, - }; + return new ByteViewSnapshot(value, buffer, sharedBufferView); } function validateByteView(snapshot) { @@ -325,7 +337,7 @@ function createBatchEntry(chunks) { views[i] = view; byteLength += view.byteLength; } - return { __proto__: null, views, byteLength }; + return new BatchEntry(views, byteLength); } /** @@ -345,15 +357,14 @@ function splitBatchEntry(entry, limit) { for (let i = 0; i < views.length; i++) { const view = views[i]; if (current.length > 0 && byteLength + view.byteLength >= limit) { - ArrayPrototypePush(entries, - { __proto__: null, views: current, byteLength }); + ArrayPrototypePush(entries, new BatchEntry(current, byteLength)); current = []; byteLength = 0; } ArrayPrototypePush(current, view); byteLength += view.byteLength; } - ArrayPrototypePush(entries, { __proto__: null, views: current, byteLength }); + ArrayPrototypePush(entries, new BatchEntry(current, byteLength)); return entries; } From 115ddacf33109aae37d4cfba092443e8b801d561 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 02:28:24 +0000 Subject: [PATCH 06/58] stream: construct stream/iter wait and merge records The records queued when a stream/iter read, write or drain has to wait, and merge()'s ready-queue entries, were `{ __proto__: null, ... }` literals, which V8 creates in dictionary mode. They can be created once per chunk: whenever the consumer is ahead of the producer, every read waits, and with a full budget every write does. Create them with constructors whose prototype is an empty, frozen, null-prototype object: PendingRequest and PendingWrite (push(), broadcast() and fromWritable()), QueuedWrite (fromWritable()) and MergeEntry (merge()). fromWritable() drain waiters now settle through resolve(false) instead of a per-waiter close() closure. merge() tells error entries apart by their missing iterator rather than by a kind string. When every push() read waits for data, or every write waits behind a full budget, reading or writing 16-byte chunks is about 30% faster. fromWritable() with queued writes is about 18% faster, and merge() of two sources about 9% faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/broadcast.js | 8 ++++--- lib/internal/streams/iter/classic.js | 32 ++++++++++++++------------ lib/internal/streams/iter/consumers.js | 27 ++++++++++++---------- lib/internal/streams/iter/push.js | 8 ++++--- lib/internal/streams/iter/utils.js | 17 ++++++++++++++ 5 files changed, 59 insertions(+), 33 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index df2a701f20c..e395b5e84f2 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -58,6 +58,8 @@ const { const { IterResult, + PendingRequest, + PendingWrite, kMultiConsumerDefaultBudget, kResolvedPromise, convertChunks, @@ -264,7 +266,7 @@ class BroadcastImpl { if (state.resolve) { const { promise, resolve, reject } = PromiseWithResolvers(); ArrayPrototypePush(state.pending, - { __proto__: null, resolve, reject }); + new PendingRequest(resolve, reject)); return promise; } @@ -625,7 +627,7 @@ class BroadcastWriter { if (canWrite === null) return null; if (canWrite) return PromiseResolve(true); const { promise, resolve, reject } = PromiseWithResolvers(); - ArrayPrototypePush(this.#pendingDrains, { __proto__: null, resolve, reject }); + ArrayPrototypePush(this.#pendingDrains, new PendingRequest(resolve, reject)); return promise; } @@ -802,7 +804,7 @@ class BroadcastWriter { */ #createPendingWrite(batch, signal) { const { promise, resolve, reject } = PromiseWithResolvers(); - const entry = { __proto__: null, batch, resolve, reject }; + const entry = new PendingWrite(batch, resolve, reject); this.#pendingWrites.push(entry); if (signal) { wireBroadcastWriteSignal(entry, signal, resolve, reject, this); diff --git a/lib/internal/streams/iter/classic.js b/lib/internal/streams/iter/classic.js index 2c1fa640714..2a7b818efd9 100644 --- a/lib/internal/streams/iter/classic.js +++ b/lib/internal/streams/iter/classic.js @@ -14,6 +14,7 @@ const { ArrayPrototypePush, FunctionPrototypeCall, + ObjectFreeze, Promise, PromisePrototypeThen, PromiseReject, @@ -61,6 +62,7 @@ const { } = require('internal/streams/iter/types'); const { + PendingRequest, convertChunks, getWriterSignal, onSignalAbort, @@ -522,6 +524,18 @@ function toReadableSync(source, options = kNullPrototype) { // Cache: one Writer adapter per Writable instance. const fromWritableCache = new SafeWeakMap(); +// A write queued by fromWritable() until the writable can accept it. Like +// the other per-write records, constructed rather than created as a +// `{ __proto__: null, ... }` literal (a dictionary-mode object). +function QueuedWrite(chunks, resolve, reject) { + this.chunks = chunks; + this.resolve = resolve; + this.reject = reject; + this.signal = undefined; + this.onAbort = undefined; +} +QueuedWrite.prototype = ObjectFreeze({ __proto__: null }); + /** * Create a stream/iter Writer adapter from a classic Writable (or duck-type). * @@ -643,7 +657,7 @@ function fromWritable(writable, options = kNullPrototype) { if (preserveReason) { pending[i].reject(reason); } else { - pending[i].close(); + pending[i].resolve(false); } } @@ -755,14 +769,7 @@ function fromWritable(writable, options = kNullPrototype) { function queueWrite(chunks, signal) { const { promise, resolve, reject } = PromiseWithResolvers(); - const entry = { - __proto__: null, - chunks, - resolve, - reject, - signal: undefined, - onAbort: undefined, - }; + const entry = new QueuedWrite(chunks, resolve, reject); pendingWrites.push(entry); installDrainListener(); @@ -1038,12 +1045,7 @@ function fromWritable(writable, options = kNullPrototype) { return PromiseResolve(true); } const { promise, resolve, reject } = PromiseWithResolvers(); - ArrayPrototypePush(drainWaiters, { - __proto__: null, - resolve, - reject, - close() { resolve(false); }, - }); + ArrayPrototypePush(drainWaiters, new PendingRequest(resolve, reject)); installDrainListener(); return promise; }; diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index a18794aa732..8678e8235ac 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -17,6 +17,7 @@ const { ArrayPrototypeShift, ArrayPrototypeSlice, FunctionPrototypeCall, + ObjectFreeze, Promise, PromisePrototypeThen, SafePromiseAllReturnVoid, @@ -456,6 +457,16 @@ function ondrain(drainable) { const kNoMergeError = Symbol('kNoMergeError'); +// An entry in merge()'s ready queue: a value from `iterator`, or, for a +// source that failed, `reason` with no iterator. Created for every merged +// chunk, so constructed rather than created as a dictionary-mode literal. +function MergeEntry(iterator, value, reason) { + this.iterator = iterator; + this.value = value; + this.reason = reason; +} +MergeEntry.prototype = ObjectFreeze({ __proto__: null }); + /** * Merge multiple async iterables by yielding values in temporal order. * @param {...(AsyncIterable|object)} args @@ -531,12 +542,8 @@ function merge(...args) { if (result.done) { activeCount--; } else { - ArrayPrototypePush(ready, { - __proto__: null, - kind: 'value', - iterator, - value: result.value, - }); + ArrayPrototypePush(ready, + new MergeEntry(iterator, result.value, undefined)); } if (waitResolve) { waitResolve(); @@ -547,11 +554,7 @@ function merge(...args) { const onRejected = (iterator, reason) => { pendingPulls.delete(iterator); if (stopped) return; - ArrayPrototypePush(ready, { - __proto__: null, - kind: 'error', - reason, - }); + ArrayPrototypePush(ready, new MergeEntry(undefined, undefined, reason)); if (waitResolve) { waitResolve(); waitResolve = null; @@ -579,7 +582,7 @@ function merge(...args) { // Drain ready queue synchronously while (ready.length > 0) { const item = ArrayPrototypeShift(ready); - if (item.kind === 'error') { + if (item.iterator === undefined) { throw item.reason; } yield item.value; diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index d15f5752fa6..b5db45a1a32 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -30,6 +30,8 @@ const { const { IterResult, + PendingRequest, + PendingWrite, kPushDefaultBudget, kResolvedPromise, createBatchEntry, @@ -294,7 +296,7 @@ class PushQueue { */ #createPendingWrite(batch, signal) { const { promise, resolve, reject } = PromiseWithResolvers(); - const entry = { __proto__: null, batch, resolve, reject }; + const entry = new PendingWrite(batch, resolve, reject); this.#pendingWrites.push(entry); if (signal) { @@ -429,7 +431,7 @@ class PushQueue { */ waitForDrain() { const { promise, resolve, reject } = PromiseWithResolvers(); - ArrayPrototypePush(this.#pendingDrains, { __proto__: null, resolve, reject }); + ArrayPrototypePush(this.#pendingDrains, new PendingRequest(resolve, reject)); return promise; } @@ -467,7 +469,7 @@ class PushQueue { } const { promise, resolve, reject } = PromiseWithResolvers(); - this.#pendingReads.push({ __proto__: null, resolve, reject }); + this.#pendingReads.push(new PendingRequest(resolve, reject)); return promise; } diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 915e5340ceb..8b0c074b4e1 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -241,6 +241,21 @@ function BatchEntry(views, byteLength) { } BatchEntry.prototype = ObjectFreeze({ __proto__: null }); +// Waiters for reads, writes and drains are queued whenever a stream has +// to wait, which can be once per chunk. +function PendingRequest(resolve, reject) { + this.resolve = resolve; + this.reject = reject; +} +PendingRequest.prototype = ObjectFreeze({ __proto__: null }); + +function PendingWrite(batch, resolve, reject) { + this.batch = batch; + this.resolve = resolve; + this.reject = reject; +} +PendingWrite.prototype = ObjectFreeze({ __proto__: null }); + function snapshotByteView(value) { const buffer = TypedArrayPrototypeGetBuffer(value); const sharedBufferView = isSharedArrayBuffer(buffer) ? @@ -576,6 +591,8 @@ function validateBackpressure(value) { module.exports = { IterResult, + PendingRequest, + PendingWrite, kMultiConsumerDefaultBudget, kPushDefaultBudget, kResolvedPromise, From 44924201fda9d0eff81643c7e30cd308a0f49d1c Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 02:35:56 +0000 Subject: [PATCH 07/58] stream: construct the options passed to stream/iter transforms pull() passes every stateless transform call a new `{ __proto__: null, signal }` options object, which V8 creates in dictionary mode, once per batch and transform. Create the options with a TransformOptions constructor instead, for stateful transforms as well so that both forms receive the same kind of object. Each call still gets its own object, as the pipeline requires. Its prototype is a single frozen object with no %Object.prototype% in its chain, so a transform cannot pass state to other transforms through it, and with a non-enumerable `constructor` so that util.inspect() prints `TransformOptions { signal }`. With 300,000 single-chunk batches, pull() is about 2% faster with one stateless transform and about 8% faster with four. This is observable: the options object's prototype is no longer null. A new test covers the options contract, and the documentation now describes it, including that pullSync() passes transforms no options. Assisted-by: OpenCode Signed-off-by: James M Snell --- doc/api/stream_iter.md | 6 ++++ lib/internal/streams/iter/pull.js | 27 ++++++++++---- test/parallel/test-stream-iter-pull-async.js | 37 ++++++++++++++++++++ 3 files changed, 64 insertions(+), 6 deletions(-) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 0976f94854c..644bcb1f9eb 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -140,6 +140,12 @@ Both forms receive an `options` parameter with the following property: can check `signal.aborted` or listen for the `'abort'` event to perform early cleanup. +In `pull()`, stateless transforms receive a new `options` object for every +call, and stateful transforms one for the pipeline, so a transform can modify +its `options` without affecting other transforms. The object does not inherit +from `Object.prototype`. Transforms passed to [`pullSync()`][] receive no +`options`. + The flush signal (`null`) is sent after the source ends, giving transforms a chance to emit trailing data (e.g., compression footers). diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 9a52a5fdfed..35836dd27d6 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -11,6 +11,8 @@ const { ArrayPrototypePush, ArrayPrototypeSlice, FunctionPrototypeCall, + ObjectDefineProperty, + ObjectFreeze, ObjectSetPrototypeOf, PromisePrototypeThen, PromiseReject, @@ -564,6 +566,19 @@ function* createSyncPipeline(source, transforms) { // Async Pipeline Implementation // ============================================================================= +// The options object passed to async transforms. Stateless transforms get a +// new one for every call, so it is constructed rather than created as a +// `{ __proto__: null, signal }` literal (a dictionary-mode object). The +// prototype is frozen and has no %Object.prototype% in its chain, and the +// non-enumerable `constructor` lets util.inspect() print the options as +// `TransformOptions { signal }`. +function TransformOptions(signal) { + this.signal = signal; +} +TransformOptions.prototype = ObjectFreeze(ObjectDefineProperty( + { __proto__: null }, 'constructor', + { __proto__: null, value: TransformOptions })); + /** * Apply a single stateless async transform to a source. * @yields {Uint8Array[]} @@ -574,7 +589,7 @@ function* createSyncPipeline(source, transforms) { * avoiding the overhead of N async generator ticks for N transforms. * * INVARIANT: This function accepts a signal, NOT a pre-built options object. - * A fresh { __proto__: null, signal } options object is created for each + * A fresh TransformOptions object is created for each * transform invocation to prevent cross-transform mutation. * @param {AsyncIterable} source * @param {Array} run - Array of stateless transform functions @@ -585,7 +600,7 @@ async function* applyFusedStatelessAsyncTransforms(source, run, signal) { for await (const chunks of source) { let current = chunks; for (let i = 0; i < run.length; i++) { - let result = run[i](current, { __proto__: null, signal }); + let result = run[i](current, new TransformOptions(signal)); if (isPromise(result)) result = await result; if (result === null) { current = null; @@ -628,14 +643,14 @@ async function* applyFusedStatelessAsyncTransforms(source, run, signal) { for (let j = 0; j < pending.length; j++) { const pendingResult = appendTransformResultAsync( next, - run[i](pending[j], { __proto__: null, signal })); + run[i](pending[j], new TransformOptions(signal))); if (pendingResult !== undefined) { await pendingResult; } } const flushResult = appendTransformResultAsync( next, - run[i](null, { __proto__: null, signal })); + run[i](null, new TransformOptions(signal))); if (flushResult !== undefined) { await flushResult; } @@ -742,7 +757,7 @@ async function* createAsyncPipeline(source, transforms, signal) { // generator layer to avoid unnecessary async generator ticks. // // INVARIANT: Each transform invocation MUST receive its own fresh options - // object ({ __proto__: null, signal }). Transforms may mutate the options + // object (new TransformOptions(signal)). Transforms may mutate the options // object, so sharing a single object across invocations would allow one // transform to corrupt the options seen by another. The signal is shared // across calls (mutations to it are acceptable), but the containing options @@ -763,7 +778,7 @@ async function* createAsyncPipeline(source, transforms, signal) { transformSignal); statelessRun = []; } - const opts = { __proto__: null, signal: transformSignal }; + const opts = new TransformOptions(transformSignal); if (transform[kValidatedTransform]) { current = applyValidatedStatefulAsyncTransform( current, transform.transform, transform.receiver, opts); diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index 71fc27426ea..5bffd49d511 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -16,6 +16,7 @@ const { } = require('stream/iter'); const { setImmediate } = require('timers/promises'); +const { inspect } = require('util'); async function testPullIdentity() { const data = await text(pull(from('hello-async'))); @@ -573,6 +574,41 @@ async function testTransformOptionsNotShared() { assert.strictEqual(seen[1].mutated, undefined); } +// Stateless transforms get a new options object for every call, and stateful +// transforms one for the pipeline. The options object has only `signal`, does +// not inherit from Object.prototype, and its prototype is frozen, so that a +// transform cannot pass state to others through it. +async function testTransformOptionsShape() { + const seen = []; + const stateless = (chunks, options) => { + seen.push(options); + return chunks; + }; + const stateful = { + async* transform(source, options) { + seen.push(options); + for await (const chunks of source) yield chunks; + }, + }; + const ac = new AbortController(); + await text(pull(from(['a', 'b']), stateless, stateful, + { signal: ac.signal })); + // Stateless: one call per batch plus the flush call. + assert.strictEqual(seen.length, 4); + assert.strictEqual(new Set(seen).size, seen.length); + for (const options of seen) { + assert.strictEqual(options instanceof Object, false); + assert.deepStrictEqual(Object.keys(options), ['signal']); + assert.ok(options.signal instanceof AbortSignal); + assert.strictEqual(Object.isFrozen(Object.getPrototypeOf(options)), true); + assert.match(inspect(options), /^TransformOptions \{ signal: /); + } + assert.strictEqual(Object.getPrototypeOf(seen[0]), + Object.getPrototypeOf(seen[3])); + assert.throws(() => { Object.getPrototypeOf(seen[0]).leak = true; }, + TypeError); +} + // Run the uncaughtException test sequentially (it installs a global handler // that would interfere with concurrent tests). (async () => { @@ -609,6 +645,7 @@ async function testTransformOptionsNotShared() { testTransformReturnsArrayBuffer(), testPipeToStringSource(), testTransformOptionsNotShared(), + testTransformOptionsShape(), ]); // Run after all concurrent tests complete to avoid global handler races await testTransformSignalListenerErrorOnSourceError(); From cdae3e119251f2a9907ac8223b43021d5580eb5f Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 02:39:06 +0000 Subject: [PATCH 08/58] stream: do not miss 'drain' in fromWritable() When a write filled the classic Writable, fromWritable() recorded that it needed to drain, but only listened for 'drain' once a later write was queued or something waited for drain. If the Writable emitted 'drain' before that, for example because its write callback ran on a microtask or with process.nextTick(), the event was missed and the flag was never cleared: the next write() or writev() never settled, canWrite stayed false and ondrain() never resolved. Listen for 'drain' as soon as a write returns false, and keep listening until it is emitted. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/classic.js | 7 ++++ ...est-stream-iter-from-writable-lifecycle.js | 33 +++++++++++++++++++ 2 files changed, 40 insertions(+) diff --git a/lib/internal/streams/iter/classic.js b/lib/internal/streams/iter/classic.js index 2a7b818efd9..0eafba0182b 100644 --- a/lib/internal/streams/iter/classic.js +++ b/lib/internal/streams/iter/classic.js @@ -621,7 +621,9 @@ function fromWritable(writable, options = kNullPrototype) { } function removeDrainListenerIfIdle() { + // Keep listening while needsDrain is set: it is only cleared by 'drain'. if (!drainListenerInstalled || + needsDrain || pendingWrites.length !== 0 || drainWaiters.length !== 0) { return; @@ -701,7 +703,12 @@ function fromWritable(writable, options = kNullPrototype) { for (let i = 0; i < chunks.length; i++) { const bytes = chunks[i]; if (!writable.write(bytes)) { + // Listen for 'drain' now, even if nothing is waiting yet: the + // Writable can emit it before the next write or wait (for example + // when its write callback runs on a microtask), and needsDrain + // would never be cleared. needsDrain = true; + installDrainListener(); ok = false; } totalBytes += TypedArrayPrototypeGetByteLength(bytes); diff --git a/test/parallel/test-stream-iter-from-writable-lifecycle.js b/test/parallel/test-stream-iter-from-writable-lifecycle.js index d5be23fdd03..450fe8d7359 100644 --- a/test/parallel/test-stream-iter-from-writable-lifecycle.js +++ b/test/parallel/test-stream-iter-from-writable-lifecycle.js @@ -277,7 +277,40 @@ async function testSignalAbortedByUnderlyingEnd() { assert.strictEqual(await ending, 0); } +// A write that fills the Writable resolves before 'drain'. If the Writable's +// write callback runs on a microtask or a tick, 'drain' can be emitted before +// the next write or wait; the adapter must not miss it. +async function testDrainBeforeNextWrite() { + for (const defer of [queueMicrotask, process.nextTick]) { + const writable = new Writable({ + highWaterMark: 4, + write(chunk, encoding, callback) { defer(callback); }, + }); + const writer = fromWritable(writable); + + for (let i = 0; i < 3; i++) { + await writer.write('abcd'); + } + await writer.writev(['ab', 'cd']); + await writer.writev(['ab', 'cd']); + + await setImmediate(); + assert.strictEqual(writer.canWrite, true); + // Nothing is waiting any more, so the 'drain' listener is removed. + assert.strictEqual(writable.listenerCount('drain'), 0); + + writer.write('abcd'); + assert.strictEqual(writer.canWrite, false); + assert.strictEqual(await ondrain(writer), true); + assert.strictEqual(writer.canWrite, true); + + await writer.end(); + assert.strictEqual(writable.listenerCount('drain'), 0); + } +} + Promise.all([ + testDrainBeforeNextWrite(), testQueuedWriteThrowsDuringFlush(), testQueuedWriteErrorsDuringFlush(), testWriteErrorsSynchronously(), From c3803d676887ed238b204b1be9ad289ab968ac4d Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 02:47:38 +0000 Subject: [PATCH 09/58] stream: share the once option for stream/iter abort listeners stream/iter registered its one-time 'abort' listeners with a new `{ __proto__: null, once: true }` options object every time, which can be once per chunk (abortableNext()) or per waiting write. Use a single shared kNullOnceOption instead. It is frozen, because signals can come from user code and a patched addEventListener() must not be able to change the options for every later registration. The difference is small, since the listener registration itself costs much more: with 300,000 16-byte chunks, pull() with a signal is about 1.7% faster, push() writes that wait with a signal about 3% faster, and pipeTo() and bytes() with a signal are unchanged. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/broadcast.js | 5 +++-- lib/internal/streams/iter/classic.js | 6 ++---- lib/internal/streams/iter/consumers.js | 6 ++---- lib/internal/streams/iter/duplex.js | 4 ++-- lib/internal/streams/iter/pull.js | 6 +++--- lib/internal/streams/iter/push.js | 8 +++----- lib/internal/streams/iter/transform.js | 3 ++- lib/internal/streams/iter/utils.js | 10 ++++++++-- 8 files changed, 25 insertions(+), 23 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index e395b5e84f2..00ed899d653 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -61,6 +61,7 @@ const { PendingRequest, PendingWrite, kMultiConsumerDefaultBudget, + kNullOnceOption, kResolvedPromise, convertChunks, createBatchEntry, @@ -99,7 +100,7 @@ function raceEndWithSignal(promise, signal) { const { promise: aborted, reject } = PromiseWithResolvers(); const onAbort = () => reject(signal.reason); - signal.addEventListener('abort', onAbort, { __proto__: null, once: true }); + signal.addEventListener('abort', onAbort, kNullOnceOption); if (signal.aborted) onAbort(); return SafePromisePrototypeFinally( @@ -873,7 +874,7 @@ function wireBroadcastWriteSignal(entry, signal, resolve, reject, self) { entry.batch = null; reject(reason); }; - signal.addEventListener('abort', onAbort, { __proto__: null, once: true }); + signal.addEventListener('abort', onAbort, kNullOnceOption); } // ============================================================================= diff --git a/lib/internal/streams/iter/classic.js b/lib/internal/streams/iter/classic.js index 0eafba0182b..c1ed8ad51f5 100644 --- a/lib/internal/streams/iter/classic.js +++ b/lib/internal/streams/iter/classic.js @@ -63,6 +63,7 @@ const { const { PendingRequest, + kNullOnceOption, convertChunks, getWriterSignal, onSignalAbort, @@ -103,10 +104,7 @@ function raceWithSignal(promise, signal) { reject, } = PromiseWithResolvers(); const onAbort = () => reject(signal.reason); - signal.addEventListener('abort', onAbort, { - __proto__: null, - once: true, - }); + signal.addEventListener('abort', onAbort, kNullOnceOption); PromisePrototypeThen( promise, (value) => { diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index 8678e8235ac..0751d2cc95f 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -54,6 +54,7 @@ const { } = require('internal/streams/iter/from'); const { + kNullOnceOption, concatBytes, getProtocolMethod, recordChunk, @@ -528,10 +529,7 @@ function merge(...args) { waitResolve = null; } }; - signal.addEventListener('abort', onAbort, { - __proto__: null, - once: true, - }); + signal.addEventListener('abort', onAbort, kNullOnceOption); } // Called when a source's .next() settles. Pushes the result into diff --git a/lib/internal/streams/iter/duplex.js b/lib/internal/streams/iter/duplex.js index 95a117d8141..ce5ccadd060 100644 --- a/lib/internal/streams/iter/duplex.js +++ b/lib/internal/streams/iter/duplex.js @@ -18,6 +18,7 @@ const { const { converters, } = require('internal/streams/iter/webidl'); +const { kNullOnceOption } = require('internal/streams/iter/utils'); /** * Create a pair of connected duplex channels for bidirectional communication. @@ -66,8 +67,7 @@ function duplex(options = { __proto__: null }) { if (signal.aborted) { abortBoth(); } else { - signal.addEventListener('abort', abortBoth, - { __proto__: null, once: true }); + signal.addEventListener('abort', abortBoth, kNullOnceOption); } } diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 35836dd27d6..fc8fa9902c1 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -53,6 +53,7 @@ const { const { IterResult, + kNullOnceOption, createBatchEntry, isTransformObject, parsePullArgs, @@ -748,7 +749,7 @@ async function* createAsyncPipeline(source, transforms, signal) { abortHandler = () => { abortSignal(controller.signal, signal.reason); }; - signal.addEventListener('abort', abortHandler, { __proto__: null, once: true }); + signal.addEventListener('abort', abortHandler, kNullOnceOption); } let completed = false; @@ -992,8 +993,7 @@ function pullWithConsumerCleanup(source, transforms, signal) { if (signal !== undefined) { abortHandler = () => closeSource('throw', signal.reason); - signal.addEventListener('abort', abortHandler, - { __proto__: null, once: true }); + signal.addEventListener('abort', abortHandler, kNullOnceOption); if (signal.aborted) abortHandler(); } diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index b5db45a1a32..c99253af43e 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -32,6 +32,7 @@ const { IterResult, PendingRequest, PendingWrite, + kNullOnceOption, kPushDefaultBudget, kResolvedPromise, createBatchEntry, @@ -72,10 +73,7 @@ function raceEndWithSignal(promise, signal) { } = PromiseWithResolvers(); const onAbort = () => reject(signal.reason); - signal.addEventListener('abort', onAbort, { - __proto__: null, - once: true, - }); + signal.addEventListener('abort', onAbort, kNullOnceOption); PromisePrototypeThen( promise, (value) => { @@ -320,7 +318,7 @@ class PushQueue { reject(reason); }; - signal.addEventListener('abort', onAbort, { __proto__: null, once: true }); + signal.addEventListener('abort', onAbort, kNullOnceOption); } return promise; diff --git a/lib/internal/streams/iter/transform.js b/lib/internal/streams/iter/transform.js index 2cd06957b80..c6c315cf6bd 100644 --- a/lib/internal/streams/iter/transform.js +++ b/lib/internal/streams/iter/transform.js @@ -40,6 +40,7 @@ const { } = require('internal/errors'); const { isArrayBufferView, isAnyArrayBuffer } = require('internal/util/types'); const { kValidatedTransform } = require('internal/streams/iter/types'); +const { kNullOnceOption } = require('internal/streams/iter/utils'); const { checkRangesOrGetDefault, kValidateObjectAllowArray, @@ -413,7 +414,7 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { reject(signal.reason); } }; - signal.addEventListener('abort', onAbort, { __proto__: null, once: true }); + signal.addEventListener('abort', onAbort, kNullOnceOption); function continueInputAsync() { const { promise, resolve, reject } = PromiseWithResolvers(); diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 8b0c074b4e1..8c2835acd8b 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -47,6 +47,11 @@ const { kValidatedTransform, } = require('internal/streams/iter/types'); +// Shared `addEventListener()` options for one-time 'abort' listeners. Frozen +// because signals can come from user code, and a patched addEventListener() +// must not be able to change the options for every later registration. +const kNullOnceOption = ObjectFreeze({ __proto__: null, once: true }); + // Cached resolved promise to avoid allocating a new one on every sync fast-path. const kResolvedPromise = PromiseResolve(); @@ -98,7 +103,7 @@ function onSignalAbort(signal, handler) { if (signal.aborted) { handler(); } else { - signal.addEventListener('abort', handler, { __proto__: null, once: true }); + signal.addEventListener('abort', handler, kNullOnceOption); } } @@ -122,7 +127,7 @@ function abortableNext(iterator, signal) { const next = iterator.next(); const { promise, reject } = PromiseWithResolvers(); const onAbort = getOnAbort(reject, signal); - signal.addEventListener('abort', onAbort, { __proto__: null, once: true }); + signal.addEventListener('abort', onAbort, kNullOnceOption); if (signal.aborted) { onAbort(); } @@ -594,6 +599,7 @@ module.exports = { PendingRequest, PendingWrite, kMultiConsumerDefaultBudget, + kNullOnceOption, kPushDefaultBudget, kResolvedPromise, concatBytes, From 15c62f055796738fe64359cfc54ee5e104da7bf3 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 03:09:04 +0000 Subject: [PATCH 10/58] stream: make stream/iter from() cancellation waits cheaper To stay cancellable while a source is pending, from() waits for every value of an async source through waitForNormalization(). For each value it created a PromiseWithResolvers(), raced it against the value with SafePromiseRace(), which wraps both in new promises, and ran an async function with try/finally. This was the largest per-batch cost of normalizing an async source. Wait with a single promise and a single reaction on the value instead, and reject that promise directly on cancellation. The outcome is unchanged: the value's result, its rejection, or the cancellation reason, whichever comes first, and the cancellation reason if the normalization was cancelled by the time the value fulfills. With an async generator yielding 16-byte chunks, pipeTo() is about 1.7x faster and iterating from() about 1.75x faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/from.js | 58 ++++++++++++++++++++++--------- 1 file changed, 41 insertions(+), 17 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 78b61b267de..c89b7ea6d08 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -19,7 +19,6 @@ const { PromisePrototypeThen, PromiseResolve, PromiseWithResolvers, - SafePromiseRace, Symbol, SymbolAsyncIterator, SymbolIterator, @@ -54,6 +53,7 @@ const { const { IterResult, + kResolvedPromise, getProtocolMethod, toUint8Array, } = require('internal/streams/iter/utils'); @@ -62,7 +62,6 @@ const { // Bounds peak memory when arrays flow through transforms, which must // allocate output for the entire batch at once. const FROM_BATCH_SIZE = 128; -const kNormalizationCancelled = Symbol('kNormalizationCancelled'); // Yielded by normalizeAsyncValue() (only when `emitFlush` is true) right // before it waits on a promise or on a nested async iterable. Callers that // batch chunks yield whatever they have collected so far, so that chunks that @@ -83,31 +82,56 @@ function cancelNormalization(context, reason, suppressCleanup = false) { context.cancelled = true; context.reason = reason; context.suppressCleanup = suppressCleanup; - context.resolve?.(kNormalizationCancelled); + context.resolve?.(); } function throwIfNormalizationCancelled(context) { if (context?.cancelled) throw context.reason; } -async function waitForNormalization(value, context) { +/** + * Wait for `value`, but stop waiting if the normalization is cancelled. + * Settles with the first of: + * - `value` fulfilling: its value, or the cancellation reason if the + * normalization has been cancelled by then; + * - `value` rejecting: its rejection reason; + * - cancellation: the cancellation reason. + * This runs for every value of a normalized async source, so it uses a + * single promise and a single reaction on `value` rather than racing + * promises. + * @param {any} value + * @param {object} [context] + * @returns {Promise|any} + */ +function waitForNormalization(value, context) { if (context === undefined) return value; - const { promise, resolve } = PromiseWithResolvers(); + const { promise, resolve, reject } = PromiseWithResolvers(); + const onCancel = () => { + if (context.resolve === onCancel) context.resolve = null; + reject(context.reason); + }; + PromisePrototypeThen( + PromiseResolve(value), + (result) => { + if (context.resolve === onCancel) context.resolve = null; + if (context.cancelled) { + reject(context.reason); + } else { + resolve(result); + } + }, + (error) => { + if (context.resolve === onCancel) context.resolve = null; + reject(error); + }); if (context.cancelled) { - resolve(kNormalizationCancelled); + // Already cancelled: a `value` that has already settled still takes + // precedence, as its reaction above runs first. + PromisePrototypeThen(kResolvedPromise, onCancel); } else { - context.resolve = resolve; - } - try { - const result = await SafePromiseRace([ - PromiseResolve(value), - promise, - ]); - throwIfNormalizationCancelled(context); - return result; - } finally { - if (context.resolve === resolve) context.resolve = null; + context.resolve = onCancel; } + return promise; } function createNormalizationIterator(createIterator) { From 17ae0b0de1fd3bad01db8db8fe95e2562e54d06f Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 03:09:21 +0000 Subject: [PATCH 11/58] stream: yield bounded batches directly in stream/iter from() When a source value is already a Uint8Array[] batch, from() and fromSync() yielded it through `yield* yieldBoundedBatch(value)`, which creates a generator for every batch only to split batches larger than 128 chunks. In the async normalization, yield* of a sync generator also costs several extra promise ticks per batch. Yield batches within the bound directly, and delegate to yieldBoundedBatch() only for larger ones. Empty batches are still skipped. With 16-byte chunks, one per batch: pipeTo() from a sync iterable is about 2x faster and iterating from() over it about 2.8x faster; from an async generator, pipeTo() is about 1.5x and iteration about 1.7x faster; pipeToSync() is about 1.4x faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/from.js | 22 ++++++++++++++++++---- 1 file changed, 18 insertions(+), 4 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index c89b7ea6d08..5f0103b7baa 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -332,7 +332,11 @@ function* normalizeSyncSource(source) { yield batch; batch = []; } - yield* yieldBoundedBatch(value); + if (value.length <= FROM_BATCH_SIZE) { + if (value.length !== 0) yield value; + } else { + yield* yieldBoundedBatch(value); + } continue; } // Fast path 2: value is a single Uint8Array (very common) @@ -581,9 +585,15 @@ async function* normalizeAsyncSource(source, context) { if (isAsyncIterable(source)) { const iterable = yieldNormalizationAbortable(source, context); for await (const value of iterable) { - // Fast path 1: value is already a Uint8Array[] batch + // Fast path 1: value is already a Uint8Array[] batch. Yield a batch + // within the bound directly: yield* of a sync generator from an async + // generator costs several extra promise ticks per batch. if (isUint8ArrayBatch(value)) { - yield* yieldBoundedBatch(value); + if (value.length <= FROM_BATCH_SIZE) { + if (value.length !== 0) yield value; + } else { + yield* yieldBoundedBatch(value); + } continue; } // Fast path 2: value is a single Uint8Array (very common) @@ -630,7 +640,11 @@ async function* normalizeAsyncSource(source, context) { yield batch; batch = []; } - yield* yieldBoundedBatch(value); + if (value.length <= FROM_BATCH_SIZE) { + if (value.length !== 0) yield value; + } else { + yield* yieldBoundedBatch(value); + } continue; } // Fast path 2: value is a single Uint8Array (very common) From d0da1f74f7c96e0cc71e8e1ce5ac14ecdc7dadd8 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 03:21:12 +0000 Subject: [PATCH 12/58] stream: avoid batch entries for single-chunk pipeTo() writes pipeTo() and pipeToSync() created a batch entry for every batch, with an array and a seven-field snapshot per chunk, to reject a chunk that is resized or detached after being accepted. For the common single-chunk batch, check the view around writeSync() with the snapshot kept in locals instead, and create a batch entry in pipeTo() only to fall back to write(). Views on SharedArrayBuffers and batches of more chunks are still snapshotted as before, since writing one chunk can change another. With 16-byte chunks, one per batch, this saves about 100-175 bytes of allocation per chunk: pipeTo() from a sync source is about 17% faster, pipeToSync() about 15% faster and pipeTo() from an async generator about 7% faster. A new test covers detaching, resizing and growing a shared view in writeSync() for both, and the fallback to write(). Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/pull.js | 23 ++++++++++ lib/internal/streams/iter/utils.js | 37 ++++++++++++++++ .../test-stream-iter-resizable-buffers.js | 44 +++++++++++++++++++ 3 files changed, 104 insertions(+) diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index fc8fa9902c1..f4b1bc4e7bb 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -19,6 +19,7 @@ const { PromiseResolve, SymbolAsyncIterator, SymbolIterator, + TypedArrayPrototypeGetByteLength, Uint8Array, } = primordials; @@ -54,6 +55,7 @@ const { const { IterResult, kNullOnceOption, + callWithByteView, createBatchEntry, isTransformObject, parsePullArgs, @@ -1063,6 +1065,16 @@ function pipeToSync(source, ...args) { try { for (const batch of pipeline) { + // Single chunk, the common case: no batch entry needed. + if (batch.length === 1) { + const chunk = batch[0]; + if (callWithByteView(chunk, writer.writeSync, writer) === false) { + throw new ERR_OUT_OF_RANGE( + 'write', 'within byte budget', 'budget exhausted'); + } + totalBytes += TypedArrayPrototypeGetByteLength(chunk); + continue; + } const entry = createBatchEntry(batch); if (hasWritevSync && batch.length > 1) { const accepted = writer.writevSync(validateBatchEntry(entry)); @@ -1190,6 +1202,17 @@ async function pipeTo(source, ...args) { // Returns undefined on sync success, or a Promise when async fallback // is required. Callers must check: const p = writeBatch(b); if (p) await p; function writeBatch(batch) { + // Single chunk, the common case: check the view around writeSync() + // without allocating a batch entry, and create one only to fall back to + // the async path. + if (batch.length === 1 && hasWriteSync) { + const chunk = batch[0]; + if (callWithByteView(chunk, writer.writeSync, writer)) { + totalBytes += TypedArrayPrototypeGetByteLength(chunk); + return; + } + return writeBatchAsyncFallback(createBatchEntry(batch), 0); + } const entry = createBatchEntry(batch); if (hasWritev && batch.length > 1) { if (!hasWritevSync || diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 8c2835acd8b..bc57920bcbc 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -7,6 +7,7 @@ const { ArrayBufferPrototypeGetResizable, ArrayPrototypePush, ArrayPrototypeSlice, + FunctionPrototypeCall, ObjectDefineProperty, ObjectFreeze, PromiseResolve, @@ -295,6 +296,41 @@ function validateByteView(snapshot) { return value; } +/** + * Call `method` on `receiver` with the byte view `value`, and throw if the + * call resized or detached it, as validateByteView() does for a snapshot + * taken just before the call. Used for single-chunk writes, the common case: + * for views on an ArrayBuffer the snapshot is kept in locals instead of a + * ByteViewSnapshot, so that nothing is allocated per chunk. + * @param {Uint8Array} value + * @param {Function} method + * @param {object} receiver + * @returns {any} The result of the call. + */ +function callWithByteView(value, method, receiver) { + const buffer = TypedArrayPrototypeGetBuffer(value); + if (isSharedArrayBuffer(buffer)) { + const snapshot = snapshotByteView(value); + const result = FunctionPrototypeCall(method, receiver, value); + validateByteView(snapshot); + return result; + } + const bufferByteLength = ArrayBufferPrototypeGetByteLength(buffer); + const byteLength = TypedArrayPrototypeGetByteLength(value); + const byteOffset = TypedArrayPrototypeGetByteOffset(value); + const detached = ArrayBufferPrototypeGetDetached(buffer); + const result = FunctionPrototypeCall(method, receiver, value); + if (TypedArrayPrototypeGetBuffer(value) !== buffer || + ArrayBufferPrototypeGetByteLength(buffer) !== bufferByteLength || + TypedArrayPrototypeGetByteLength(value) !== byteLength || + TypedArrayPrototypeGetByteOffset(value) !== byteOffset || + ArrayBufferPrototypeGetDetached(buffer) !== detached) { + throw new ERR_INVALID_STATE.TypeError( + 'Byte view was resized or detached after being accepted'); + } + return result; +} + /** * Validate an explicit `budget` option. The spec only requires the default * budget to be at least 16384 bytes; any positive explicit budget is valid. @@ -602,6 +638,7 @@ module.exports = { kNullOnceOption, kPushDefaultBudget, kResolvedPromise, + callWithByteView, concatBytes, convertChunks, createBatchEntry, diff --git a/test/parallel/test-stream-iter-resizable-buffers.js b/test/parallel/test-stream-iter-resizable-buffers.js index 91f7429ec12..ab75d07c455 100644 --- a/test/parallel/test-stream-iter-resizable-buffers.js +++ b/test/parallel/test-stream-iter-resizable-buffers.js @@ -167,6 +167,49 @@ async function testPipeRejectsWriterResize() { ); } +// Single-chunk batches are checked around writeSync() without a snapshot +// object. Detaching or resizing the view in writeSync() must still be +// rejected, for views on ArrayBuffers and on growable SharedArrayBuffers. +async function testPipeRejectsSyncWriteDetachOrResize() { + const cases = [ + ['detach', () => new ArrayBuffer(2), (buffer) => buffer.transfer()], + ['resize', () => new ArrayBuffer(1, { maxByteLength: 2 }), + (buffer) => buffer.resize(2)], + ['grow shared', () => new SharedArrayBuffer(1, { maxByteLength: 2 }), + (buffer) => buffer.grow(2)], + ]; + for (const [, createBuffer, change] of cases) { + const asyncBuffer = createBuffer(); + await assert.rejects(pipeTo([new Uint8Array(asyncBuffer)], { + writeSync() { + change(asyncBuffer); + return true; + }, + write: common.mustNotCall(), + fail: common.mustCall(), + }), kResizeError); + + const syncBuffer = createBuffer(); + assert.throws(() => pipeToSync([new Uint8Array(syncBuffer)], { + writeSync() { + change(syncBuffer); + return true; + }, + fail: common.mustCall(), + }, { preventClose: true }), kResizeError); + } + + // An unchanged view is written; when writeSync() declines it, pipeTo() + // falls back to write() for it. + const chunk = Uint8Array.of(1, 2); + const written = []; + assert.strictEqual(await pipeTo([chunk], { + writeSync: () => false, + write(value) { written.push(value); }, + }, { preventClose: true }), 2); + assert.deepStrictEqual(written, [chunk]); +} + async function testConsumersRejectDetachedViews() { // Views of fixed-length buffers are tracked without a full snapshot; they // must still be rejected when detached after being accepted. @@ -205,4 +248,5 @@ Promise.all([ testConsumersRejectResizedViews(), testConsumersRejectDetachedViews(), testPipeRejectsWriterResize(), + testPipeRejectsSyncWriteDetachOrResize(), ]).then(common.mustCall()); From 773f42550f348e86bef08de39c107b22a35c0957 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 03:24:10 +0000 Subject: [PATCH 13/58] stream: wait without an async function in stream/iter from() The iterator through which from() reads an async source, to stay cancellable while the source is pending, had an async next() that awaited waitForNormalization() for every value: an async function frame and promise, plus another promise and reactions for the wait. Make next() a plain function that waits for the source's result with a single promise, which a cancellation rejects directly. The result checks, the closing of the source on cancellation and the precedence between the source's result and a cancellation are unchanged. With an async generator yielding 16-byte chunks, this allocates about 450 bytes less per chunk; pipeTo() is about 9% faster and iterating from() about 11% faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/from.js | 105 +++++++++++++++++++++++------- 1 file changed, 81 insertions(+), 24 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 5f0103b7baa..15ed395a42e 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -17,6 +17,7 @@ const { FunctionPrototypeCall, ObjectSetPrototypeOf, PromisePrototypeThen, + PromiseReject, PromiseResolve, PromiseWithResolvers, Symbol, @@ -414,38 +415,94 @@ function yieldNormalizationAbortable(source, context) { } } + // Settle a pending next() with a result of the source's next(). Throws + // (to be handled by the caller) like the checks it replaces would. + function toIterResult(result) { + throwIfNormalizationCancelled(context); + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); + } + if (result.done) { + reading = false; + throwIfNormalizationCancelled(context); + completed = true; + closed = true; + return new IterResult(true, result.value); + } + const value = result.value; + reading = false; + throwIfNormalizationCancelled(context); + return new IterResult(false, value); + } + + // Reject a pending next(), closing the source first if the + // normalization has been cancelled. + function rejectNext(reject, error) { + if (context.cancelled) { + PromisePrototypeThen(closeSource(true), () => { + reading = false; + reject(error); + }); + return; + } + reading = false; + reject(error); + } + return ObjectSetPrototypeOf({ - async next() { + // next() runs for every value of the source, so instead of being an + // async function awaiting waitForNormalization(), it waits for the + // source with a single promise, which a cancellation rejects + // directly. + next() { if (completed) { - return new IterResult(true, undefined); + return PromiseResolve(new IterResult(true, undefined)); } - throwIfNormalizationCancelled(context); + if (context.cancelled) return PromiseReject(context.reason); reading = true; + const { promise, resolve, reject } = PromiseWithResolvers(); + let next; try { - const next = FunctionPrototypeCall(nextMethod, iterator); - const result = await waitForNormalization(next, context); - if ((typeof result !== 'object' && typeof result !== 'function') || - result === null) { - throw new ERR_INVALID_RETURN_VALUE( - 'an object', 'iterator.next()', result); - } - if (result.done) { - reading = false; - throwIfNormalizationCancelled(context); - completed = true; - closed = true; - return new IterResult(true, result.value); - } - const value = result.value; - reading = false; - throwIfNormalizationCancelled(context); - return new IterResult(false, value); + next = FunctionPrototypeCall(nextMethod, iterator); } catch (error) { - if (context.cancelled) await closeSource(true); - reading = false; - throw error; + rejectNext(reject, error); + return promise; } + // The first of the source's result and a cancellation settles + // next(); whichever comes later is ignored. + let settled = false; + const onCancel = () => { + if (settled) return; + settled = true; + if (context.resolve === onCancel) context.resolve = null; + rejectNext(reject, context.reason); + }; + context.resolve = onCancel; + PromisePrototypeThen( + PromiseResolve(next), + (result) => { + if (settled) return; + settled = true; + if (context.resolve === onCancel) context.resolve = null; + let iterResult; + try { + iterResult = toIterResult(result); + } catch (error) { + rejectNext(reject, error); + return; + } + resolve(iterResult); + }, + (error) => { + if (settled) return; + settled = true; + if (context.resolve === onCancel) context.resolve = null; + rejectNext(reject, error); + }); + return promise; }, async return(value) { await closeSource( From d2d813933746f5fc21336fdc2640f7931c89dbc2 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 03:34:12 +0000 Subject: [PATCH 14/58] stream: normalize async sources without a generator in from() from() normalized an async iterable source with an async generator looping over it with for await, which costs several promises and an async frame for every batch. Replace it with an iterator written out by hand that behaves the same way: nothing happens until the first next(), calls made while one is in progress are queued, an error from the source ends the iteration without closing the source, an error normalizing a value closes the source first, return() closes the value being normalized and the source and propagates errors from closing them, and throw() closes them ignoring such errors. Values that are already Uint8Array[] batches or Uint8Arrays take one promise per batch; any other value is normalized by an async generator as before. Sync iterable sources are unchanged. With an async generator yielding 16-byte chunks, this allocates about 440 bytes less per chunk; pipeTo() is about 28% faster and iterating from() about 38% faster. The results of the iterator are now IterResult objects, like those of the other stream/iter iterators, rather than ordinary objects; one test compared them as such. New tests cover the behavior above. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/from.js | 294 +++++++++++++++--- test/parallel/test-stream-iter-from-async.js | 158 +++++++++- .../test-stream-iter-iterator-result.js | 3 + 3 files changed, 409 insertions(+), 46 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 15ed395a42e..d7716352d07 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -631,59 +631,263 @@ async function* normalizeAsyncValue( } /** - * Normalize an async streamable source, yielding batches of Uint8Array. - * @param {AsyncIterable|Iterable} source + * Normalize a value of an async source that is neither a Uint8Array nor a + * Uint8Array[] batch, yielding batches of Uint8Array. Chunks are batched, + * but whatever has been collected is yielded before waiting on async content + * (such as a nested async iterable), so data is not held back until it ends. + * @param {any} value + * @param {object} context * @yields {Uint8Array[]} */ -async function* normalizeAsyncSource(source, context) { - throwIfNormalizationCancelled(context); - - // Prefer async iteration if available - if (isAsyncIterable(source)) { - const iterable = yieldNormalizationAbortable(source, context); - for await (const value of iterable) { - // Fast path 1: value is already a Uint8Array[] batch. Yield a batch - // within the bound directly: yield* of a sync generator from an async - // generator costs several extra promise ticks per batch. - if (isUint8ArrayBatch(value)) { - if (value.length <= FROM_BATCH_SIZE) { - if (value.length !== 0) yield value; - } else { - yield* yieldBoundedBatch(value); - } - continue; - } - // Fast path 2: value is a single Uint8Array (very common) - if (isUint8Array(value)) { - yield [value]; - continue; - } - // Slow path: normalize the value. Chunks are batched, but whatever has - // been collected is yielded before waiting on async content (such as a - // nested async iterable), so data is not held back until it ends. - let batch = []; - for await (const chunk of - normalizeAsyncValue(value, true, context, true)) { - if (chunk === kFlushBatch) { - if (batch.length > 0) { - yield batch; - batch = []; - } - continue; - } - ArrayPrototypePush(batch, chunk); - if (batch.length === FROM_BATCH_SIZE) { - yield batch; - batch = []; - } - } +async function* normalizeAsyncSourceValue(value, context) { + let batch = []; + for await (const chunk of + normalizeAsyncValue(value, true, context, true)) { + if (chunk === kFlushBatch) { if (batch.length > 0) { yield batch; + batch = []; } + continue; } - return; + ArrayPrototypePush(batch, chunk); + if (batch.length === FROM_BATCH_SIZE) { + yield batch; + batch = []; + } + } + if (batch.length > 0) { + yield batch; + } +} + +const kStart = 0; +const kActive = 1; +const kDone = 2; + +/** + * Iterator normalizing an async iterable source into batches of Uint8Array. + * + * This is what an async generator looping over the source with for await + * would do, written out by hand: the source is read for every batch, and an + * async generator layer costs several promises and an async frame each + * time. Values that are already batches or Uint8Arrays (the common case) + * are handled with one promise per batch; any other value is normalized by + * normalizeAsyncSourceValue(). As with an async generator: + * - nothing happens until the first next(); + * - calls made while one is in progress are queued; + * - an error from the source ends the iteration without closing the source, + * and an error normalizing a value closes the source first; + * - return() closes the value being normalized and the source, propagating + * errors from closing them, and throw() closes them ignoring such errors. + * @param {AsyncIterable} source + * @param {object} context + * @returns {object} An object with next(), return() and throw(). + */ +function createAsyncSourceNormalizer(source, context) { + let state = kStart; + // The source iterator, from yieldNormalizationAbortable(). Its return() + // is an async function, so calling it cannot throw synchronously. + let iterator; + // The value being normalized: a sync iterator of the sub-batches of an + // oversized batch, or a normalizeAsyncSourceValue() generator. + let boundedBatches = null; + let valueBatches = null; + let busy = false; + let queue = null; + + function drain() { + if (busy || queue === null) return; + const { 0: method, 1: arg, 2: resolve, 3: reject } = queue.shift(); + if (queue.length === 0) queue = null; + busy = true; + PromisePrototypeThen(method(arg), resolve, reject); + } + + function settled() { + busy = false; + if (queue !== null) PromisePrototypeThen(kResolvedPromise, drain); + } + + function finish(result) { + settled(); + return result; + } + + function fail(error) { + state = kDone; + settled(); + throw error; + } + + function run(method, arg) { + if (busy || queue !== null) { + const { promise, resolve, reject } = PromiseWithResolvers(); + queue ??= []; + ArrayPrototypePush(queue, [method, arg, resolve, reject]); + return promise; + } + busy = true; + return method(arg); + } + + // Closing for a throw completion: wait, but ignore errors. + function closeQuietly(it) { + return PromisePrototypeThen(it.return(), undefined, () => {}); } + function onBodyError(error) { + // Like for await when its body throws: close the source, keep the error. + valueBatches = null; + boundedBatches = null; + state = kDone; + return PromisePrototypeThen( + closeQuietly(iterator), () => fail(error)); + } + + function onSourceResult(result) { + if (result.done) { + state = kDone; + return finish(new IterResult(true, undefined)); + } + try { + return handleValue(result.value); + } catch (error) { + return onBodyError(error); + } + } + + function handleValue(value) { + if (isUint8ArrayBatch(value)) { + if (value.length <= FROM_BATCH_SIZE) { + if (value.length === 0) return pullSource(); + return finish(new IterResult(false, value)); + } + boundedBatches = yieldBoundedBatch(value); + return pullValue(); + } + if (isUint8Array(value)) return finish(new IterResult(false, [value])); + valueBatches = normalizeAsyncSourceValue(value, context); + return pullValue(); + } + + function onValueResult(result) { + if (result.done) { + valueBatches = null; + return pullSource(); + } + return finish(new IterResult(false, result.value)); + } + + function pullSource() { + return PromisePrototypeThen(iterator.next(), onSourceResult, fail); + } + + function pullValue() { + if (boundedBatches !== null) { + const result = boundedBatches.next(); + if (!result.done) return finish(new IterResult(false, result.value)); + boundedBatches = null; + return pullSource(); + } + return PromisePrototypeThen( + valueBatches.next(), onValueResult, onBodyError); + } + + function doNext() { + if (state === kDone) { + return PromiseResolve(finish(new IterResult(true, undefined))); + } + if (state === kStart) { + try { + throwIfNormalizationCancelled(context); + const iterable = yieldNormalizationAbortable(source, context); + iterator = iterable[SymbolAsyncIterator](); + } catch (error) { + state = kDone; + settled(); + return PromiseReject(error); + } + state = kActive; + } + if (boundedBatches !== null || valueBatches !== null) { + return PromiseResolve(pullValue()); + } + return pullSource(); + } + + function doReturn(value) { + const result = new IterResult(true, value); + if (state !== kActive) { + state = kDone; + return PromiseResolve(finish(result)); + } + state = kDone; + boundedBatches = null; + const pending = valueBatches; + valueBatches = null; + // Close the value being normalized, then the source. If closing the + // value fails, the source is still closed and the error kept. + let closed; + if (pending === null) { + closed = iterator.return(); + } else { + closed = PromisePrototypeThen( + pending.return(), + () => iterator.return(), + (error) => PromisePrototypeThen(closeQuietly(iterator), () => { + throw error; + })); + } + return PromisePrototypeThen(closed, () => finish(result), fail); + } + + function doThrow(error) { + if (state !== kActive) { + state = kDone; + settled(); + return PromiseReject(error); + } + state = kDone; + boundedBatches = null; + const pending = valueBatches; + valueBatches = null; + const closed = pending === null ? closeQuietly(iterator) : + PromisePrototypeThen(closeQuietly(pending), () => closeQuietly(iterator)); + return PromisePrototypeThen(closed, () => fail(error)); + } + + return ObjectSetPrototypeOf({ + next() { return run(doNext); }, + return(value) { return run(doReturn, value); }, + throw(error) { return run(doThrow, error); }, + }, null); +} + +/** + * Normalize an async streamable source, yielding batches of Uint8Array. + * @param {AsyncIterable|Iterable} source + * @param {object} context + * @returns {object} An async iterator. + */ +function normalizeAsyncSource(source, context) { + // Prefer async iteration if available. + if (isAsyncIterable(source)) { + return createAsyncSourceNormalizer(source, context); + } + return normalizeSyncSourceAsync(source, context); +} + +/** + * Normalize a sync streamable source for from(), yielding batches of + * Uint8Array. + * @param {Iterable} source + * @param {object} context + * @yields {Uint8Array[]} + */ +async function* normalizeSyncSourceAsync(source, context) { + throwIfNormalizationCancelled(context); + // Fall back to sync iteration - batch sync values together with a bound. if (isSyncIterable(source)) { let batch = []; diff --git a/test/parallel/test-stream-iter-from-async.js b/test/parallel/test-stream-iter-from-async.js index 1b58fd6603c..66a921549d7 100644 --- a/test/parallel/test-stream-iter-from-async.js +++ b/test/parallel/test-stream-iter-from-async.js @@ -123,7 +123,7 @@ async function testFromDoesNotHoldBackNestedAsyncIterable() { } const iterator = from(source())[Symbol.asyncIterator](); - assert.deepStrictEqual(await iterator.next(), + assert.deepStrictEqual({ ...await iterator.next() }, { done: false, value: [new Uint8Array([1])] }); resolve(); const rest = []; @@ -529,6 +529,156 @@ async function testFromFunctionWithProtocols() { assert.throws(() => from(() => {}), { code: 'ERR_INVALID_ARG_TYPE' }); } +// from() reads async sources like an async generator looping over them with +// for await. The tests below check the parts of that behavior that are +// observable from the source: when it is read and closed. + +// Creates an async iterable source of `values`, recording calls in `log`. +function createLoggedSource(log, values, { failAt = -1, returnError } = {}) { + let i = 0; + return { + [Symbol.asyncIterator]() { + log.push('iterator'); + return { + async next() { + log.push(`next ${i}`); + if (i === failAt) { + i++; + throw new Error('source failed'); + } + if (i >= values.length) return { done: true, value: undefined }; + return { done: false, value: values[i++] }; + }, + async return() { + log.push('return'); + if (returnError !== undefined) throw returnError; + return { done: true, value: undefined }; + }, + }; + }, + }; +} + +async function settle(log, label, promise) { + try { + const result = await promise; + log.push(`${label}: ${result.done ? 'done' : result.value.length}`); + } catch (error) { + log.push(`${label}: ${error.code ?? error.message}`); + } +} + +async function testFromQueuesConcurrentNext() { + const log = []; + const iterator = from(createLoggedSource(log, [ + Uint8Array.of(1), Uint8Array.of(2), + ]))[Symbol.asyncIterator](); + const results = [iterator.next(), iterator.next(), iterator.next()]; + for (let i = 0; i < results.length; i++) { + await settle(log, `result ${i}`, results[i]); + } + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'next 1', 'result 0: 1', 'next 2', 'result 1: 1', + 'result 2: done', + ]); +} + +async function testFromSplitsOversizedBatches() { + const log = []; + const batch = Array.from({ length: 300 }, () => new Uint8Array(1)); + const iterator = from(createLoggedSource(log, [batch, [], [batch[0]]]))[ + Symbol.asyncIterator](); + for (let i = 0; i < 5; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result 0: 128', 'result 1: 128', 'result 2: 44', + 'next 1', 'next 2', 'result 3: 1', 'next 3', 'result 4: done', + ]); +} + +async function testFromSourceErrorDoesNotCloseSource() { + const log = []; + const iterator = from(createLoggedSource(log, [Uint8Array.of(1)], { + failAt: 1, + }))[Symbol.asyncIterator](); + for (let i = 0; i < 3; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result 0: 1', 'next 1', 'result 1: source failed', + 'result 2: done', + ]); +} + +async function testFromNormalizationErrorClosesSource() { + // A value that cannot be normalized. + let log = []; + let iterator = from(createLoggedSource(log, [Uint8Array.of(1), 42]))[ + Symbol.asyncIterator](); + for (let i = 0; i < 3; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result 0: 1', 'next 1', 'return', + 'result 1: ERR_INVALID_ARG_TYPE', 'result 2: done', + ]); + + // A nested async iterable that fails. + async function* nested() { + yield Uint8Array.of(2); + throw new Error('nested failed'); + } + log = []; + iterator = from(createLoggedSource(log, [nested()]))[Symbol.asyncIterator](); + for (let i = 0; i < 3; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result 0: 1', 'return', 'result 1: nested failed', + 'result 2: done', + ]); +} + +async function testFromReturnAndThrowBeforeStart() { + const log = []; + let iterator = from(createLoggedSource(log, [Uint8Array.of(1)]))[ + Symbol.asyncIterator](); + assert.deepStrictEqual({ ...await iterator.return('v') }, + { done: true, value: 'v' }); + await settle(log, 'after return', iterator.next()); + iterator = from(createLoggedSource(log, [Uint8Array.of(1)]))[ + Symbol.asyncIterator](); + await settle(log, 'throw', iterator.throw(new Error('thrown'))); + await settle(log, 'after throw', iterator.next()); + // The source is never read. + assert.deepStrictEqual(log, [ + 'after return: done', 'throw: thrown', 'after throw: done', + ]); +} + +async function testFromReturnAndThrowCloseSource() { + const returnError = new Error('return failed'); + for (const method of ['return', 'throw']) { + const log = []; + let nestedClosed = false; + async function* nested() { + try { + yield Uint8Array.of(1); + yield Uint8Array.of(2); + } finally { + nestedClosed = true; + } + } + const iterator = from(createLoggedSource(log, [nested()], { + returnError, + }))[Symbol.asyncIterator](); + await settle(log, 'result', iterator.next()); + await settle(log, method, iterator[method](new Error('thrown'))); + await settle(log, 'after', iterator.next()); + assert.strictEqual(nestedClosed, true); + // return() propagates errors from closing the source; throw() keeps its + // own error. + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result: 1', 'return', + method === 'return' ? 'return: return failed' : 'throw: thrown', + 'after: done', + ]); + } +} + Promise.all([ testFromString(), testFromAsyncGenerator(), @@ -563,4 +713,10 @@ Promise.all([ testConsumerAbortClosesPendingNestedIterator(), testFromCancellationHandlesCleanupRejection(), testFromDataView(), + testFromQueuesConcurrentNext(), + testFromSplitsOversizedBatches(), + testFromSourceErrorDoesNotCloseSource(), + testFromNormalizationErrorClosesSource(), + testFromReturnAndThrowBeforeStart(), + testFromReturnAndThrowCloseSource(), ]).then(common.mustCall()); diff --git a/test/parallel/test-stream-iter-iterator-result.js b/test/parallel/test-stream-iter-iterator-result.js index 4f29bf05534..ba4c35686de 100644 --- a/test/parallel/test-stream-iter-iterator-result.js +++ b/test/parallel/test-stream-iter-iterator-result.js @@ -33,6 +33,9 @@ async function testAsyncIterators() { return readable; }, 'share()': () => share(from('a')).pull(), + 'from() of an async iterable': () => from((async function*() { + yield 'a'; + })()), 'broadcast()': () => { const { writer, broadcast: bc } = broadcast(); const consumer = bc.push(); From fdc4c0239280aa1210e24ba0129c16239d1bfee9 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 03:52:16 +0000 Subject: [PATCH 15/58] stream: normalize sync sources without an async generator in from() from() normalized a sync iterable source with an async generator, which costs several promises and an async frame for every batch, even though the source is read synchronously. Read the source with a sync generator instead, which collects chunks into batches as before, and normalize the values that need it, such as promises, in an async iterator written out by hand around it. The sync generator's for...of reads and closes the source as before: return() and throw() are passed to it, after closing the value being normalized, and an error normalizing a value is thrown into it, closing the source as for an error in the loop body. As an async generator does, the iterator stays busy until the tick after a result, so that calls made synchronously after one are queued behind it. With 16-byte chunks, one per batch, this allocates about 150 bytes less per chunk; iterating from() is about 15% faster and pipeTo() about 11% faster. Sources of single chunks, which are batched, are unchanged. The queue of operations is now shared with the iterator for async sources, and uses ArrayPrototypeShift(). New tests cover batching, errors, closing and queuing for sync sources. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/from.js | 339 ++++++++++++++----- test/parallel/test-stream-iter-from-async.js | 128 ++++++- 2 files changed, 377 insertions(+), 90 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index d7716352d07..54b3db2a2f6 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -10,11 +10,13 @@ const { ArrayIsArray, ArrayPrototypeEvery, ArrayPrototypePush, + ArrayPrototypeShift, ArrayPrototypeSlice, DataViewPrototypeGetBuffer, DataViewPrototypeGetByteLength, DataViewPrototypeGetByteOffset, FunctionPrototypeCall, + ObjectFreeze, ObjectSetPrototypeOf, PromisePrototypeThen, PromiseReject, @@ -637,12 +639,15 @@ async function* normalizeAsyncValue( * (such as a nested async iterable), so data is not held back until it ends. * @param {any} value * @param {object} context + * @param {boolean} [allowNestedAsync] Whether nested async iterables and + * toAsyncStreamable are allowed (they are not in sync sources). * @yields {Uint8Array[]} */ -async function* normalizeAsyncSourceValue(value, context) { +async function* normalizeAsyncSourceValue(value, context, + allowNestedAsync = true) { let batch = []; for await (const chunk of - normalizeAsyncValue(value, true, context, true)) { + normalizeAsyncValue(value, allowNestedAsync, context, true)) { if (chunk === kFlushBatch) { if (batch.length > 0) { yield batch; @@ -665,6 +670,52 @@ const kStart = 0; const kActive = 1; const kDone = 2; +/** + * Serializes the operations of a hand-written async iterator the way an + * async generator queues its requests: an operation started while another + * is in progress waits until it has settled. + * @returns {{ run: Function, settled: Function, release: Function }} + */ +function createOperationQueue() { + let busy = false; + let queue = null; + + function drain() { + if (busy || queue === null) return; + const { 0: method, 1: arg, 2: resolve, 3: reject } = + ArrayPrototypeShift(queue); + if (queue.length === 0) queue = null; + busy = true; + PromisePrototypeThen(method(arg), resolve, reject); + } + + return { + __proto__: null, + // Run `method(arg)`, which returns a promise and must call settled() + // once its result is known, now or after the operations before it. + run(method, arg) { + if (busy || queue !== null) { + const { promise, resolve, reject } = PromiseWithResolvers(); + queue ??= []; + ArrayPrototypePush(queue, [method, arg, resolve, reject]); + return promise; + } + busy = true; + return method(arg); + }, + settled() { + busy = false; + if (queue !== null) PromisePrototypeThen(kResolvedPromise, drain); + }, + // Like settled(), but start the next queued operation now, as an async + // generator does once the await of a yield or return completes. + release() { + busy = false; + drain(); + }, + }; +} + /** * Iterator normalizing an async iterable source into batches of Uint8Array. * @@ -693,21 +744,7 @@ function createAsyncSourceNormalizer(source, context) { // oversized batch, or a normalizeAsyncSourceValue() generator. let boundedBatches = null; let valueBatches = null; - let busy = false; - let queue = null; - - function drain() { - if (busy || queue === null) return; - const { 0: method, 1: arg, 2: resolve, 3: reject } = queue.shift(); - if (queue.length === 0) queue = null; - busy = true; - PromisePrototypeThen(method(arg), resolve, reject); - } - - function settled() { - busy = false; - if (queue !== null) PromisePrototypeThen(kResolvedPromise, drain); - } + const { run, settled } = createOperationQueue(); function finish(result) { settled(); @@ -720,17 +757,6 @@ function createAsyncSourceNormalizer(source, context) { throw error; } - function run(method, arg) { - if (busy || queue !== null) { - const { promise, resolve, reject } = PromiseWithResolvers(); - queue ??= []; - ArrayPrototypePush(queue, [method, arg, resolve, reject]); - return promise; - } - busy = true; - return method(arg); - } - // Closing for a throw completion: wait, but ignore errors. function closeQuietly(it) { return PromisePrototypeThen(it.return(), undefined, () => {}); @@ -875,86 +901,221 @@ function normalizeAsyncSource(source, context) { if (isAsyncIterable(source)) { return createAsyncSourceNormalizer(source, context); } - return normalizeSyncSourceAsync(source, context); + return createSyncSourceNormalizer(source, context); } +// A value of a sync source that is not a Uint8Array or a Uint8Array[] +// batch, yielded by readSyncSource() to be normalized asynchronously. +function SyncSourceValue(value) { + this.value = value; +} +SyncSourceValue.prototype = ObjectFreeze({ __proto__: null }); + /** - * Normalize a sync streamable source for from(), yielding batches of - * Uint8Array. + * Read a sync iterable source for from(): yields Uint8Array[] batches, + * collecting single Uint8Arrays into batches of up to FROM_BATCH_SIZE + * chunks, and a SyncSourceValue for any other value, after the chunks + * collected before it. * @param {Iterable} source * @param {object} context - * @yields {Uint8Array[]} + * @yields {Uint8Array[]|SyncSourceValue} */ -async function* normalizeSyncSourceAsync(source, context) { +function* readSyncSource(source, context) { throwIfNormalizationCancelled(context); - - // Fall back to sync iteration - batch sync values together with a bound. - if (isSyncIterable(source)) { - let batch = []; - - for (const value of source) { - throwIfNormalizationCancelled(context); - // Fast path 1: value is already a Uint8Array[] batch - if (isUint8ArrayBatch(value)) { - // Flush any accumulated batch first - if (batch.length > 0) { - yield batch; - batch = []; - } - if (value.length <= FROM_BATCH_SIZE) { - if (value.length !== 0) yield value; - } else { - yield* yieldBoundedBatch(value); - } - continue; - } - // Fast path 2: value is a single Uint8Array (very common) - if (isUint8Array(value)) { - ArrayPrototypePush(batch, value); - if (batch.length === FROM_BATCH_SIZE) { - yield batch; - batch = []; - } - continue; - } - // Slow path: normalize the value - must flush and yield individually + if (!isSyncIterable(source)) { + throw new ERR_INVALID_ARG_TYPE( + 'source', ['Iterable', 'AsyncIterable'], source); + } + let batch = []; + for (const value of source) { + throwIfNormalizationCancelled(context); + // Fast path 1: value is already a Uint8Array[] batch + if (isUint8ArrayBatch(value)) { + // Flush any accumulated batch first if (batch.length > 0) { yield batch; batch = []; } - let asyncBatch = []; - for await (const chunk of - normalizeAsyncValue(value, false, context, true)) { - if (chunk === kFlushBatch) { - if (asyncBatch.length > 0) { - yield asyncBatch; - asyncBatch = []; - } - continue; - } - ArrayPrototypePush(asyncBatch, chunk); - if (asyncBatch.length === FROM_BATCH_SIZE) { - yield asyncBatch; - asyncBatch = []; - } + if (value.length <= FROM_BATCH_SIZE) { + if (value.length !== 0) yield value; + } else { + yield* yieldBoundedBatch(value); } - if (asyncBatch.length > 0) { - yield asyncBatch; + continue; + } + // Fast path 2: value is a single Uint8Array (very common) + if (isUint8Array(value)) { + ArrayPrototypePush(batch, value); + if (batch.length === FROM_BATCH_SIZE) { + yield batch; + batch = []; } + continue; } - - // Yield any remaining batched values + // Slow path: flush, then have the value normalized if (batch.length > 0) { yield batch; + batch = []; } - return; + yield new SyncSourceValue(value); + } + // Yield any remaining batched values + if (batch.length > 0) { + yield batch; } +} - throw new ERR_INVALID_ARG_TYPE( - 'source', - ['Iterable', 'AsyncIterable'], - source, - ); +/** + * Iterator normalizing a sync iterable source for from() into batches of + * Uint8Array, without an async generator layer for every batch (see + * createAsyncSourceNormalizer()). + * + * The source is read by the sync generator readSyncSource(), so for...of + * reads and closes it, and this iterator only adds the asynchronous + * normalization of values that need it. Operations on the generator mirror + * those on the async generator this replaces: return() and throw() are + * passed to it, after closing the value being normalized, and an error + * normalizing a value is thrown into it, so that for...of closes the source + * as for an error in the loop body. + * @param {Iterable} source + * @param {object} context + * @returns {object} An object with next(), return() and throw(). + */ +function createSyncSourceNormalizer(source, context) { + const reader = readSyncSource(source, context); + let done = false; + // A normalizeAsyncSourceValue() generator for the value being normalized. + let valueBatches = null; + const { run, settled, release } = createOperationQueue(); + + // Results are often produced synchronously. An async generator stays busy + // until the tick after a yield or return (both await their operand), so + // that a next(), return() or throw() made synchronously after this one is + // queued behind it; do the same. + function finish(result) { + PromisePrototypeThen(kResolvedPromise, release); + return result; + } + + function fail(error) { + done = true; + settled(); + throw error; + } + + // Throw `error` into the reader, closing the source as for...of does when + // its body throws: errors from closing it are ignored. + function throwIntoReader(error) { + try { + reader.throw(error); + } catch { + // The reader rethrows `error`. + } + } + + function onValueResult(result) { + if (result.done) { + valueBatches = null; + return produce(); + } + return finish(new IterResult(false, result.value)); + } + + function onValueError(error) { + valueBatches = null; + throwIntoReader(error); + return fail(error); + } + + // Produce the next result: an IterResult, or a promise for one when a + // value is normalized asynchronously. Throws, after settling, on error. + function produce() { + if (valueBatches !== null) { + return PromisePrototypeThen( + valueBatches.next(), onValueResult, onValueError); + } + let result; + try { + result = reader.next(); + } catch (error) { + return fail(error); + } + if (result.done) { + done = true; + return finish(new IterResult(true, undefined)); + } + const value = result.value; + if (ArrayIsArray(value)) return finish(new IterResult(false, value)); + valueBatches = normalizeAsyncSourceValue(value.value, context, false); + return PromisePrototypeThen( + valueBatches.next(), onValueResult, onValueError); + } + + function doNext() { + if (done) return PromiseResolve(finish(new IterResult(true, undefined))); + try { + return PromiseResolve(produce()); + } catch (error) { + return PromiseReject(error); + } + } + + function doReturn(value) { + const result = new IterResult(true, value); + if (done) return PromiseResolve(finish(result)); + done = true; + const pending = valueBatches; + valueBatches = null; + if (pending === null) { + try { + reader.return(); + } catch (error) { + settled(); + return PromiseReject(error); + } + return PromiseResolve(finish(result)); + } + // Close the value being normalized, then the source. If closing the + // value fails, the source is still closed and the error kept. + return PromisePrototypeThen(pending.return(), () => { + try { + reader.return(); + } catch (error) { + return fail(error); + } + return finish(result); + }, (error) => { + throwIntoReader(error); + return fail(error); + }); + } + + function doThrow(error) { + if (done) { + settled(); + return PromiseReject(error); + } + done = true; + const pending = valueBatches; + valueBatches = null; + if (pending === null) { + throwIntoReader(error); + settled(); + return PromiseReject(error); + } + return PromisePrototypeThen( + PromisePrototypeThen(pending.return(), undefined, () => {}), + () => { + throwIntoReader(error); + return fail(error); + }); + } + + return ObjectSetPrototypeOf({ + next() { return run(doNext); }, + return(value) { return run(doReturn, value); }, + throw(error) { return run(doThrow, error); }, + }, null); } async function* normalizeAsyncStreamableResult(result, context) { diff --git a/test/parallel/test-stream-iter-from-async.js b/test/parallel/test-stream-iter-from-async.js index 66a921549d7..356f07353d7 100644 --- a/test/parallel/test-stream-iter-from-async.js +++ b/test/parallel/test-stream-iter-from-async.js @@ -564,7 +564,9 @@ async function settle(log, label, promise) { const result = await promise; log.push(`${label}: ${result.done ? 'done' : result.value.length}`); } catch (error) { - log.push(`${label}: ${error.code ?? error.message}`); + const reason = error.name === 'AbortError' ? error.name : + error.code ?? error.message; + log.push(`${label}: ${reason}`); } } @@ -679,6 +681,126 @@ async function testFromReturnAndThrowCloseSource() { } } +// Creates a sync iterable source of `values`, recording calls in `log`. +function createLoggedSyncSource(log, values, { failAt = -1, returnError } = {}) { + let i = 0; + return { + [Symbol.iterator]() { + log.push('iterator'); + return { + next() { + log.push(`next ${i}`); + if (i === failAt) { + i++; + throw new Error('source failed'); + } + if (i >= values.length) return { done: true, value: undefined }; + return { done: false, value: values[i++] }; + }, + return() { + log.push('return'); + if (returnError !== undefined) throw returnError; + return { done: true, value: undefined }; + }, + }; + }, + }; +} + +async function testFromSyncSourceBatching() { + // Single chunks are collected into batches of up to 128 chunks; any other + // value is yielded after the chunks collected before it. + const log = []; + const chunks = Array.from({ length: 130 }, () => new Uint8Array(1)); + const iterator = from(createLoggedSyncSource(log, [ + ...chunks, [Uint8Array.of(1), Uint8Array.of(2)], Uint8Array.of(3), 'ab', + ]))[Symbol.asyncIterator](); + for (let i = 0; i < 6; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', ...Array.from({ length: 128 }, (_, i) => `next ${i}`), + 'result 0: 128', 'next 128', 'next 129', 'next 130', 'result 1: 2', + 'result 2: 2', 'next 131', 'next 132', 'result 3: 1', 'result 4: 1', + 'next 133', 'result 5: done', + ]); +} + +async function testFromSyncSourceErrors() { + // An error from the source does not close it. + let log = []; + let iterator = from(createLoggedSyncSource(log, [[Uint8Array.of(1)]], { + failAt: 1, + }))[Symbol.asyncIterator](); + for (let i = 0; i < 3; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result 0: 1', 'next 1', 'result 1: source failed', + 'result 2: done', + ]); + + // An error processing a value closes it, also after flushing the chunks + // collected before the value, and when normalizing a promise fails. + for (const value of [42, Promise.reject(new Error('rejected'))]) { + log = []; + iterator = from(createLoggedSyncSource(log, [Uint8Array.of(1), value]))[ + Symbol.asyncIterator](); + for (let i = 0; i < 3; i++) { + await settle(log, `result ${i}`, iterator.next()); + } + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'next 1', 'result 0: 1', 'return', + `result 1: ${value === 42 ? 'ERR_INVALID_ARG_TYPE' : 'rejected'}`, + 'result 2: done', + ]); + } +} + +async function testFromSyncSourceReturnAndThrow() { + const returnError = new Error('return failed'); + for (const method of ['return', 'throw']) { + // While reading the source, it is closed: return() propagates errors + // from closing it, throw() keeps its own error. + let log = []; + let iterator = from(createLoggedSyncSource(log, [ + [Uint8Array.of(1)], [Uint8Array.of(2)], + ], { returnError }))[Symbol.asyncIterator](); + await settle(log, 'result', iterator.next()); + await settle(log, method, iterator[method](new Error('thrown'))); + await settle(log, 'after', iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result: 1', 'return', + method === 'return' ? 'return: return failed' : 'throw: thrown', + 'after: done', + ]); + + // Once the source has ended, it is not closed. + log = []; + iterator = from(createLoggedSyncSource(log, [Uint8Array.of(1)]))[ + Symbol.asyncIterator](); + await settle(log, 'result', iterator.next()); + await settle(log, method, iterator[method](new Error('thrown'))); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'next 1', 'result: 1', + method === 'return' ? 'return: done' : 'throw: thrown', + ]); + } +} + +async function testFromSyncSourceQueuesBehindReturn() { + // A next() made synchronously after another waits for it, and so sees a + // return() made synchronously after it as well. + const log = []; + const iterator = from(createLoggedSyncSource(log, [ + [Uint8Array.of(1)], 'a', [Uint8Array.of(2)], + ]))[Symbol.asyncIterator](); + const results = [iterator.next(), iterator.next(), iterator.return()]; + for (let i = 0; i < results.length; i++) { + await settle(log, `result ${i}`, results[i]); + } + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'next 1', 'return', 'result 0: 1', + 'result 1: AbortError', 'result 2: done', + ]); +} + Promise.all([ testFromString(), testFromAsyncGenerator(), @@ -719,4 +841,8 @@ Promise.all([ testFromNormalizationErrorClosesSource(), testFromReturnAndThrowBeforeStart(), testFromReturnAndThrowCloseSource(), + testFromSyncSourceBatching(), + testFromSyncSourceErrors(), + testFromSyncSourceReturnAndThrow(), + testFromSyncSourceQueuesBehindReturn(), ]).then(common.mustCall()); From 2fbcad3b793cf88776f43e677e1daa3072b5f783 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 04:20:59 +0000 Subject: [PATCH 16/58] stream: run stream/iter transforms without async generators pipeTo() and pull() ran a pipeline of transforms through two async generator layers for every batch: one applying each run of stateless transforms, and the pipeline itself, which checks for an abort before passing each batch on and cleans up when done. Each costs several promises and an async frame per batch. Write both out by hand, behaving the same way: the source is opened by the first next(), calls made while one is in progress are queued, an error from the source ends the pipeline without closing it, an error from a transform closes it, return() and throw() close what is being read, and the transforms' signal is aborted when the pipeline fails or is stopped early. Transforms returning batches or chunks synchronously take a single promise per batch for each layer; results that have to be waited for or normalized asynchronously, the flush, and stateful transforms are still handled by async generators. With an async generator yielding 16-byte chunks, one per batch, pipeTo() through one or two stateless transforms allocates about 900 bytes less per chunk and is about 38% faster. The queue of operations for these iterators moves to utils.js. New tests cover closing the source and aborting the transforms' signal. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/from.js | 55 +- lib/internal/streams/iter/pull.js | 672 +++++++++++++++---- lib/internal/streams/iter/utils.js | 58 ++ test/parallel/test-stream-iter-pull-async.js | 118 ++++ 4 files changed, 729 insertions(+), 174 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 54b3db2a2f6..47b71e09a49 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -10,7 +10,6 @@ const { ArrayIsArray, ArrayPrototypeEvery, ArrayPrototypePush, - ArrayPrototypeShift, ArrayPrototypeSlice, DataViewPrototypeGetBuffer, DataViewPrototypeGetByteLength, @@ -56,7 +55,11 @@ const { const { IterResult, + kActive, + kDone, kResolvedPromise, + kStart, + createOperationQueue, getProtocolMethod, toUint8Array, } = require('internal/streams/iter/utils'); @@ -666,56 +669,6 @@ async function* normalizeAsyncSourceValue(value, context, } } -const kStart = 0; -const kActive = 1; -const kDone = 2; - -/** - * Serializes the operations of a hand-written async iterator the way an - * async generator queues its requests: an operation started while another - * is in progress waits until it has settled. - * @returns {{ run: Function, settled: Function, release: Function }} - */ -function createOperationQueue() { - let busy = false; - let queue = null; - - function drain() { - if (busy || queue === null) return; - const { 0: method, 1: arg, 2: resolve, 3: reject } = - ArrayPrototypeShift(queue); - if (queue.length === 0) queue = null; - busy = true; - PromisePrototypeThen(method(arg), resolve, reject); - } - - return { - __proto__: null, - // Run `method(arg)`, which returns a promise and must call settled() - // once its result is known, now or after the operations before it. - run(method, arg) { - if (busy || queue !== null) { - const { promise, resolve, reject } = PromiseWithResolvers(); - queue ??= []; - ArrayPrototypePush(queue, [method, arg, resolve, reject]); - return promise; - } - busy = true; - return method(arg); - }, - settled() { - busy = false; - if (queue !== null) PromisePrototypeThen(kResolvedPromise, drain); - }, - // Like settled(), but start the next queued operation now, as an async - // generator does once the await of a yield or return completes. - release() { - busy = false; - drain(); - }, - }; -} - /** * Iterator normalizing an async iterable source into batches of Uint8Array. * diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index f4b1bc4e7bb..e300364a820 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -17,6 +17,7 @@ const { PromisePrototypeThen, PromiseReject, PromiseResolve, + Symbol, SymbolAsyncIterator, SymbolIterator, TypedArrayPrototypeGetByteLength, @@ -27,6 +28,7 @@ const { codes: { ERR_INVALID_ARG_TYPE, ERR_INVALID_ARG_VALUE, + ERR_INVALID_RETURN_VALUE, ERR_INVALID_STATE, ERR_OUT_OF_RANGE, }, @@ -54,9 +56,14 @@ const { const { IterResult, + kActive, + kDone, kNullOnceOption, + kResolvedPromise, + kStart, callWithByteView, createBatchEntry, + createOperationQueue, isTransformObject, parsePullArgs, snapshotTransform, @@ -583,85 +590,348 @@ TransformOptions.prototype = ObjectFreeze(ObjectDefineProperty( { __proto__: null, value: TransformOptions })); /** - * Apply a single stateless async transform to a source. + * Close an async iterator as for await does: for a throw completion + * (`quiet`), wait for it but ignore errors; otherwise reject on errors and + * on a result that is not an object. + * @param {object} iterator + * @param {boolean} quiet + * @returns {Promise} + */ +function closeAsyncIterator(iterator, quiet) { + let promise; + try { + const returnMethod = iterator.return; + if (returnMethod === undefined || returnMethod === null) { + return kResolvedPromise; + } + promise = PromiseResolve(FunctionPrototypeCall(returnMethod, iterator)); + } catch (error) { + return quiet ? kResolvedPromise : PromiseReject(error); + } + if (quiet) return PromisePrototypeThen(promise, undefined, () => {}); + return PromisePrototypeThen(promise, (result) => { + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.return()', result); + } + }); +} + +/** + * Normalize the output of a fused run of stateless transforms for a batch. + * @yields {Uint8Array[]} + */ +async function* yieldFusedStatelessOutput(current) { + if (isUint8ArrayBatch(current)) { + if (current.length > 0) yield current; + } else if (isUint8Array(current)) { + yield [current]; + } else if (typeof current === 'string') { + yield [toUint8Array(current)]; + } else if (isAnyArrayBuffer(current)) { + yield [new Uint8Array(current)]; + } else if (ArrayBufferIsView(current)) { + yield [arrayBufferViewToUint8Array(current)]; + } else { + yield* processTransformResultAsync(current); + } +} + +/** + * Apply a fused run of stateless transforms to a batch from transform + * `index` on, given the result of that transform, when the result has to be + * waited for or normalized asynchronously. + * @param {Array} run + * @param {number} index + * @param {any} result - The result of `run[index]` + * @param {AbortSignal} signal * @yields {Uint8Array[]} */ +async function* continueFusedStatelessBatch(run, index, result, signal) { + let current; + for (let i = index; i < run.length; i++) { + if (i !== index) result = run[i](current, new TransformOptions(signal)); + if (isPromise(result)) result = await result; + if (result === null) return; + if (i === run.length - 1) { + current = result; + break; + } + current = normalizeTransformResultFast(result); + if (current === undefined) { + const normalized = []; + const pendingResult = appendTransformResultAsync(normalized, result); + if (pendingResult !== undefined) await pendingResult; + current = normalized.length === 0 ? null : normalized[0]; + } + if (current === null) return; + } + yield* yieldFusedStatelessOutput(current); +} + +/** + * Flush a fused run of stateless transforms once the source has ended: + * flush each transform after all upstream data, including data emitted by + * earlier flushes, has been processed by that transform. + * @param {Array} run + * @param {AbortSignal} signal + * @yields {Uint8Array[]} + */ +async function* flushFusedStatelessAsyncTransforms(run, signal) { + let pending = []; + for (let i = 0; i < run.length; i++) { + const next = []; + for (let j = 0; j < pending.length; j++) { + const pendingResult = appendTransformResultAsync( + next, + run[i](pending[j], new TransformOptions(signal))); + if (pendingResult !== undefined) { + await pendingResult; + } + } + const flushResult = appendTransformResultAsync( + next, + run[i](null, new TransformOptions(signal))); + if (flushResult !== undefined) { + await flushResult; + } + pending = next; + } + for (let i = 0; i < pending.length; i++) { + yield pending[i]; + } +} + +// Returned by applyBatch() below when a batch is processed by a delegate. +const kDelegated = Symbol('kDelegated'); + /** * Apply a fused run of stateless async transforms to a source. * All transforms in the run are applied in a tight synchronous loop per batch, * avoiding the overhead of N async generator ticks for N transforms. * + * This is what an async generator looping over the source with for await + * would do, written out by hand, as an async generator layer costs several + * promises and an async frame for every batch: when the transforms return + * batches or chunks synchronously, a batch takes a single promise. Results + * that have to be waited for or normalized asynchronously, and the flush + * once the source has ended, are handled by async generators. As with an + * async generator looping over the source: + * - the source is opened by the first next(), and calls made while one is in + * progress are queued; + * - an error from the source ends the iteration without closing it, and an + * error from a transform closes it, ignoring errors from closing it; + * - return() and throw() are passed to the async generator handling a batch, + * if any, and close the source unless it has ended; return() propagates + * errors from closing it, throw() ignores them. + * * INVARIANT: This function accepts a signal, NOT a pre-built options object. * A fresh TransformOptions object is created for each * transform invocation to prevent cross-transform mutation. * @param {AsyncIterable} source * @param {Array} run - Array of stateless transform functions * @param {AbortSignal} signal - The pipeline's abort signal - * @yields {Uint8Array[]} + * @returns {AsyncIterator} */ -async function* applyFusedStatelessAsyncTransforms(source, run, signal) { - for await (const chunks of source) { +function applyFusedStatelessAsyncTransforms(source, run, signal) { + let state = kStart; + let iterator; + let nextMethod; + // Whether the source has ended, and the run is being flushed. + let sourceDone = false; + // An async generator yielding the output of a batch, or of the flush. + let delegate = null; + const operations = createOperationQueue(); + + function finish(result) { + operations.settled(); + return result; + } + + function fail(error) { + state = kDone; + operations.settled(); + throw error; + } + + // Like for await when its body throws: close the source, keep the error. + function onBodyError(error) { + delegate = null; + state = kDone; + return PromisePrototypeThen( + closeAsyncIterator(iterator, true), () => fail(error)); + } + + function onDelegateError(error) { + if (!sourceDone) return onBodyError(error); + delegate = null; + return fail(error); + } + + function onDelegateResult(result) { + if (!result.done) return finish(new IterResult(false, result.value)); + delegate = null; + if (!sourceDone) return pullSource(); + state = kDone; + return finish(new IterResult(true, undefined)); + } + + function pullDelegate() { + return PromisePrototypeThen( + delegate.next(), onDelegateResult, onDelegateError); + } + + // Apply the run to a batch synchronously: returns the output batch, null + // if there is none, or kDelegated if `delegate` is to produce it. + function applyBatch(chunks) { let current = chunks; for (let i = 0; i < run.length; i++) { - let result = run[i](current, new TransformOptions(signal)); - if (isPromise(result)) result = await result; - if (result === null) { - current = null; - break; + const result = run[i](current, new TransformOptions(signal)); + if (isPromise(result)) { + delegate = continueFusedStatelessBatch(run, i, result, signal); + return kDelegated; } + if (result === null) return null; if (i === run.length - 1) { current = result; - continue; + break; } current = normalizeTransformResultFast(result); if (current === undefined) { - const normalized = []; - const pendingResult = appendTransformResultAsync(normalized, result); - if (pendingResult !== undefined) await pendingResult; - current = normalized.length === 0 ? null : normalized[0]; + delegate = continueFusedStatelessBatch(run, i, result, signal); + return kDelegated; } - if (current === null) break; + if (current === null) return null; } - if (current === null) continue; - // Normalize the final output if (isUint8ArrayBatch(current)) { - if (current.length > 0) yield current; - } else if (isUint8Array(current)) { - yield [current]; - } else if (typeof current === 'string') { - yield [toUint8Array(current)]; - } else if (isAnyArrayBuffer(current)) { - yield [new Uint8Array(current)]; - } else if (ArrayBufferIsView(current)) { - yield [arrayBufferViewToUint8Array(current)]; - } else { - yield* processTransformResultAsync(current); + return current.length > 0 ? current : null; + } + if (isUint8Array(current)) return [current]; + if (typeof current === 'string') return [toUint8Array(current)]; + if (isAnyArrayBuffer(current)) return [new Uint8Array(current)]; + if (ArrayBufferIsView(current)) { + return [arrayBufferViewToUint8Array(current)]; } + delegate = processTransformResultAsync(current); + return kDelegated; } - // Flush each transform after all upstream data, including data emitted by - // earlier flushes, has been processed by that transform. - let pending = []; - for (let i = 0; i < run.length; i++) { - const next = []; - for (let j = 0; j < pending.length; j++) { - const pendingResult = appendTransformResultAsync( - next, - run[i](pending[j], new TransformOptions(signal))); - if (pendingResult !== undefined) { - await pendingResult; + + function onSourceResult(result) { + let value; + try { + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); } + if (result.done) { + sourceDone = true; + delegate = flushFusedStatelessAsyncTransforms(run, signal); + return pullDelegate(); + } + value = result.value; + } catch (error) { + return fail(error); } - const flushResult = appendTransformResultAsync( - next, - run[i](null, new TransformOptions(signal))); - if (flushResult !== undefined) { - await flushResult; + let batch; + try { + batch = applyBatch(value); + } catch (error) { + return onBodyError(error); } - pending = next; + if (batch === kDelegated) return pullDelegate(); + if (batch === null) return pullSource(); + return finish(new IterResult(false, batch)); } - for (let i = 0; i < pending.length; i++) { - yield pending[i]; + + function pullSource() { + let promise; + try { + promise = PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)); + } catch (error) { + return fail(error); + } + return PromisePrototypeThen(promise, onSourceResult, fail); } + + function doNext() { + if (state === kDone) { + return PromiseResolve(finish(new IterResult(true, undefined))); + } + if (state === kStart) { + try { + iterator = source[SymbolAsyncIterator](); + nextMethod = iterator.next; + } catch (error) { + state = kDone; + operations.settled(); + return PromiseReject(error); + } + state = kActive; + } + try { + return PromiseResolve(delegate !== null ? pullDelegate() : pullSource()); + } catch (error) { + return PromiseReject(error); + } + } + + function doReturn(value) { + const result = new IterResult(true, value); + if (state !== kActive) { + state = kDone; + return PromiseResolve(finish(result)); + } + state = kDone; + const pending = delegate; + delegate = null; + const open = sourceDone ? null : iterator; + let closed; + if (pending === null) { + closed = open === null ? kResolvedPromise : + closeAsyncIterator(open, false); + } else { + // Close the delegate, then the source. If closing the delegate fails, + // the source is still closed and the error kept. + closed = PromisePrototypeThen( + closeAsyncIterator(pending, false), + () => (open === null ? undefined : closeAsyncIterator(open, false)), + (error) => { + if (open === null) throw error; + return PromisePrototypeThen(closeAsyncIterator(open, true), () => { + throw error; + }); + }); + } + return PromisePrototypeThen(closed, () => finish(result), fail); + } + + function doThrow(error) { + if (state !== kActive) { + state = kDone; + operations.settled(); + return PromiseReject(error); + } + state = kDone; + const pending = delegate; + delegate = null; + // The delegate rethrows `error` (as yield* would see it do). + let closed = pending === null ? kResolvedPromise : + PromisePrototypeThen(pending.throw(error), undefined, () => {}); + if (!sourceDone) { + closed = PromisePrototypeThen( + closed, () => closeAsyncIterator(iterator, true)); + } + return PromisePrototypeThen(closed, () => fail(error)); + } + + return ObjectSetPrototypeOf({ + next() { return operations.run(doNext); }, + return(value) { return operations.run(doReturn, value); }, + throw(error) { return operations.run(doThrow, error); }, + [SymbolAsyncIterator]() { return this; }, + }, null); } /** @@ -729,93 +999,105 @@ async function* applyValidatedStatefulAsyncTransform( /** * Create an async pipeline from source through transforms. - * @yields {Uint8Array[]} + * @param {AsyncIterable} source + * @param {Array} transforms + * @param {AbortSignal} [signal] + * @returns {AsyncIterator} */ -async function* createAsyncPipeline(source, transforms, signal) { - // Check for abort - signal?.throwIfAborted(); - - // Fast path: no transforms, just yield normalized source directly +function createAsyncPipeline(source, transforms, signal) { if (transforms.length === 0) { - yield* yieldAbortable(source, signal); - return; + return createAsyncPipelineWithoutTransforms(source, signal); } + return createAsyncTransformPipeline(source, transforms, signal); +} - const normalized = yieldAbortable(source, signal); +async function* createAsyncPipelineWithoutTransforms(source, signal) { + // Check for abort + signal?.throwIfAborted(); + yield* yieldAbortable(source, signal); +} - // Create internal controller for transform cancellation. - // Note: if signal was already aborted, we threw above - no need to check here. - const controller = new AbortController(); - let abortHandler; - if (signal) { - abortHandler = () => { - abortSignal(controller.signal, signal.reason); - }; - signal.addEventListener('abort', abortHandler, kNullOnceOption); - } +/** + * Build the chain of transform layers of a pipeline. + * @param {AsyncIterable} normalized + * @param {Array} transforms + * @param {AbortSignal} transformSignal + * @returns {AsyncIterable} + */ +function createAsyncTransformLayers(normalized, transforms, transformSignal) { + // Apply transforms - fuse consecutive stateless transforms into a single + // layer to avoid unnecessary async ticks. + // + // INVARIANT: Each transform invocation MUST receive its own fresh options + // object (new TransformOptions(signal)). Transforms may mutate the options + // object, so sharing a single object across invocations would allow one + // transform to corrupt the options seen by another. The signal is shared + // across calls (mutations to it are acceptable), but the containing options + // object must be unique per call. This is enforced inside + // applyFusedStatelessAsyncTransforms and applyStatefulAsyncTransform, which + // accept the signal directly and create the options object per invocation. + // DO NOT pass a pre-built options object. + let current = normalized; + let statelessRun = []; - let completed = false; - try { - // Apply transforms - fuse consecutive stateless transforms into a single - // generator layer to avoid unnecessary async generator ticks. - // - // INVARIANT: Each transform invocation MUST receive its own fresh options - // object (new TransformOptions(signal)). Transforms may mutate the options - // object, so sharing a single object across invocations would allow one - // transform to corrupt the options seen by another. The signal is shared - // across calls (mutations to it are acceptable), but the containing options - // object must be unique per call. This is enforced inside - // applyFusedStatelessAsyncTransforms and applyStatefulAsyncTransform, which - // accept the signal directly and create the options object per invocation. - // DO NOT pass a pre-built options object. - let current = normalized; - const transformSignal = controller.signal; - let statelessRun = []; - - for (let i = 0; i < transforms.length; i++) { - const transform = transforms[i]; - if (isTransformObject(transform)) { - // Flush any accumulated stateless run before the stateful transform - if (statelessRun.length > 0) { - current = applyFusedStatelessAsyncTransforms(current, statelessRun, - transformSignal); - statelessRun = []; - } - const opts = new TransformOptions(transformSignal); - if (transform[kValidatedTransform]) { - current = applyValidatedStatefulAsyncTransform( - current, transform.transform, transform.receiver, opts); - } else { - current = applyStatefulAsyncTransform( - current, transform.transform, transform.receiver, opts); - } + for (let i = 0; i < transforms.length; i++) { + const transform = transforms[i]; + if (isTransformObject(transform)) { + // Flush any accumulated stateless run before the stateful transform + if (statelessRun.length > 0) { + current = applyFusedStatelessAsyncTransforms(current, statelessRun, + transformSignal); + statelessRun = []; + } + const opts = new TransformOptions(transformSignal); + if (transform[kValidatedTransform]) { + current = applyValidatedStatefulAsyncTransform( + current, transform.transform, transform.receiver, opts); } else { - ArrayPrototypePush(statelessRun, transform); + current = applyStatefulAsyncTransform( + current, transform.transform, transform.receiver, opts); } + } else { + ArrayPrototypePush(statelessRun, transform); } - // Flush remaining stateless run - if (statelessRun.length > 0) { - current = applyFusedStatelessAsyncTransforms(current, statelessRun, - transformSignal); - } + } + // Flush remaining stateless run + if (statelessRun.length > 0) { + current = applyFusedStatelessAsyncTransforms(current, statelessRun, + transformSignal); + } + return current; +} - for await (const batch of current) { - controller.signal.throwIfAborted(); - yield batch; - } - // A transform can abort while completing without producing a final batch, - // for example when an async flush resolves to null. In that case the loop - // body has no opportunity to observe the abort. - controller.signal.throwIfAborted(); - completed = true; - } catch (error) { - if (!controller.signal.aborted) { - abortSignal(controller.signal, error); - } - throw error; - } finally { +/** + * The pipeline through one or more transforms: an async iterator doing what + * an async generator would, written out by hand to avoid an async generator + * layer for every batch (see applyFusedStatelessAsyncTransforms()). + * + * When started by the first next(), it checks `signal`, then creates the + * controller whose signal the transforms get, aborted when `signal` aborts. + * Each batch, and the end, is passed on only if the transforms' signal has + * not been aborted. If the pipeline fails, the transforms' signal is aborted + * with the error; if it is stopped early by return() or throw(), the + * transforms are closed and their signal is aborted. + * @param {AsyncIterable} source + * @param {Array} transforms + * @param {AbortSignal} [signal] + * @returns {AsyncIterator} + */ +function createAsyncTransformPipeline(source, transforms, signal) { + let state = kStart; + let controller; + let abortHandler; + let completed = false; + let iterator; + let nextMethod; + const operations = createOperationQueue(); + + // What an async generator would do in its `finally` block. + function cleanup() { if (!completed && !controller.signal.aborted) { - // Consumer stopped early or generator return() was called. + // Consumer stopped early or return() was called. // If a transform listener throws here, let it propagate. controller.abort(lazyDOMException('Aborted', 'AbortError')); } @@ -824,6 +1106,150 @@ async function* createAsyncPipeline(source, transforms, signal) { signal.removeEventListener('abort', abortHandler); } } + + function finish(result) { + operations.settled(); + return result; + } + + function complete(result) { + state = kDone; + try { + cleanup(); + } finally { + operations.settled(); + } + return result; + } + + // What an async generator would do in its `catch` block, then `finally`. + function fail(error) { + state = kDone; + try { + try { + if (!controller.signal.aborted) { + abortSignal(controller.signal, error); + } + } finally { + cleanup(); + } + } finally { + operations.settled(); + } + throw error; + } + + function onResult(result) { + let value; + try { + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); + } + if (result.done) { + // A transform can abort while completing without producing a final + // batch, for example when an async flush resolves to null. In that + // case there is no batch with which to observe the abort. + controller.signal.throwIfAborted(); + completed = true; + return complete(new IterResult(true, undefined)); + } + value = result.value; + } catch (error) { + return fail(error); + } + try { + controller.signal.throwIfAborted(); + } catch (error) { + // Like for await when its body throws: close the transforms first. + return PromisePrototypeThen( + closeAsyncIterator(iterator, true), () => fail(error)); + } + return finish(new IterResult(false, value)); + } + + function pullTransforms() { + let promise; + try { + promise = PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)); + } catch (error) { + return fail(error); + } + return PromisePrototypeThen(promise, onResult, fail); + } + + function doNext() { + if (state === kDone) { + return PromiseResolve(finish(new IterResult(true, undefined))); + } + if (state === kStart) { + try { + // Check for abort + signal?.throwIfAborted(); + } catch (error) { + state = kDone; + operations.settled(); + return PromiseReject(error); + } + state = kActive; + const normalized = yieldAbortable(source, signal); + // Create internal controller for transform cancellation. + controller = new AbortController(); + if (signal) { + abortHandler = () => { + abortSignal(controller.signal, signal.reason); + }; + signal.addEventListener('abort', abortHandler, kNullOnceOption); + } + try { + const current = createAsyncTransformLayers( + normalized, transforms, controller.signal); + iterator = current[SymbolAsyncIterator](); + nextMethod = iterator.next; + } catch (error) { + try { + fail(error); + } catch (failure) { + return PromiseReject(failure); + } + } + } + try { + return PromiseResolve(pullTransforms()); + } catch (error) { + return PromiseReject(error); + } + } + + function doReturn(value) { + const result = new IterResult(true, value); + if (state !== kActive) { + state = kDone; + return PromiseResolve(finish(result)); + } + state = kDone; + return PromisePrototypeThen( + closeAsyncIterator(iterator, false), () => complete(result), fail); + } + + function doThrow(error) { + if (state !== kActive) { + state = kDone; + operations.settled(); + return PromiseReject(error); + } + state = kDone; + return PromisePrototypeThen( + closeAsyncIterator(iterator, true), () => fail(error)); + } + + return ObjectSetPrototypeOf({ + next() { return operations.run(doNext); }, + return(value) { return operations.run(doReturn, value); }, + throw(error) { return operations.run(doThrow, error); }, + [SymbolAsyncIterator]() { return this; }, + }, null); } // ============================================================================= diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index bc57920bcbc..86d9b11aab7 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -6,10 +6,12 @@ const { ArrayBufferPrototypeGetDetached, ArrayBufferPrototypeGetResizable, ArrayPrototypePush, + ArrayPrototypeShift, ArrayPrototypeSlice, FunctionPrototypeCall, ObjectDefineProperty, ObjectFreeze, + PromisePrototypeThen, PromiseResolve, PromiseWithResolvers, SafePromisePrototypeFinally, @@ -331,6 +333,58 @@ function callWithByteView(value, method, receiver) { return result; } +// States of the hand-written async iterators that replace async generators +// on hot paths. +const kStart = 0; +const kActive = 1; +const kDone = 2; + +/** + * Serializes the operations of a hand-written async iterator the way an + * async generator queues its requests: an operation started while another + * is in progress waits until it has settled. + * @returns {{ run: Function, settled: Function, release: Function }} + */ +function createOperationQueue() { + let busy = false; + let queue = null; + + function drain() { + if (busy || queue === null) return; + const { 0: method, 1: arg, 2: resolve, 3: reject } = + ArrayPrototypeShift(queue); + if (queue.length === 0) queue = null; + busy = true; + PromisePrototypeThen(method(arg), resolve, reject); + } + + return { + __proto__: null, + // Run `method(arg)`, which returns a promise and must call settled() + // once its result is known, now or after the operations before it. + run(method, arg) { + if (busy || queue !== null) { + const { promise, resolve, reject } = PromiseWithResolvers(); + queue ??= []; + ArrayPrototypePush(queue, [method, arg, resolve, reject]); + return promise; + } + busy = true; + return method(arg); + }, + settled() { + busy = false; + if (queue !== null) PromisePrototypeThen(kResolvedPromise, drain); + }, + // Like settled(), but start the next queued operation now, as an async + // generator does once the await of a yield or return completes. + release() { + busy = false; + drain(); + }, + }; +} + /** * Validate an explicit `budget` option. The spec only requires the default * budget to be at least 16384 bytes; any positive explicit budget is valid. @@ -632,6 +686,9 @@ function validateBackpressure(value) { module.exports = { IterResult, + kActive, + kDone, + kStart, PendingRequest, PendingWrite, kMultiConsumerDefaultBudget, @@ -640,6 +697,7 @@ module.exports = { kResolvedPromise, callWithByteView, concatBytes, + createOperationQueue, convertChunks, createBatchEntry, recordChunk, diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index 5bffd49d511..c971ab02ff1 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -7,6 +7,7 @@ const { broadcast, dump, from, + pipeTo, pull, push, share, @@ -611,6 +612,119 @@ async function testTransformOptionsShape() { // Run the uncaughtException test sequentially (it installs a global handler // that would interfere with concurrent tests). +// Transform pipelines read and close the source like an async generator +// looping over it with for await. The tests below check the parts of that +// behavior that are observable from the source and the transforms' signal. + +// Creates an async iterable source of batches of one-byte chunks with values +// `values`, recording calls in `log`. +function createLoggedSource(log, values, { failAt = -1 } = {}) { + let i = 0; + return { + [Symbol.asyncIterator]() { + return { + async next() { + log.push(`next ${i}`); + if (i === failAt) { + i++; + throw new Error('source failed'); + } + if (i >= values.length) return { done: true, value: undefined }; + return { done: false, value: [Uint8Array.of(values[i++])] }; + }, + async return() { + log.push('return'); + return { done: true, value: undefined }; + }, + }; + }, + }; +} + +// A transform recording its calls, and the abort of its signal, in `log`. +function createLoggedTransform(log, transform = (chunks) => chunks) { + let listening = false; + return (chunks, options) => { + log.push(chunks === null ? 'flush' : `transform ${chunks[0][0]}`); + if (!listening) { + listening = true; + options.signal.addEventListener('abort', () => { + log.push(`abort: ${options.signal.reason.message}`); + }); + } + return transform(chunks, options); + }; +} + +async function testTransformErrorClosesSource() { + const fail = (chunks) => { + if (chunks?.[0][0] === 2) throw new Error('transform failed'); + return chunks; + }; + for (const transform of [fail, async (chunks) => fail(chunks)]) { + const log = []; + await assert.rejects(async () => { + // eslint-disable-next-line no-unused-vars + for await (const _ of pull(createLoggedSource(log, [1, 2, 3]), + createLoggedTransform(log, transform))); + }, /transform failed/); + assert.deepStrictEqual(log, [ + 'next 0', 'transform 1', 'next 1', 'transform 2', 'return', + 'abort: transform failed', + ]); + } +} + +async function testTransformSourceErrorDoesNotCloseSource() { + const log = []; + await assert.rejects(async () => { + // eslint-disable-next-line no-unused-vars + for await (const _ of pull(createLoggedSource(log, [1], { failAt: 1 }), + createLoggedTransform(log))); + }, /source failed/); + assert.deepStrictEqual(log, [ + 'next 0', 'transform 1', 'next 1', 'abort: source failed', + ]); +} + +async function testPipeToTransformsStoppedEarly() { + // When the writer fails, the transforms are closed, closing the source, and + // their signal is aborted. + const log = []; + let writes = 0; + await assert.rejects(pipeTo( + createLoggedSource(log, [1, 2, 3]), createLoggedTransform(log), { + write() { + if (++writes === 2) throw new Error('write failed'); + }, + }), /write failed/); + assert.deepStrictEqual(log, [ + 'next 0', 'transform 1', 'next 1', 'transform 2', 'return', + 'abort: Aborted', + ]); +} + +async function testTransformReturnClosesOutputAndSource() { + // A transform output that is iterated asynchronously is closed, and then + // the source, when the pipeline is stopped while reading it. + const log = []; + const iterator = pull(createLoggedSource(log, [1, 2]), + createLoggedTransform(log, async function*() { + try { + yield Uint8Array.of(1); + yield Uint8Array.of(2); + } finally { + log.push('output closed'); + } + }))[Symbol.asyncIterator](); + assert.strictEqual((await iterator.next()).done, false); + assert.strictEqual((await iterator.return()).done, true); + assert.strictEqual((await iterator.next()).done, true); + assert.deepStrictEqual(log, [ + 'next 0', 'transform 1', 'output closed', 'abort: Aborted', 'return', + ]); +} + (async () => { await Promise.all([ testPullIdentity(), @@ -646,6 +760,10 @@ async function testTransformOptionsShape() { testPipeToStringSource(), testTransformOptionsNotShared(), testTransformOptionsShape(), + testTransformErrorClosesSource(), + testTransformSourceErrorDoesNotCloseSource(), + testPipeToTransformsStoppedEarly(), + testTransformReturnClosesOutputAndSource(), ]); // Run after all concurrent tests complete to avoid global handler races await testTransformSignalListenerErrorOnSourceError(); From 406ac2f83675176dc3a33f2de8779f27dac446e7 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 04:22:36 +0000 Subject: [PATCH 17/58] stream: read stream/iter sources with one abort listener To read a source until a signal aborts, yieldAbortable() used an async generator calling abortableNext() for every value, which added an abort listener, raced the value against the abort with SafePromiseRace() and removed the listener again with SafePromisePrototypeFinally(). This made pull(), which always reads through its own signal, and pipeTo() and the consumers with a signal about four times slower than without one. Write the generator out by hand with a single abort listener for the whole iteration, held weakly so that the signal does not keep the iterator alive, and wait for each value with a single promise that an abort rejects. Aborts before, while and after reading a value, closing the source on errors and aborts, and return() and throw() behave as before. With an async generator yielding 16-byte chunks, one per batch, pull() is about 3.6x faster with or without a transform, pipeTo() with a signal and a transform about 3.4x, and bytes() and array() with a signal about 4.4x; pull() allocates about 7 KB less per chunk. New tests cover an abort while the source is producing a value and the removal of the listener. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/utils.js | 254 +++++++++++++----- .../test-stream-iter-pipeto-signal.js | 56 +++- 2 files changed, 249 insertions(+), 61 deletions(-) diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 86d9b11aab7..01c21fd946d 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -11,11 +11,11 @@ const { FunctionPrototypeCall, ObjectDefineProperty, ObjectFreeze, + ObjectSetPrototypeOf, PromisePrototypeThen, + PromiseReject, PromiseResolve, PromiseWithResolvers, - SafePromisePrototypeFinally, - SafePromiseRace, SafeWeakSet, SymbolAsyncIterator, TypedArrayPrototypeGetBuffer, @@ -38,6 +38,7 @@ const { } = require('internal/errors'); const { isSharedArrayBuffer, isUint8Array } = require('internal/util/types'); +const { kWeakHandler } = require('internal/event_target'); const { validateInteger, @@ -110,36 +111,6 @@ function onSignalAbort(signal, handler) { } } -function getOnAbort(reject, signal) { - return () => reject(signal.reason); -} - -/** - * Read one item from an async iterator, rejecting early if the signal aborts. - * @param {AsyncIterator} iterator - The iterator to read from. - * @param {AbortSignal|undefined} signal - Optional abort signal. - * @returns {Promise>|IteratorResult} - */ -function abortableNext(iterator, signal) { - if (signal === undefined) { - return iterator.next(); - } - - signal.throwIfAborted(); - - const next = iterator.next(); - const { promise, reject } = PromiseWithResolvers(); - const onAbort = getOnAbort(reject, signal); - signal.addEventListener('abort', onAbort, kNullOnceOption); - if (signal.aborted) { - onAbort(); - } - - return SafePromisePrototypeFinally(SafePromiseRace([next, promise]), () => { - signal.removeEventListener('abort', onAbort); - }); -} - /** * Wrap an async source so each pending read is abort-aware. * @param {AsyncIterable} source - The source to read from. @@ -153,39 +124,202 @@ function yieldAbortable(source, signal) { return { __proto__: null, - async *[SymbolAsyncIterator]() { - const iterator = source[SymbolAsyncIterator](); - let completed = false; - let aborted = false; + [SymbolAsyncIterator]() { + return createAbortableIterator(source, signal); + }, + }; +} +/** + * Iterator reading `source` until `signal` aborts, for yieldAbortable(). + * + * This is what an async generator looping over the source with + * abortableNext() would do, written out by hand: that costs an abort + * listener, a promise race and an async generator layer for every value. + * Instead, a single abort listener is added for the whole iteration, held + * weakly so that the signal does not keep the iterator alive, and each + * next() waits with a single promise that an abort rejects. As with the + * generator: + * - the source is opened by the first next(), and calls made while one is in + * progress are queued; + * - an abort before or while reading a value, or once it has been read, + * rejects with the abort reason; + * - unless the source has ended, an error (including an abort) closes it: + * if the signal is aborted, without waiting, otherwise waiting and + * rejecting with an error from closing it instead; + * - return() closes the source, propagating errors from closing it. + * @param {AsyncIterable} source + * @param {AbortSignal} signal + * @returns {AsyncIterator} + */ +function createAbortableIterator(source, signal) { + let state = kStart; + let iterator; + let completed = false; + // The settling functions of the pending next(), if any. + let resolveNext = null; + let rejectNext = null; + const operations = createOperationQueue(); + + const self = ObjectSetPrototypeOf({ + next() { return operations.run(doNext); }, + return(value) { return operations.run(doReturn, value); }, + throw(error) { return operations.run(doThrow, error); }, + [SymbolAsyncIterator]() { return this; }, + }, null); + + function onAbort() { + if (rejectNext === null) return; + const reject = rejectNext; + resolveNext = rejectNext = null; + fail(reject, signal.reason); + } + + // What the generator did in its `catch` and `finally` blocks for `error`. + function fail(reject, error) { + state = kDone; + signal.removeEventListener('abort', onAbort); + const aborted = signal.aborted; + if (!completed && typeof iterator.return === 'function') { + let closed; try { - while (true) { - const { done, value } = await abortableNext(iterator, signal); - if (done) { - completed = true; - return; - } - signal.throwIfAborted(); - yield value; + const result = iterator.return(); + if (aborted) { + // PromiseResolve(result) can reject if result is a thenable that + // rejects, so mark it as handled even though the abort takes + // precedence over the result of iterator.return(). + markPromiseAsHandled(PromiseResolve(result)); + } else { + closed = PromiseResolve(result); } + } catch (closeError) { + operations.settled(); + reject(closeError); + return; + } + if (closed !== undefined) { + PromisePrototypeThen(closed, () => { + operations.settled(); + reject(error); + }, (closeError) => { + operations.settled(); + reject(closeError); + }); + return; + } + } + operations.settled(); + reject(error); + } + + function onResult(result) { + if (resolveNext === null) return; // Settled by an abort. + const resolve = resolveNext; + const reject = rejectNext; + resolveNext = rejectNext = null; + let done; + let value; + try { + ({ done, value } = result); + if (!done) signal.throwIfAborted(); + } catch (error) { + fail(reject, error); + return; + } + if (done) { + completed = true; + state = kDone; + signal.removeEventListener('abort', onAbort); + operations.settled(); + resolve(new IterResult(true, undefined)); + return; + } + operations.settled(); + resolve(new IterResult(false, value)); + } + + function onError(error) { + if (rejectNext === null) return; // Settled by an abort. + const reject = rejectNext; + resolveNext = rejectNext = null; + fail(reject, error); + } + + function doNext() { + if (state === kDone) { + operations.settled(); + return PromiseResolve(new IterResult(true, undefined)); + } + if (state === kStart) { + try { + iterator = source[SymbolAsyncIterator](); } catch (error) { - aborted = signal.aborted; - throw error; - } finally { - if (!completed && typeof iterator.return === 'function') { - const result = iterator.return(); - if (aborted) { - // PromiseResolve(result) can reject if result is a thenable that - // rejects, so mark it as handled even though the abort takes - // precedence over the result of iterator.return(). - markPromiseAsHandled(PromiseResolve(result)); - } else { - await result; - } - } + state = kDone; + operations.settled(); + return PromiseReject(error); } - }, - }; + state = kActive; + signal.addEventListener('abort', onAbort, + { __proto__: null, [kWeakHandler]: self }); + } + const { promise, resolve, reject } = PromiseWithResolvers(); + let next; + try { + signal.throwIfAborted(); + next = PromiseResolve(iterator.next()); + } catch (error) { + fail(reject, error); + return promise; + } + resolveNext = resolve; + rejectNext = reject; + PromisePrototypeThen(next, onResult, onError); + // The signal can abort while iterator.next() runs. + if (signal.aborted) onAbort(); + return promise; + } + + function doReturn(value) { + const result = new IterResult(true, value); + if (state !== kActive) { + state = kDone; + operations.settled(); + return PromiseResolve(result); + } + state = kDone; + signal.removeEventListener('abort', onAbort); + if (typeof iterator.return !== 'function') { + operations.settled(); + return PromiseResolve(result); + } + let closed; + try { + closed = PromiseResolve(iterator.return()); + } catch (error) { + operations.settled(); + return PromiseReject(error); + } + return PromisePrototypeThen(closed, () => { + operations.settled(); + return result; + }, (error) => { + operations.settled(); + throw error; + }); + } + + function doThrow(error) { + if (state !== kActive) { + state = kDone; + operations.settled(); + return PromiseReject(error); + } + const { promise, reject } = PromiseWithResolvers(); + fail(reject, error); + return promise; + } + + return self; } /** diff --git a/test/parallel/test-stream-iter-pipeto-signal.js b/test/parallel/test-stream-iter-pipeto-signal.js index be04c8c635d..3645db1a92f 100644 --- a/test/parallel/test-stream-iter-pipeto-signal.js +++ b/test/parallel/test-stream-iter-pipeto-signal.js @@ -7,7 +7,8 @@ const common = require('../common'); const assert = require('assert'); const { setTimeout } = require('timers/promises'); -const { pipeTo, from } = require('stream/iter'); +const { getEventListeners } = require('events'); +const { bytes, pipeTo, from } = require('stream/iter'); async function testPipeToPreAbortedSignalFailsWriter() { const reason = new Error('already aborted'); @@ -160,6 +161,57 @@ async function testPipeToLiveSignalWithTransformsCompletes() { assert.ok(written.length > 0); } +async function testSignalAbortedWhileReadingSource() { + // The signal can abort while the source is producing a value; the read + // must still reject with the abort reason, and the source be closed, even + // if the value never comes. + for (const consume of [ + (source, signal) => pipeTo(source, { write() {} }, { signal }), + (source, signal) => bytes(source, { signal }), + ]) { + const ac = new AbortController(); + const reason = new Error('aborted while reading'); + let closed = false; + const source = { + [Symbol.asyncIterator]() { + return { + next() { + ac.abort(reason); + return new Promise(() => {}); + }, + async return() { + closed = true; + return { done: true }; + }, + }; + }, + }; + await assert.rejects(consume(source, ac.signal), reason); + assert.strictEqual(closed, true); + } +} + +async function testSignalListenersRemoved() { + // No abort listener is left on the signal once reading completes, fails + // or is aborted. + const ac = new AbortController(); + await pipeTo(from('abc'), { write() {} }, { signal: ac.signal }); + await bytes(from('abc'), { signal: ac.signal }); + await assert.rejects(bytes((async function*() { + yield 'a'; + throw new Error('source failed'); + })(), { signal: ac.signal }), /source failed/); + assert.strictEqual(getEventListeners(ac.signal, 'abort').length, 0); + + const aborting = new AbortController(); + await assert.rejects(bytes((async function*() { + yield 'a'; + aborting.abort(); + yield 'b'; + })(), { signal: aborting.signal }), { name: 'AbortError' }); + assert.strictEqual(getEventListeners(aborting.signal, 'abort').length, 0); +} + Promise.all([ testPipeToPreAbortedSignalFailsWriter(), testPipeToPreAbortedSignalPreventFail(), @@ -168,4 +220,6 @@ Promise.all([ testPipeToLiveSignalWithTransforms(), testPipeToLiveSignalCompletes(), testPipeToLiveSignalWithTransformsCompletes(), + testSignalAbortedWhileReadingSource(), + testSignalListenersRemoved(), ]).then(common.mustCall()); From 978ca82b80c3ae3f1f016eef24a3f864ab2ef45a Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 05:30:30 +0000 Subject: [PATCH 18/58] stream: remove async generator layers from stream/iter pull() pull() read its pipeline through an async generator delegating to it with yield*, and without transforms the pipeline was another such generator around the source, each costing several promises per batch. The pipelines already start lazily and queue calls as an async generator does, so use the pipeline directly, and write the pipeline without transforms out by hand: it checks the signal on the first next(), then passes every call to the iterator reading the source. Results are now IterResult objects, like those of the other stream/iter iterators; one test compared them as ordinary objects. With an async generator yielding 16-byte chunks, one per batch, pull() is about 42% faster without transforms and 30% faster with a signal, and about 12% faster through a transform. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/pull.js | 69 +++++++++++++++---- .../test-stream-iter-iterator-result.js | 8 ++- test/parallel/test-stream-iter-pull-async.js | 2 +- 3 files changed, 64 insertions(+), 15 deletions(-) diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index e300364a820..b59949ff1f3 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -1011,10 +1011,60 @@ function createAsyncPipeline(source, transforms, signal) { return createAsyncTransformPipeline(source, transforms, signal); } -async function* createAsyncPipelineWithoutTransforms(source, signal) { - // Check for abort - signal?.throwIfAborted(); - yield* yieldAbortable(source, signal); +/** + * The pipeline without transforms: what an async generator doing + * `signal?.throwIfAborted(); yield* yieldAbortable(source, signal);` would + * do. With a signal, it is written out by hand to avoid an async generator + * layer for every batch: the signal is checked by the first next(), then + * every call is passed to the iterator of yieldAbortable(), which queues + * them and completes as the generator would. + * @param {AsyncIterable} source + * @param {AbortSignal} [signal] + * @returns {AsyncIterator} + */ +function createAsyncPipelineWithoutTransforms(source, signal) { + if (signal === undefined) return yieldFrom(source); + let state = kStart; + let iterator; + return ObjectSetPrototypeOf({ + next() { + if (state === kStart) { + try { + // Check for abort + signal.throwIfAborted(); + iterator = yieldAbortable(source, signal)[SymbolAsyncIterator](); + } catch (error) { + state = kDone; + return PromiseReject(error); + } + state = kActive; + } else if (state === kDone) { + return PromiseResolve(new IterResult(true, undefined)); + } + return iterator.next(); + }, + return(value) { + if (state !== kActive) { + state = kDone; + return PromiseResolve(new IterResult(true, value)); + } + return iterator.return(value); + }, + throw(error) { + if (state !== kActive) { + state = kDone; + return PromiseReject(error); + } + return iterator.throw(error); + }, + [SymbolAsyncIterator]() { + return this; + }, + }, null); +} + +async function* yieldFrom(source) { + yield* source; } /** @@ -1303,10 +1353,8 @@ function pull(source, ...args) { [SymbolAsyncIterator]() { if (signal === undefined) { const controller = new AbortController(); - async function* pipeline() { - yield* createAsyncPipeline(normalized, transforms, controller.signal); - } - const iterator = pipeline(); + const iterator = createAsyncPipeline( + normalized, transforms, controller.signal); return ObjectSetPrototypeOf({ next(value) { return iterator.next(value); @@ -1341,10 +1389,7 @@ function createAbortablePullIterator(source, transforms, signal) { if (!aborted) { controller = new AbortController(); const iteratorSignal = AbortSignal.any([signal, controller.signal]); - async function* pipeline() { - yield* createAsyncPipeline(source, transforms, iteratorSignal); - } - iterator = pipeline(); + iterator = createAsyncPipeline(source, transforms, iteratorSignal); } function onRejected(error) { diff --git a/test/parallel/test-stream-iter-iterator-result.js b/test/parallel/test-stream-iter-iterator-result.js index ba4c35686de..96cbdc6ec18 100644 --- a/test/parallel/test-stream-iter-iterator-result.js +++ b/test/parallel/test-stream-iter-iterator-result.js @@ -3,8 +3,6 @@ // Iterator results created by the stream/iter iterators themselves do not // inherit from Object.prototype, so prototype pollution cannot affect them. -// (Iterators implemented as async generators, such as the one returned by -// pull(), return ordinary iterator results created by the engine.) const common = require('../common'); const assert = require('assert'); @@ -12,6 +10,7 @@ const { inspect } = require('util'); const { broadcast, from, + pull, push, share, shareSync, @@ -33,6 +32,11 @@ async function testAsyncIterators() { return readable; }, 'share()': () => share(from('a')).pull(), + 'pull()': () => pull(from('a')), + 'pull() with a signal': () => pull(from('a'), { + signal: new AbortController().signal, + }), + 'pull() with a transform': () => pull(from('a'), (chunks) => chunks), 'from() of an async iterable': () => from((async function*() { yield 'a'; })()), diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index c971ab02ff1..308e3a0ebb9 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -314,7 +314,7 @@ async function testPullReturnWhileSourceNextPending() { ]); assert.notStrictEqual(result, timeout); - assert.deepStrictEqual(result, { value: undefined, done: true }); + assert.deepStrictEqual({ ...result }, { value: undefined, done: true }); await next; } From 15fba7684dba901c5e3050c6720b5e44c2c68fd5 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 05:57:01 +0000 Subject: [PATCH 19/58] stream: avoid quadratic batching in zlib/iter transforms When the output collected by a zlib/iter transform exceeded a batch, drainBatch() took buffers off the front of the pending array with ArrayPrototypeShift(), which copies the rest of the array. The sync transforms collect all output for an input chunk before draining it, so a small input that decompresses to many buffers took quadratic time: with a chunkSize of 1024, decompressing 128 MiB took 5.6 seconds, growing four times when the output doubles. Take each batch from the front with a single slice, advancing an index, and clear the slots taken so that the buffers can be collected. The batches are unchanged; decompressing 128 MiB as above takes 151 ms. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/transform.js | 83 +++++++++++++------ .../test-stream-iter-transform-buffering.js | 38 ++++++++- 2 files changed, 94 insertions(+), 27 deletions(-) diff --git a/lib/internal/streams/iter/transform.js b/lib/internal/streams/iter/transform.js index c6c315cf6bd..8e60589aba3 100644 --- a/lib/internal/streams/iter/transform.js +++ b/lib/internal/streams/iter/transform.js @@ -11,7 +11,7 @@ const { ArrayPrototypeMap, ArrayPrototypePush, - ArrayPrototypeShift, + ArrayPrototypeSlice, FunctionPrototypeCall, MathMax, NumberIsNaN, @@ -311,6 +311,8 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { let outOffset = 0; let chunkSize; let pending = []; + // Index of the first buffer of `pending` not yet taken by drainBatch(). + let pendingHead = 0; let pendingBytes = 0; // Current write operation state (read by the callback for looping). @@ -442,20 +444,33 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { function drainBatch() { if (pendingBytes <= BATCH_HWM) { - // Swap instead of splice - avoids copying the array. - const batch = pending; + // Take everything: swap instead of copying, unless batches have + // already been taken from the front. + const batch = pendingHead === 0 ? pending : + ArrayPrototypeSlice(pending, pendingHead); pending = []; + pendingHead = 0; pendingBytes = 0; return batch; } - const batch = []; + // Take up to BATCH_HWM bytes from the front by advancing pendingHead: + // shifting them off instead would copy the rest of the array, which + // makes draining a large output quadratic. + let end = pendingHead; let batchBytes = 0; - while (pending.length > 0 && batchBytes < BATCH_HWM) { - const buf = ArrayPrototypeShift(pending); - ArrayPrototypePush(batch, buf); - const len = TypedArrayPrototypeGetByteLength(buf); - batchBytes += len; - pendingBytes -= len; + while (end < pending.length && batchBytes < BATCH_HWM) { + batchBytes += TypedArrayPrototypeGetByteLength(pending[end]); + end++; + } + const batch = ArrayPrototypeSlice(pending, pendingHead, end); + pendingBytes -= batchBytes; + if (end === pending.length) { + pending = []; + pendingHead = 0; + } else { + // Do not keep the buffers taken alive. + for (let i = pendingHead; i < end; i++) pending[i] = undefined; + pendingHead = end; } return batch; } @@ -511,7 +526,7 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { processInputBatches(kEmpty, finishFlag)) { yield batch; } - while (pending.length > 0) { + while (pending.length > pendingHead) { yield drainBatch(); } } @@ -531,11 +546,12 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { } if (pendingBytes >= BATCH_HWM) { - while (pending.length > 0 && pendingBytes >= BATCH_HWM) { + while (pending.length > pendingHead && + pendingBytes >= BATCH_HWM) { yield drainBatch(); } } - if (pending.length > 0) { + if (pending.length > pendingHead) { yield drainBatch(); } } @@ -547,7 +563,7 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { processInputBatches(kEmpty, finishFlag)) { yield batch; } - while (pending.length > 0) { + while (pending.length > pendingHead) { yield drainBatch(); } } @@ -602,6 +618,8 @@ function makeZlibTransformSync(createHandleFn, processFlag, finishFlag) { let outBuf = Buffer.allocUnsafe(chunkSize); let outOffset = 0; let pending = []; + // Index of the first buffer of `pending` not yet taken by drainBatch(). + let pendingHead = 0; let pendingBytes = 0; function processSyncInput(input, flushFlag) { @@ -666,19 +684,33 @@ function makeZlibTransformSync(createHandleFn, processFlag, finishFlag) { function drainBatch() { if (pendingBytes <= BATCH_HWM) { - const batch = pending; + // Take everything: swap instead of copying, unless batches have + // already been taken from the front. + const batch = pendingHead === 0 ? pending : + ArrayPrototypeSlice(pending, pendingHead); pending = []; + pendingHead = 0; pendingBytes = 0; return batch; } - const batch = []; + // Take up to BATCH_HWM bytes from the front by advancing pendingHead: + // shifting them off instead would copy the rest of the array, which + // makes draining a large output quadratic. + let end = pendingHead; let batchBytes = 0; - while (pending.length > 0 && batchBytes < BATCH_HWM) { - const buf = ArrayPrototypeShift(pending); - const len = TypedArrayPrototypeGetByteLength(buf); - ArrayPrototypePush(batch, buf); - batchBytes += len; - pendingBytes -= len; + while (end < pending.length && batchBytes < BATCH_HWM) { + batchBytes += TypedArrayPrototypeGetByteLength(pending[end]); + end++; + } + const batch = ArrayPrototypeSlice(pending, pendingHead, end); + pendingBytes -= batchBytes; + if (end === pending.length) { + pending = []; + pendingHead = 0; + } else { + // Do not keep the buffers taken alive. + for (let i = pendingHead; i < end; i++) pending[i] = undefined; + pendingHead = end; } return batch; } @@ -688,7 +720,7 @@ function makeZlibTransformSync(createHandleFn, processFlag, finishFlag) { if (batch === null) { // Flush signal - finalize the engine. processSyncInput(Buffer.alloc(0), finishFlag); - while (pending.length > 0) { + while (pending.length > pendingHead) { yield drainBatch(); } continue; @@ -699,11 +731,12 @@ function makeZlibTransformSync(createHandleFn, processFlag, finishFlag) { } if (pendingBytes >= BATCH_HWM) { - while (pending.length > 0 && pendingBytes >= BATCH_HWM) { + while (pending.length > pendingHead && + pendingBytes >= BATCH_HWM) { yield drainBatch(); } } - if (pending.length > 0) { + if (pending.length > pendingHead) { yield drainBatch(); } } diff --git a/test/parallel/test-stream-iter-transform-buffering.js b/test/parallel/test-stream-iter-transform-buffering.js index b1b813f60b9..95b9a71f65c 100644 --- a/test/parallel/test-stream-iter-transform-buffering.js +++ b/test/parallel/test-stream-iter-transform-buffering.js @@ -4,8 +4,13 @@ const common = require('../common'); const assert = require('assert'); const { brotliDecompressSync, gunzipSync, gzipSync } = require('zlib'); -const { compressBrotli, compressGzip, decompressGzip } = require('zlib/iter'); -const { bytes, from, pull } = require('stream/iter'); +const { + compressBrotli, + compressGzip, + decompressGzip, + decompressGzipSync, +} = require('zlib/iter'); +const { bytes, from, fromSync, pull, pullSync } = require('stream/iter'); async function testDecompressionOutputIsBounded() { let input = Buffer.alloc(32 * 1024 * 1024, 0x61); @@ -38,6 +43,34 @@ async function testSmallChunkSizeDecompression() { assert.deepStrictEqual(Buffer.from(output), input); } +// With a chunk size that does not divide the batch size, output collected +// beyond a batch is taken a batch at a time, from the front. +async function testPartialBatchDrains() { + const input = Buffer.alloc(1024 * 1024); + for (let i = 0; i < input.length; i++) input[i] = (i * 7 + (i >> 12)) & 0xff; + const compressed = gzipSync(input); + const batchSizes = (batches) => batches.map((batch) => { + return batch.reduce((total, chunk) => total + chunk.length, 0); + }); + + const asyncBatches = []; + for await (const batch of pull(from(compressed), + decompressGzip({ chunkSize: 1000 }))) { + asyncBatches.push(batch); + } + const syncBatches = [ + ...pullSync(fromSync(compressed), decompressGzipSync({ chunkSize: 1000 })), + ]; + for (const batches of [asyncBatches, syncBatches]) { + assert.deepStrictEqual(Buffer.concat(batches.flat()), input); + assert.ok(batches.length > 1); + // A batch is closed once it holds 64 KiB. + for (const size of batchSizes(batches)) { + assert.ok(size <= 64 * 1024 + 1000, `batch of ${size} bytes`); + } + } +} + // Deterministic incompressible data. Brotli's buffering decisions depend on // content, so random data would make the tests below flaky. function incompressible(size) { @@ -108,6 +141,7 @@ async function testSourceReturnFailuresAreIgnored() { await testDecompressionOutputIsBounded(); await Promise.all([ testSmallChunkSizeDecompression(), + testPartialBatchDrains(), testLargeFinishOutput(), testExplicitFlushSignal(), testSourceReturnFailuresAreIgnored(), From a9fc11f57d10ad043ee95f9eb98e3d5718cf0459 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 05:59:33 +0000 Subject: [PATCH 20/58] stream: use a RingBuffer for stream/iter merge()'s ready queue merge() queued every batch from its sources in an array, taking each off the front with ArrayPrototypeShift(), which copies the rest of the queue: up to one entry per source. Use a RingBuffer. Merging async generators yielding 16-byte chunks, one per batch, is about 6% faster with 2 sources, 9% with 8 and 36% with 64. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/consumers.js | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index 0751d2cc95f..459eedb49bb 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -14,7 +14,6 @@ const { ArrayBufferPrototypeSlice, ArrayPrototypeMap, ArrayPrototypePush, - ArrayPrototypeShift, ArrayPrototypeSlice, FunctionPrototypeCall, ObjectFreeze, @@ -38,6 +37,7 @@ const { }, } = require('internal/errors'); const { TextDecoder } = require('internal/encoding'); +const { RingBuffer } = require('internal/streams/iter/ringbuffer'); const { validateFunction, } = require('internal/validators'); @@ -515,7 +515,7 @@ function merge(...args) { // between consumer pulls are drained synchronously without an extra // async tick per batch. Each source has at most one pending .next() // at a time. Every batch from every source is preserved. - const ready = []; + const ready = new RingBuffer(); const pendingPulls = new SafeSet(); let activeCount = normalized.length; let waitResolve = null; @@ -540,8 +540,7 @@ function merge(...args) { if (result.done) { activeCount--; } else { - ArrayPrototypePush(ready, - new MergeEntry(iterator, result.value, undefined)); + ready.push(new MergeEntry(iterator, result.value, undefined)); } if (waitResolve) { waitResolve(); @@ -552,7 +551,7 @@ function merge(...args) { const onRejected = (iterator, reason) => { pendingPulls.delete(iterator); if (stopped) return; - ArrayPrototypePush(ready, new MergeEntry(undefined, undefined, reason)); + ready.push(new MergeEntry(undefined, undefined, reason)); if (waitResolve) { waitResolve(); waitResolve = null; @@ -579,7 +578,7 @@ function merge(...args) { // Drain ready queue synchronously while (ready.length > 0) { - const item = ArrayPrototypeShift(ready); + const item = ready.shift(); if (item.iterator === undefined) { throw item.reason; } From 883571e38c53586f119c25c4fc89ec4630e3cbae Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 06:03:17 +0000 Subject: [PATCH 21/58] stream: use a RingBuffer for stream/iter broadcast pending reads Reads requested from a broadcast() consumer while another one is pending were queued in an array and taken off the front with ArrayPrototypeShift(), which copies the rest of the queue. Settling many of them was quadratic: ending the writer with 160,000 reads pending took 22 seconds. Use a RingBuffer, starting small since the queue is rarely used. The same case takes 56 ms. A new test covers the order in which several pending reads are settled. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/broadcast.js | 14 +++++++------- .../test-stream-iter-broadcast-basic.js | 19 +++++++++++++++++++ 2 files changed, 26 insertions(+), 7 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index 00ed899d653..5dfb297a5fd 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -9,7 +9,6 @@ const { ArrayIsArray, ArrayPrototypePush, - ArrayPrototypeShift, FunctionPrototypeCall, ObjectSetPrototypeOf, PromisePrototypeThen, @@ -194,7 +193,9 @@ class BroadcastImpl { cursor: this.#bufferStart, resolve: null, reject: null, - pending: [], + // Reads requested while another one is pending. Rarely used, so it + // starts small. + pending: new RingBuffer(1), detached: false, error: kNoBroadcastError, }, null); @@ -266,8 +267,7 @@ class BroadcastImpl { if (state.resolve) { const { promise, resolve, reject } = PromiseWithResolvers(); - ArrayPrototypePush(state.pending, - new PendingRequest(resolve, reject)); + state.pending.push(new PendingRequest(resolve, reject)); return promise; } @@ -563,7 +563,7 @@ class BroadcastImpl { } #promotePending(consumer) { - const next = ArrayPrototypeShift(consumer.pending); + const next = consumer.pending.shift(); if (next === undefined) return false; consumer.resolve = next.resolve; consumer.reject = next.reject; @@ -576,14 +576,14 @@ class BroadcastImpl { consumer.reject = null; } while (consumer.pending.length > 0) { - ArrayPrototypeShift(consumer.pending).resolve( + consumer.pending.shift().resolve( new IterResult(true, undefined)); } } #rejectPending(consumer, reason) { while (consumer.pending.length > 0) { - ArrayPrototypeShift(consumer.pending).reject(reason); + consumer.pending.shift().reject(reason); } } } diff --git a/test/parallel/test-stream-iter-broadcast-basic.js b/test/parallel/test-stream-iter-broadcast-basic.js index c59689c2d12..7b2187f94e3 100644 --- a/test/parallel/test-stream-iter-broadcast-basic.js +++ b/test/parallel/test-stream-iter-broadcast-basic.js @@ -392,8 +392,27 @@ async function testOverlappingNextKeepsEarlierRead() { assert.strictEqual(bc.consumerCount, 0); } +async function testOverlappingNextResolvedInOrder() { + const { writer, broadcast: bc } = broadcast(); + const it = bc.push()[Symbol.asyncIterator](); + const reads = [it.next(), it.next(), it.next(), it.next(), it.next()]; + + for (const value of ['a', 'b', 'c']) await writer.write(value); + const error = new Error('failed'); + writer.fail(error); + + for (const [i, value] of ['a', 'b', 'c'].entries()) { + const result = await reads[i]; + assert.strictEqual(result.done, false); + assert.strictEqual(Buffer.concat(result.value).toString(), value); + } + await assert.rejects(reads[3], error); + await assert.rejects(reads[4], error); +} + Promise.all([ testBasicBroadcast(), + testOverlappingNextResolvedInOrder(), testMultipleWrites(), testConsumerCount(), testWriteSync(), From ee5aa7fb7ab3d5ab9c0a013518e2d82acecb90b4 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 06:05:28 +0000 Subject: [PATCH 22/58] stream: use a RingBuffer for the stream/iter operation queue The hand-written stream/iter iterators queue calls made while another one is in progress, as async generators do. They were queued in an array taken off the front with ArrayPrototypeShift(), which copies the rest of the queue, so calling next() many times without waiting was quadratic: 160,000 concurrent next() calls on a from() iterator took 7 seconds. Use a RingBuffer, kept once created, of QueuedOperation objects. The same case takes 330 ms. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/utils.js | 29 ++++++++++++++++++++--------- 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 01c21fd946d..2d85d888ddf 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -6,7 +6,6 @@ const { ArrayBufferPrototypeGetDetached, ArrayBufferPrototypeGetResizable, ArrayPrototypePush, - ArrayPrototypeShift, ArrayPrototypeSlice, FunctionPrototypeCall, ObjectDefineProperty, @@ -39,6 +38,7 @@ const { const { isSharedArrayBuffer, isUint8Array } = require('internal/util/types'); const { kWeakHandler } = require('internal/event_target'); +const { RingBuffer } = require('internal/streams/iter/ringbuffer'); const { validateInteger, @@ -473,6 +473,15 @@ const kStart = 0; const kActive = 1; const kDone = 2; +class QueuedOperation { + constructor(method, arg, resolve, reject) { + this.method = method; + this.arg = arg; + this.resolve = resolve; + this.reject = reject; + } +} + /** * Serializes the operations of a hand-written async iterator the way an * async generator queues its requests: an operation started while another @@ -481,13 +490,13 @@ const kDone = 2; */ function createOperationQueue() { let busy = false; + // Created on first use, then kept: queued operations, a RingBuffer of + // QueuedOperation, since as many can be queued as calls are made. let queue = null; function drain() { - if (busy || queue === null) return; - const { 0: method, 1: arg, 2: resolve, 3: reject } = - ArrayPrototypeShift(queue); - if (queue.length === 0) queue = null; + if (busy || queue === null || queue.length === 0) return; + const { method, arg, resolve, reject } = queue.shift(); busy = true; PromisePrototypeThen(method(arg), resolve, reject); } @@ -497,10 +506,10 @@ function createOperationQueue() { // Run `method(arg)`, which returns a promise and must call settled() // once its result is known, now or after the operations before it. run(method, arg) { - if (busy || queue !== null) { + if (busy || (queue !== null && queue.length !== 0)) { const { promise, resolve, reject } = PromiseWithResolvers(); - queue ??= []; - ArrayPrototypePush(queue, [method, arg, resolve, reject]); + queue ??= new RingBuffer(); + queue.push(new QueuedOperation(method, arg, resolve, reject)); return promise; } busy = true; @@ -508,7 +517,9 @@ function createOperationQueue() { }, settled() { busy = false; - if (queue !== null) PromisePrototypeThen(kResolvedPromise, drain); + if (queue !== null && queue.length !== 0) { + PromisePrototypeThen(kResolvedPromise, drain); + } }, // Like settled(), but start the next queued operation now, as an async // generator does once the await of a yield or return completes. From d510b07b3e21d680dc7660450f8fb1bb499d2f19 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 06:09:36 +0000 Subject: [PATCH 23/58] stream: append to arrays with indexed stores in stream/iter hot paths ArrayPrototypePush() is listed among the primordials with known performance issues. Append with an indexed store instead where stream/iter collects chunks or batches: batching sources in from() and fromSync(), flattening transform output in pull() and pullSync(), the batches of push() and of Readable sources, the output of the zlib/iter transforms and the chunks collected by bytes() and similar consumers. Like ArrayPrototypePush(), an indexed store is unaffected by changes to Array.prototype.push and runs setters defined for indices on Array.prototype. bytes() is about 13% faster, pullSync() through a generator transform and push() about 6%, and the other paths up to 4%. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/classic.js | 6 ++-- lib/internal/streams/iter/from.js | 9 +++--- lib/internal/streams/iter/pull.js | 38 +++++++++++++------------- lib/internal/streams/iter/push.js | 2 +- lib/internal/streams/iter/transform.js | 25 +++++++---------- lib/internal/streams/iter/utils.js | 13 ++++----- 6 files changed, 43 insertions(+), 50 deletions(-) diff --git a/lib/internal/streams/iter/classic.js b/lib/internal/streams/iter/classic.js index c1ed8ad51f5..2b39bc92787 100644 --- a/lib/internal/streams/iter/classic.js +++ b/lib/internal/streams/iter/classic.js @@ -160,14 +160,14 @@ async function normalizeBatch(raw) { for (let i = 0; i < raw.length; i++) { const value = raw[i]; if (isUint8Array(value)) { - ArrayPrototypePush(batch, value); + batch[batch.length] = value; } else { // normalizeAsyncValue may await for async protocols (e.g. // toAsyncStreamable on yielded objects). Stream events during // the suspension are queued, not lost -- errors will surface // on the next loop iteration after this yield completes. for await (const normalized of normalizeAsyncValue(value)) { - ArrayPrototypePush(batch, normalized); + batch[batch.length] = normalized; } } } @@ -229,7 +229,7 @@ function createBatchedAsyncIterator(stream, normalize) { stream._readableState?.length > 0) { const c = stream.read(); if (c === null) break; - ArrayPrototypePush(batch, c); + batch[batch.length] = c; } if (normalize !== null) { const result = await normalize(batch); diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 47b71e09a49..c249b0e86f1 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -9,7 +9,6 @@ const { ArrayBufferIsView, ArrayIsArray, ArrayPrototypeEvery, - ArrayPrototypePush, ArrayPrototypeSlice, DataViewPrototypeGetBuffer, DataViewPrototypeGetByteLength, @@ -347,7 +346,7 @@ function* normalizeSyncSource(source) { } // Fast path 2: value is a single Uint8Array (very common) if (isUint8Array(value)) { - ArrayPrototypePush(batch, value); + batch[batch.length] = value; if (batch.length === FROM_BATCH_SIZE) { yield batch; batch = []; @@ -361,7 +360,7 @@ function* normalizeSyncSource(source) { } let valueBatch = []; for (const chunk of normalizeSyncValue(value)) { - ArrayPrototypePush(valueBatch, chunk); + valueBatch[valueBatch.length] = chunk; if (valueBatch.length === FROM_BATCH_SIZE) { yield valueBatch; valueBatch = []; @@ -658,7 +657,7 @@ async function* normalizeAsyncSourceValue(value, context, } continue; } - ArrayPrototypePush(batch, chunk); + batch[batch.length] = chunk; if (batch.length === FROM_BATCH_SIZE) { yield batch; batch = []; @@ -898,7 +897,7 @@ function* readSyncSource(source, context) { } // Fast path 2: value is a single Uint8Array (very common) if (isUint8Array(value)) { - ArrayPrototypePush(batch, value); + batch[batch.length] = value; if (batch.length === FROM_BATCH_SIZE) { yield batch; batch = []; diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index b59949ff1f3..7cc7f1169c4 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -264,7 +264,7 @@ function* processTransformResultSync(result) { const batch = []; for (const item of result) { for (const chunk of flattenTransformYieldSync(item)) { - ArrayPrototypePush(batch, chunk); + batch[batch.length] = chunk; } } if (batch.length > 0) { @@ -290,28 +290,28 @@ function appendTransformResultSync(target, result) { } if (isUint8ArrayBatch(result)) { if (result.length > 0) { - ArrayPrototypePush(target, result); + target[target.length] = result; } return; } if (isUint8Array(result)) { - ArrayPrototypePush(target, [result]); + target[target.length] = [result]; return; } if (typeof result === 'string') { - ArrayPrototypePush(target, [toUint8Array(result)]); + target[target.length] = [toUint8Array(result)]; return; } if (isAnyArrayBuffer(result)) { - ArrayPrototypePush(target, [new Uint8Array(result)]); + target[target.length] = [new Uint8Array(result)]; return; } if (ArrayBufferIsView(result)) { - ArrayPrototypePush(target, [arrayBufferViewToUint8Array(result)]); + target[target.length] = [arrayBufferViewToUint8Array(result)]; return; } for (const batch of processTransformResultSync(result)) { - ArrayPrototypePush(target, batch); + target[target.length] = batch; } } @@ -360,11 +360,11 @@ async function* processTransformResultAsync(result) { const batch = []; for await (const item of result) { if (isUint8Array(item)) { - ArrayPrototypePush(batch, item); + batch[batch.length] = item; continue; } for await (const chunk of flattenTransformYieldAsync(item)) { - ArrayPrototypePush(batch, chunk); + batch[batch.length] = chunk; } } if (batch.length > 0) { @@ -377,13 +377,13 @@ async function* processTransformResultAsync(result) { const batch = []; for (const item of result) { if (isUint8Array(item)) { - ArrayPrototypePush(batch, item); + batch[batch.length] = item; continue; } // Note: This iteration is synchronous, since async iterables // may not be nested within sync iterables. for (const chunk of flattenTransformYieldSync(item)) { - ArrayPrototypePush(batch, chunk); + batch[batch.length] = chunk; } } if (batch.length > 0) { @@ -410,24 +410,24 @@ function appendTransformResultAsync(target, result) { } if (isUint8ArrayBatch(result)) { if (result.length > 0) { - ArrayPrototypePush(target, result); + target[target.length] = result; } return; } if (isUint8Array(result)) { - ArrayPrototypePush(target, [result]); + target[target.length] = [result]; return; } if (typeof result === 'string') { - ArrayPrototypePush(target, [toUint8Array(result)]); + target[target.length] = [toUint8Array(result)]; return; } if (isAnyArrayBuffer(result)) { - ArrayPrototypePush(target, [new Uint8Array(result)]); + target[target.length] = [new Uint8Array(result)]; return; } if (ArrayBufferIsView(result)) { - ArrayPrototypePush(target, [arrayBufferViewToUint8Array(result)]); + target[target.length] = [arrayBufferViewToUint8Array(result)]; return; } return appendTransformResultAsyncSlow(target, result); @@ -435,7 +435,7 @@ function appendTransformResultAsync(target, result) { async function appendTransformResultAsyncSlow(target, result) { for await (const batch of processTransformResultAsync(result)) { - ArrayPrototypePush(target, batch); + target[target.length] = batch; } } @@ -533,7 +533,7 @@ function* applyStatefulSyncTransform(source, transform, receiver) { if (item === null) continue; const batch = []; for (const chunk of flattenTransformYieldSync(item)) { - ArrayPrototypePush(batch, chunk); + batch[batch.length] = chunk; } if (batch.length > 0) { yield batch; @@ -968,7 +968,7 @@ async function* applyStatefulAsyncTransform( // Slow path: flatten arbitrary transform yield const batch = []; for await (const chunk of flattenTransformYieldAsync(item)) { - ArrayPrototypePush(batch, chunk); + batch[batch.length] = chunk; } if (batch.length > 0) { yield batch; diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index c99253af43e..8b6eb2da536 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -508,7 +508,7 @@ class PushQueue { for (let i = 0; i < this.#slots.length; i++) { const batch = validateBatchEntry(this.#slots.get(i)); for (let j = 0; j < batch.length; j++) { - ArrayPrototypePush(result, batch[j]); + result[result.length] = batch[j]; } } this.#slots.clear(); diff --git a/lib/internal/streams/iter/transform.js b/lib/internal/streams/iter/transform.js index 8e60589aba3..e7239278500 100644 --- a/lib/internal/streams/iter/transform.js +++ b/lib/internal/streams/iter/transform.js @@ -10,7 +10,6 @@ const { ArrayPrototypeMap, - ArrayPrototypePush, ArrayPrototypeSlice, FunctionPrototypeCall, MathMax, @@ -333,18 +332,16 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { if (have > 0) { if (bufferExhausted && outOffset === 0) { // Entire buffer filled from start - yield directly, no copy. - ArrayPrototypePush(pending, outBuf); + pending[pending.length] = outBuf; } else if (bufferExhausted) { // Tail of buffer filled and buffer is being replaced - // subarray is safe since outBuf reference is overwritten below. - ArrayPrototypePush(pending, - outBuf.subarray(outOffset, outOffset + have)); + pending[pending.length] = + outBuf.subarray(outOffset, outOffset + have); } else { // Partial fill, buffer will be reused - must copy. - ArrayPrototypePush(pending, - TypedArrayPrototypeSlice(outBuf, - outOffset, - outOffset + have)); + pending[pending.length] = + TypedArrayPrototypeSlice(outBuf, outOffset, outOffset + have); } pendingBytes += have; outOffset += have; @@ -642,17 +639,15 @@ function makeZlibTransformSync(createHandleFn, processFlag, finishFlag) { if (have > 0) { if (bufferExhausted && outOffset === 0) { // Entire buffer filled - yield directly, no copy. - ArrayPrototypePush(pending, outBuf); + pending[pending.length] = outBuf; } else if (bufferExhausted) { // Tail filled, buffer being replaced - subarray is safe. - ArrayPrototypePush(pending, - outBuf.subarray(outOffset, outOffset + have)); + pending[pending.length] = + outBuf.subarray(outOffset, outOffset + have); } else { // Partial fill, buffer reused - must copy. - ArrayPrototypePush(pending, - TypedArrayPrototypeSlice(outBuf, - outOffset, - outOffset + have)); + pending[pending.length] = + TypedArrayPrototypeSlice(outBuf, outOffset, outOffset + have); } pendingBytes += have; outOffset += have; diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 2d85d888ddf..82dace3d458 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -5,7 +5,6 @@ const { ArrayBufferPrototypeGetByteLength, ArrayBufferPrototypeGetDetached, ArrayBufferPrototypeGetResizable, - ArrayPrototypePush, ArrayPrototypeSlice, FunctionPrototypeCall, ObjectDefineProperty, @@ -555,12 +554,12 @@ function validateBudget(budget) { function recordChunk(chunks, checks, value) { const buffer = TypedArrayPrototypeGetBuffer(value); const byteLength = TypedArrayPrototypeGetByteLength(value); - ArrayPrototypePush(chunks, value); + chunks[chunks.length] = value; if (byteLength === 0 || isSharedArrayBuffer(buffer) || ArrayBufferPrototypeGetResizable(buffer)) { - ArrayPrototypePush(checks, snapshotByteView(value)); + checks[checks.length] = snapshotByteView(value); } else { - ArrayPrototypePush(checks, byteLength); + checks[checks.length] = byteLength; } return byteLength; } @@ -612,14 +611,14 @@ function splitBatchEntry(entry, limit) { for (let i = 0; i < views.length; i++) { const view = views[i]; if (current.length > 0 && byteLength + view.byteLength >= limit) { - ArrayPrototypePush(entries, new BatchEntry(current, byteLength)); + entries[entries.length] = new BatchEntry(current, byteLength); current = []; byteLength = 0; } - ArrayPrototypePush(current, view); + current[current.length] = view; byteLength += view.byteLength; } - ArrayPrototypePush(entries, new BatchEntry(current, byteLength)); + entries[entries.length] = new BatchEntry(current, byteLength); return entries; } From f7be69de59031297b7c15c5f16cd4229172c75d5 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 06:18:09 +0000 Subject: [PATCH 24/58] stream: read stream/iter push() readables without normalizing The readable of push() yields batches that it has validated already, but from(), and so pipeTo(), normalized them again through another async iterator layer. Mark the readable with kValidatedSource, as for Readable sources, so that they are read directly. Piping a push() stream written with 64 KiB chunks is about 30% faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/push.js | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index 8b6eb2da536..f5d5e38f006 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -26,6 +26,7 @@ const { const { drainableProtocol, + kValidatedSource, } = require('internal/streams/iter/types'); const { @@ -734,6 +735,9 @@ class PushWriter { function createReadable(queue) { return { __proto__: null, + // The batches read are validated already: from() and pipeTo() can read + // them without normalizing them again. + [kValidatedSource]: true, [SymbolAsyncIterator]() { return ObjectSetPrototypeOf({ async next() { From 8e2dba6524a73c25adecc031b075bd4586a04fed Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 06:23:06 +0000 Subject: [PATCH 25/58] stream: avoid allocations on every stream/iter push() write Every write() to a push() or broadcast() writer converted its options with the WriteOptions dictionary converter to look for a signal, which creates an empty dictionary when there are no options, and every write replaced the array of pending drains, even when there were none. Return no signal for undefined or null options without converting them, and leave the array of pending drains alone when it is empty. Writing 16-byte chunks to a push() stream with await write() and reading them is about 30% faster, and about 18% faster when piping them; 64 KiB chunks are about 11% faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/broadcast.js | 2 ++ lib/internal/streams/iter/push.js | 2 ++ lib/internal/streams/iter/utils.js | 3 +++ 3 files changed, 7 insertions(+) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index 5dfb297a5fd..e107927730c 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -839,6 +839,7 @@ class BroadcastWriter { } #resolvePendingDrains(canWrite) { + if (this.#pendingDrains.length === 0) return; const drains = this.#pendingDrains; this.#pendingDrains = []; for (let i = 0; i < drains.length; i++) { @@ -847,6 +848,7 @@ class BroadcastWriter { } #rejectPendingDrains(error) { + if (this.#pendingDrains.length === 0) return; const drains = this.#pendingDrains; this.#pendingDrains = []; for (let i = 0; i < drains.length; i++) { diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index f5d5e38f006..6039ab08803 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -594,6 +594,7 @@ class PushQueue { } #resolvePendingDrains(canWrite) { + if (this.#pendingDrains.length === 0) return; const drains = this.#pendingDrains; this.#pendingDrains = []; for (let i = 0; i < drains.length; i++) { @@ -602,6 +603,7 @@ class PushQueue { } #rejectPendingDrains(error) { + if (this.#pendingDrains.length === 0) return; const drains = this.#pendingDrains; this.#pendingDrains = []; for (let i = 0; i < drains.length; i++) { diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 82dace3d458..f52255fc445 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -707,6 +707,9 @@ function convertChunks(chunks) { * @returns {AbortSignal|undefined} */ function getWriterSignal(options) { + // Writes are hot, and converting undefined or null creates an empty + // dictionary, which has no signal. + if (options === undefined || options === null) return undefined; return converters.WriteOptions(options, kWriteOptionsContext).signal; } From e58a927aeb96245b7d7dd7c4ba7f822e8237b26b Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 06:24:00 +0000 Subject: [PATCH 26/58] stream: snapshot common stream/iter byte views in fewer fields To detect that a chunk accepted by a writer was resized or detached before it is read, a ByteViewSnapshot of its buffer, byteLength, byteOffset and detached state is taken for every chunk written, and checked when it is read. A non-empty view of a fixed-length, non-shared ArrayBuffer can only change by the buffer being detached, which makes its byteLength 0, as recordChunk() already relies on. Snapshot such views, the common case, as a FixedByteView of the view and its byteLength, which is all that needs to be checked. Writing 16-byte chunks to a push() stream and reading them is about 17% faster with await write() and 37% faster with writeSync(), and piping them about 29% and 51%; 64 KiB chunks are about 7-13% faster. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/utils.js | 30 +++++++++++++++++++++++++++--- 1 file changed, 27 insertions(+), 3 deletions(-) diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index f52255fc445..ad71de638ab 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -376,6 +376,17 @@ function ByteViewSnapshot(value, buffer, sharedBufferView) { } ByteViewSnapshot.prototype = ObjectFreeze({ __proto__: null }); +// The snapshot of a non-empty view of a fixed-length, non-shared ArrayBuffer, +// the common case. Such a view can only change by the buffer being detached, +// which makes its byteLength 0, so its byteLength is all that needs to be +// recorded. `buffer` is null to tell it from a ByteViewSnapshot. +function FixedByteView(value, byteLength) { + this.value = value; + this.byteLength = byteLength; + this.buffer = null; +} +FixedByteView.prototype = ObjectFreeze({ __proto__: null }); + function BatchEntry(views, byteLength) { this.views = views; this.byteLength = byteLength; @@ -399,12 +410,25 @@ PendingWrite.prototype = ObjectFreeze({ __proto__: null }); function snapshotByteView(value) { const buffer = TypedArrayPrototypeGetBuffer(value); - const sharedBufferView = isSharedArrayBuffer(buffer) ? - new Uint8Array(buffer) : undefined; - return new ByteViewSnapshot(value, buffer, sharedBufferView); + if (isSharedArrayBuffer(buffer)) { + return new ByteViewSnapshot(value, buffer, new Uint8Array(buffer)); + } + const byteLength = TypedArrayPrototypeGetByteLength(value); + if (byteLength !== 0 && !ArrayBufferPrototypeGetResizable(buffer)) { + return new FixedByteView(value, byteLength); + } + return new ByteViewSnapshot(value, buffer, undefined); } function validateByteView(snapshot) { + if (snapshot.buffer === null) { + const { value } = snapshot; + if (TypedArrayPrototypeGetByteLength(value) !== snapshot.byteLength) { + throw new ERR_INVALID_STATE.TypeError( + 'Byte view was resized or detached after being accepted'); + } + return value; + } const { value, buffer, From 639d13a9ad1c596fc8dfe4d3e81ca75c5a6d5227 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 07:01:01 +0000 Subject: [PATCH 27/58] stream: resolve stream/iter push() return() with its value The iterator of a push() readable resolved return(value) with an undefined value, unlike async generators and the other stream/iter iterators. Since push() readables are no longer wrapped by from(), this also applied to from() and pipeTo(). Resolve it with `value`. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/push.js | 4 ++-- test/parallel/test-stream-iter-push-basic.js | 18 +++++++++++++++++- 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index 6039ab08803..a5174ad7e55 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -745,9 +745,9 @@ function createReadable(queue) { async next() { return queue.read(); }, - async return() { + async return(value) { queue.consumerReturn(); - return new IterResult(true, undefined); + return new IterResult(true, value); }, async throw(error) { queue.consumerThrow(error); diff --git a/test/parallel/test-stream-iter-push-basic.js b/test/parallel/test-stream-iter-push-basic.js index 9b1c044cac4..ad7eab5f53a 100644 --- a/test/parallel/test-stream-iter-push-basic.js +++ b/test/parallel/test-stream-iter-push-basic.js @@ -3,7 +3,7 @@ const common = require('../common'); const assert = require('assert'); -const { push, text } = require('stream/iter'); +const { from, push, text } = require('stream/iter'); async function testBasicWriteRead() { const { writer, readable } = push(); @@ -174,6 +174,21 @@ async function testInvalidBackpressure() { } } +async function testReturnValue() { + // return(value) resolves with `value`, read directly or through from(). + for (const getIterator of [ + (readable) => readable[Symbol.asyncIterator](), + (readable) => from(readable)[Symbol.asyncIterator](), + ]) { + const { writer, readable } = push(); + const iterator = getIterator(readable); + const result = await iterator.return('value'); + assert.strictEqual(result.done, true); + assert.strictEqual(result.value, 'value'); + assert.strictEqual(writer.canWrite, null); + } +} + Promise.all([ testBasicWriteRead(), testMultipleWrites(), @@ -187,4 +202,5 @@ Promise.all([ testConsumerBreakWriteSyncReturnsFalse(), testPushWithTransforms(), testInvalidBackpressure(), + testReturnValue(), ]).then(common.mustCall()); From 2b4f9ca014b9efc0f84e5b563842c08d63b819ae Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 07:03:04 +0000 Subject: [PATCH 28/58] doc: document stream/iter iterator results and from() identity Some stream/iter iterators return iterator results that do not inherit from Object.prototype, and from() returns validated sources unchanged. Document both. Assisted-by: OpenCode Signed-off-by: James M Snell --- doc/api/stream_iter.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 644bcb1f9eb..7e121aeb1c5 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -119,6 +119,10 @@ async function run() { } ``` +Some iterators of this module return iterator results (`{ done, value }` +objects) that do not inherit from `Object.prototype`. Code should only rely on +their `done` and `value` properties, as `for await...of` does. + ### Transforms Transforms come in two forms: @@ -598,6 +602,10 @@ Objects implementing `Symbol.for('Stream.toAsyncStreamable')` or precedence over the iteration protocols (`Symbol.asyncIterator`, `Symbol.iterator`). +The readable of a [`push()`][] stream without transforms and the iterables +returned by [`fromReadable()`][] already yield normalized batches, so `from()` +returns them unchanged. + ```mjs import { Buffer } from 'node:buffer'; import { from, text } from 'node:stream/iter'; @@ -2415,6 +2423,7 @@ console.log(textSync(stream)); // 'hello world' [`pipeTo()`]: #pipetosource-transforms-writer-options [`pull()`]: #pullsource-transforms-options [`pullSync()`]: #pullsyncsource-transforms +[`push()`]: #pushtransforms-options [`share()`]: #sharesource-options [`stream.Readable`]: stream.md#class-streamreadable [`stream.Writable`]: stream.md#class-streamwritable From 6b43ce414c415c7e7ea7cf1b952256a02f0086a7 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 09:24:41 +0000 Subject: [PATCH 29/58] stream: check stream/iter pipe writes more cheaply pipeTo() and pipeToSync() check that each chunk is not resized or detached while the writer uses it: around writeSync() with callWithByteView() for batches of one chunk, and with a batch entry of snapshots, one object for every chunk, for larger batches written one chunk at a time. A non-empty view of a fixed-length, non-shared ArrayBuffer can only change by the buffer being detached, which makes its byteLength 0 (see FixedByteView). For such views, check only the byteLength around writeSync(), and snapshot batches as a FixedBatch of their chunks and byteLengths, checking each buffer once when chunks share it. Other views are checked as before. Piping a sync generator yielding 16-byte chunks one per batch is about 15% faster with pipeTo() and 19% faster with pipeToSync(); with batches of 128 chunks, pipeTo() is about 28% faster and pipeToSync() 20%. Assisted-by: OpenCode Signed-off-by: James M Snell --- lib/internal/streams/iter/pull.js | 40 ++++++++++ lib/internal/streams/iter/utils.js | 79 +++++++++++++++++++ .../test-stream-iter-resizable-buffers.js | 48 +++++++++++ 3 files changed, 167 insertions(+) diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 7cc7f1169c4..0193535e176 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -62,10 +62,13 @@ const { kResolvedPromise, kStart, callWithByteView, + checkFixedBatchChunk, createBatchEntry, createOperationQueue, + fixedBatchToEntry, isTransformObject, parsePullArgs, + snapshotFixedBatch, snapshotTransform, toUint8Array, validateBatchEntry, @@ -1546,6 +1549,24 @@ function pipeToSync(source, ...args) { totalBytes += TypedArrayPrototypeGetByteLength(chunk); continue; } + if (!hasWritevSync) { + // Chunks written one at a time: snapshot them without an object for + // every chunk when possible. + const fixed = snapshotFixedBatch(batch); + if (fixed !== null) { + for (let i = 0; i < batch.length; i++) { + const chunk = checkFixedBatchChunk(fixed, i); + const accepted = writer.writeSync(chunk); + checkFixedBatchChunk(fixed, i); + if (accepted === false) { + throw new ERR_OUT_OF_RANGE( + 'write', 'within byte budget', 'budget exhausted'); + } + totalBytes += fixed.byteLengths[i]; + } + continue; + } + } const entry = createBatchEntry(batch); if (hasWritevSync && batch.length > 1) { const accepted = writer.writevSync(validateBatchEntry(entry)); @@ -1672,6 +1693,21 @@ async function pipeTo(source, ...args) { // Write a batch using try-fallback: sync first, async if needed. // Returns undefined on sync success, or a Promise when async fallback // is required. Callers must check: const p = writeBatch(b); if (p) await p; + // Write a FixedBatch with writeSync(), like the loop at the end of + // writeBatch(), falling back to the async path for the rest of the batch. + function writeFixedBatch(fixed) { + const { chunks } = fixed; + for (let i = 0; i < chunks.length; i++) { + const chunk = checkFixedBatchChunk(fixed, i); + if (!writer.writeSync(chunk)) { + checkFixedBatchChunk(fixed, i); + return writeBatchAsyncFallback(fixedBatchToEntry(fixed), i); + } + checkFixedBatchChunk(fixed, i); + totalBytes += fixed.byteLengths[i]; + } + } + function writeBatch(batch) { // Single chunk, the common case: check the view around writeSync() // without allocating a batch entry, and create one only to fall back to @@ -1684,6 +1720,10 @@ async function pipeTo(source, ...args) { } return writeBatchAsyncFallback(createBatchEntry(batch), 0); } + if (!hasWritev && hasWriteSync) { + const fixed = snapshotFixedBatch(batch); + if (fixed !== null) return writeFixedBatch(fixed); + } const entry = createBatchEntry(batch); if (hasWritev && batch.length > 1) { if (!hasWritevSync || diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index ad71de638ab..568c9afe854 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -468,6 +468,20 @@ function validateByteView(snapshot) { */ function callWithByteView(value, method, receiver) { const buffer = TypedArrayPrototypeGetBuffer(value); + if (!isSharedArrayBuffer(buffer) && !ArrayBufferPrototypeGetResizable(buffer)) { + // A non-empty view of a fixed-length, non-shared ArrayBuffer can only + // change by the buffer being detached, which makes its byteLength 0 (see + // FixedByteView), so checking its byteLength is enough. + const byteLength = TypedArrayPrototypeGetByteLength(value); + if (byteLength !== 0) { + const result = FunctionPrototypeCall(method, receiver, value); + if (TypedArrayPrototypeGetByteLength(value) !== byteLength) { + throw new ERR_INVALID_STATE.TypeError( + 'Byte view was resized or detached after being accepted'); + } + return result; + } + } if (isSharedArrayBuffer(buffer)) { const snapshot = snapshotByteView(value); const result = FunctionPrototypeCall(method, receiver, value); @@ -607,6 +621,68 @@ function validateRecordedChunks(chunks, checks) { return chunks; } +// A batch accepted for writing one chunk at a time: its chunks, and their +// byteLengths when accepted (see snapshotFixedBatch()). +function FixedBatch(chunks, byteLengths, byteLength) { + this.chunks = chunks; + this.byteLengths = byteLengths; + this.byteLength = byteLength; +} +FixedBatch.prototype = ObjectFreeze({ __proto__: null }); + +/** + * Snapshot a batch of Uint8Arrays like createBatchEntry(), without an object + * for every chunk, when every chunk is a non-empty view of a fixed-length, + * non-shared ArrayBuffer: such views can only change by being detached, + * which makes their byteLength 0 (see FixedByteView), so their byteLengths + * are all that needs to be recorded. Check a chunk with checkFixedBatchChunk() + * before and after using it. + * @param {Uint8Array[]} chunks + * @returns {FixedBatch|null} null if a chunk needs a full snapshot. + */ +function snapshotFixedBatch(chunks) { + const count = chunks.length; + const byteLengths = new Array(count); + let byteLength = 0; + let checkedBuffer; + for (let i = 0; i < count; i++) { + const view = chunks[i]; + const buffer = TypedArrayPrototypeGetBuffer(view); + // Chunks often share a buffer: check each buffer once. + if (buffer !== checkedBuffer) { + if (isSharedArrayBuffer(buffer) || + ArrayBufferPrototypeGetResizable(buffer)) { + return null; + } + checkedBuffer = buffer; + } + const length = TypedArrayPrototypeGetByteLength(view); + if (length === 0) return null; + byteLengths[i] = length; + byteLength += length; + } + return new FixedBatch(ArrayPrototypeSlice(chunks), byteLengths, byteLength); +} + +function checkFixedBatchChunk(batch, index) { + const chunk = batch.chunks[index]; + if (TypedArrayPrototypeGetByteLength(chunk) !== batch.byteLengths[index]) { + throw new ERR_INVALID_STATE.TypeError( + 'Byte view was resized or detached after being accepted'); + } + return chunk; +} + +// The batch entry of a FixedBatch, to continue writing it asynchronously. +function fixedBatchToEntry(batch) { + const { chunks, byteLengths } = batch; + const views = new Array(chunks.length); + for (let i = 0; i < chunks.length; i++) { + views[i] = new FixedByteView(chunks[i], byteLengths[i]); + } + return new BatchEntry(views, batch.byteLength); +} + function createBatchEntry(chunks) { const views = new Array(chunks.length); let byteLength = 0; @@ -870,7 +946,10 @@ module.exports = { concatBytes, createOperationQueue, convertChunks, + checkFixedBatchChunk, createBatchEntry, + fixedBatchToEntry, + snapshotFixedBatch, recordChunk, splitBatchEntry, getProtocolMethod, diff --git a/test/parallel/test-stream-iter-resizable-buffers.js b/test/parallel/test-stream-iter-resizable-buffers.js index ab75d07c455..1691893d347 100644 --- a/test/parallel/test-stream-iter-resizable-buffers.js +++ b/test/parallel/test-stream-iter-resizable-buffers.js @@ -210,6 +210,53 @@ async function testPipeRejectsSyncWriteDetachOrResize() { assert.deepStrictEqual(written, [chunk]); } +// Batches of several chunks written one at a time are checked against the +// byteLengths recorded when the batch was accepted, before and after each +// writeSync(). Detaching a later chunk, or the chunk being written, must be +// rejected. +async function testPipeRejectsDetachInMultiChunkBatch() { + for (const detachIndex of [1, 0]) { + const asyncBuffers = [new ArrayBuffer(2), new ArrayBuffer(2)]; + const asyncWritten = []; + await assert.rejects(pipeTo([asyncBuffers.map((b) => new Uint8Array(b))], { + writeSync(chunk) { + asyncWritten.push(chunk.byteLength); + if (asyncWritten.length === 1) asyncBuffers[detachIndex].transfer(); + return true; + }, + write: common.mustNotCall(), + fail: common.mustCall(), + }), kResizeError); + assert.deepStrictEqual(asyncWritten, [2]); + + const syncBuffers = [new ArrayBuffer(2), new ArrayBuffer(2)]; + const syncWritten = []; + assert.throws(() => pipeToSync([syncBuffers.map((b) => new Uint8Array(b))], { + writeSync(chunk) { + syncWritten.push(chunk.byteLength); + if (syncWritten.length === 1) syncBuffers[detachIndex].transfer(); + return true; + }, + fail: common.mustCall(), + }, { preventClose: true }), kResizeError); + assert.deepStrictEqual(syncWritten, [2]); + } + + // When writeSync() declines a chunk, pipeTo() writes the rest of the batch + // with write(). + const chunks = [Uint8Array.of(1), Uint8Array.of(2, 3), Uint8Array.of(4)]; + const written = []; + assert.strictEqual(await pipeTo([chunks], { + writeSync(chunk) { + if (chunk === chunks[1]) return false; + written.push(chunk); + return true; + }, + write(chunk) { written.push(chunk); }, + }, { preventClose: true }), 4); + assert.deepStrictEqual(written, chunks); +} + async function testConsumersRejectDetachedViews() { // Views of fixed-length buffers are tracked without a full snapshot; they // must still be rejected when detached after being accepted. @@ -249,4 +296,5 @@ Promise.all([ testConsumersRejectDetachedViews(), testPipeRejectsWriterResize(), testPipeRejectsSyncWriteDetachOrResize(), + testPipeRejectsDetachInMultiChunkBatch(), ]).then(common.mustCall()); From c84710a3471d61cdaeca3a6ce9b9678d78b129b1 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 09:33:06 +0000 Subject: [PATCH 30/58] stream: read sync sources synchronously in stream/iter pipeTo() pipeTo() iterated the from() normalization of a sync iterable source with for await...of, which costs three promises and several ticks per batch, even though the batches are read synchronously. Let the iterator returned by from() for a sync iterable read the next batch synchronously when no operation is running or queued and the value read needs no asynchronous normalization, and have pipeTo() use it when there are no transforms and no signal. Other values are normalized through next() as before, a write error still closes the source, and an error reading the source still does not. The source and the writer see the same calls in the same order; the batches are no longer written on separate ticks unless a write is asynchronous. Piping a sync generator yielding 16-byte chunks, one per batch, is about 2.5 times faster, the same as pipeToSync(). Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 56 +++++++++++++++++++++++- lib/internal/streams/iter/pull.js | 31 +++++++++++++ lib/internal/streams/iter/utils.js | 4 ++ test/parallel/test-stream-iter-pipeto.js | 47 ++++++++++++++++++++ 4 files changed, 137 insertions(+), 1 deletion(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index c249b0e86f1..759a89bb9a7 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -139,9 +139,36 @@ function waitForNormalization(value, context) { return promise; } +// The method of an iterator returned by from() for a sync iterable that reads +// the next batch synchronously when it can, for pipeTo(): see nextSyncBatch() +// in createSyncSourceNormalizer(). +const kNextSyncBatch = Symbol('kNextSyncBatch'); + function createNormalizationIterator(createIterator) { const context = createNormalizationContext(); const iterator = createIterator(context); + if (iterator[kNextSyncBatch] !== undefined) { + return ObjectSetPrototypeOf({ + next(value) { + return FunctionPrototypeCall(iterator.next, iterator, value); + }, + return(value) { + cancelNormalization( + context, lazyDOMException('Aborted', 'AbortError')); + return FunctionPrototypeCall(iterator.return, iterator, value); + }, + throw(error) { + cancelNormalization(context, error, true); + return FunctionPrototypeCall(iterator.throw, iterator, error); + }, + [kNextSyncBatch]() { + return iterator[kNextSyncBatch](); + }, + [SymbolAsyncIterator]() { + return this; + }, + }, null); + } return ObjectSetPrototypeOf({ next(value) { return FunctionPrototypeCall(iterator.next, iterator, value); @@ -938,7 +965,7 @@ function createSyncSourceNormalizer(source, context) { let done = false; // A normalizeAsyncSourceValue() generator for the value being normalized. let valueBatches = null; - const { run, settled, release } = createOperationQueue(); + const { run, settled, release, idle } = createOperationQueue(); // Results are often produced synchronously. An async generator stays busy // until the tick after a yield or return (both await their operand), so @@ -1063,10 +1090,36 @@ function createSyncSourceNormalizer(source, context) { }); } + // Read the next batch like next() but synchronously, when it is read + // synchronously from the source and no operation is running or queued: + // returns the batch, or null when done. Returns undefined, having started + // nothing but the normalization of a value, when next() must be used. + // Throws on error, like next() rejects. + function nextSyncBatch() { + if (!idle() || valueBatches !== null) return undefined; + if (done) return null; + let result; + try { + result = reader.next(); + } catch (error) { + done = true; + throw error; + } + if (result.done) { + done = true; + return null; + } + const value = result.value; + if (ArrayIsArray(value)) return value; + valueBatches = normalizeAsyncSourceValue(value.value, context, false); + return undefined; + } + return ObjectSetPrototypeOf({ next() { return run(doNext); }, return(value) { return run(doReturn, value); }, throw(error) { return run(doThrow, error); }, + [kNextSyncBatch]: nextSyncBatch, }, null); } @@ -1290,6 +1343,7 @@ module.exports = { isPrimitiveChunk, isSyncIterable, isUint8ArrayBatch, + kNextSyncBatch, normalizeAsyncSource, normalizeAsyncValue, normalizeSyncSource, diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 0193535e176..1f0b3659693 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -52,6 +52,7 @@ const { isSyncIterable, isAsyncIterable, isUint8ArrayBatch, + kNextSyncBatch, } = require('internal/streams/iter/from'); const { @@ -1708,6 +1709,34 @@ async function pipeTo(source, ...args) { } } + // Write the batches of a sync source as the for await...of loop below + // does, reading them synchronously when possible: an error reading a + // batch ends the loop, and an error writing one closes the source first, + // ignoring errors closing it. + async function pipeSyncSource(iterator) { + for (;;) { + let batch = iterator[kNextSyncBatch](); + if (batch === undefined) { + const result = await iterator.next(); + if (result.done) return; + batch = result.value; + } else if (batch === null) { + return; + } + try { + const p = writeBatch(batch); + if (p) await p; + } catch (error) { + try { + await iterator.return(); + } catch { + // The error writing the batch is thrown. + } + throw error; + } + } + } + function writeBatch(batch) { // Single chunk, the common case: check the view around writeSync() // without allocating a batch entry, and create one only to fall back to @@ -1767,6 +1796,8 @@ async function pipeTo(source, ...args) { const p = writeBatch(batch); if (p) await p; } + } else if (normalized[kNextSyncBatch] !== undefined) { + await pipeSyncSource(normalized); } else { for await (const batch of normalized) { const p = writeBatch(batch); diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 568c9afe854..6b40f5a2903 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -564,6 +564,10 @@ function createOperationQueue() { busy = false; drain(); }, + // Whether no operation is running or queued. + idle() { + return !busy && (queue === null || queue.length === 0); + }, }; } diff --git a/test/parallel/test-stream-iter-pipeto.js b/test/parallel/test-stream-iter-pipeto.js index ed1614552f9..f1db05d0e6e 100644 --- a/test/parallel/test-stream-iter-pipeto.js +++ b/test/parallel/test-stream-iter-pipeto.js @@ -345,6 +345,52 @@ async function testPipeToSyncIterableAsyncValue() { assert.strictEqual(result, 'ab'); } +// pipeTo() reads sync iterables synchronously when it can. An error writing +// a batch must still close the source, ignoring an error closing it, and an +// error reading the source must not close it. +async function testPipeToSyncIterableWriteError() { + for (const returnThrows of [false, true]) { + const error = new Error('write'); + let closed = false; + const source = { + [Symbol.iterator]() { + let i = 0; + return { + next() { + return { done: false, value: [new Uint8Array([i++])] }; + }, + return() { + closed = true; + if (returnThrows) throw new Error('return'); + return { done: true }; + }, + }; + }, + }; + await assert.rejects(pipeTo(source, { + write: common.mustNotCall(), + writeSync: common.mustCall(() => { throw error; }), + fail: common.mustCall((reason) => assert.strictEqual(reason, error)), + }), error); + assert.strictEqual(closed, true); + } + + const error = new Error('source'); + const source = { + [Symbol.iterator]() { + return { + next() { throw error; }, + return: common.mustNotCall(), + }; + }, + }; + await assert.rejects(pipeTo(source, { + write: common.mustNotCall(), + writeSync: common.mustNotCall(), + fail: common.mustCall((reason) => assert.strictEqual(reason, error)), + }), error); +} + Promise.all([ testPipeToSync(), testPipeTo(), @@ -365,5 +411,6 @@ Promise.all([ testPipeToSyncIterableUsesFromBatching(), testPipeToSyncIterableWriteFallback(), testPipeToSyncIterableAsyncValue(), + testPipeToSyncIterableWriteError(), testPipeToSourceNormalizationIndependentOfWriter(), ]).then(common.mustCall()); From e5523892aab25fe888a50875095621354b403ebc Mon Sep 17 00:00:00 2001 From: James M Snell Date: Sun, 4 Oct 2026 09:43:20 +0000 Subject: [PATCH 31/58] stream: read async sources more cheaply in stream/iter pipeTo() from() reads an async iterable source through a layer that lets a cancellation of the normalization reject a pending read: for every batch, it waits with a PromiseWithResolvers() and three closures, and the normalizer handles the source's result in a second reaction. pipeTo() never cancels the normalization while a read is pending: it only calls return() after an error writing a batch. Without transforms or a signal, have it read through a method of the iterator returned by from() that reads the source without waiting for a cancellation, and handles the source's result in a single reaction. The source's results are checked as before, a write error still closes the source, and an error reading the source still does not. Piping an async generator yielding 16-byte chunks, one per batch, is about 1.5 times faster, with one promise and 620 bytes less per batch. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 98 ++++++++++++++++++++++++ lib/internal/streams/iter/pull.js | 24 ++++++ test/parallel/test-stream-iter-pipeto.js | 58 ++++++++++++++ 3 files changed, 180 insertions(+) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 759a89bb9a7..51d17c4e087 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -144,9 +144,44 @@ function waitForNormalization(value, context) { // in createSyncSourceNormalizer(). const kNextSyncBatch = Symbol('kNextSyncBatch'); +// The method of an iterator returned by from() for an async iterable that +// reads like next() when nothing can cancel the normalization while a read +// is pending, for pipeTo(): see createAsyncSourceNormalizer(). +const kNextUncancellable = Symbol('kNextUncancellable'); + +// Methods of the iterator returned by yieldNormalizationAbortable(), used by +// createAsyncSourceNormalizer() for kNextUncancellable. +const kReadUncancellable = Symbol('kReadUncancellable'); +const kToIterResult = Symbol('kToIterResult'); +const kReadFailed = Symbol('kReadFailed'); +// The result kReadUncancellable gives once the source is done. +const kDoneSourceResult = ObjectFreeze({ __proto__: null, done: true, value: undefined }); + function createNormalizationIterator(createIterator) { const context = createNormalizationContext(); const iterator = createIterator(context); + if (iterator[kNextUncancellable] !== undefined) { + return ObjectSetPrototypeOf({ + next(value) { + return FunctionPrototypeCall(iterator.next, iterator, value); + }, + return(value) { + cancelNormalization( + context, lazyDOMException('Aborted', 'AbortError')); + return FunctionPrototypeCall(iterator.return, iterator, value); + }, + throw(error) { + cancelNormalization(context, error, true); + return FunctionPrototypeCall(iterator.throw, iterator, error); + }, + [kNextUncancellable]() { + return iterator[kNextUncancellable](); + }, + [SymbolAsyncIterator]() { + return this; + }, + }, null); + } if (iterator[kNextSyncBatch] !== undefined) { return ObjectSetPrototypeOf({ next(value) { @@ -468,6 +503,15 @@ function yieldNormalizationAbortable(source, context) { return new IterResult(false, value); } + function onUncancellableResult(result) { + try { + return toIterResult(result); + } catch (error) { + reading = false; + throw error; + } + } + // Reject a pending next(), closing the source first if the // normalization has been cancelled. function rejectNext(reject, error) { @@ -535,6 +579,30 @@ function yieldNormalizationAbortable(source, context) { }); return promise; }, + // Like next(), when nothing cancels the normalization while the read + // is pending, in two parts so that the caller handles the result in + // the same reaction: kReadUncancellable starts the read, returning a + // promise for the source's result, and the caller passes that result + // to kToIterResult, or calls kReadFailed if the promise rejects. + [kReadUncancellable]() { + if (completed) { + return PromiseResolve(kDoneSourceResult); + } + if (context.cancelled) return PromiseReject(context.reason); + reading = true; + let next; + try { + next = FunctionPrototypeCall(nextMethod, iterator); + } catch (error) { + reading = false; + return PromiseReject(error); + } + return PromiseResolve(next); + }, + [kToIterResult]: onUncancellableResult, + [kReadFailed]() { + reading = false; + }, async return(value) { await closeSource( context.suppressCleanup || (context.cancelled && reading)); @@ -723,6 +791,10 @@ function createAsyncSourceNormalizer(source, context) { // oversized batch, or a normalizeAsyncSourceValue() generator. let boundedBatches = null; let valueBatches = null; + // Whether the source is read with kNextUncancellable: set by the + // kNextUncancellable method, for callers that never cancel the + // normalization while a read is pending. + let uncancellable = false; const { run, settled } = createOperationQueue(); function finish(result) { @@ -784,7 +856,28 @@ function createAsyncSourceNormalizer(source, context) { return finish(new IterResult(false, result.value)); } + // A kReadUncancellable result: handled as the result of the source's + // next() is, in the same reaction. + function onUncancellableResult(result) { + let iterResult; + try { + iterResult = iterator[kToIterResult](result); + } catch (error) { + return fail(error); + } + return onSourceResult(iterResult); + } + + function onUncancellableError(error) { + iterator[kReadFailed](); + return fail(error); + } + function pullSource() { + if (uncancellable) { + return PromisePrototypeThen(iterator[kReadUncancellable](), + onUncancellableResult, onUncancellableError); + } return PromisePrototypeThen(iterator.next(), onSourceResult, fail); } @@ -866,6 +959,10 @@ function createAsyncSourceNormalizer(source, context) { next() { return run(doNext); }, return(value) { return run(doReturn, value); }, throw(error) { return run(doThrow, error); }, + [kNextUncancellable]() { + uncancellable = true; + return run(doNext); + }, }, null); } @@ -1344,6 +1441,7 @@ module.exports = { isSyncIterable, isUint8ArrayBatch, kNextSyncBatch, + kNextUncancellable, normalizeAsyncSource, normalizeAsyncValue, normalizeSyncSource, diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 1f0b3659693..31bd3acf26a 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -53,6 +53,7 @@ const { isAsyncIterable, isUint8ArrayBatch, kNextSyncBatch, + kNextUncancellable, } = require('internal/streams/iter/from'); const { @@ -1737,6 +1738,27 @@ async function pipeTo(source, ...args) { } } + // Write the batches of an async source as the for await...of loop below + // does. Nothing cancels the normalization while a batch is read: return() + // is only called after an error writing one. + async function pipeAsyncSource(iterator) { + for (;;) { + const result = await iterator[kNextUncancellable](); + if (result.done) return; + try { + const p = writeBatch(result.value); + if (p) await p; + } catch (error) { + try { + await iterator.return(); + } catch { + // The error writing the batch is thrown. + } + throw error; + } + } + } + function writeBatch(batch) { // Single chunk, the common case: check the view around writeSync() // without allocating a batch entry, and create one only to fall back to @@ -1798,6 +1820,8 @@ async function pipeTo(source, ...args) { } } else if (normalized[kNextSyncBatch] !== undefined) { await pipeSyncSource(normalized); + } else if (normalized[kNextUncancellable] !== undefined) { + await pipeAsyncSource(normalized); } else { for await (const batch of normalized) { const p = writeBatch(batch); diff --git a/test/parallel/test-stream-iter-pipeto.js b/test/parallel/test-stream-iter-pipeto.js index f1db05d0e6e..c5dc0a3f133 100644 --- a/test/parallel/test-stream-iter-pipeto.js +++ b/test/parallel/test-stream-iter-pipeto.js @@ -391,6 +391,63 @@ async function testPipeToSyncIterableWriteError() { }), error); } +// pipeTo() reads async iterables without waiting for a cancellation, since +// nothing cancels them while a read is pending. An error writing a batch +// must still close the source, ignoring an error closing it, an error +// reading the source must not close it, and the source's results are still +// checked. +async function testPipeToAsyncIterableErrors() { + for (const returnRejects of [false, true]) { + const error = new Error('write'); + let closed = false; + const source = { + [Symbol.asyncIterator]() { + let i = 0; + return { + async next() { + return { done: false, value: [new Uint8Array([i++])] }; + }, + async return() { + closed = true; + if (returnRejects) throw new Error('return'); + return { done: true }; + }, + }; + }, + }; + await assert.rejects(pipeTo(source, { + write: common.mustNotCall(), + writeSync: common.mustCall(() => { throw error; }), + fail: common.mustCall((reason) => assert.strictEqual(reason, error)), + }), error); + assert.strictEqual(closed, true); + } + + const error = new Error('source'); + await assert.rejects(pipeTo({ + [Symbol.asyncIterator]() { + return { + next() { return Promise.reject(error); }, + return: common.mustNotCall(), + }; + }, + }, { + write: common.mustNotCall(), + writeSync: common.mustNotCall(), + fail: common.mustCall((reason) => assert.strictEqual(reason, error)), + }), error); + + await assert.rejects(pipeTo({ + [Symbol.asyncIterator]() { + return { next: async () => 42 }; + }, + }, { + write: common.mustNotCall(), + writeSync: common.mustNotCall(), + fail: common.mustCall(), + }), { code: 'ERR_INVALID_RETURN_VALUE' }); +} + Promise.all([ testPipeToSync(), testPipeTo(), @@ -412,5 +469,6 @@ Promise.all([ testPipeToSyncIterableWriteFallback(), testPipeToSyncIterableAsyncValue(), testPipeToSyncIterableWriteError(), + testPipeToAsyncIterableErrors(), testPipeToSourceNormalizationIndependentOfWriter(), ]).then(common.mustCall()); From 468fed1556fbc276b5a720ee8ac1a8381f137a44 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 01:57:35 +0000 Subject: [PATCH 32/58] stream: do not retain a promise per read in stream/iter share() share() read its source by racing the source's next() with a promise that cancel() resolves, created once for the share. That promise does not settle until the share is cancelled, so every race added reactions to it that were kept for as long as the share was in use: about 720 bytes per batch read, including three promises, two promise reactions and six closures. Sharing a source of 800,000 16-byte chunks between two consumers kept 574 MiB alive after garbage collection by the end. Instead, keep the resolver of the pending read in a field, and have cancel() settle that read with the same result as before. A read still settles as soon as the share is cancelled, without waiting for the source. What a share keeps alive no longer grows with the batches read: the peak heap sharing 200,000 16-byte chunks between two consumers drops from 180 MB to 9 MB, and the share is about 3 times faster. Assisted-by: OpenCode --- lib/internal/streams/iter/share.js | 42 ++++++++++++----- .../test-stream-iter-share-read-retention.js | 47 +++++++++++++++++++ 2 files changed, 77 insertions(+), 12 deletions(-) create mode 100644 test/parallel/test-stream-iter-share-read-retention.js diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index 937528b3929..484b923f025 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -12,7 +12,6 @@ const { PromisePrototypeThen, PromiseResolve, PromiseWithResolvers, - SafePromiseRace, SafeSet, Symbol, SymbolAsyncIterator, @@ -89,8 +88,9 @@ class ShareImpl { #cancelled = false; #pulling = false; #pullWaiters = []; - #cancelPromise; - #resolveCancel; + // Settles the pending read of the source with kShareCancelled when the + // share is cancelled, or null if no read is pending. + #cancelPendingRead = null; #cancelError = kNoShareError; #cachedMinCursor = 0; #cachedMinCursorConsumers = 0; @@ -101,9 +101,6 @@ class ShareImpl { constructor(source, options) { this.#source = source; this.#options = options; - const { promise, resolve } = PromiseWithResolvers(); - this.#cancelPromise = promise; - this.#resolveCancel = resolve; } get consumerCount() { @@ -280,8 +277,11 @@ class ShareImpl { this.#cancelError = reason; } - this.#resolveCancel(kShareCancelled); - this.#resolveCancel = null; + const cancelPendingRead = this.#cancelPendingRead; + if (cancelPendingRead !== null) { + this.#cancelPendingRead = null; + cancelPendingRead(kShareCancelled); + } try { const returnMethod = this.#sourceIterator?.return; @@ -416,10 +416,7 @@ class ShareImpl { } } - const result = await SafePromiseRace([ - this.#sourceIterator.next(), - this.#cancelPromise, - ]); + const result = await this.#readSource(); if (this.#cancelled || result === kShareCancelled) return; @@ -444,6 +441,27 @@ class ShareImpl { })(); } + // Read the next result of the source, settling early with kShareCancelled + // if the share is cancelled first. Racing the read with a promise settled + // by cancel() would add reactions to that promise on every read, which + // would be kept until the share is cancelled or collected. + #readSource() { + const next = this.#sourceIterator.next(); + const { promise, resolve, reject } = PromiseWithResolvers(); + this.#cancelPendingRead = resolve; + PromisePrototypeThen( + PromiseResolve(next), + (result) => { + if (this.#cancelPendingRead === resolve) this.#cancelPendingRead = null; + resolve(result); + }, + (error) => { + if (this.#cancelPendingRead === resolve) this.#cancelPendingRead = null; + reject(error); + }); + return promise; + } + #bufferBatch(batch) { const entry = createBatchEntry(batch); // 'drop-oldest' evicts whole entries. A single pulled batch can be much diff --git a/test/parallel/test-stream-iter-share-read-retention.js b/test/parallel/test-stream-iter-share-read-retention.js new file mode 100644 index 00000000000..d95fee33642 --- /dev/null +++ b/test/parallel/test-stream-iter-share-read-retention.js @@ -0,0 +1,47 @@ +// Flags: --experimental-stream-iter --expose-gc --no-warnings +'use strict'; + +// Reading the source of a share() must not leave objects behind for every +// read: what a share keeps alive must not grow with the number of batches +// read from its source. + +const common = require('../common'); +const assert = require('assert'); +const { queryObjects } = require('v8'); +const { share } = require('stream/iter'); + +async function test() { + const reads = 4000; + const chunk = new Uint8Array(1); + let i = 0; + const source = { + [Symbol.asyncIterator]() { return this; }, + async next() { + return i++ < reads ? + { done: false, value: [chunk] } : + { done: true, value: undefined }; + }, + }; + const shared = share(source, { backpressure: 'unbounded' }); + const a = shared.pull()[Symbol.asyncIterator](); + const b = shared.pull()[Symbol.asyncIterator](); + + const promises = []; + for (let n = 0; n < reads; n++) { + assert.strictEqual((await a.next()).done, false); + assert.strictEqual((await b.next()).done, false); + if (n === 1000 || n === 3000) { + globalThis.gc(); + promises.push(queryObjects(Promise, { format: 'count' })); + } + } + assert.strictEqual((await a.next()).done, true); + assert.strictEqual((await b.next()).done, true); + + // 2000 reads happened between the two counts. Retaining even one promise + // per read would add 2000. + const growth = promises[1] - promises[0]; + assert.ok(growth < 200, `${growth} promises retained across 2000 reads`); +} + +test().then(common.mustCall()); From 564ad6015ec9d3e2277883ab3d6b01f291c720ea Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 02:18:38 +0000 Subject: [PATCH 33/58] stream: return stream/iter from() results unchanged from from() from() returns sources known to yield normalized batches unchanged, such as the readable of a push() stream, but not its own results. Since pipeTo(), pull() and the consumers call from() on their source, a stream created with from() was normalized a second time when passed to them, adding a layer to every read and hiding the fast paths that pipeTo() uses to read sync and async sources. Mark the results of from() as validated sources, so that from() returns them unchanged, as the specification allows for async iterables created by the implementation that yield only normalized batches. Piping from(source) is now as fast as piping source: for a generator yielding 16-byte chunks one per batch, about 2.8 times faster when it is a sync generator and 1.7 times faster when it is an async one. Assisted-by: OpenCode --- doc/api/stream_iter.md | 6 ++--- lib/internal/streams/iter/from.js | 7 +++++ test/parallel/test-stream-iter-from-async.js | 28 ++++++++++++++++++++ 3 files changed, 38 insertions(+), 3 deletions(-) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 7e121aeb1c5..133affd828d 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -602,9 +602,9 @@ Objects implementing `Symbol.for('Stream.toAsyncStreamable')` or precedence over the iteration protocols (`Symbol.asyncIterator`, `Symbol.iterator`). -The readable of a [`push()`][] stream without transforms and the iterables -returned by [`fromReadable()`][] already yield normalized batches, so `from()` -returns them unchanged. +The readable of a [`push()`][] stream without transforms, the iterables +returned by [`fromReadable()`][], and the results of `from()` itself already +yield normalized batches, so `from()` returns them unchanged. ```mjs import { Buffer } from 'node:buffer'; diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 51d17c4e087..d882e2c5a78 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -180,6 +180,7 @@ function createNormalizationIterator(createIterator) { [SymbolAsyncIterator]() { return this; }, + [kValidatedSource]: true, }, null); } if (iterator[kNextSyncBatch] !== undefined) { @@ -202,6 +203,7 @@ function createNormalizationIterator(createIterator) { [SymbolAsyncIterator]() { return this; }, + [kValidatedSource]: true, }, null); } return ObjectSetPrototypeOf({ @@ -220,6 +222,7 @@ function createNormalizationIterator(createIterator) { [SymbolAsyncIterator]() { return this; }, + [kValidatedSource]: true, }, null); } @@ -229,6 +232,7 @@ function createNormalizationSource(createIterator) { [SymbolAsyncIterator]() { return createNormalizationIterator(createIterator); }, + [kValidatedSource]: true, }; } @@ -1349,6 +1353,7 @@ function from(input) { async *[SymbolAsyncIterator]() { yield [chunk]; }, + [kValidatedSource]: true, }; } @@ -1392,6 +1397,7 @@ function from(input) { async *[SymbolAsyncIterator]() { // Empty - yield nothing }, + [kValidatedSource]: true, }; } if (isUint8Array(input[0])) { @@ -1409,6 +1415,7 @@ function from(input) { } } }, + [kValidatedSource]: true, }; } } diff --git a/test/parallel/test-stream-iter-from-async.js b/test/parallel/test-stream-iter-from-async.js index 356f07353d7..e9199a3654d 100644 --- a/test/parallel/test-stream-iter-from-async.js +++ b/test/parallel/test-stream-iter-from-async.js @@ -801,7 +801,35 @@ async function testFromSyncSourceQueuesBehindReturn() { ]); } +// from() returns its own results unchanged: they already yield normalized +// batches, and normalizing them again would add a layer to every read. +async function testFromReturnsItsOwnResults() { + async function* asyncSource() { + yield 'a'; + yield [new Uint8Array([98])]; + } + + function* syncSource() { + yield 'c'; + yield new Uint8Array([100]); + } + const inputs = [ + [asyncSource(), 'ab'], + [syncSource(), 'cd'], + ['ef', 'ef'], + [[new Uint8Array([103]), new Uint8Array([104])], 'gh'], + [[], ''], + [{ [Symbol.for('Stream.toAsyncStreamable')]() { return 'ij'; } }, 'ij'], + ]; + for (const [input, expected] of inputs) { + const normalized = from(input); + assert.strictEqual(from(normalized), normalized); + assert.strictEqual(await text(from(from(normalized))), expected); + } +} + Promise.all([ + testFromReturnsItsOwnResults(), testFromString(), testFromAsyncGenerator(), testFromAsyncIteratorResultShapes(), From 626786e61ef4100273fcc0ce43b8fe5721a5050c Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 02:22:12 +0000 Subject: [PATCH 34/58] stream: create stream/iter from() values without async generators from() of a string, ArrayBuffer, ArrayBufferView or array of Uint8Arrays returns an iterable whose iterators are async generators yielding the value's batches. Creating and resuming an async generator costs a generator object, its frame and several promises, which dominates the cost of a short stream: creating and reading a stream of one 16-byte chunk spent a quarter of its time collecting garbage. Iterate such values with a small iterator that does what the generator did: each iteration starts over and yields the same batches, bounded as before, and return() and throw() end it, throw() rejecting with its argument. Creating and reading a stream of one 16-byte chunk with from() is about 1.8 times faster, with 900 bytes less allocated per stream. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 90 +++++++++++++------- test/parallel/test-stream-iter-from-async.js | 30 +++++++ 2 files changed, 91 insertions(+), 29 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index d882e2c5a78..a926eb982bd 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -1335,6 +1335,64 @@ function fromSync(input) { }; } +/** + * An async iterable yielding the chunks of `batch`, a Uint8Array[], in + * batches of at most FROM_BATCH_SIZE chunks, for from() of values that need + * no normalization. Each iterator does what an async generator yielding the + * batches would, without the generator object and the promises and frames + * of resuming it: next() gives the batches, then done; return() and throw() + * end the iteration, throw() rejecting with its argument. + * @param {Uint8Array[]} batch + * @returns {AsyncIterable} + */ +function createBatchSource(batch) { + return { + __proto__: null, + [SymbolAsyncIterator]() { + return createBatchIterator(batch); + }, + [kValidatedSource]: true, + }; +} + +function createBatchIterator(batch) { + let index = 0; + + // The next batch, or null once done. + function nextBatch() { + if (index >= batch.length) { + index = batch.length; + return null; + } + if (index === 0 && batch.length <= FROM_BATCH_SIZE) { + index = batch.length; + return batch; + } + const start = index; + index += FROM_BATCH_SIZE; + return ArrayPrototypeSlice(batch, start, index); + } + + return ObjectSetPrototypeOf({ + next() { + const value = nextBatch(); + return PromiseResolve(value === null ? + new IterResult(true, undefined) : new IterResult(false, value)); + }, + return(value) { + index = batch.length; + return PromiseResolve(new IterResult(true, value)); + }, + throw(error) { + index = batch.length; + return PromiseReject(error); + }, + [SymbolAsyncIterator]() { + return this; + }, + }, null); +} + /** * Create a ByteStreamReadable from a ByteInput or Streamable. * @param {string|ArrayBuffer|ArrayBufferView|Iterable|AsyncIterable} input @@ -1347,14 +1405,7 @@ function from(input) { // Check for primitives first (ByteInput) if (isPrimitiveChunk(input)) { - const chunk = primitiveToUint8Array(input); - return { - __proto__: null, - async *[SymbolAsyncIterator]() { - yield [chunk]; - }, - [kValidatedSource]: true, - }; + return createBatchSource([primitiveToUint8Array(input)]); } // Check toAsyncStreamable protocol (takes precedence over toStreamable and @@ -1392,31 +1443,12 @@ function from(input) { // the throughput benefit of batched processing. if (ArrayIsArray(input)) { if (input.length === 0) { - return { - __proto__: null, - async *[SymbolAsyncIterator]() { - // Empty - yield nothing - }, - [kValidatedSource]: true, - }; + return createBatchSource(input); } if (isUint8Array(input[0])) { const allUint8 = ArrayPrototypeEvery(input, isUint8Array); if (allUint8) { - const batch = input; - return { - __proto__: null, - async *[SymbolAsyncIterator]() { - if (batch.length <= FROM_BATCH_SIZE) { - yield batch; - } else { - for (let i = 0; i < batch.length; i += FROM_BATCH_SIZE) { - yield ArrayPrototypeSlice(batch, i, i + FROM_BATCH_SIZE); - } - } - }, - [kValidatedSource]: true, - }; + return createBatchSource(input); } } } diff --git a/test/parallel/test-stream-iter-from-async.js b/test/parallel/test-stream-iter-from-async.js index e9199a3654d..cfbb46fa9b6 100644 --- a/test/parallel/test-stream-iter-from-async.js +++ b/test/parallel/test-stream-iter-from-async.js @@ -828,7 +828,37 @@ async function testFromReturnsItsOwnResults() { } } +// from() of a value needing no normalization reads like an async generator +// yielding its batches: each iteration starts over, return() and throw() +// end it, and arrays are yielded in bounded batches. +async function testFromValueIteration() { + const chunk = new Uint8Array([1]); + const source = from(chunk); + for (let i = 0; i < 2; i++) { + const batches = await Array.fromAsync(source); + assert.deepStrictEqual(batches, [[chunk]]); + } + + let iterator = source[Symbol.asyncIterator](); + assert.deepStrictEqual({ ...await iterator.return(5) }, + { done: true, value: 5 }); + assert.strictEqual((await iterator.next()).done, true); + + iterator = source[Symbol.asyncIterator](); + assert.deepStrictEqual((await iterator.next()).value, [chunk]); + const error = new Error('thrown'); + await assert.rejects(iterator.throw(error), error); + assert.strictEqual((await iterator.next()).done, true); + + const chunks = Array.from({ length: 300 }, () => chunk); + const lengths = (await Array.fromAsync(from(chunks))) + .map((batch) => batch.length); + assert.deepStrictEqual(lengths, [128, 128, 44]); + assert.deepStrictEqual(await Array.fromAsync(from([])), []); +} + Promise.all([ + testFromValueIteration(), testFromReturnsItsOwnResults(), testFromString(), testFromAsyncGenerator(), From 9178f77751d2685938c3f454566f75b6b2d07350 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 02:45:17 +0000 Subject: [PATCH 35/58] stream: create stream/iter transform signals and options lazily A pull() pipeline with transforms creates an AbortController for the signal passed to its transforms, and each call of a stateless transform is passed a new options object holding it. Most transforms never read the signal, and a new options object for every call allocates on every batch unless the call is inlined. Keep the abort state of a pipeline in a PipelineAbort, which creates the AbortController when the signal is first read, already aborted with the same reason if the pipeline has been aborted; nothing can have listened to the signal before, so this cannot be told apart from a signal created with the pipeline. Code in the pipeline checks the abort state instead of reading the signal. Give each transform of a pipeline one options object, passed to every call of a stateless transform. A transform still cannot change the options another transform sees. `signal` is an accessor until it is first read or assigned, after which it is a data property. This departs from the specification, where TransformCallbackOptions is a dictionary, converted to a new object with a `signal` data property for every call. Calling four stateless transforms that do not read the signal over a sync source of 16-byte chunks, one per batch, is about 4% faster, and about 9% faster when the transforms are not inlined. Assisted-by: OpenCode --- doc/api/stream_iter.md | 11 +- lib/internal/streams/iter/pull.js | 141 +++++++++++-------- lib/internal/streams/iter/utils.js | 49 +++++++ test/parallel/test-stream-iter-pull-async.js | 74 ++++++++-- 4 files changed, 205 insertions(+), 70 deletions(-) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 133affd828d..67d1dffb91b 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -144,10 +144,13 @@ Both forms receive an `options` parameter with the following property: can check `signal.aborted` or listen for the `'abort'` event to perform early cleanup. -In `pull()`, stateless transforms receive a new `options` object for every -call, and stateful transforms one for the pipeline, so a transform can modify -its `options` without affecting other transforms. The object does not inherit -from `Object.prototype`. Transforms passed to [`pullSync()`][] receive no +Each transform of a pipeline receives its own `options` object, the same one +for every call of a stateless transform, so a transform can modify its +`options` without affecting other transforms. The object does not inherit +from `Object.prototype`. The signal is created when `options.signal` is +first read, so a pipeline whose transforms never read it does not create one; +until then, `signal` is an accessor property. It is the same signal for every +transform of the pipeline. Transforms passed to [`pullSync()`][] receive no `options`. The flush signal (`null`) is sent after the source ends, giving transforms diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 31bd3acf26a..1cb0276fc87 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -58,6 +58,7 @@ const { const { IterResult, + PipelineAbort, kActive, kDone, kNullOnceOption, @@ -581,19 +582,53 @@ function* createSyncPipeline(source, transforms) { // Async Pipeline Implementation // ============================================================================= -// The options object passed to async transforms. Stateless transforms get a -// new one for every call, so it is constructed rather than created as a -// `{ __proto__: null, signal }` literal (a dictionary-mode object). The -// prototype is frozen and has no %Object.prototype% in its chain, and the -// non-enumerable `constructor` lets util.inspect() print the options as -// `TransformOptions { signal }`. -function TransformOptions(signal) { - this.signal = signal; +// The options object passed to an async transform: one for each transform of +// a pipeline, passed to every call of a stateless transform. Its `signal` is +// the pipeline's signal, created by `abort` when first read (see +// PipelineAbort): until then it is an accessor, which replaces itself with a +// data property holding the signal. The prototype is frozen and has no +// %Object.prototype% in its chain, and the non-enumerable `constructor` lets +// util.inspect() print the options as `TransformOptions { signal }`. +function TransformOptions(abort) { + ObjectDefineProperty(this, 'signal', { + __proto__: null, + configurable: true, + enumerable: true, + get() { + const signal = abort.signal; + ObjectDefineProperty(this, 'signal', { + __proto__: null, + configurable: true, + enumerable: true, + value: signal, + writable: true, + }); + return signal; + }, + set(value) { + ObjectDefineProperty(this, 'signal', { + __proto__: null, + configurable: true, + enumerable: true, + value, + writable: true, + }); + }, + }); } TransformOptions.prototype = ObjectFreeze(ObjectDefineProperty( { __proto__: null }, 'constructor', { __proto__: null, value: TransformOptions })); +// The options for each function of a fused run of stateless transforms. +function createRunOptions(run, abort) { + const options = []; + for (let i = 0; i < run.length; i++) { + options[i] = new TransformOptions(abort); + } + return options; +} + /** * Close an async iterator as for await does: for a throw completion * (`quiet`), wait for it but ignore errors; otherwise reject on errors and @@ -650,13 +685,13 @@ async function* yieldFusedStatelessOutput(current) { * @param {Array} run * @param {number} index * @param {any} result - The result of `run[index]` - * @param {AbortSignal} signal + * @param {TransformOptions[]} runOptions - The options of each function * @yields {Uint8Array[]} */ -async function* continueFusedStatelessBatch(run, index, result, signal) { +async function* continueFusedStatelessBatch(run, index, result, runOptions) { let current; for (let i = index; i < run.length; i++) { - if (i !== index) result = run[i](current, new TransformOptions(signal)); + if (i !== index) result = run[i](current, runOptions[i]); if (isPromise(result)) result = await result; if (result === null) return; if (i === run.length - 1) { @@ -680,24 +715,24 @@ async function* continueFusedStatelessBatch(run, index, result, signal) { * flush each transform after all upstream data, including data emitted by * earlier flushes, has been processed by that transform. * @param {Array} run - * @param {AbortSignal} signal + * @param {TransformOptions[]} runOptions - The options of each function * @yields {Uint8Array[]} */ -async function* flushFusedStatelessAsyncTransforms(run, signal) { +async function* flushFusedStatelessAsyncTransforms(run, runOptions) { let pending = []; for (let i = 0; i < run.length; i++) { const next = []; for (let j = 0; j < pending.length; j++) { const pendingResult = appendTransformResultAsync( next, - run[i](pending[j], new TransformOptions(signal))); + run[i](pending[j], runOptions[i])); if (pendingResult !== undefined) { await pendingResult; } } const flushResult = appendTransformResultAsync( next, - run[i](null, new TransformOptions(signal))); + run[i](null, runOptions[i])); if (flushResult !== undefined) { await flushResult; } @@ -731,15 +766,14 @@ const kDelegated = Symbol('kDelegated'); * if any, and close the source unless it has ended; return() propagates * errors from closing it, throw() ignores them. * - * INVARIANT: This function accepts a signal, NOT a pre-built options object. - * A fresh TransformOptions object is created for each - * transform invocation to prevent cross-transform mutation. + * Each function of the run is passed its own options object on every call. * @param {AsyncIterable} source * @param {Array} run - Array of stateless transform functions - * @param {AbortSignal} signal - The pipeline's abort signal + * @param {PipelineAbort} abort - The pipeline's abort state * @returns {AsyncIterator} */ -function applyFusedStatelessAsyncTransforms(source, run, signal) { +function applyFusedStatelessAsyncTransforms(source, run, abort) { + const runOptions = createRunOptions(run, abort); let state = kStart; let iterator; let nextMethod; @@ -792,9 +826,9 @@ function applyFusedStatelessAsyncTransforms(source, run, signal) { function applyBatch(chunks) { let current = chunks; for (let i = 0; i < run.length; i++) { - const result = run[i](current, new TransformOptions(signal)); + const result = run[i](current, runOptions[i]); if (isPromise(result)) { - delegate = continueFusedStatelessBatch(run, i, result, signal); + delegate = continueFusedStatelessBatch(run, i, result, runOptions); return kDelegated; } if (result === null) return null; @@ -804,7 +838,7 @@ function applyFusedStatelessAsyncTransforms(source, run, signal) { } current = normalizeTransformResultFast(result); if (current === undefined) { - delegate = continueFusedStatelessBatch(run, i, result, signal); + delegate = continueFusedStatelessBatch(run, i, result, runOptions); return kDelegated; } if (current === null) return null; @@ -832,7 +866,7 @@ function applyFusedStatelessAsyncTransforms(source, run, signal) { } if (result.done) { sourceDone = true; - delegate = flushFusedStatelessAsyncTransforms(run, signal); + delegate = flushFusedStatelessAsyncTransforms(run, runOptions); return pullDelegate(); } value = result.value; @@ -988,7 +1022,7 @@ async function* applyStatefulAsyncTransform( * @yields {Uint8Array[]} */ async function* applyValidatedStatefulAsyncTransform( - source, transform, receiver, options) { + source, transform, receiver, options, abort) { const output = FunctionPrototypeCall( transform, receiver, source, options); for await (const batch of output) { @@ -999,7 +1033,7 @@ async function* applyValidatedStatefulAsyncTransform( // Check abort after the transform completes - without the // withFlushAsync wrapper there is no extra yield to give // the outer pipeline a chance to see the abort. - options.signal?.throwIfAborted(); + abort.throwIfAborted(); } /** @@ -1076,22 +1110,17 @@ async function* yieldFrom(source) { * Build the chain of transform layers of a pipeline. * @param {AsyncIterable} normalized * @param {Array} transforms - * @param {AbortSignal} transformSignal + * @param {PipelineAbort} abort - The pipeline's abort state * @returns {AsyncIterable} */ -function createAsyncTransformLayers(normalized, transforms, transformSignal) { +function createAsyncTransformLayers(normalized, transforms, abort) { // Apply transforms - fuse consecutive stateless transforms into a single // layer to avoid unnecessary async ticks. // - // INVARIANT: Each transform invocation MUST receive its own fresh options - // object (new TransformOptions(signal)). Transforms may mutate the options - // object, so sharing a single object across invocations would allow one - // transform to corrupt the options seen by another. The signal is shared - // across calls (mutations to it are acceptable), but the containing options - // object must be unique per call. This is enforced inside - // applyFusedStatelessAsyncTransforms and applyStatefulAsyncTransform, which - // accept the signal directly and create the options object per invocation. - // DO NOT pass a pre-built options object. + // Each transform gets its own options object, created here once and passed + // to every call of the transform: transforms may mutate the options + // object, and must not be able to change the options another transform + // sees. let current = normalized; let statelessRun = []; @@ -1101,13 +1130,13 @@ function createAsyncTransformLayers(normalized, transforms, transformSignal) { // Flush any accumulated stateless run before the stateful transform if (statelessRun.length > 0) { current = applyFusedStatelessAsyncTransforms(current, statelessRun, - transformSignal); + abort); statelessRun = []; } - const opts = new TransformOptions(transformSignal); + const opts = new TransformOptions(abort); if (transform[kValidatedTransform]) { current = applyValidatedStatefulAsyncTransform( - current, transform.transform, transform.receiver, opts); + current, transform.transform, transform.receiver, opts, abort); } else { current = applyStatefulAsyncTransform( current, transform.transform, transform.receiver, opts); @@ -1119,7 +1148,7 @@ function createAsyncTransformLayers(normalized, transforms, transformSignal) { // Flush remaining stateless run if (statelessRun.length > 0) { current = applyFusedStatelessAsyncTransforms(current, statelessRun, - transformSignal); + abort); } return current; } @@ -1130,9 +1159,9 @@ function createAsyncTransformLayers(normalized, transforms, transformSignal) { * layer for every batch (see applyFusedStatelessAsyncTransforms()). * * When started by the first next(), it checks `signal`, then creates the - * controller whose signal the transforms get, aborted when `signal` aborts. - * Each batch, and the end, is passed on only if the transforms' signal has - * not been aborted. If the pipeline fails, the transforms' signal is aborted + * abort state whose signal the transforms get, aborted when `signal` aborts. + * Each batch, and the end, is passed on only if the pipeline has not been + * aborted. If the pipeline fails, the transforms' signal is aborted * with the error; if it is stopped early by return() or throw(), the * transforms are closed and their signal is aborted. * @param {AsyncIterable} source @@ -1142,7 +1171,7 @@ function createAsyncTransformLayers(normalized, transforms, transformSignal) { */ function createAsyncTransformPipeline(source, transforms, signal) { let state = kStart; - let controller; + let abort; let abortHandler; let completed = false; let iterator; @@ -1151,10 +1180,10 @@ function createAsyncTransformPipeline(source, transforms, signal) { // What an async generator would do in its `finally` block. function cleanup() { - if (!completed && !controller.signal.aborted) { + if (!completed && !abort.aborted) { // Consumer stopped early or return() was called. // If a transform listener throws here, let it propagate. - controller.abort(lazyDOMException('Aborted', 'AbortError')); + abort.abort(lazyDOMException('Aborted', 'AbortError')); } // Clean up user signal listener to prevent holding controller alive if (signal && abortHandler) { @@ -1182,9 +1211,7 @@ function createAsyncTransformPipeline(source, transforms, signal) { state = kDone; try { try { - if (!controller.signal.aborted) { - abortSignal(controller.signal, error); - } + abort.abort(error); } finally { cleanup(); } @@ -1206,7 +1233,7 @@ function createAsyncTransformPipeline(source, transforms, signal) { // A transform can abort while completing without producing a final // batch, for example when an async flush resolves to null. In that // case there is no batch with which to observe the abort. - controller.signal.throwIfAborted(); + abort.throwIfAborted(); completed = true; return complete(new IterResult(true, undefined)); } @@ -1214,9 +1241,8 @@ function createAsyncTransformPipeline(source, transforms, signal) { } catch (error) { return fail(error); } - try { - controller.signal.throwIfAborted(); - } catch (error) { + if (abort.aborted) { + const error = abort.reason; // Like for await when its body throws: close the transforms first. return PromisePrototypeThen( closeAsyncIterator(iterator, true), () => fail(error)); @@ -1249,17 +1275,16 @@ function createAsyncTransformPipeline(source, transforms, signal) { } state = kActive; const normalized = yieldAbortable(source, signal); - // Create internal controller for transform cancellation. - controller = new AbortController(); + abort = new PipelineAbort(); if (signal) { abortHandler = () => { - abortSignal(controller.signal, signal.reason); + abort.abort(signal.reason); }; signal.addEventListener('abort', abortHandler, kNullOnceOption); } try { const current = createAsyncTransformLayers( - normalized, transforms, controller.signal); + normalized, transforms, abort); iterator = current[SymbolAsyncIterator](); nextMethod = iterator.next; } catch (error) { diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 6b40f5a2903..b4eb5e9c662 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -37,6 +37,10 @@ const { const { isSharedArrayBuffer, isUint8Array } = require('internal/util/types'); const { kWeakHandler } = require('internal/event_target'); +const { + AbortController, + abortSignal, +} = require('internal/abort_controller'); const { RingBuffer } = require('internal/streams/iter/ringbuffer'); const { @@ -110,6 +114,50 @@ function onSignalAbort(signal, handler) { } } +/** + * The abort state of a pipeline: what the AbortController of a pipeline in + * the specification holds, without creating an AbortController unless its + * signal is used. Most transforms never read the signal passed to them, so + * most pipelines never need one. When the signal is first read, it is + * created aborted with the same reason if the pipeline has been aborted: + * nothing can have listened to it before, so this cannot be told apart from + * a signal created with the pipeline. Code inside the pipeline checks + * `aborted` instead of reading the signal. + */ +class PipelineAbort { + aborted = false; + reason = undefined; + #controller = null; + + get signal() { + let controller = this.#controller; + if (controller === null) { + controller = this.#controller = new AbortController(); + if (this.aborted) abortSignal(controller.signal, this.reason); + } + return controller.signal; + } + + /** + * Abort the pipeline with `reason`, if it is not aborted already. Abort + * listeners on the signal, if it has been created, run synchronously, and + * errors they throw propagate. + * @param {any} reason + */ + abort(reason) { + if (this.aborted) return; + this.aborted = true; + this.reason = reason; + if (this.#controller !== null) { + abortSignal(this.#controller.signal, reason); + } + } + + throwIfAborted() { + if (this.aborted) throw this.reason; + } +} + /** * Wrap an async source so each pending read is abort-aware. * @param {AsyncIterable} source - The source to read from. @@ -942,6 +990,7 @@ module.exports = { kStart, PendingRequest, PendingWrite, + PipelineAbort, kMultiConsumerDefaultBudget, kNullOnceOption, kPushDefaultBudget, diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index 308e3a0ebb9..4c1bc0a7a14 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -553,7 +553,7 @@ async function testPipeToStringSource() { assert.strictEqual(data, 'hello-pipe'); } -// INVARIANT: Each transform invocation receives its own options object. +// Each transform receives its own options object. // A transform that mutates options must not affect subsequent transforms. async function testTransformOptionsNotShared() { const seen = []; @@ -575,41 +575,97 @@ async function testTransformOptionsNotShared() { assert.strictEqual(seen[1].mutated, undefined); } -// Stateless transforms get a new options object for every call, and stateful -// transforms one for the pipeline. The options object has only `signal`, does +// Each transform of a pipeline gets its own options object, passed to every +// call of a stateless transform. The options object has only `signal`, does // not inherit from Object.prototype, and its prototype is frozen, so that a -// transform cannot pass state to others through it. +// transform cannot pass state to others through it. The signal is the same +// for every transform and every call. async function testTransformOptionsShape() { const seen = []; + const statelessSeen = []; + let statefulOptions; const stateless = (chunks, options) => { seen.push(options); + statelessSeen.push(options); return chunks; }; const stateful = { async* transform(source, options) { seen.push(options); + statefulOptions = options; for await (const chunks of source) yield chunks; }, }; const ac = new AbortController(); await text(pull(from(['a', 'b']), stateless, stateful, { signal: ac.signal })); - // Stateless: one call per batch plus the flush call. + // Stateless: one call per batch plus the flush call; stateful: one call. + assert.strictEqual(statelessSeen.length, 3); + assert.strictEqual(new Set(statelessSeen).size, 1); assert.strictEqual(seen.length, 4); - assert.strictEqual(new Set(seen).size, seen.length); + assert.notStrictEqual(statefulOptions, statelessSeen[0]); for (const options of seen) { assert.strictEqual(options instanceof Object, false); assert.deepStrictEqual(Object.keys(options), ['signal']); assert.ok(options.signal instanceof AbortSignal); + assert.strictEqual(options.signal, seen[0].signal); assert.strictEqual(Object.isFrozen(Object.getPrototypeOf(options)), true); assert.match(inspect(options), /^TransformOptions \{ signal: /); } - assert.strictEqual(Object.getPrototypeOf(seen[0]), - Object.getPrototypeOf(seen[3])); + assert.strictEqual(Object.getPrototypeOf(statefulOptions), + Object.getPrototypeOf(statelessSeen[0])); assert.throws(() => { Object.getPrototypeOf(seen[0]).leak = true; }, TypeError); } +// `options.signal` can be assigned, as a data property could, whether or +// not it has been read. +async function testTransformOptionsSignalAssignable() { + const seen = []; + let calls = 0; + const transform = (chunks, options) => { + seen.push(options.signal); + options.signal = ++calls; + return chunks; + }; + const unread = (chunks, options) => { + options.signal = 'unread'; + seen.push(options.signal); + return chunks; + }; + await text(pull(from(['a', 'b']), transform, unread)); + assert.ok(seen[0] instanceof AbortSignal); + assert.deepStrictEqual(seen.slice(1), ['unread', 1, 'unread', 2, 'unread']); +} + +// The transforms' signal is the pipeline's: first read after the pipeline +// has been aborted, it is already aborted with the same reason, and read +// before, it is aborted when the pipeline is. +async function testTransformSignalReadAfterAbort() { + const reason = new Error('stop'); + let options; + const transform = (chunks, opts) => { + options = opts; + throw reason; + }; + await assert.rejects(text(pull(from('a'), transform)), reason); + assert.strictEqual(options.signal.aborted, true); + assert.strictEqual(options.signal.reason, reason); + assert.strictEqual(options.signal, options.signal); + + const ac = new AbortController(); + let signal; + const iterator = pull(from(['a', 'b']), (chunks, opts) => { + signal ??= opts.signal; + return chunks; + }, { signal: ac.signal })[Symbol.asyncIterator](); + await iterator.next(); + assert.strictEqual(signal.aborted, false); + ac.abort(reason); + assert.strictEqual(signal.aborted, true); + assert.strictEqual(signal.reason, reason); +} + // Run the uncaughtException test sequentially (it installs a global handler // that would interfere with concurrent tests). // Transform pipelines read and close the source like an async generator @@ -760,6 +816,8 @@ async function testTransformReturnClosesOutputAndSource() { testPipeToStringSource(), testTransformOptionsNotShared(), testTransformOptionsShape(), + testTransformSignalReadAfterAbort(), + testTransformOptionsSignalAssignable(), testTransformErrorClosesSource(), testTransformSourceErrorDoesNotCloseSource(), testPipeToTransformsStoppedEarly(), From d200edae4a9293c9a77d04217dada86fc08a76e1 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 03:07:24 +0000 Subject: [PATCH 36/58] stream: handle stream/iter pipeline aborts in one place Every pull() pipeline handled an abort in several layers. Without a signal, pull() created an AbortController for the consumer stopping early; with one, an AbortController followed by AbortSignal.any() and a reaction to every pull; and the pipeline created another AbortController for its transforms and read its source through an abortable iterator, which added a promise, a reaction and an iterator result to every batch. Handle an abort in the pipeline's iterator instead. Its pending pull is a promise of its own, which an abort rejects at once with the abort reason, after calling the source's return() without waiting for it. The source is read through a PipelineSource, which passes reads on as they are, rejects them once the pipeline is aborted, and calls the source's return() at most once. Each batch is checked against the pipeline's abort state, as before. This also fixes three cases. A pending pull now rejects when the signal aborts while the pipeline waits on a transform, not only on the source. When the signal aborts while no pull is pending, the source is now closed, as the specification requires before the next pull. And the pipeline's listener on the signal is now removed however the pipeline ends. When the consumer of pipeTo() stops early, the transforms' signal is now aborted before the transforms are closed, as it was for pull(). pipeTo() without a signal never stops the pipeline while a pull is pending, so its pulls stay the promises of the reactions to the transforms' results. For 16-byte chunks one per batch, iterating pull() with one or three stateless transforms is 8-17% faster, piping with transforms and a signal 5-8% faster, and iterating pull() with a signal about 1.55 times faster, with or without a transform. Assisted-by: OpenCode --- lib/internal/streams/iter/pull.js | 526 ++++++++++-------- .../test-stream-iter-pipeto-signal.js | 6 +- test/parallel/test-stream-iter-pull-async.js | 75 ++- 3 files changed, 365 insertions(+), 242 deletions(-) diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 1cb0276fc87..ae68d2b3c13 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -17,6 +17,7 @@ const { PromisePrototypeThen, PromiseReject, PromiseResolve, + PromiseWithResolvers, Symbol, SymbolAsyncIterator, SymbolIterator, @@ -39,11 +40,7 @@ const { isPromise, isUint8Array, } = require('internal/util/types'); -const { - AbortController, - AbortSignal, - abortSignal, -} = require('internal/abort_controller'); +const { kWeakHandler } = require('internal/event_target'); const { arrayBufferViewToUint8Array, @@ -1037,73 +1034,72 @@ async function* applyValidatedStatefulAsyncTransform( } /** - * Create an async pipeline from source through transforms. - * @param {AsyncIterable} source - * @param {Array} transforms - * @param {AbortSignal} [signal] - * @returns {AsyncIterator} + * The source of a pipeline, as its transforms read it: passes reads to the + * source's iterator as they are, so that reading costs nothing more, and + * lets the pipeline close the source's iterator when it is aborted, without + * waiting for a pending read to complete. The source's return() is called + * at most once. */ -function createAsyncPipeline(source, transforms, signal) { - if (transforms.length === 0) { - return createAsyncPipelineWithoutTransforms(source, signal); +class PipelineSource { + #source; + #abort; + #iterator = null; + #closed = false; + + constructor(source, abort) { + this.#source = source; + this.#abort = abort; } - return createAsyncTransformPipeline(source, transforms, signal); -} -/** - * The pipeline without transforms: what an async generator doing - * `signal?.throwIfAborted(); yield* yieldAbortable(source, signal);` would - * do. With a signal, it is written out by hand to avoid an async generator - * layer for every batch: the signal is checked by the first next(), then - * every call is passed to the iterator of yieldAbortable(), which queues - * them and completes as the generator would. - * @param {AsyncIterable} source - * @param {AbortSignal} [signal] - * @returns {AsyncIterator} - */ -function createAsyncPipelineWithoutTransforms(source, signal) { - if (signal === undefined) return yieldFrom(source); - let state = kStart; - let iterator; - return ObjectSetPrototypeOf({ - next() { - if (state === kStart) { - try { - // Check for abort - signal.throwIfAborted(); - iterator = yieldAbortable(source, signal)[SymbolAsyncIterator](); - } catch (error) { - state = kDone; - return PromiseReject(error); - } - state = kActive; - } else if (state === kDone) { - return PromiseResolve(new IterResult(true, undefined)); - } - return iterator.next(); - }, - return(value) { - if (state !== kActive) { - state = kDone; - return PromiseResolve(new IterResult(true, value)); - } - return iterator.return(value); - }, - throw(error) { - if (state !== kActive) { - state = kDone; - return PromiseReject(error); - } - return iterator.throw(error); - }, - [SymbolAsyncIterator]() { - return this; - }, - }, null); -} + [SymbolAsyncIterator]() { + return this; + } -async function* yieldFrom(source) { - yield* source; + next() { + if (this.#abort.aborted) return PromiseReject(this.#abort.reason); + if (this.#closed) return PromiseResolve(new IterResult(true, undefined)); + this.#iterator ??= this.#source[SymbolAsyncIterator](); + return this.#iterator.next(); + } + + return(value) { + const iterator = this.#iterator; + if (this.#closed || iterator === null) { + this.#closed = true; + return PromiseResolve(new IterResult(true, value)); + } + this.#closed = true; + const returnMethod = iterator.return; + if (returnMethod === undefined || returnMethod === null) { + return PromiseResolve(new IterResult(true, value)); + } + return FunctionPrototypeCall(returnMethod, iterator, value); + } + + throw(error) { + // As for await closes the source when its body throws. + return PromisePrototypeThen(this.return(), () => { throw error; }); + } + + // Close the source's iterator for an abort, without waiting, ignoring + // errors from closing it. + close() { + const iterator = this.#iterator; + if (this.#closed || iterator === null) { + this.#closed = true; + return; + } + this.#closed = true; + try { + const returnMethod = iterator.return; + if (returnMethod !== undefined && returnMethod !== null) { + markPromiseAsHandled(PromiseResolve( + FunctionPrototypeCall(returnMethod, iterator))); + } + } catch { + // The abort takes precedence over errors closing the source. + } + } } /** @@ -1154,163 +1150,280 @@ function createAsyncTransformLayers(normalized, transforms, abort) { } /** - * The pipeline through one or more transforms: an async iterator doing what - * an async generator would, written out by hand to avoid an async generator - * layer for every batch (see applyFusedStatelessAsyncTransforms()). + * The iterator of a pull() pipeline, also used by pipeTo() for transforms: an + * async iterator doing what an async generator would, written out by hand to + * avoid an async generator layer for every batch (see + * applyFusedStatelessAsyncTransforms()). + * + * The pipeline's abort state (a PipelineAbort) gives the transforms their + * signal. The pipeline is aborted when `signal` aborts, when the consumer + * stops early with return() or throw(), and with the error when it fails. + * An abort is handled here, once, instead of in every layer: + * - a pull that is pending rejects at once with the abort reason, after the + * source's return() is called, without waiting for it (any later result of + * the transforms is ignored); + * - the source is read through a PipelineSource, which rejects reads once + * the pipeline is aborted; + * - each batch, and the end, is passed on only if the pipeline has not been + * aborted. + * When `signal` aborts, the pull that observes it and every later pull reject + * with its reason, and the transforms are closed. When the consumer stops, + * later pulls report the end. * - * When started by the first next(), it checks `signal`, then creates the - * abort state whose signal the transforms get, aborted when `signal` aborts. - * Each batch, and the end, is passed on only if the pipeline has not been - * aborted. If the pipeline fails, the transforms' signal is aborted - * with the error; if it is stopped early by return() or throw(), the - * transforms are closed and their signal is aborted. + * A pending pull can only be rejected at once if it is a promise of its own, + * one more for every batch. Without a signal, pipeTo() never stops the + * pipeline while a pull is pending, so it passes `interruptible` as false + * and each pull is the promise of the reaction to the transforms' result. * @param {AsyncIterable} source * @param {Array} transforms * @param {AbortSignal} [signal] + * @param {boolean} [interruptible] * @returns {AsyncIterator} */ -function createAsyncTransformPipeline(source, transforms, signal) { +function createAsyncPipeline(source, transforms, signal, + interruptible = true) { let state = kStart; + // Whether `signal` has aborted the pipeline: pulls then reject. + let signalAborted = signal !== undefined && signal.aborted; let abort; - let abortHandler; - let completed = false; + let pipelineSource; let iterator; let nextMethod; + // The settling functions of the pending pull, if any. The operation queue + // runs one pull at a time. + let resolvePull = null; + let rejectPull = null; const operations = createOperationQueue(); - // What an async generator would do in its `finally` block. - function cleanup() { - if (!completed && !abort.aborted) { - // Consumer stopped early or return() was called. - // If a transform listener throws here, let it propagate. - abort.abort(lazyDOMException('Aborted', 'AbortError')); - } - // Clean up user signal listener to prevent holding controller alive - if (signal && abortHandler) { - signal.removeEventListener('abort', abortHandler); - } - } + const self = ObjectSetPrototypeOf({ + next() { return operations.run(doNext); }, + return(value) { + stop(lazyDOMException('Aborted', 'AbortError')); + return operations.run(doReturn, value); + }, + throw(error) { + stop(error); + return operations.run(doThrow, error); + }, + [SymbolAsyncIterator]() { return this; }, + }, null); - function finish(result) { - operations.settled(); - return result; + function onSignalAbort() { + if (state === kDone) return; + signalAborted = true; + // Close the transforms too; their errors are ignored, the pull rejects + // with the abort reason. + abortPipeline(signal.reason, true); } - function complete(result) { - state = kDone; - try { - cleanup(); - } finally { - operations.settled(); + function removeSignalListener() { + if (signal !== undefined && state !== kStart) { + signal.removeEventListener('abort', onSignalAbort); } - return result; } - // What an async generator would do in its `catch` block, then `finally`. - function fail(error) { + // Abort the pipeline: the transforms' signal at once, then close the + // source and reject the pending pull, if any. The signal can abort while + // the source or a transform is called, so these wait for a microtask, for + // that call to return. A result of the transforms that comes later is + // ignored. + function abortPipeline(reason, closeTransforms) { state = kDone; - try { - try { - abort.abort(error); - } finally { - cleanup(); + removeSignalListener(); + abort.abort(reason); + const reject = rejectPull; + resolvePull = rejectPull = null; + PromisePrototypeThen(kResolvedPromise, () => { + pipelineSource.close(); + if (reject !== null) { + operations.settled(); + reject(reason); } - } finally { - operations.settled(); + if (closeTransforms) { + markPromiseAsHandled(closeAsyncIterator(iterator, true)); + } + }); + } + + // The consumer stops early: abort the pipeline at once, so that the + // transforms see their signal aborted before they are closed and a pending + // pull rejects; doReturn() or doThrow() then closes the transforms. + function stop(reason) { + if (state !== kActive) return; + if (rejectPull !== null) { + abortPipeline(reason, false); + state = kActive; + } else { + abort.abort(reason); } - throw error; } - function onResult(result) { - let value; - try { - if ((typeof result !== 'object' && typeof result !== 'function') || - result === null) { - throw new ERR_INVALID_RETURN_VALUE( - 'an object', 'iterator.next()', result); - } - if (result.done) { - // A transform can abort while completing without producing a final - // batch, for example when an async flush resolves to null. In that - // case there is no batch with which to observe the abort. - abort.throwIfAborted(); - completed = true; - return complete(new IterResult(true, undefined)); - } - value = result.value; - } catch (error) { - return fail(error); + // The end of the pipeline, for an error: what an async generator would do + // in its `catch` block, then `finally`. + function failed(error) { + state = kDone; + removeSignalListener(); + abort.abort(error); + } + + // What a pull settles with, given the result of the transforms: a result, + // or a promise for one, or it throws. The pipeline is done if it throws or + // the result is done. + function handleResult(result) { + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); + } + if (result.done) { + // A transform can abort while completing without producing a final + // batch, for example when an async flush resolves to null. In that + // case there is no batch with which to observe the abort. + abort.throwIfAborted(); + state = kDone; + removeSignalListener(); + return new IterResult(true, undefined); } if (abort.aborted) { const error = abort.reason; // Like for await when its body throws: close the transforms first. return PromisePrototypeThen( - closeAsyncIterator(iterator, true), () => fail(error)); + closeAsyncIterator(iterator, true), () => { throw error; }); } - return finish(new IterResult(false, value)); + return new IterResult(false, result.value); } - function pullTransforms() { - let promise; + function onResult(result) { + const resolve = resolvePull; + if (resolve === null) return; // Settled by an abort. + const reject = rejectPull; + let settled; try { - promise = PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)); + settled = handleResult(result); } catch (error) { - return fail(error); + onError(error); + return; + } + if (isPromise(settled)) { + PromisePrototypeThen(settled, undefined, (error) => { + if (rejectPull === reject) onError(error); + }); + return; + } + resolvePull = rejectPull = null; + operations.settled(); + resolve(settled); + } + + function onError(error) { + const reject = rejectPull; + if (reject === null) return; // Settled by an abort. + resolvePull = rejectPull = null; + operations.settled(); + failed(error); + reject(error); + } + + // onResult() and onError() for a pull that is the promise of the reaction + // to the transforms' result (see `interruptible`). + function onDirectResult(result) { + let settled; + try { + settled = handleResult(result); + } catch (error) { + return onDirectError(error); } - return PromisePrototypeThen(promise, onResult, fail); + if (isPromise(settled)) { + return PromisePrototypeThen(settled, undefined, onDirectError); + } + operations.settled(); + return settled; + } + + function onDirectError(error) { + operations.settled(); + failed(error); + throw error; } function doNext() { + if (signalAborted) { + operations.settled(); + return PromiseReject(signal.reason); + } if (state === kDone) { - return PromiseResolve(finish(new IterResult(true, undefined))); + operations.settled(); + return PromiseResolve(new IterResult(true, undefined)); } if (state === kStart) { + state = kActive; + abort = new PipelineAbort(); + pipelineSource = new PipelineSource(source, abort); + if (signal !== undefined) { + signal.addEventListener('abort', onSignalAbort, + { __proto__: null, once: true, + [kWeakHandler]: self }); + } try { - // Check for abort - signal?.throwIfAborted(); + if (transforms.length === 0) { + iterator = pipelineSource; + } else { + iterator = createAsyncTransformLayers( + pipelineSource, transforms, abort)[SymbolAsyncIterator](); + } + nextMethod = iterator.next; } catch (error) { state = kDone; + removeSignalListener(); operations.settled(); + abort.abort(error); return PromiseReject(error); } - state = kActive; - const normalized = yieldAbortable(source, signal); - abort = new PipelineAbort(); - if (signal) { - abortHandler = () => { - abort.abort(signal.reason); - }; - signal.addEventListener('abort', abortHandler, kNullOnceOption); - } + } + if (!interruptible) { + let next; try { - const current = createAsyncTransformLayers( - normalized, transforms, abort); - iterator = current[SymbolAsyncIterator](); - nextMethod = iterator.next; + next = PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)); } catch (error) { - try { - fail(error); - } catch (failure) { - return PromiseReject(failure); - } + operations.settled(); + failed(error); + return PromiseReject(error); } - } + return PromisePrototypeThen(next, onDirectResult, onDirectError); + } + // The pull is pending from now on: the signal can abort while the + // transforms or the source are called. + const { promise, resolve, reject } = PromiseWithResolvers(); + resolvePull = resolve; + rejectPull = reject; + let next; try { - return PromiseResolve(pullTransforms()); + next = PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)); } catch (error) { - return PromiseReject(error); + onError(error); + return promise; } + PromisePrototypeThen(next, onResult, onError); + return promise; } function doReturn(value) { const result = new IterResult(true, value); if (state !== kActive) { state = kDone; - return PromiseResolve(finish(result)); + operations.settled(); + return PromiseResolve(result); } state = kDone; + removeSignalListener(); return PromisePrototypeThen( - closeAsyncIterator(iterator, false), () => complete(result), fail); + closeAsyncIterator(iterator, false), () => { + operations.settled(); + return result; + }, (error) => { + operations.settled(); + throw error; + }); } function doThrow(error) { @@ -1320,16 +1433,15 @@ function createAsyncTransformPipeline(source, transforms, signal) { return PromiseReject(error); } state = kDone; + removeSignalListener(); return PromisePrototypeThen( - closeAsyncIterator(iterator, true), () => fail(error)); + closeAsyncIterator(iterator, true), () => { + operations.settled(); + throw error; + }); } - return ObjectSetPrototypeOf({ - next() { return operations.run(doNext); }, - return(value) { return operations.run(doReturn, value); }, - throw(error) { return operations.run(doThrow, error); }, - [SymbolAsyncIterator]() { return this; }, - }, null); + return self; } // ============================================================================= @@ -1381,76 +1493,15 @@ function pull(source, ...args) { return { __proto__: null, [SymbolAsyncIterator]() { - if (signal === undefined) { - const controller = new AbortController(); - const iterator = createAsyncPipeline( - normalized, transforms, controller.signal); - return ObjectSetPrototypeOf({ - next(value) { - return iterator.next(value); - }, - return(value) { - controller.abort(lazyDOMException('Aborted', 'AbortError')); - return iterator.return(value); - }, - throw(error) { - abortSignal(controller.signal, error); - return iterator.throw(error); - }, - [SymbolAsyncIterator]() { - return this; - }, - }, null); + // Without transforms or a signal, the pipeline is the source. + if (transforms.length === 0 && signal === undefined) { + return normalized[SymbolAsyncIterator](); } - return createAbortablePullIterator(normalized, transforms, signal); + return createAsyncPipeline(normalized, transforms, signal); }, }; } -// Once `signal` aborts the pipeline, the pull that observed it rejects and -// so does every later pull, with the abort reason. That includes the case of -// an already-aborted signal, where the pipeline is never started. A plain -// async generator would instead complete after throwing, so later pulls -// would report a clean end of the stream. -function createAbortablePullIterator(source, transforms, signal) { - let aborted = signal.aborted; - let controller; - let iterator; - if (!aborted) { - controller = new AbortController(); - const iteratorSignal = AbortSignal.any([signal, controller.signal]); - iterator = createAsyncPipeline(source, transforms, iteratorSignal); - } - - function onRejected(error) { - if (signal.aborted) aborted = true; - throw error; - } - - return ObjectSetPrototypeOf({ - next(value) { - if (aborted) return PromiseReject(signal.reason); - return PromisePrototypeThen(iterator.next(value), undefined, onRejected); - }, - return(value) { - if (aborted) { - return PromiseResolve(new IterResult(true, value)); - } - controller.abort(lazyDOMException('Aborted', 'AbortError')); - return iterator.return(value); - }, - throw(error) { - if (aborted) return PromiseReject(error); - abortSignal(controller.signal, error); - return PromisePrototypeThen(iterator.throw(error), undefined, - onRejected); - }, - [SymbolAsyncIterator]() { - return this; - }, - }, null); -} - // Keep ownership of a bonded consumer outside the transform pipeline so it can // be detached even when the pipeline never starts or terminates early. function pullWithConsumerCleanup(source, transforms, signal) { @@ -1854,7 +1905,8 @@ async function pipeTo(source, ...args) { } } } else { - const pipeline = createAsyncPipeline(normalized, transforms, signal); + const pipeline = createAsyncPipeline( + normalized, transforms, signal, signal !== undefined); if (signal) { for await (const batch of pipeline) { diff --git a/test/parallel/test-stream-iter-pipeto-signal.js b/test/parallel/test-stream-iter-pipeto-signal.js index 3645db1a92f..48385710d8e 100644 --- a/test/parallel/test-stream-iter-pipeto-signal.js +++ b/test/parallel/test-stream-iter-pipeto-signal.js @@ -8,7 +8,7 @@ const common = require('../common'); const assert = require('assert'); const { setTimeout } = require('timers/promises'); const { getEventListeners } = require('events'); -const { bytes, pipeTo, from } = require('stream/iter'); +const { bytes, pipeTo, pull, from } = require('stream/iter'); async function testPipeToPreAbortedSignalFailsWriter() { const reason = new Error('already aborted'); @@ -165,9 +165,13 @@ async function testSignalAbortedWhileReadingSource() { // The signal can abort while the source is producing a value; the read // must still reject with the abort reason, and the source be closed, even // if the value never comes. + const identity = (chunks) => chunks; for (const consume of [ (source, signal) => pipeTo(source, { write() {} }, { signal }), + (source, signal) => pipeTo(source, identity, { write() {} }, { signal }), (source, signal) => bytes(source, { signal }), + (source, signal) => bytes(pull(source, { signal })), + (source, signal) => bytes(pull(source, identity, { signal })), ]) { const ac = new AbortController(); const reason = new Error('aborted while reading'); diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index 4c1bc0a7a14..c8293257712 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -18,6 +18,7 @@ const { const { setImmediate } = require('timers/promises'); const { inspect } = require('util'); +const { getEventListeners } = require('events'); async function testPullIdentity() { const data = await text(pull(from('hello-async'))); @@ -365,6 +366,68 @@ async function testPullSignalAbortWithTransformWhileSourceNextPending() { await assert.rejects(next, { name: 'AbortError' }); } +// An abort rejects a pending pull at once, wherever the pipeline is waiting: +// here, on a transform that never settles. +async function testPullSignalAbortWhileTransformPending() { + const ac = new AbortController(); + let transformSignal; + const iter = pull(from('a'), (chunks, options) => { + transformSignal = options.signal; + return new Promise(() => {}); + }, { signal: ac.signal })[Symbol.asyncIterator](); + const next = iter.next(); + await setImmediate(); + const reason = new Error('stop'); + ac.abort(reason); + await assert.rejects(next, reason); + assert.strictEqual(transformSignal.reason, reason); + await assert.rejects(iter.next(), reason); +} + +// When the signal aborts while no pull is pending, the source is closed once, +// and the next pull rejects. +async function testPullSignalAbortWhileIdleClosesSource() { + const log = []; + const ac = new AbortController(); + const iter = pull(createLoggedSource(log, [1, 2, 3]), (chunks) => chunks, + { signal: ac.signal })[Symbol.asyncIterator](); + assert.strictEqual((await iter.next()).done, false); + const reason = new Error('stop'); + ac.abort(reason); + await setImmediate(); + assert.deepStrictEqual(log, ['next 0', 'return']); + await assert.rejects(iter.next(), reason); + assert.strictEqual((await iter.return()).done, true); + assert.deepStrictEqual(log, ['next 0', 'return']); +} + +// The pipeline's listener on the signal is removed once the pipeline is +// done, however it ends. +async function testPullSignalListenerRemoved() { + const identity = (chunks) => chunks; + const ac = new AbortController(); + const { signal } = ac; + await text(pull(from(['a', 'b']), identity, { signal })); + assert.strictEqual(getEventListeners(signal, 'abort').length, 0); + + let iter = pull(from(['a', 'b']), identity, { signal })[Symbol.asyncIterator](); + await iter.next(); + assert.strictEqual(getEventListeners(signal, 'abort').length, 1); + await iter.return(); + assert.strictEqual(getEventListeners(signal, 'abort').length, 0); + + iter = pull(from(['a', 'b']), () => { throw new Error('failed'); }, + { signal })[Symbol.asyncIterator](); + await assert.rejects(iter.next(), /failed/); + assert.strictEqual(getEventListeners(signal, 'abort').length, 0); + + iter = pull(from(['a', 'b']), identity, { signal })[Symbol.asyncIterator](); + await iter.next(); + ac.abort(); + assert.strictEqual(getEventListeners(signal, 'abort').length, 0); + await assert.rejects(iter.next(), { name: 'AbortError' }); +} + // Pull consumer break (return()) cleans up transform signal async function testPullConsumerBreakCleanup() { let signalAborted = false; @@ -744,8 +807,9 @@ async function testTransformSourceErrorDoesNotCloseSource() { } async function testPipeToTransformsStoppedEarly() { - // When the writer fails, the transforms are closed, closing the source, and - // their signal is aborted. + // When the writer fails, the transforms' signal is aborted, and the + // transforms are closed, closing the source, as when the consumer of pull() + // stops early. const log = []; let writes = 0; await assert.rejects(pipeTo( @@ -755,8 +819,8 @@ async function testPipeToTransformsStoppedEarly() { }, }), /write failed/); assert.deepStrictEqual(log, [ - 'next 0', 'transform 1', 'next 1', 'transform 2', 'return', - 'abort: Aborted', + 'next 0', 'transform 1', 'next 1', 'transform 2', 'abort: Aborted', + 'return', ]); } @@ -817,6 +881,9 @@ async function testTransformReturnClosesOutputAndSource() { testTransformOptionsNotShared(), testTransformOptionsShape(), testTransformSignalReadAfterAbort(), + testPullSignalAbortWhileTransformPending(), + testPullSignalAbortWhileIdleClosesSource(), + testPullSignalListenerRemoved(), testTransformOptionsSignalAssignable(), testTransformErrorClosesSource(), testTransformSourceErrorDoesNotCloseSource(), From 0ccea7aef7518039d20cd38310fc28a9abfe1939 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 03:19:03 +0000 Subject: [PATCH 37/58] stream: keep stream/iter pipeTo() fast paths with a signal Without transforms, pipeTo() with a signal read its source through an abortable iterator, which adds a promise, a reaction and an iterator result to every batch, and gave up reading sync sources synchronously. Every write(), writev() and end() was passed a new options object. Instead, read the source as without a signal, synchronously when possible, checking between batches whether the signal has aborted, and race the whole pipe once with the abort: one listener and one promise for the pipe. When the signal aborts, the pipe rejects at once with the abort reason, after calling the source's return() without waiting for it, as a pull() pipeline with the signal rejects its pending pull, and no other batch is read or written. Async sources are read with next(), which return() can cancel while it is pending. Pass the same options object to every write(), writev() and end(). This departs from the specification, where WriteOptions is a dictionary, converted to a new object for every call. For 16-byte chunks one per batch, piping a sync generator with a signal is about 3.9 times faster, as fast as without one, and piping an async generator with a signal about 1.55 times faster. Assisted-by: OpenCode --- doc/api/stream_iter.md | 4 +- lib/internal/streams/iter/pull.js | 112 +++++++++++++++--- .../test-stream-iter-pipeto-signal.js | 81 +++++++++++++ 3 files changed, 178 insertions(+), 19 deletions(-) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 67d1dffb91b..02c0719e943 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -674,7 +674,9 @@ added: * `writer` {Object} Destination with `write(chunk)` method. * `options` {Object} * `signal` {AbortSignal} Abort the pipeline. Aborting fails the destination - writer unless `preventFail` is `true`. + writer unless `preventFail` is `true`. The signal is passed to the + writer's `write()`, `writev()` and `end()` in an options object, the same + object for every call. * `preventClose` {boolean} If `true`, do not call `writer.end()` when the source ends. **Default:** `false`. * `preventFail` {boolean} If `true`, do not call `writer.fail()` on diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index ae68d2b3c13..c80068a7a25 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -73,7 +73,6 @@ const { toUint8Array, validateBatchEntry, validateByteView, - yieldAbortable, } = require('internal/streams/iter/utils'); const { converters, @@ -1737,6 +1736,11 @@ async function pipeTo(source, ...args) { const hasWriteSync = typeof writer.writeSync === 'function'; const normalized = from(source); + // The options passed to every write(), writev() and end(). + const writeOptions = signal ? { __proto__: null, signal } : undefined; + // Without transforms, whether `signal` has aborted the pipe: the batches + // are then no longer read or written (see raceSignal()). + let signalAborted = false; let totalBytes = 0; const hasWritev = typeof writer.writev === 'function'; @@ -1757,9 +1761,7 @@ async function pipeTo(source, ...args) { } validateByteView(view); } - const result = writer.write( - validateByteView(view), - signal ? { __proto__: null, signal } : undefined); + const result = writer.write(validateByteView(view), writeOptions); if (result !== undefined) { await result; } @@ -1792,10 +1794,11 @@ async function pipeTo(source, ...args) { // ignoring errors closing it. async function pipeSyncSource(iterator) { for (;;) { + if (signalAborted) return; let batch = iterator[kNextSyncBatch](); if (batch === undefined) { const result = await iterator.next(); - if (result.done) return; + if (result.done || signalAborted) return; batch = result.value; } else if (batch === null) { return; @@ -1804,16 +1807,87 @@ async function pipeTo(source, ...args) { const p = writeBatch(batch); if (p) await p; } catch (error) { - try { - await iterator.return(); - } catch { - // The error writing the batch is thrown. + if (!signalAborted) { + try { + await iterator.return(); + } catch { + // The error writing the batch is thrown. + } + } + throw error; + } + } + } + + // Write the batches of a source read with next(), as the for await...of + // loop below does, stopping once `signal` aborts. + async function pipeSourceWithSignal(iterator) { + for (;;) { + const result = await iterator.next(); + if (result.done || signalAborted) return; + try { + const p = writeBatch(result.value); + if (p) await p; + } catch (error) { + if (!signalAborted) { + try { + await iterator.return(); + } catch { + // The error writing the batch is thrown. + } } throw error; } } } + // Wait for `pipe(iterator)` to write the batches of `iterator`, the + // normalized source, unless `signal` aborts first. On an abort, the source + // is closed, without waiting, and the abort reason is thrown at once, as a + // pull() pipeline with `signal` would reject its pending pull; `piped` + // then stops before reading or writing another batch. This takes one + // listener and one promise for the whole pipe, instead of a layer reading + // the source for every batch. The listener is added first, as the signal + // can abort while the source is read. + function raceSignal(iterator, pipe) { + const { promise, resolve, reject } = PromiseWithResolvers(); + let settled = false; + function closeAndReject() { + try { + const returnMethod = iterator.return; + if (typeof returnMethod === 'function') { + markPromiseAsHandled(PromiseResolve( + FunctionPrototypeCall(returnMethod, iterator))); + } + } catch { + // The abort takes precedence over errors closing the source. + } + reject(signal.reason); + } + + function onAbort() { + if (settled) return; + settled = true; + signalAborted = true; + // The signal can abort while the source is called: close it once that + // call has returned. + PromisePrototypeThen(kResolvedPromise, closeAndReject); + } + signal.addEventListener('abort', onAbort, kNullOnceOption); + PromisePrototypeThen(pipe(iterator), (value) => { + if (settled) return; + settled = true; + signal.removeEventListener('abort', onAbort); + resolve(value); + }, (error) => { + if (settled) return; + settled = true; + signal.removeEventListener('abort', onAbort); + reject(error); + }); + return promise; + } + // Write the batches of an async source as the for await...of loop below // does. Nothing cancels the normalization while a batch is read: return() // is only called after an error writing one. @@ -1856,8 +1930,8 @@ async function pipeTo(source, ...args) { if (!hasWritevSync || !writer.writevSync(validateBatchEntry(entry))) { validateBatchEntry(entry); - const opts = signal ? { __proto__: null, signal } : undefined; - const writevResult = writer.writev(validateBatchEntry(entry), opts); + const writevResult = writer.writev( + validateBatchEntry(entry), writeOptions); if (writevResult === undefined) { validateBatchEntry(entry); totalBytes += entry.byteLength; @@ -1889,11 +1963,12 @@ async function pipeTo(source, ...args) { if (transforms.length === 0) { // Fast path: no transforms - iterate normalized source directly if (signal) { - for await (const batch of yieldAbortable(normalized, signal)) { - signal.throwIfAborted(); - const p = writeBatch(batch); - if (p) await p; - } + // Batches are read synchronously when possible, as without a signal. + // Async reads go through next(), which return() can cancel while it + // is pending, unlike kNextUncancellable. + const iterator = normalized[SymbolAsyncIterator](); + await raceSignal(iterator, iterator[kNextSyncBatch] !== undefined ? + pipeSyncSource : pipeSourceWithSignal); } else if (normalized[kNextSyncBatch] !== undefined) { await pipeSyncSource(normalized); } else if (normalized[kNextUncancellable] !== undefined) { @@ -1909,8 +1984,9 @@ async function pipeTo(source, ...args) { normalized, transforms, signal, signal !== undefined); if (signal) { + // The pipeline rejects its pending pull when `signal` aborts. for await (const batch of pipeline) { - signal.throwIfAborted(); + if (signal.aborted) throw signal.reason; const p = writeBatch(batch); if (p) await p; } @@ -1924,7 +2000,7 @@ async function pipeTo(source, ...args) { if (!options.preventClose) { if (!hasEndSync || writer.endSync() < 0) { - await writer.end?.(signal ? { __proto__: null, signal } : undefined); + await writer.end?.(writeOptions); } } } catch (error) { diff --git a/test/parallel/test-stream-iter-pipeto-signal.js b/test/parallel/test-stream-iter-pipeto-signal.js index 48385710d8e..57fa9fed530 100644 --- a/test/parallel/test-stream-iter-pipeto-signal.js +++ b/test/parallel/test-stream-iter-pipeto-signal.js @@ -195,6 +195,84 @@ async function testSignalAbortedWhileReadingSource() { } } +// write(), writev() and end() are all passed the same options object. +async function testWriteOptionsShared() { + const ac = new AbortController(); + const seen = []; + const writer = { + async write(chunk, options) { seen.push(options); }, + async writev(chunks, options) { seen.push(options); }, + async end(options) { seen.push(options); }, + }; + async function* source() { + yield [new Uint8Array([1])]; + yield [new Uint8Array([2]), new Uint8Array([3])]; + yield [new Uint8Array([4])]; + } + await pipeTo(source(), writer, { signal: ac.signal }); + assert.strictEqual(seen.length, 4); + assert.strictEqual(new Set(seen).size, 1); + assert.strictEqual(seen[0].signal, ac.signal); +} + +// The signal aborting while a chunk is written synchronously stops the pipe: +// no other chunk is written, the source is closed, and the pipe rejects with +// the abort reason. +async function testAbortDuringWriteSync() { + for (const sync of [true, false]) { + const ac = new AbortController(); + const reason = new Error('abort in writeSync'); + const written = []; + let closed = false; + function* syncSource() { + try { + for (let i = 0; i < 5; i++) yield [new Uint8Array([i])]; + } finally { + closed = true; + } + } + + async function* asyncSource() { + try { + for (let i = 0; i < 5; i++) yield [new Uint8Array([i])]; + } finally { + closed = true; + } + } + const writer = { + write: common.mustNotCall(), + writeSync(chunk) { + written.push(chunk[0]); + if (chunk[0] === 1) ac.abort(reason); + return true; + }, + fail: common.mustCall((error) => assert.strictEqual(error, reason)), + }; + await assert.rejects( + pipeTo(sync ? syncSource() : asyncSource(), writer, + { signal: ac.signal }), + reason); + assert.deepStrictEqual(written, [0, 1]); + await setTimeout(1); + assert.strictEqual(closed, true); + } +} + +// The signal aborting while a write is pending rejects the pipe at once, +// even if the write never completes. +async function testAbortWhileWritePending() { + const ac = new AbortController(); + const reason = new Error('abort while writing'); + const writer = { + write() { + ac.abort(reason); + return new Promise(() => {}); + }, + }; + await assert.rejects(pipeTo(from('a'), writer, { signal: ac.signal }), + reason); +} + async function testSignalListenersRemoved() { // No abort listener is left on the signal once reading completes, fails // or is aborted. @@ -226,4 +304,7 @@ Promise.all([ testPipeToLiveSignalWithTransformsCompletes(), testSignalAbortedWhileReadingSource(), testSignalListenersRemoved(), + testWriteOptionsShared(), + testAbortDuringWriteSync(), + testAbortWhileWritePending(), ]).then(common.mustCall()); From a9e4a40add3d29d6e64e2632cb3520a865509b51 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 03:25:26 +0000 Subject: [PATCH 38/58] stream: read stream/iter consumer sources without a layer per signal bytes(), text(), arrayBuffer() and array() with a signal read their source through an abortable iterator, which adds a promise, a reaction and an iterator result to every batch. Read the source as for await...of does, checking between batches whether the signal has aborted, and race the whole read once with the abort, as pipeTo() does without transforms: move that race to raceSignal() in utils.js and use it for both. When the signal aborts, the consumer rejects at once with the abort reason, after calling the source's return() without waiting for it, and the source is not read further. For 16-byte chunks one per batch, array() of a sync generator with a signal is about 1.9 times faster, and of an async generator about 1.35 times faster: about as fast as without a signal. Assisted-by: OpenCode --- lib/internal/streams/iter/consumers.js | 33 ++++++++-- lib/internal/streams/iter/pull.js | 65 +++---------------- lib/internal/streams/iter/utils.js | 62 ++++++++++++++++++ .../test-stream-iter-consumers-bytes.js | 21 ++++++ 4 files changed, 122 insertions(+), 59 deletions(-) diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index 459eedb49bb..7cdc7d38e82 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -57,6 +57,7 @@ const { kNullOnceOption, concatBytes, getProtocolMethod, + raceSignal, recordChunk, validateRecordedChunks, yieldAbortable, @@ -149,10 +150,7 @@ async function collectAsync(source, signal, limit) { // Slow path: with signal or limit checks let totalBytes = 0; - const iterable = signal ? yieldAbortable(normalized, signal) : normalized; - - for await (const batch of iterable) { - signal?.throwIfAborted(); + function recordBatch(batch) { for (let i = 0; i < batch.length; i++) { totalBytes += recordChunk(chunks, checks, batch[i]); if (limit !== undefined && totalBytes > limit) { @@ -161,6 +159,33 @@ async function collectAsync(source, signal, limit) { } } + if (!signal) { + for await (const batch of normalized) recordBatch(batch); + return validateRecordedChunks(chunks, checks); + } + + // Read as for await...of does, stopping once the signal aborts; see + // raceSignal(). + const state = { __proto__: null, aborted: false }; + async function consume(iterator) { + for (;;) { + const result = await iterator.next(); + if (result.done || state.aborted) return; + try { + recordBatch(result.value); + } catch (error) { + // As for await closes the source when its body throws, ignoring + // errors from closing it. + try { + await iterator.return?.(); + } catch { + // The error recording the batch is thrown. + } + throw error; + } + } + } + await raceSignal(normalized[SymbolAsyncIterator](), signal, state, consume); return validateRecordedChunks(chunks, checks); } diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index c80068a7a25..37c86f9740c 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -68,6 +68,7 @@ const { fixedBatchToEntry, isTransformObject, parsePullArgs, + raceSignal, snapshotFixedBatch, snapshotTransform, toUint8Array, @@ -1740,7 +1741,7 @@ async function pipeTo(source, ...args) { const writeOptions = signal ? { __proto__: null, signal } : undefined; // Without transforms, whether `signal` has aborted the pipe: the batches // are then no longer read or written (see raceSignal()). - let signalAborted = false; + const abortState = { __proto__: null, aborted: false }; let totalBytes = 0; const hasWritev = typeof writer.writev === 'function'; @@ -1794,11 +1795,11 @@ async function pipeTo(source, ...args) { // ignoring errors closing it. async function pipeSyncSource(iterator) { for (;;) { - if (signalAborted) return; + if (abortState.aborted) return; let batch = iterator[kNextSyncBatch](); if (batch === undefined) { const result = await iterator.next(); - if (result.done || signalAborted) return; + if (result.done || abortState.aborted) return; batch = result.value; } else if (batch === null) { return; @@ -1807,7 +1808,7 @@ async function pipeTo(source, ...args) { const p = writeBatch(batch); if (p) await p; } catch (error) { - if (!signalAborted) { + if (!abortState.aborted) { try { await iterator.return(); } catch { @@ -1824,12 +1825,12 @@ async function pipeTo(source, ...args) { async function pipeSourceWithSignal(iterator) { for (;;) { const result = await iterator.next(); - if (result.done || signalAborted) return; + if (result.done || abortState.aborted) return; try { const p = writeBatch(result.value); if (p) await p; } catch (error) { - if (!signalAborted) { + if (!abortState.aborted) { try { await iterator.return(); } catch { @@ -1841,53 +1842,6 @@ async function pipeTo(source, ...args) { } } - // Wait for `pipe(iterator)` to write the batches of `iterator`, the - // normalized source, unless `signal` aborts first. On an abort, the source - // is closed, without waiting, and the abort reason is thrown at once, as a - // pull() pipeline with `signal` would reject its pending pull; `piped` - // then stops before reading or writing another batch. This takes one - // listener and one promise for the whole pipe, instead of a layer reading - // the source for every batch. The listener is added first, as the signal - // can abort while the source is read. - function raceSignal(iterator, pipe) { - const { promise, resolve, reject } = PromiseWithResolvers(); - let settled = false; - function closeAndReject() { - try { - const returnMethod = iterator.return; - if (typeof returnMethod === 'function') { - markPromiseAsHandled(PromiseResolve( - FunctionPrototypeCall(returnMethod, iterator))); - } - } catch { - // The abort takes precedence over errors closing the source. - } - reject(signal.reason); - } - - function onAbort() { - if (settled) return; - settled = true; - signalAborted = true; - // The signal can abort while the source is called: close it once that - // call has returned. - PromisePrototypeThen(kResolvedPromise, closeAndReject); - } - signal.addEventListener('abort', onAbort, kNullOnceOption); - PromisePrototypeThen(pipe(iterator), (value) => { - if (settled) return; - settled = true; - signal.removeEventListener('abort', onAbort); - resolve(value); - }, (error) => { - if (settled) return; - settled = true; - signal.removeEventListener('abort', onAbort); - reject(error); - }); - return promise; - } - // Write the batches of an async source as the for await...of loop below // does. Nothing cancels the normalization while a batch is read: return() // is only called after an error writing one. @@ -1967,8 +1921,9 @@ async function pipeTo(source, ...args) { // Async reads go through next(), which return() can cancel while it // is pending, unlike kNextUncancellable. const iterator = normalized[SymbolAsyncIterator](); - await raceSignal(iterator, iterator[kNextSyncBatch] !== undefined ? - pipeSyncSource : pipeSourceWithSignal); + await raceSignal(iterator, signal, abortState, + iterator[kNextSyncBatch] !== undefined ? + pipeSyncSource : pipeSourceWithSignal); } else if (normalized[kNextSyncBatch] !== undefined) { await pipeSyncSource(normalized); } else if (normalized[kNextUncancellable] !== undefined) { diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index b4eb5e9c662..c22497064df 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -158,6 +158,67 @@ class PipelineAbort { } } +/** + * Wait for `consume(iterator)`, which reads `iterator` (a normalized source) + * to the end, unless `signal` aborts first. On an abort, `state.aborted` is + * set, the source is closed without waiting, ignoring errors, and the + * returned promise rejects with the abort reason at once, as a pull() + * pipeline with `signal` would reject its pending pull; `consume` must check + * `state.aborted` after each read and stop. Closing and rejecting wait for a + * microtask, as the signal can abort while the source is called. + * + * This takes one listener and one promise for the whole read, instead of a + * layer around the source adding a promise to every read. + * @param {AsyncIterator} iterator + * @param {AbortSignal} signal + * @param {{ aborted: boolean }} state + * @param {Function} consume + * @returns {Promise} + */ +function raceSignal(iterator, signal, state, consume) { + const { promise, resolve, reject } = PromiseWithResolvers(); + let settled = false; + function closeAndReject() { + try { + const returnMethod = iterator.return; + if (typeof returnMethod === 'function') { + markPromiseAsHandled(PromiseResolve( + FunctionPrototypeCall(returnMethod, iterator))); + } + } catch { + // The abort takes precedence over errors closing the source. + } + reject(signal.reason); + } + + function onAbort() { + if (settled) return; + settled = true; + state.aborted = true; + PromisePrototypeThen(kResolvedPromise, closeAndReject); + } + // Added first, as the signal can abort while the source is read. + signal.addEventListener('abort', onAbort, kNullOnceOption); + let consumed; + try { + consumed = consume(iterator); + } catch (error) { + consumed = PromiseReject(error); + } + PromisePrototypeThen(consumed, (value) => { + if (settled) return; + settled = true; + signal.removeEventListener('abort', onAbort); + resolve(value); + }, (error) => { + if (settled) return; + settled = true; + signal.removeEventListener('abort', onAbort); + reject(error); + }); + return promise; +} + /** * Wrap an async source so each pending read is abort-aware. * @param {AsyncIterable} source - The source to read from. @@ -1013,6 +1074,7 @@ module.exports = { isTransformObject, onSignalAbort, parsePullArgs, + raceSignal, snapshotTransform, toUint8Array, toWriterUint8Array, diff --git a/test/parallel/test-stream-iter-consumers-bytes.js b/test/parallel/test-stream-iter-consumers-bytes.js index 531917eb973..5bf06a19ea8 100644 --- a/test/parallel/test-stream-iter-consumers-bytes.js +++ b/test/parallel/test-stream-iter-consumers-bytes.js @@ -3,6 +3,7 @@ const common = require('../common'); const assert = require('assert'); +const { getEventListeners } = require('events'); const { from, fromSync, @@ -272,7 +273,27 @@ async function testTextStringSource() { assert.strictEqual(result, 'direct-string'); } +// With a signal, exceeding the limit closes the source, rejects with the +// limit error, and leaves no listener on the signal. +async function testBytesSignalAndLimit() { + const ac = new AbortController(); + let closed = false; + async function* source() { + try { + for (;;) yield [new Uint8Array(8)]; + } finally { + closed = true; + } + } + await assert.rejects( + bytes(source(), { signal: ac.signal, limit: 20 }), + { code: 'ERR_OUT_OF_RANGE' }); + assert.strictEqual(closed, true); + assert.strictEqual(getEventListeners(ac.signal, 'abort').length, 0); +} + Promise.all([ + testBytesSignalAndLimit(), testBytesSyncBasic(), testBytesSyncLimit(), testBytesAsync(), From 2909ce66c27f35d5e29f53bbc3d5c134d89c0c61 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 04:02:16 +0000 Subject: [PATCH 39/58] stream: read stream/iter from() async sources in one reaction from() read an async source through yieldNormalizationAbortable(), whose next() a cancellation can interrupt: for every batch it created a promise with three closures, a reaction to the source's result, and an iterator result, and the normalizer then took another reaction and another iterator result. Only pipeTo() without a signal avoided this, with kNextUncancellable. Give yieldNormalizationAbortable() a read, kRead, that calls handlers set once by the normalizer, so that it takes no closure, and handle the source's result in the same reaction. The normalizer's pending read is a promise of its own, which a cancellation rejects as before. A batch passed on unchanged keeps the iterator result of the read. For 16-byte chunks one per batch, iterating from() of an async generator is 13-20% faster and allocates about 370 bytes less per batch; piping one with a signal is 11-15% faster. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 138 +++++++++++++++++++++++++++++- 1 file changed, 136 insertions(+), 2 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index a926eb982bd..23b7021a649 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -154,6 +154,10 @@ const kNextUncancellable = Symbol('kNextUncancellable'); const kReadUncancellable = Symbol('kReadUncancellable'); const kToIterResult = Symbol('kToIterResult'); const kReadFailed = Symbol('kReadFailed'); +// A read of the source that a cancellation can interrupt, for +// createAsyncSourceNormalizer(): see kRead in yieldNormalizationAbortable(). +const kSetReadHandlers = Symbol('kSetReadHandlers'); +const kRead = Symbol('kRead'); // The result kReadUncancellable gives once the source is done. const kDoneSourceResult = ObjectFreeze({ __proto__: null, done: true, value: undefined }); @@ -516,6 +520,55 @@ function yieldNormalizationAbortable(source, context) { } } + // The handlers of kRead, set once by kSetReadHandlers, and whether a + // kRead is pending. + let onRead = null; + let onReadError = null; + let readPending = false; + + // Fail a kRead, closing the source first if the normalization has + // been cancelled. + function failRead(error) { + if (context.cancelled) { + PromisePrototypeThen(closeSource(true), () => { + reading = false; + onReadError(error); + }); + return; + } + reading = false; + onReadError(error); + } + + // context.resolve while a kRead is pending. + function onReadCancelled() { + if (!readPending) return; + readPending = false; + context.resolve = null; + failRead(context.reason); + } + + function onReadResult(result) { + if (!readPending) return; + readPending = false; + if (context.resolve === onReadCancelled) context.resolve = null; + let iterResult; + try { + iterResult = toIterResult(result); + } catch (error) { + failRead(error); + return; + } + onRead(iterResult); + } + + function onReadRejected(error) { + if (!readPending) return; + readPending = false; + if (context.resolve === onReadCancelled) context.resolve = null; + failRead(error); + } + // Reject a pending next(), closing the source first if the // normalization has been cancelled. function rejectNext(reject, error) { @@ -604,6 +657,38 @@ function yieldNormalizationAbortable(source, context) { return PromiseResolve(next); }, [kToIterResult]: onUncancellableResult, + // Like next(), calling the handlers set by kSetReadHandlers instead + // of settling a promise of its own: onRead with the result, or + // onReadError with the error, possibly synchronously. A cancellation + // calls onReadError with its reason. The handlers are the same for + // every read, so that a read takes no closure, and the caller can + // handle the result in the same reaction. + [kSetReadHandlers](resultHandler, errorHandler) { + onRead = resultHandler; + onReadError = errorHandler; + }, + [kRead]() { + if (completed) { + onRead(new IterResult(true, undefined)); + return; + } + if (context.cancelled) { + onReadError(context.reason); + return; + } + reading = true; + let next; + try { + next = FunctionPrototypeCall(nextMethod, iterator); + } catch (error) { + failRead(error); + return; + } + readPending = true; + context.resolve = onReadCancelled; + PromisePrototypeThen(PromiseResolve(next), onReadResult, + onReadRejected); + }, [kReadFailed]() { reading = false; }, @@ -799,6 +884,11 @@ function createAsyncSourceNormalizer(source, context) { // kNextUncancellable method, for callers that never cancel the // normalization while a read is pending. let uncancellable = false; + // Whether the source is read with kRead, and the settling functions of + // the read's promise while it is pending. + let readable = false; + let resolveRead = null; + let rejectRead = null; const { run, settled } = createOperationQueue(); function finish(result) { @@ -832,16 +922,23 @@ function createAsyncSourceNormalizer(source, context) { return finish(new IterResult(true, undefined)); } try { - return handleValue(result.value); + return handleValue(result.value, result); } catch (error) { return onBodyError(error); } } - function handleValue(value) { + // `result` is the source's result for `value`. When the source is read + // with kRead or kReadUncancellable, it is an IterResult of + // yieldNormalizationAbortable(), passed on as it is for a batch passed on + // unchanged. + function handleValue(value, result) { if (isUint8ArrayBatch(value)) { if (value.length <= FROM_BATCH_SIZE) { if (value.length === 0) return pullSource(); + if (readable || uncancellable) { + return finish(result); + } return finish(new IterResult(false, value)); } boundedBatches = yieldBoundedBatch(value); @@ -877,11 +974,44 @@ function createAsyncSourceNormalizer(source, context) { return fail(error); } + // The kRead handlers: the result is handled in the reaction to the + // source's result, and settles the read's promise. + function onRead(result) { + const resolve = resolveRead; + const reject = rejectRead; + resolveRead = rejectRead = null; + let settledWith; + try { + settledWith = onSourceResult(result); + } catch (error) { + reject(error); + return; + } + resolve(settledWith); + } + + function onReadError(error) { + const reject = rejectRead; + resolveRead = rejectRead = null; + state = kDone; + settled(); + reject(error); + } + function pullSource() { if (uncancellable) { return PromisePrototypeThen(iterator[kReadUncancellable](), onUncancellableResult, onUncancellableError); } + if (readable) { + // One promise for the read, which a cancellation rejects, instead of + // next()'s promise and a reaction to it. + const { promise, resolve, reject } = PromiseWithResolvers(); + resolveRead = resolve; + rejectRead = reject; + iterator[kRead](); + return promise; + } return PromisePrototypeThen(iterator.next(), onSourceResult, fail); } @@ -905,6 +1035,10 @@ function createAsyncSourceNormalizer(source, context) { throwIfNormalizationCancelled(context); const iterable = yieldNormalizationAbortable(source, context); iterator = iterable[SymbolAsyncIterator](); + if (iterator[kSetReadHandlers] !== undefined) { + iterator[kSetReadHandlers](onRead, onReadError); + readable = true; + } } catch (error) { state = kDone; settled(); From c3a0e47ef5efcbb28bff45fb9bfbb18b0b644ff5 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 04:32:46 +0000 Subject: [PATCH 40/58] stream: check stream/iter byte views by their byteLength alone To detect that a byte view was resized or detached after being accepted, stream/iter recorded, for views not known to be of a fixed-length, non-shared buffer, a snapshot object with the view's buffer, the buffer's byteLength and detached state, and the view's byteLength and byteOffset. Telling the two cases apart read every view's buffer, which for a small typed array that V8 keeps on the heap, such as one just allocated by a transform, moves its contents to a new ArrayBuffer. What stream/iter accounts for a view is its byteLength, and a view whose bytes are no longer those accounted for has a different byteLength: a detached view, a view out of bounds of a shrunk buffer, and a length-tracking view of a resized or grown buffer. Record and check the byteLength alone, without reading the buffer, for every view. A fixed-length view of a resizable buffer that is resized while the view stays in bounds, whose bytes are unchanged, is no longer rejected. For 16-byte chunks, piping through a transform that copies each chunk is about 1.6 times faster, and 2.8 times with pipeToSync(); piping a sync generator of 64-chunk batches about 1.75 times faster; and piping a sync generator one chunk per batch to a writer with writeSync() about 1.2-1.3 times faster. Assisted-by: OpenCode --- lib/internal/streams/iter/consumers.js | 20 +- lib/internal/streams/iter/pull.js | 31 ++- lib/internal/streams/iter/utils.js | 180 ++++-------------- .../test-stream-iter-resizable-buffers.js | 21 ++ 4 files changed, 83 insertions(+), 169 deletions(-) diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index 7cdc7d38e82..aa79795e93e 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -108,19 +108,19 @@ function collectSync(source, limit) { // Normalize source via fromSync() - accepts strings, ArrayBuffers, protocols, etc. const normalized = fromSync(source); const chunks = []; - const checks = []; + const byteLengths = []; let totalBytes = 0; for (const batch of normalized) { for (let i = 0; i < batch.length; i++) { - totalBytes += recordChunk(chunks, checks, batch[i]); + totalBytes += recordChunk(chunks, byteLengths, batch[i]); if (limit !== undefined && totalBytes > limit) { throw new ERR_OUT_OF_RANGE('totalBytes', `<= ${limit}`, totalBytes); } } } - return validateRecordedChunks(chunks, checks); + return validateRecordedChunks(chunks, byteLengths); } /** @@ -136,23 +136,23 @@ async function collectAsync(source, signal, limit) { // Normalize source via from() - accepts strings, ArrayBuffers, protocols, etc. const normalized = from(source); const chunks = []; - const checks = []; + const byteLengths = []; // Fast path: no signal and no limit if (!signal && limit === undefined) { for await (const batch of normalized) { for (let i = 0; i < batch.length; i++) { - recordChunk(chunks, checks, batch[i]); + recordChunk(chunks, byteLengths, batch[i]); } } - return validateRecordedChunks(chunks, checks); + return validateRecordedChunks(chunks, byteLengths); } - // Slow path: with signal or limit checks + // Slow path: with signal or limit byteLengths let totalBytes = 0; function recordBatch(batch) { for (let i = 0; i < batch.length; i++) { - totalBytes += recordChunk(chunks, checks, batch[i]); + totalBytes += recordChunk(chunks, byteLengths, batch[i]); if (limit !== undefined && totalBytes > limit) { throw new ERR_OUT_OF_RANGE('totalBytes', `<= ${limit}`, totalBytes); } @@ -161,7 +161,7 @@ async function collectAsync(source, signal, limit) { if (!signal) { for await (const batch of normalized) recordBatch(batch); - return validateRecordedChunks(chunks, checks); + return validateRecordedChunks(chunks, byteLengths); } // Read as for await...of does, stopping once the signal aborts; see @@ -186,7 +186,7 @@ async function collectAsync(source, signal, limit) { } } await raceSignal(normalized[SymbolAsyncIterator](), signal, state, consume); - return validateRecordedChunks(chunks, checks); + return validateRecordedChunks(chunks, byteLengths); } /** diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 37c86f9740c..b8dcd03c761 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -1629,21 +1629,19 @@ function pipeToSync(source, ...args) { } if (!hasWritevSync) { // Chunks written one at a time: snapshot them without an object for - // every chunk when possible. + // every chunk. const fixed = snapshotFixedBatch(batch); - if (fixed !== null) { - for (let i = 0; i < batch.length; i++) { - const chunk = checkFixedBatchChunk(fixed, i); - const accepted = writer.writeSync(chunk); - checkFixedBatchChunk(fixed, i); - if (accepted === false) { - throw new ERR_OUT_OF_RANGE( - 'write', 'within byte budget', 'budget exhausted'); - } - totalBytes += fixed.byteLengths[i]; + for (let i = 0; i < batch.length; i++) { + const chunk = checkFixedBatchChunk(fixed, i); + const accepted = writer.writeSync(chunk); + checkFixedBatchChunk(fixed, i); + if (accepted === false) { + throw new ERR_OUT_OF_RANGE( + 'write', 'within byte budget', 'budget exhausted'); } - continue; + totalBytes += fixed.byteLengths[i]; } + continue; } const entry = createBatchEntry(batch); if (hasWritevSync && batch.length > 1) { @@ -1771,9 +1769,6 @@ async function pipeTo(source, ...args) { } } - // Write a batch using try-fallback: sync first, async if needed. - // Returns undefined on sync success, or a Promise when async fallback - // is required. Callers must check: const p = writeBatch(b); if (p) await p; // Write a FixedBatch with writeSync(), like the loop at the end of // writeBatch(), falling back to the async path for the rest of the batch. function writeFixedBatch(fixed) { @@ -1863,6 +1858,9 @@ async function pipeTo(source, ...args) { } } + // Write a batch using try-fallback: sync first, async if needed. + // Returns undefined on sync success, or a Promise when async fallback + // is required. Callers must check: const p = writeBatch(b); if (p) await p; function writeBatch(batch) { // Single chunk, the common case: check the view around writeSync() // without allocating a batch entry, and create one only to fall back to @@ -1876,8 +1874,7 @@ async function pipeTo(source, ...args) { return writeBatchAsyncFallback(createBatchEntry(batch), 0); } if (!hasWritev && hasWriteSync) { - const fixed = snapshotFixedBatch(batch); - if (fixed !== null) return writeFixedBatch(fixed); + return writeFixedBatch(snapshotFixedBatch(batch)); } const entry = createBatchEntry(batch); if (hasWritev && batch.length > 1) { diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index c22497064df..c4d1e24f0f0 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -3,8 +3,6 @@ const { Array, ArrayBufferPrototypeGetByteLength, - ArrayBufferPrototypeGetDetached, - ArrayBufferPrototypeGetResizable, ArrayPrototypeSlice, FunctionPrototypeCall, ObjectDefineProperty, @@ -467,32 +465,20 @@ function toUint8Array(chunk) { return chunk; } -// Byte view snapshots and batch entries are created for every chunk and -// every batch, so like IterResult they are constructed rather than created -// as `{ __proto__: null, ... }` literals (which are dictionary-mode objects), +// Byte views are accepted with their byteLength, and rejected when used if +// it has changed: if they were detached (byteLength 0), resized with a +// length-tracking view, or shrunk out of bounds. What was accounted for a +// view is then what is there. The byteLength is read from the view alone, +// without its buffer: reading the buffer of a small typed array that V8 +// keeps on the heap moves its contents to a new ArrayBuffer. +// +// Byte views and batch entries are created for every chunk and every batch, +// so like IterResult they are constructed rather than created as +// `{ __proto__: null, ... }` literals (which are dictionary-mode objects), // and their prototype is an empty null-prototype object. -function ByteViewSnapshot(value, buffer, sharedBufferView) { - this.value = value; - this.buffer = buffer; - this.bufferByteLength = sharedBufferView === undefined ? - ArrayBufferPrototypeGetByteLength(buffer) : - TypedArrayPrototypeGetByteLength(sharedBufferView); - this.byteLength = TypedArrayPrototypeGetByteLength(value); - this.byteOffset = TypedArrayPrototypeGetByteOffset(value); - this.detached = sharedBufferView === undefined && - ArrayBufferPrototypeGetDetached(buffer); - this.sharedBufferView = sharedBufferView; -} -ByteViewSnapshot.prototype = ObjectFreeze({ __proto__: null }); - -// The snapshot of a non-empty view of a fixed-length, non-shared ArrayBuffer, -// the common case. Such a view can only change by the buffer being detached, -// which makes its byteLength 0, so its byteLength is all that needs to be -// recorded. `buffer` is null to tell it from a ByteViewSnapshot. function FixedByteView(value, byteLength) { this.value = value; this.byteLength = byteLength; - this.buffer = null; } FixedByteView.prototype = ObjectFreeze({ __proto__: null }); @@ -518,48 +504,18 @@ function PendingWrite(batch, resolve, reject) { PendingWrite.prototype = ObjectFreeze({ __proto__: null }); function snapshotByteView(value) { - const buffer = TypedArrayPrototypeGetBuffer(value); - if (isSharedArrayBuffer(buffer)) { - return new ByteViewSnapshot(value, buffer, new Uint8Array(buffer)); - } - const byteLength = TypedArrayPrototypeGetByteLength(value); - if (byteLength !== 0 && !ArrayBufferPrototypeGetResizable(buffer)) { - return new FixedByteView(value, byteLength); - } - return new ByteViewSnapshot(value, buffer, undefined); + return new FixedByteView(value, TypedArrayPrototypeGetByteLength(value)); +} + +function throwByteViewChanged() { + throw new ERR_INVALID_STATE.TypeError( + 'Byte view was resized or detached after being accepted'); } function validateByteView(snapshot) { - if (snapshot.buffer === null) { - const { value } = snapshot; - if (TypedArrayPrototypeGetByteLength(value) !== snapshot.byteLength) { - throw new ERR_INVALID_STATE.TypeError( - 'Byte view was resized or detached after being accepted'); - } - return value; - } - const { - value, - buffer, - bufferByteLength, - byteLength, - byteOffset, - detached, - sharedBufferView, - } = snapshot; - const currentBufferByteLength = sharedBufferView === undefined ? - ArrayBufferPrototypeGetByteLength(buffer) : - TypedArrayPrototypeGetByteLength(sharedBufferView); - const currentDetached = sharedBufferView === undefined && - ArrayBufferPrototypeGetDetached(buffer); - - if (TypedArrayPrototypeGetBuffer(value) !== buffer || - currentBufferByteLength !== bufferByteLength || - TypedArrayPrototypeGetByteLength(value) !== byteLength || - TypedArrayPrototypeGetByteOffset(value) !== byteOffset || - currentDetached !== detached) { - throw new ERR_INVALID_STATE.TypeError( - 'Byte view was resized or detached after being accepted'); + const { value } = snapshot; + if (TypedArrayPrototypeGetByteLength(value) !== snapshot.byteLength) { + throwByteViewChanged(); } return value; } @@ -567,48 +523,17 @@ function validateByteView(snapshot) { /** * Call `method` on `receiver` with the byte view `value`, and throw if the * call resized or detached it, as validateByteView() does for a snapshot - * taken just before the call. Used for single-chunk writes, the common case: - * for views on an ArrayBuffer the snapshot is kept in locals instead of a - * ByteViewSnapshot, so that nothing is allocated per chunk. + * taken just before the call, without allocating one. * @param {Uint8Array} value * @param {Function} method * @param {object} receiver * @returns {any} The result of the call. */ function callWithByteView(value, method, receiver) { - const buffer = TypedArrayPrototypeGetBuffer(value); - if (!isSharedArrayBuffer(buffer) && !ArrayBufferPrototypeGetResizable(buffer)) { - // A non-empty view of a fixed-length, non-shared ArrayBuffer can only - // change by the buffer being detached, which makes its byteLength 0 (see - // FixedByteView), so checking its byteLength is enough. - const byteLength = TypedArrayPrototypeGetByteLength(value); - if (byteLength !== 0) { - const result = FunctionPrototypeCall(method, receiver, value); - if (TypedArrayPrototypeGetByteLength(value) !== byteLength) { - throw new ERR_INVALID_STATE.TypeError( - 'Byte view was resized or detached after being accepted'); - } - return result; - } - } - if (isSharedArrayBuffer(buffer)) { - const snapshot = snapshotByteView(value); - const result = FunctionPrototypeCall(method, receiver, value); - validateByteView(snapshot); - return result; - } - const bufferByteLength = ArrayBufferPrototypeGetByteLength(buffer); const byteLength = TypedArrayPrototypeGetByteLength(value); - const byteOffset = TypedArrayPrototypeGetByteOffset(value); - const detached = ArrayBufferPrototypeGetDetached(buffer); const result = FunctionPrototypeCall(method, receiver, value); - if (TypedArrayPrototypeGetBuffer(value) !== buffer || - ArrayBufferPrototypeGetByteLength(buffer) !== bufferByteLength || - TypedArrayPrototypeGetByteLength(value) !== byteLength || - TypedArrayPrototypeGetByteOffset(value) !== byteOffset || - ArrayBufferPrototypeGetDetached(buffer) !== detached) { - throw new ERR_INVALID_STATE.TypeError( - 'Byte view was resized or detached after being accepted'); + if (TypedArrayPrototypeGetByteLength(value) !== byteLength) { + throwByteViewChanged(); } return result; } @@ -690,45 +615,31 @@ function validateBudget(budget) { } /** - * Append a chunk to `chunks` for later concatenation, and the information - * needed to detect that it was resized or detached in the meantime to - * `checks`. This provides the same guarantee as snapshotByteView() and - * validateByteView() without allocating a snapshot for every chunk in the - * common case: a non-empty view of a fixed-length, non-shared ArrayBuffer can - * only change by the buffer being detached, which makes its byteLength 0, so - * its byteLength is all that needs to be recorded. + * Append a chunk to `chunks` for later concatenation, and its byteLength to + * `byteLengths`, to detect that it was resized or detached in the meantime + * as validateByteView() does, without an object for every chunk. * @param {Uint8Array[]} chunks - * @param {Array} checks + * @param {number[]} byteLengths * @param {Uint8Array} value * @returns {number} The byteLength of `value`. */ -function recordChunk(chunks, checks, value) { - const buffer = TypedArrayPrototypeGetBuffer(value); +function recordChunk(chunks, byteLengths, value) { const byteLength = TypedArrayPrototypeGetByteLength(value); chunks[chunks.length] = value; - if (byteLength === 0 || isSharedArrayBuffer(buffer) || - ArrayBufferPrototypeGetResizable(buffer)) { - checks[checks.length] = snapshotByteView(value); - } else { - checks[checks.length] = byteLength; - } + byteLengths[byteLengths.length] = byteLength; return byteLength; } /** * Validate chunks recorded with recordChunk(). * @param {Uint8Array[]} chunks - * @param {Array} checks + * @param {number[]} byteLengths * @returns {Uint8Array[]} `chunks` */ -function validateRecordedChunks(chunks, checks) { +function validateRecordedChunks(chunks, byteLengths) { for (let i = 0; i < chunks.length; i++) { - const check = checks[i]; - if (typeof check !== 'number') { - validateByteView(check); - } else if (TypedArrayPrototypeGetByteLength(chunks[i]) !== check) { - throw new ERR_INVALID_STATE.TypeError( - 'Byte view was resized or detached after being accepted'); + if (TypedArrayPrototypeGetByteLength(chunks[i]) !== byteLengths[i]) { + throwByteViewChanged(); } } return chunks; @@ -745,32 +656,17 @@ FixedBatch.prototype = ObjectFreeze({ __proto__: null }); /** * Snapshot a batch of Uint8Arrays like createBatchEntry(), without an object - * for every chunk, when every chunk is a non-empty view of a fixed-length, - * non-shared ArrayBuffer: such views can only change by being detached, - * which makes their byteLength 0 (see FixedByteView), so their byteLengths - * are all that needs to be recorded. Check a chunk with checkFixedBatchChunk() - * before and after using it. + * for every chunk: their byteLengths are recorded in an array. Check a chunk + * with checkFixedBatchChunk() before and after using it. * @param {Uint8Array[]} chunks - * @returns {FixedBatch|null} null if a chunk needs a full snapshot. + * @returns {FixedBatch} */ function snapshotFixedBatch(chunks) { const count = chunks.length; const byteLengths = new Array(count); let byteLength = 0; - let checkedBuffer; for (let i = 0; i < count; i++) { - const view = chunks[i]; - const buffer = TypedArrayPrototypeGetBuffer(view); - // Chunks often share a buffer: check each buffer once. - if (buffer !== checkedBuffer) { - if (isSharedArrayBuffer(buffer) || - ArrayBufferPrototypeGetResizable(buffer)) { - return null; - } - checkedBuffer = buffer; - } - const length = TypedArrayPrototypeGetByteLength(view); - if (length === 0) return null; + const length = TypedArrayPrototypeGetByteLength(chunks[i]); byteLengths[i] = length; byteLength += length; } @@ -780,8 +676,7 @@ function snapshotFixedBatch(chunks) { function checkFixedBatchChunk(batch, index) { const chunk = batch.chunks[index]; if (TypedArrayPrototypeGetByteLength(chunk) !== batch.byteLengths[index]) { - throw new ERR_INVALID_STATE.TypeError( - 'Byte view was resized or detached after being accepted'); + throwByteViewChanged(); } return chunk; } @@ -1082,6 +977,7 @@ module.exports = { validateBatchEntry, validateBudget, validateRecordedChunks, + throwByteViewChanged, validateByteView, yieldAbortable, }; diff --git a/test/parallel/test-stream-iter-resizable-buffers.js b/test/parallel/test-stream-iter-resizable-buffers.js index 1691893d347..3704e73e487 100644 --- a/test/parallel/test-stream-iter-resizable-buffers.js +++ b/test/parallel/test-stream-iter-resizable-buffers.js @@ -286,7 +286,28 @@ async function testConsumersRejectDetachedViews() { assert.strictEqual(result, chunk); } +// Only what was accounted for a view is checked: its byteLength. A +// fixed-length view of a resizable buffer that stays in bounds is unchanged. +async function testFixedLengthViewOfResizedBuffer() { + const buffer = new ArrayBuffer(4, { maxByteLength: 8 }); + const view = new Uint8Array(buffer, 0, 2); + const { writer, readable } = push(); + assert.strictEqual(writer.writeSync(view), true); + buffer.resize(8); + writer.endSync(); + const [chunk] = await array(readable); + assert.strictEqual(chunk, view); + + // Shrinking the buffer so that the view is out of bounds makes its + // byteLength 0: rejected. + const { writer: writer2, readable: readable2 } = push(); + assert.strictEqual(writer2.writeSync(view), true); + buffer.resize(1); + await assert.rejects(readable2[Symbol.asyncIterator]().next(), kResizeError); +} + Promise.all([ + testFixedLengthViewOfResizedBuffer(), testBufferedViewMutationRejected(), testDropOldestUsesAcceptedByteLength(), testPendingWritesRejectResizedViews(), From bd5e9e6de163a652d3e7aaeea7b90fc95ccb5085 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 04:33:42 +0000 Subject: [PATCH 41/58] stream: write to stream/iter writers with only write() inline pipeTo() to a writer without writeSync() created a batch entry with an object for every chunk, and wrote it in an async function, which every batch then awaited, even when write() returned undefined. Write single chunks directly, checking the view's byteLength around write() and after the promise it returns, and batches of several chunks from a FixedBatch, which records their byteLengths in an array. Only a promise returned by write() is awaited. For 16-byte chunks one per batch, piping a sync generator to a writer with only write() is about 2 times faster, as fast as to one with writeSync(), and piping an async generator about 1.3 times faster. Assisted-by: OpenCode --- lib/internal/streams/iter/pull.js | 47 +++++++++++++++++++ .../test-stream-iter-resizable-buffers.js | 27 +++++++++++ 2 files changed, 74 insertions(+) diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index b8dcd03c761..9dca617e7f5 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -70,6 +70,7 @@ const { parsePullArgs, raceSignal, snapshotFixedBatch, + throwByteViewChanged, snapshotTransform, toUint8Array, validateBatchEntry, @@ -1769,6 +1770,46 @@ async function pipeTo(source, ...args) { } } + // Write a FixedBatch with write(), for a writer without writeSync(), + // starting at `index`: returns undefined if every write() returned + // undefined, or a promise for writing the rest once one returns a + // promise. A write() that returns undefined is not awaited. + function writeFixedBatchAsync(fixed, index) { + const { chunks } = fixed; + for (let i = index; i < chunks.length; i++) { + const result = writer.write(checkFixedBatchChunk(fixed, i), writeOptions); + checkFixedBatchChunk(fixed, i); + if (result !== undefined) { + return PromisePrototypeThen(PromiseResolve(result), () => { + checkFixedBatchChunk(fixed, i); + totalBytes += fixed.byteLengths[i]; + return writeFixedBatchAsync(fixed, i + 1); + }); + } + totalBytes += fixed.byteLengths[i]; + } + } + + // Write a single chunk with write(), for a writer without writeSync(), + // as writeFixedBatchAsync() does, without a FixedBatch. + function writeChunkAsync(chunk) { + const byteLength = TypedArrayPrototypeGetByteLength(chunk); + const result = writer.write(chunk, writeOptions); + if (TypedArrayPrototypeGetByteLength(chunk) !== byteLength) { + throwByteViewChanged(); + } + if (result === undefined) { + totalBytes += byteLength; + return; + } + return PromisePrototypeThen(PromiseResolve(result), () => { + if (TypedArrayPrototypeGetByteLength(chunk) !== byteLength) { + throwByteViewChanged(); + } + totalBytes += byteLength; + }); + } + // Write a FixedBatch with writeSync(), like the loop at the end of // writeBatch(), falling back to the async path for the rest of the batch. function writeFixedBatch(fixed) { @@ -1862,6 +1903,12 @@ async function pipeTo(source, ...args) { // Returns undefined on sync success, or a Promise when async fallback // is required. Callers must check: const p = writeBatch(b); if (p) await p; function writeBatch(batch) { + // A writer with only write() (and writev(), used for batches of several + // chunks): no batch entry. + if (!hasWriteSync && (!hasWritev || batch.length === 1)) { + if (batch.length === 1) return writeChunkAsync(batch[0]); + return writeFixedBatchAsync(snapshotFixedBatch(batch), 0); + } // Single chunk, the common case: check the view around writeSync() // without allocating a batch entry, and create one only to fall back to // the async path. diff --git a/test/parallel/test-stream-iter-resizable-buffers.js b/test/parallel/test-stream-iter-resizable-buffers.js index 3704e73e487..80375ffaf2d 100644 --- a/test/parallel/test-stream-iter-resizable-buffers.js +++ b/test/parallel/test-stream-iter-resizable-buffers.js @@ -286,6 +286,32 @@ async function testConsumersRejectDetachedViews() { assert.strictEqual(result, chunk); } +// A writer with only write() is checked the same way, before and after each +// write(), and after the promise it returns, if any. +async function testPipeRejectsDetachWithWriteOnly() { + for (const detachIndex of [1, 0]) { + const buffers = [new ArrayBuffer(2), new ArrayBuffer(2)]; + const written = []; + await assert.rejects(pipeTo([buffers.map((b) => new Uint8Array(b))], { + write(chunk) { + written.push(chunk.byteLength); + if (written.length === 1) buffers[detachIndex].transfer(); + }, + fail: common.mustCall(), + }), kResizeError); + assert.deepStrictEqual(written, [2]); + } + + const buffer = new ArrayBuffer(2); + await assert.rejects(pipeTo([new Uint8Array(buffer)], { + async write() { + await null; + buffer.transfer(); + }, + fail: common.mustCall(), + }), kResizeError); +} + // Only what was accounted for a view is checked: its byteLength. A // fixed-length view of a resizable buffer that stays in bounds is unchanged. async function testFixedLengthViewOfResizedBuffer() { @@ -307,6 +333,7 @@ async function testFixedLengthViewOfResizedBuffer() { } Promise.all([ + testPipeRejectsDetachWithWriteOnly(), testFixedLengthViewOfResizedBuffer(), testBufferedViewMutationRejected(), testDropOldestUsesAcceptedByteLength(), From 0675a04ae48f9b332ef8401713b0b7f827eefdb7 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 10:27:36 +0000 Subject: [PATCH 42/58] stream: encode small stream/iter strings into a pool stream/iter encoded every string written, yielded or piped with TextEncoder.encode(), into an ArrayBuffer of its own. Writing many small strings, as a server-side renderer does, was then mostly the cost of allocating them. Encode strings whose UTF-8 encoding certainly fits in 8 KiB into a 64 KiB pool, as Buffer.from() does, with the fast API utf8WriteStatic(), and pass on views of it. The pool is untransferable, so that transferring the buffer of one chunk cannot detach the others. For SSR-like strings of about 65 bytes, writing to a push() stream with writeSync() is about 4 times faster, and with await write() about 2.2 times faster. Assisted-by: OpenCode --- doc/api/stream_iter.md | 5 +++ lib/internal/streams/iter/utils.js | 35 ++++++++++++++++++- test/parallel/test-stream-iter-push-writer.js | 31 +++++++++++++++- 3 files changed, 69 insertions(+), 2 deletions(-) diff --git a/doc/api/stream_iter.md b/doc/api/stream_iter.md index 02c0719e943..52fcde48cdb 100644 --- a/doc/api/stream_iter.md +++ b/doc/api/stream_iter.md @@ -94,6 +94,10 @@ are automatically UTF-8 encoded when passed to `from()`, `push()`, or `pipeTo()`. This removes ambiguity around encodings and enables zero-copy transfers between streams and native code. +Like [`Buffer.from()`][] for strings, small strings are encoded into a shared +pool: the resulting {Uint8Array} is a view of a larger {ArrayBuffer}, which +cannot be transferred. + ### Batching Each iteration yields a **batch** -- an {Array} of {Uint8Array} chunks @@ -2413,6 +2417,7 @@ console.log(textSync(stream)); // 'hello world' [Iterable Streams API]: https://iter-streams.proposal.wintertc.org/ [`--experimental-stream-iter`]: cli.md#--experimental-stream-iter [`Broadcast.from()`]: #broadcastfrominput-options +[`Buffer.from()`]: buffer.md#static-method-bufferfromstring-encoding [`Share.from()`]: #static-method-sharefrominput-options [`SyncShare.fromSync()`]: #static-method-syncsharefromsyncinput-options [`array()`]: #arraysource-options diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index c4d1e24f0f0..9a94b0db2b6 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -26,6 +26,8 @@ const { } = internalBinding('util'); const { TextEncoder } = require('internal/encoding'); +const { markAsUntransferable } = require('internal/buffer'); +const { utf8WriteStatic } = internalBinding('buffer'); const { codes: { ERR_INVALID_ARG_TYPE, @@ -63,6 +65,37 @@ const kResolvedPromise = PromiseResolve(); // Shared TextEncoder instance for string conversion. const encoder = new TextEncoder(); +// Small strings are encoded into a pool, like Buffer.from(string), instead +// of each into an ArrayBuffer of its own: writing many small strings, as a +// server-side renderer does, is otherwise mostly the cost of allocating +// them. The pool is untransferable, so that transferring the buffer of one +// chunk cannot detach the others. A string is pooled when its UTF-8 encoding +// fits in kMaxPooledByteLength bytes for certain: every UTF-16 code unit +// encodes to at most 3 bytes. +const kStringPoolSize = 64 * 1024; +const kMaxPooledLength = 8 * 1024 / 3; +let stringPool = null; +let stringPoolBuffer; +let stringPoolOffset = 0; + +function encodeString(string) { + const length = string.length; + if (length > kMaxPooledLength) return encoder.encode(string); + const maxByteLength = length * 3; + if (stringPool === null || + maxByteLength > kStringPoolSize - stringPoolOffset) { + stringPool = new Uint8Array(kStringPoolSize); + stringPoolBuffer = TypedArrayPrototypeGetBuffer(stringPool); + markAsUntransferable(stringPoolBuffer); + stringPoolOffset = 0; + } + const offset = stringPoolOffset; + const written = utf8WriteStatic(stringPool, string, offset, maxByteLength); + // Keep the chunks 8-byte aligned, as Buffer does. + stringPoolOffset = (offset + written + 7) & ~7; + return new Uint8Array(stringPoolBuffer, offset, written); +} + // Default high water marks for push and multi-consumer streams. These values // are somewhat arbitrary but have been tested across various workloads and // appear to yield the best overall throughput/latency balance. @@ -457,7 +490,7 @@ function getMinCursor(consumers, fallback) { */ function toUint8Array(chunk) { if (typeof chunk === 'string') { - return encoder.encode(chunk); + return encodeString(chunk); } if (!isUint8Array(chunk)) { throw new ERR_INVALID_ARG_TYPE('chunk', ['string', 'Uint8Array'], chunk); diff --git a/test/parallel/test-stream-iter-push-writer.js b/test/parallel/test-stream-iter-push-writer.js index a37c5eac2aa..0278ce569b9 100644 --- a/test/parallel/test-stream-iter-push-writer.js +++ b/test/parallel/test-stream-iter-push-writer.js @@ -3,7 +3,7 @@ const common = require('../common'); const assert = require('assert'); -const { dump, ondrain, push, text } = require('stream/iter'); +const { dump, push, ondrain, text, array } = require('stream/iter'); const { setImmediate } = require('timers/promises'); async function testOndrain() { @@ -663,7 +663,36 @@ async function testFailRejectsPendingReadWithFalsyReason() { ); } +// Strings are encoded as UTF-8 (lone surrogates as U+FFFD), small ones into +// a shared, untransferable pool, which must not mix up their bytes, also +// when the pool fills up; large ones each into an ArrayBuffer of their own. +async function testWriteStrings() { + const strings = []; + for (let i = 0; i < 3000; i++) { + strings.push(`row ${i} héllo 😀 \ud800 ${'x'.repeat(i % 97)}`); + } + strings.push('y'.repeat(10000), 'z'); + const { writer, readable } = push(); + const read = array(readable); + for (const string of strings) { + if (!writer.writeSync(string)) await writer.write(string); + } + await writer.end(); + const chunks = await read; + const decoder = new TextDecoder(); + assert.deepStrictEqual(chunks.map((chunk) => decoder.decode(chunk)), + strings.map((s) => s.replace('\ud800', '\ufffd'))); + for (const chunk of chunks) { + assert.strictEqual(Object.getPrototypeOf(chunk), Uint8Array.prototype); + } + const pooled = chunks[0]; + assert.ok(pooled.buffer.byteLength > pooled.byteLength); + assert.throws(() => pooled.buffer.transfer(), TypeError); + assert.strictEqual(chunks[3000].buffer.byteLength, 10000); +} + Promise.all([ + testWriteStrings(), testOndrain(), testDropPoliciesReportPhysicalCapacity(), testOndrainNonDrainable(), From e8619cf191d94299dbc7e92cf22be15f35e95c51 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 10:32:53 +0000 Subject: [PATCH 43/58] stream: read stream/iter share() consumers without a chain per read Every next() of a share() consumer chained on the consumer's previous next() and ran an async function, also when the batch was already buffered, as it is for every consumer but the first to read it. Every read of the source ran an async function, and each consumer waiting for it queued a promise of its own. Recomputing the slowest consumer's cursor, once per batch with several consumers, allocated an object. A next() that finds its batch buffered, or the end, now settles at once, without waiting for anything. Only a next() that waits for the source is waited for by the next one; when there is room in the buffer it waits for the read with one reaction, without an async function. Consumers waiting for the same read of the source share one promise, which a cancellation resolves at once, and the read is handled by functions created once per share. getMinCursor() reuses its result. With two consumers of a share of 16-byte chunks one per batch, reading a sync generator is about 1.8 times faster, and an async generator about 1.7 times faster, with 60% less allocated per chunk. Assisted-by: OpenCode --- lib/internal/streams/iter/share.js | 368 +++++++++++++++++------------ lib/internal/streams/iter/utils.js | 9 +- 2 files changed, 227 insertions(+), 150 deletions(-) diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index 484b923f025..ba457e9023c 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -10,6 +10,7 @@ const { FunctionPrototypeCall, ObjectSetPrototypeOf, PromisePrototypeThen, + PromiseReject, PromiseResolve, PromiseWithResolvers, SafeSet, @@ -73,7 +74,8 @@ const { markPromiseAsHandled } = internalBinding('util'); // ============================================================================= const kNoShareError = Symbol('kNoShareError'); -const kShareCancelled = Symbol('kShareCancelled'); +// Returned by #tryRead() when the consumer must wait for the source. +const kPull = Symbol('kPull'); const kSetFactorySignal = Symbol('kSetFactorySignal'); class ShareImpl { @@ -88,9 +90,15 @@ class ShareImpl { #cancelled = false; #pulling = false; #pullWaiters = []; - // Settles the pending read of the source with kShareCancelled when the - // share is cancelled, or null if no read is pending. - #cancelPendingRead = null; + // While the source is read: the promise that every consumer waiting for + // the read waits for, the function resolving it, and whether the batch + // read is to be discarded ('drop-newest'). A cancellation resolves it at + // once: racing the read with a promise settled by cancel() would add + // reactions to that promise on every read, which would be kept until the + // share is cancelled or collected. + #pullDone = null; + #resolvePull = null; + #pullDiscard = false; #cancelError = kNoShareError; #cachedMinCursor = 0; #cachedMinCursorConsumers = 0; @@ -148,7 +156,14 @@ class ShareImpl { reject: null, detached: false, error: kNoShareError, - pendingNext: PromiseResolve(), + // Set by #readAsync() for next(): the read waits for the source. + waiting: false, + // Created by #readAfterPull() on first use. + afterPull: undefined, + // The last pending next() of the consumer, if it is waiting for the + // source: the next one waits for it, as next() calls of an async + // generator do. + pendingNext: null, }, null); this.#consumers.add(state); @@ -165,82 +180,39 @@ class ShareImpl { return { __proto__: null, [SymbolAsyncIterator]() { - const getNext = async () => { - // Loop until we get data, source is exhausted, or - // consumer is detached. Multiple consumers may be woken - // after a single pull - those that find no data at their - // cursor must re-pull rather than terminating prematurely. - for (;;) { - if (state.detached) { - if (state.error !== kNoShareError) throw state.error; - return new IterResult(true, undefined); - } - - if (self.#cancelled) { - state.detached = true; - state.error = self.#cancelError; - self.#deleteConsumer(state); - if (state.error !== kNoShareError) throw state.error; - return new IterResult(true, undefined); - } - - // Check if data is available in buffer - const bufferIndex = state.cursor - self.#bufferStart; - if (bufferIndex < self.#buffer.length) { - const chunk = self.#readEntry(self.#buffer.get(bufferIndex)); - const cursor = state.cursor; - state.cursor++; - if (cursor === self.#cachedMinCursor && - --self.#cachedMinCursorConsumers === 0) { - self.#tryTrimBuffer(); - } - return new IterResult(false, chunk); - } - - if (self.#sourceExhausted) { - state.detached = true; - self.#deleteConsumer(state); - if (self.#sourceError !== kNoShareError) { - state.error = self.#sourceError; - throw state.error; - } - return new IterResult(true, undefined); - } - - // Need to pull from source - check buffer limit - let shouldBuffer; - try { - shouldBuffer = await self.#waitForBufferSpace(); - } catch (error) { - state.detached = true; - if (self.#deleteConsumer(state)) { - self.#tryTrimBuffer(); - } - throw error; - } - if (shouldBuffer === null) { - state.detached = true; - state.error = self.#cancelError; - self.#deleteConsumer(state); - if (state.error !== kNoShareError) throw state.error; - return new IterResult(true, undefined); - } - - await self.#pullFromSource(!shouldBuffer); - if (!shouldBuffer) { - await self.#waitForBufferSpaceAfterDrop(); - } + // A read that finds data, or the end, settles at once; one that + // waits for the source becomes pendingNext. + const read = () => { + let result; + try { + result = self.#tryRead(state); + } catch (error) { + return PromiseReject(error); } + if (result !== kPull) return PromiseResolve(result); + return self.#readAsync(state); + }; + const clearPending = (pending) => { + const onSettled = () => { + if (state.pendingNext === pending) state.pendingNext = null; + }; + PromisePrototypeThen(pending, onSettled, onSettled); }; return ObjectSetPrototypeOf({ next() { - const next = PromisePrototypeThen( - state.pendingNext, - getNext, - getNext); + let next; + if (state.pendingNext !== null) { + next = PromisePrototypeThen(state.pendingNext, read, read); + } else { + state.waiting = false; + next = read(); + // Settled at once: nothing for a later next() to wait for. + if (!state.waiting) return next; + } state.pendingNext = next; - markPromiseAsHandled(state.pendingNext); + markPromiseAsHandled(next); + clearPending(next); return next; }, @@ -277,11 +249,7 @@ class ShareImpl { this.#cancelError = reason; } - const cancelPendingRead = this.#cancelPendingRead; - if (cancelPendingRead !== null) { - this.#cancelPendingRead = null; - cancelPendingRead(kShareCancelled); - } + if (this.#resolvePull !== null) this.#finishPull(); try { const returnMethod = this.#sourceIterator?.return; @@ -327,6 +295,105 @@ class ShareImpl { // Internal methods + // Read the consumer's next batch if it is buffered, or the end: returns + // its result, or kPull if the source must be read first. + #tryRead(state) { + if (state.detached) { + if (state.error !== kNoShareError) throw state.error; + return new IterResult(true, undefined); + } + + if (this.#cancelled) { + state.detached = true; + state.error = this.#cancelError; + this.#deleteConsumer(state); + if (state.error !== kNoShareError) throw state.error; + return new IterResult(true, undefined); + } + + // Check if data is available in buffer + const bufferIndex = state.cursor - this.#bufferStart; + if (bufferIndex < this.#buffer.length) { + const chunk = this.#readEntry(this.#buffer.get(bufferIndex)); + const cursor = state.cursor; + state.cursor++; + if (cursor === this.#cachedMinCursor && + --this.#cachedMinCursorConsumers === 0) { + this.#tryTrimBuffer(); + } + return new IterResult(false, chunk); + } + + if (this.#sourceExhausted) { + state.detached = true; + this.#deleteConsumer(state); + if (this.#sourceError !== kNoShareError) { + state.error = this.#sourceError; + throw state.error; + } + return new IterResult(true, undefined); + } + return kPull; + } + + // Read the consumer's next batch once the source has been read. Loops + // until it gets data, the source is exhausted, or the consumer is + // detached: multiple consumers may be woken after a single pull, and + // those that find no data at their cursor must pull again rather than + // terminate prematurely. + #readAsync(state) { + state.waiting = true; + if (this.#bufferedBytes < this.#options.budget) { + return this.#readAfterPull(state); + } + return this.#readAsyncLoop(state); + } + + async #readAsyncLoop(state) { + for (;;) { + // Need to pull from source - check buffer limit + let shouldBuffer; + if (this.#bufferedBytes < this.#options.budget) { + shouldBuffer = true; + } else { + try { + shouldBuffer = await this.#waitForBufferSpace(); + } catch (error) { + state.detached = true; + if (this.#deleteConsumer(state)) { + this.#tryTrimBuffer(); + } + throw error; + } + } + if (shouldBuffer === null) { + state.detached = true; + state.error = this.#cancelError; + this.#deleteConsumer(state); + if (state.error !== kNoShareError) throw state.error; + return new IterResult(true, undefined); + } + + await this.#pullFromSource(!shouldBuffer); + if (!shouldBuffer) { + await this.#waitForBufferSpaceAfterDrop(); + } + + const result = this.#tryRead(state); + if (result !== kPull) return result; + } + } + + // The common case of #readAsyncLoop(), without an async function: there + // is room in the buffer, so read the source and then the consumer's batch. + #readAfterPull(state) { + state.afterPull ??= () => { + const result = this.#tryRead(state); + return result !== kPull ? result : this.#readAsyncLoop(state); + }; + return PromisePrototypeThen(this.#pullFromSource(false), state.afterPull); + } + async #waitForBufferSpace() { while (this.#bufferedBytes >= this.#options.budget) { if (this.#cancelled || @@ -384,82 +451,85 @@ class ShareImpl { return PromiseResolve(); } - if (this.#pulling) { - const { promise, resolve } = PromiseWithResolvers(); - ArrayPrototypePush(this.#pullWaiters, resolve); - return promise; - } + if (this.#pulling) return this.#pullDone; this.#pulling = true; + const { promise, resolve } = PromiseWithResolvers(); + this.#pullDone = promise; + this.#resolvePull = resolve; + this.#pullDiscard = discard; - return (async () => { - try { - if (!this.#sourceIterator) { - if (isAsyncIterable(this.#source)) { - this.#sourceIterator = - this.#source[SymbolAsyncIterator](); - } else if (isSyncIterable(this.#source)) { - const syncIterator = - this.#source[SymbolIterator](); - this.#sourceIterator = ObjectSetPrototypeOf({ - async next() { - return syncIterator.next(); - }, - async return() { - return syncIterator.return?.() ?? - new IterResult(true, undefined); - }, - }, null); - } else { - throw new ERR_INVALID_ARG_TYPE( - 'source', ['AsyncIterable', 'Iterable'], this.#source); - } + let next; + try { + if (!this.#sourceIterator) { + if (isAsyncIterable(this.#source)) { + this.#sourceIterator = + this.#source[SymbolAsyncIterator](); + } else if (isSyncIterable(this.#source)) { + const syncIterator = + this.#source[SymbolIterator](); + this.#sourceIterator = ObjectSetPrototypeOf({ + // Its result is passed to PromiseResolve(). + next() { + return syncIterator.next(); + }, + async return() { + return syncIterator.return?.() ?? + new IterResult(true, undefined); + }, + }, null); + } else { + throw new ERR_INVALID_ARG_TYPE( + 'source', ['AsyncIterable', 'Iterable'], this.#source); } + } + next = this.#sourceIterator.next(); + } catch (error) { + this.#onReadError(error); + return promise; + } + PromisePrototypeThen(PromiseResolve(next), this.#onReadResult, + this.#onReadError); + return promise; + } - const result = await this.#readSource(); - - if (this.#cancelled || result === kShareCancelled) return; - - if (result.done) { - this.#sourceExhausted = true; - } else if (!discard) { - this.#bufferBatch(result.value); - } - } catch (error) { - this.#sourceError = error; + // The handlers of a read of the source, created once per share. A read + // ended by a cancellation is ignored. + #onReadResult = (result) => { + if (this.#resolvePull === null || this.#cancelled) return; + try { + if (result.done) { this.#sourceExhausted = true; - } finally { - if (this.#sourceExhausted && this.#consumers.size === 0) { - this.#cleanupFactorySignal(); - } - this.#pulling = false; - for (let i = 0; i < this.#pullWaiters.length; i++) { - this.#pullWaiters[i](); - } - this.#pullWaiters = []; + } else if (!this.#pullDiscard) { + this.#bufferBatch(result.value); } - })(); - } - - // Read the next result of the source, settling early with kShareCancelled - // if the share is cancelled first. Racing the read with a promise settled - // by cancel() would add reactions to that promise on every read, which - // would be kept until the share is cancelled or collected. - #readSource() { - const next = this.#sourceIterator.next(); - const { promise, resolve, reject } = PromiseWithResolvers(); - this.#cancelPendingRead = resolve; - PromisePrototypeThen( - PromiseResolve(next), - (result) => { - if (this.#cancelPendingRead === resolve) this.#cancelPendingRead = null; - resolve(result); - }, - (error) => { - if (this.#cancelPendingRead === resolve) this.#cancelPendingRead = null; - reject(error); - }); - return promise; + } catch (error) { + this.#sourceError = error; + this.#sourceExhausted = true; + } + this.#finishPull(); + }; + + #onReadError = (error) => { + if (this.#resolvePull === null || this.#cancelled) return; + this.#sourceError = error; + this.#sourceExhausted = true; + this.#finishPull(); + }; + + #finishPull() { + const resolve = this.#resolvePull; + this.#resolvePull = null; + this.#pullDone = null; + if (this.#sourceExhausted && this.#consumers.size === 0) { + this.#cleanupFactorySignal(); + } + this.#pulling = false; + for (let i = 0; i < this.#pullWaiters.length; i++) { + this.#pullWaiters[i](); + } + this.#pullWaiters = []; + resolve(); } #bufferBatch(batch) { diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 9a94b0db2b6..7dd6a3fe172 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -468,6 +468,11 @@ function createAbortableIterator(source, signal) { * @param {number} fallback - Cursor to return when set is empty * @returns {{ minCursor: number, minCursorConsumers: number }} */ +// The result of getMinCursor(), reused: it runs whenever the slowest +// consumer of a share or broadcast advances, which can be once per batch, +// and its callers read the result at once. +const minCursorResult = { __proto__: null, minCursor: 0, minCursorConsumers: 0 }; + function getMinCursor(consumers, fallback) { let minCursor = fallback; let minCursorConsumers = 0; @@ -479,7 +484,9 @@ function getMinCursor(consumers, fallback) { minCursorConsumers++; } } - return { __proto__: null, minCursor, minCursorConsumers }; + minCursorResult.minCursor = minCursor; + minCursorResult.minCursorConsumers = minCursorConsumers; + return minCursorResult; } /** From c7fb80be9afb8fbc679b12628319404664673a67 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 10:47:03 +0000 Subject: [PATCH 44/58] stream: apply stream/iter stateful transforms without async generators Every stateful (generator) transform in an async pipeline added two async generators of stream/iter's own to every batch, besides the transform's: one to append the null flush signal to its source, and one to read and normalize its output. Validated transforms, such as compression, added one. Write both out by hand. The source of the transform passes reads to the pipeline with one reaction each, then yields null once, and passes return() and throw() on as yield* does. The output is read with one reaction per item, and normalized as before; outputs that are not async iterables are still read by an async generator, as for await reads them. The transform is still called on the first pull. For 16-byte chunks one per batch, iterating pull() with one stateful transform is 23-31% faster, and with three 36-45% faster; compression is unchanged. Assisted-by: OpenCode --- lib/internal/streams/iter/pull.js | 251 ++++++++++++++++--- test/parallel/test-stream-iter-pull-async.js | 59 +++++ 2 files changed, 275 insertions(+), 35 deletions(-) diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 9dca617e7f5..2cc643eafb2 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -972,22 +972,98 @@ function applyFusedStatelessAsyncTransforms(source, run, abort) { } /** - * Append a null flush signal after the source is exhausted. - * @yields {Uint8Array[]} - */ -/** - * Append a null flush signal after the source is exhausted. - * @yields {Uint8Array[]} + * The source of a stateful transform: what an async generator doing + * `yield* source; yield null;` would yield, written out by hand, as it is + * read for every batch. Reads are passed to + * the source's iterator, with one reaction each; once it is done, null is + * yielded once, as the flush signal. return() and throw() are passed on as + * yield* does. + * @param {AsyncIterable} source + * @returns {AsyncIterator} */ -async function* withFlushAsync(source) { - yield* source; - yield null; +function createFlushSource(source) { + let iterator = null; + // Whether the source is done, and the flush signal yielded; then the + // source is ended. + let sourceDone = false; + let done = false; + + function onResult(result) { + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + done = true; + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); + } + if (result.done) { + sourceDone = true; + return new IterResult(false, null); + } + return new IterResult(false, result.value); + } + + function onError(error) { + done = true; + throw error; + } + + return ObjectSetPrototypeOf({ + next() { + if (done || sourceDone) { + done = true; + return PromiseResolve(new IterResult(true, undefined)); + } + try { + iterator ??= source[SymbolAsyncIterator](); + return PromisePrototypeThen(PromiseResolve(iterator.next()), + onResult, onError); + } catch (error) { + done = true; + return PromiseReject(error); + } + }, + return(value) { + const result = new IterResult(true, value); + if (done || sourceDone || iterator === null) { + done = true; + return PromiseResolve(result); + } + done = true; + return PromisePrototypeThen(closeAsyncIterator(iterator, false), + () => result); + }, + throw(error) { + if (done || sourceDone || iterator === null) { + done = true; + return PromiseReject(error); + } + try { + const throwMethod = iterator.throw; + if (throwMethod === undefined || throwMethod === null) { + // As yield*: close the source, then throw a TypeError. + done = true; + return PromisePrototypeThen(closeAsyncIterator(iterator, false), + () => { + throw new ERR_INVALID_STATE.TypeError( + 'The iterator does not provide ' + + 'a \'throw\' method'); + }); + } + return PromisePrototypeThen( + PromiseResolve(FunctionPrototypeCall(throwMethod, iterator, error)), + onResult, onError); + } catch (thrown) { + done = true; + return PromiseReject(thrown); + } + }, + [SymbolAsyncIterator]() { return this; }, + }, null); } -async function* applyStatefulAsyncTransform( - source, transform, receiver, options) { - const output = FunctionPrototypeCall( - transform, receiver, withFlushAsync(source), options); +// Iterate the output of a stateful transform as for await does, for outputs +// that are not async iterables: see createStatefulTransformIterator(). +async function* yieldStatefulOutput(output) { for await (const item of output) { if (item === null) continue; // Fast path: item is already a Uint8Array[] batch (e.g. compression transforms) @@ -1014,24 +1090,133 @@ async function* applyStatefulAsyncTransform( } /** - * Fast path for validated stateful transforms (e.g. compression). - * Skips withFlushAsync (transform handles done internally) and - * skips isUint8ArrayBatch validation (transform guarantees valid output). - * @yields {Uint8Array[]} + * A stateful transform applied to `source`: what an async generator doing + * `for await (const item of output)` over the transform's output would + * yield, written out by hand, as it runs for every batch. The transform is + * called on the first next(), with createFlushSource(source), or with + * `source` itself for a validated transform (e.g. compression), whose + * output needs no normalization. An output that is not an async iterable + * is read by the async generator instead (yieldStatefulOutput()). + * @param {AsyncIterable} source + * @param {Function} transform + * @param {object} receiver + * @param {TransformOptions} options + * @param {PipelineAbort} abort + * @param {boolean} validated + * @returns {AsyncIterator} */ -async function* applyValidatedStatefulAsyncTransform( - source, transform, receiver, options, abort) { - const output = FunctionPrototypeCall( - transform, receiver, source, options); - for await (const batch of output) { - if (batch.length > 0) { - yield batch; +function createStatefulTransformIterator( + source, transform, receiver, options, abort, validated) { + let state = kStart; + let iterator; + let nextMethod; + + function pull() { + return PromisePrototypeThen( + PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)), + onItem, onError); + } + + function onItem(result) { + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + state = kDone; + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); + } + if (result.done) { + state = kDone; + // Without the flush signal there is no extra read to give the + // pipeline a chance to see an abort. + if (validated) abort.throwIfAborted(); + return new IterResult(true, undefined); + } + const item = result.value; + if (validated) { + return item.length > 0 ? new IterResult(false, item) : pull(); + } + if (item === null) return pull(); + if (isUint8ArrayBatch(item)) { + return item.length > 0 ? new IterResult(false, item) : pull(); + } + if (isUint8Array(item)) return new IterResult(false, [item]); + // Slow path: flatten arbitrary transform yield. An error doing so + // closes the output, as for await does when its body throws. + return PromisePrototypeThen(flattenItem(item), (batch) => { + return batch.length > 0 ? new IterResult(false, batch) : pull(); + }, (error) => { + state = kDone; + return PromisePrototypeThen(closeAsyncIterator(iterator, true), () => { + throw error; + }); + }); + } + + async function flattenItem(item) { + const batch = []; + for await (const chunk of flattenTransformYieldAsync(item)) { + batch[batch.length] = chunk; } + return batch; } - // Check abort after the transform completes - without the - // withFlushAsync wrapper there is no extra yield to give - // the outer pipeline a chance to see the abort. - abort.throwIfAborted(); + + function onError(error) { + state = kDone; + throw error; + } + + return ObjectSetPrototypeOf({ + next() { + if (state === kDone) { + return PromiseResolve(new IterResult(true, undefined)); + } + if (state === kStart) { + state = kActive; + try { + const output = FunctionPrototypeCall( + transform, receiver, + validated ? source : createFlushSource(source), options); + if (output != null && + typeof output[SymbolAsyncIterator] === 'function') { + iterator = output[SymbolAsyncIterator](); + } else { + // Iterated as for await would: sync iterables, and errors. + iterator = yieldStatefulOutput(output); + } + nextMethod = iterator.next; + } catch (error) { + state = kDone; + return PromiseReject(error); + } + } + try { + return pull(); + } catch (error) { + state = kDone; + return PromiseReject(error); + } + }, + return(value) { + const result = new IterResult(true, value); + if (state !== kActive) { + state = kDone; + return PromiseResolve(result); + } + state = kDone; + return PromisePrototypeThen(closeAsyncIterator(iterator, false), + () => result); + }, + throw(error) { + if (state !== kActive) { + state = kDone; + return PromiseReject(error); + } + state = kDone; + return PromisePrototypeThen(closeAsyncIterator(iterator, true), + () => { throw error; }); + }, + [SymbolAsyncIterator]() { return this; }, + }, null); } /** @@ -1131,13 +1316,9 @@ function createAsyncTransformLayers(normalized, transforms, abort) { statelessRun = []; } const opts = new TransformOptions(abort); - if (transform[kValidatedTransform]) { - current = applyValidatedStatefulAsyncTransform( - current, transform.transform, transform.receiver, opts, abort); - } else { - current = applyStatefulAsyncTransform( - current, transform.transform, transform.receiver, opts); - } + current = createStatefulTransformIterator( + current, transform.transform, transform.receiver, opts, abort, + transform[kValidatedTransform] === true); } else { ArrayPrototypePush(statelessRun, transform); } diff --git a/test/parallel/test-stream-iter-pull-async.js b/test/parallel/test-stream-iter-pull-async.js index c8293257712..53480cb653d 100644 --- a/test/parallel/test-stream-iter-pull-async.js +++ b/test/parallel/test-stream-iter-pull-async.js @@ -4,6 +4,7 @@ const common = require('../common'); const assert = require('assert'); const { + array, broadcast, dump, from, @@ -428,6 +429,63 @@ async function testPullSignalListenerRemoved() { await assert.rejects(iter.next(), { name: 'AbortError' }); } +// Stateful transforms: called on the first pull, with a source that ends +// with one null flush signal; their output is normalized; stopping early +// closes the output and the source. +async function testStatefulTransformProtocol() { + const log = []; + const stateful = { + async *transform(source) { + log.push('called'); + try { + for await (const batch of source) { + log.push(batch === null ? 'flush' : `batch ${batch[0][0]}`); + if (batch === null) { + yield 'end'; + } else { + yield batch; + yield batch[0]; + yield null; + yield []; + } + } + } finally { + log.push('output closed'); + } + }, + }; + const iterable = pull(createLoggedSource(log, [1, 2]), stateful); + assert.deepStrictEqual(log, []); + const chunks = await array(iterable); + assert.deepStrictEqual(chunks.map((chunk) => chunk[0]), [1, 1, 2, 2, 101]); + assert.deepStrictEqual(log, [ + 'called', 'next 0', 'batch 1', 'next 1', 'batch 2', 'next 2', 'flush', + 'output closed', + ]); + + log.length = 0; + const iterator = pull(createLoggedSource(log, [1, 2, 3]), + stateful)[Symbol.asyncIterator](); + assert.strictEqual((await iterator.next()).value[0][0], 1); + assert.strictEqual((await iterator.return()).done, true); + assert.deepStrictEqual(log, [ + 'called', 'next 0', 'batch 1', 'return', 'output closed', + ]); + assert.strictEqual((await iterator.next()).done, true); + + // Outputs that are sync iterables are read as for await reads them. + const syncOutput = { transform: () => [[Uint8Array.of(7)], 'x'] }; + assert.deepStrictEqual((await array(pull(from('a'), syncOutput))).map( + (chunk) => chunk[0]), [7, 120]); + + // Errors from the transform reject the pull. + const failing = { + // eslint-disable-next-line require-yield + async *transform() { throw new Error('stateful failed'); }, + }; + await assert.rejects(array(pull(from('a'), failing)), /stateful failed/); +} + // Pull consumer break (return()) cleans up transform signal async function testPullConsumerBreakCleanup() { let signalAborted = false; @@ -884,6 +942,7 @@ async function testTransformReturnClosesOutputAndSource() { testPullSignalAbortWhileTransformPending(), testPullSignalAbortWhileIdleClosesSource(), testPullSignalListenerRemoved(), + testStatefulTransformProtocol(), testTransformOptionsSignalAssignable(), testTransformErrorClosesSource(), testTransformSourceErrorDoesNotCloseSource(), From 7e718b6b1cb8a76c1a5d29c77ec1b5da7ed42fe3 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 12:30:17 +0000 Subject: [PATCH 45/58] stream: read stream/iter from() sync sources without a generator from() read sync iterable sources with a sync generator doing for...of over the source, resumed for every batch. Its next() calls went through FunctionPrototypeCall(), which V8 does not inline for the next() of a generator. Read the source with SyncSourceReader, which does what that generator did, written out by hand: it collects single chunks into batches, splits oversized batches, flushes before other values, and closes the source as for...of does on return(), throw() and cancellation. It calls the next() of a generator as the constant %GeneratorPrototype%.next, and an own next() method as iterator.next(), so that V8 can inline either. Like for...of, it reads next() only once. Iterating from() over a sync source yielding one-chunk batches is about 15% faster, and over a generator yielding single chunks about 6-11% faster. Piping a sync source to a writer is about 20% faster. The microtask that keeps the normalizer busy until the tick after each batch, for async generator parity, is the remaining per-batch cost; note it. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 280 +++++++++++++++---- test/parallel/test-stream-iter-from-async.js | 73 +++++ 2 files changed, 304 insertions(+), 49 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 23b7021a649..d32d2353bdc 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -15,6 +15,8 @@ const { DataViewPrototypeGetByteOffset, FunctionPrototypeCall, ObjectFreeze, + ObjectGetOwnPropertyDescriptor, + ObjectGetPrototypeOf, ObjectSetPrototypeOf, PromisePrototypeThen, PromiseReject, @@ -41,6 +43,7 @@ const { lazyDOMException } = require('internal/util'); const { isAnyArrayBuffer, + isGeneratorObject, isPromise, isTypedArray, isUint8Array, @@ -1119,63 +1122,237 @@ function normalizeAsyncSource(source, context) { } // A value of a sync source that is not a Uint8Array or a Uint8Array[] -// batch, yielded by readSyncSource() to be normalized asynchronously. +// batch, returned by SyncSourceReader to be normalized asynchronously. function SyncSourceValue(value) { this.value = value; } SyncSourceValue.prototype = ObjectFreeze({ __proto__: null }); /** - * Read a sync iterable source for from(): yields Uint8Array[] batches, + * Read a sync iterable source for from(): returns Uint8Array[] batches, * collecting single Uint8Arrays into batches of up to FROM_BATCH_SIZE * chunks, and a SyncSourceValue for any other value, after the chunks * collected before it. - * @param {Iterable} source - * @param {object} context - * @yields {Uint8Array[]|SyncSourceValue} + * + * This is what a sync generator doing `for (const value of source)` would + * yield, written out by hand, as it runs for every batch: next(), return() + * and throw() behave as that generator's do. The source is closed, as + * for...of closes it, when the reader is returned or thrown into while it + * is reading the source, or when the normalization is found cancelled; + * errors closing it are ignored for throw() and cancellation. An error from + * the source's next() or [Symbol.iterator]() ends the reader without + * closing the source. */ -function* readSyncSource(source, context) { - throwIfNormalizationCancelled(context); - if (!isSyncIterable(source)) { - throw new ERR_INVALID_ARG_TYPE( - 'source', ['Iterable', 'AsyncIterable'], source); - } - let batch = []; - for (const value of source) { - throwIfNormalizationCancelled(context); - // Fast path 1: value is already a Uint8Array[] batch - if (isUint8ArrayBatch(value)) { - // Flush any accumulated batch first - if (batch.length > 0) { - yield batch; - batch = []; +// %GeneratorPrototype%.next, taken before user code runs. +const GeneratorPrototypeNext = + ObjectGetPrototypeOf(function*() {}).prototype.next; + +// How a SyncSourceReader calls the next() of its source's iterator. +const kNextOwn = 1; +const kNextGenerator = 2; + +// The states of a SyncSourceReader: not started, reading the source, done. +const kReaderStart = 0; +const kReaderActive = 1; +const kReaderDone = 2; + +class SyncSourceReader { + #source; + #context; + #iterator = null; + #next = null; + // kNextGenerator if the source's iterator is a generator whose next() is + // %GeneratorPrototype%.next, which is then called as a constant; kNextOwn + // if its next() is an own data property, which is then called as + // `iterator.next()`; otherwise 0, and #next is called with + // FunctionPrototypeCall(). Calls V8 can inline make reading single chunks + // faster. (Generators of different generator functions don't share a map, + // so for them `iterator.next()` would be a megamorphic load per chunk.) + #direct = 0; + #state = kReaderStart; + // Single Uint8Arrays read and not yet returned. + #batch = []; + // A value to return on the next call, after the batch returned before it. + #pending = undefined; + // A batch larger than FROM_BATCH_SIZE being returned in slices, and the + // offset of the next slice. + #large = null; + #offset = 0; + + constructor(source, context) { + this.#source = source; + this.#context = context; + } + + next() { + switch (this.#state) { + case kReaderDone: + return new IterResult(true, undefined); + case kReaderStart: + this.#start(); + break; + } + if (this.#pending !== undefined) { + const value = this.#pending; + this.#pending = undefined; + return new IterResult(false, value); + } + if (this.#large !== null) return new IterResult(false, this.#nextSlice()); + // Read into locals: the loop runs for every chunk of the source. + const iterator = this.#iterator; + const next = this.#next; + const direct = this.#direct; + const context = this.#context; + const batch = this.#batch; + for (;;) { + let value; + try { + const result = direct === kNextGenerator ? + FunctionPrototypeCall(GeneratorPrototypeNext, iterator) : + direct === kNextOwn ? iterator.next() : + FunctionPrototypeCall(next, iterator); + if ((typeof result !== 'object' && typeof result !== 'function') || + result === null) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.next()', result); + } + if (result.done) { + this.#state = kReaderDone; + if (batch.length > 0) { + return new IterResult(false, this.#takeBatch()); + } + return new IterResult(true, undefined); + } + value = result.value; + } catch (error) { + this.#state = kReaderDone; + throw error; } - if (value.length <= FROM_BATCH_SIZE) { - if (value.length !== 0) yield value; - } else { - yield* yieldBoundedBatch(value); + if (context?.cancelled) { + this.#close(true); + throw context.reason; } - continue; - } - // Fast path 2: value is a single Uint8Array (very common) - if (isUint8Array(value)) { - batch[batch.length] = value; - if (batch.length === FROM_BATCH_SIZE) { - yield batch; - batch = []; + // Fast path 1: value is a single Uint8Array (very common). Checking + // ArrayIsArray() first keeps batches from paying for isUint8Array(). + const isArray = ArrayIsArray(value); + if (!isArray && isUint8Array(value)) { + batch[batch.length] = value; + if (batch.length === FROM_BATCH_SIZE) { + return new IterResult(false, this.#takeBatch()); + } + continue; } - continue; + // Fast path 2: value is already a Uint8Array[] batch + if (isArray && isUint8ArrayBatch(value)) { + if (value.length === 0) { + if (batch.length > 0) { + return new IterResult(false, this.#takeBatch()); + } + continue; + } + let ready = value; + if (value.length > FROM_BATCH_SIZE) { + this.#large = value; + this.#offset = 0; + ready = undefined; + } + // Flush any accumulated batch first + if (batch.length > 0) { + this.#pending = ready; + return new IterResult(false, this.#takeBatch()); + } + return new IterResult(false, ready ?? this.#nextSlice()); + } + // Slow path: flush, then have the value normalized + const sourceValue = new SyncSourceValue(value); + if (batch.length > 0) { + this.#pending = sourceValue; + return new IterResult(false, this.#takeBatch()); + } + return new IterResult(false, sourceValue); } - // Slow path: flush, then have the value normalized - if (batch.length > 0) { - yield batch; - batch = []; + } + + return() { + const state = this.#state; + this.#state = kReaderDone; + this.#pending = undefined; + this.#large = null; + if (state === kReaderActive) this.#close(false); + return new IterResult(true, undefined); + } + + throw(error) { + const state = this.#state; + this.#state = kReaderDone; + this.#pending = undefined; + this.#large = null; + if (state === kReaderActive) this.#close(true); + throw error; + } + + #start() { + this.#state = kReaderActive; + try { + throwIfNormalizationCancelled(this.#context); + const source = this.#source; + if (!isSyncIterable(source)) { + throw new ERR_INVALID_ARG_TYPE( + 'source', ['Iterable', 'AsyncIterable'], source); + } + const iterator = source[SymbolIterator](); + const next = iterator.next; + this.#iterator = iterator; + this.#next = next; + if (isGeneratorObject(iterator) && next === GeneratorPrototypeNext) { + this.#direct = kNextGenerator; + } else if (typeof iterator === 'object' && iterator !== null) { + const descriptor = ObjectGetOwnPropertyDescriptor(iterator, 'next'); + if (descriptor !== undefined && descriptor.value === next) { + this.#direct = kNextOwn; + } + } + } catch (error) { + this.#state = kReaderDone; + throw error; } - yield new SyncSourceValue(value); } - // Yield any remaining batched values - if (batch.length > 0) { - yield batch; + + #takeBatch() { + const batch = this.#batch; + this.#batch = []; + return batch; + } + + #nextSlice() { + const large = this.#large; + const offset = this.#offset; + this.#offset = offset + FROM_BATCH_SIZE; + if (this.#offset >= large.length) this.#large = null; + return ArrayPrototypeSlice(large, offset, offset + FROM_BATCH_SIZE); + } + + // Close the source as for...of does: errors are ignored if `quiet` (an + // error is being thrown), and otherwise thrown, as is a result of + // return() that is not an object. + #close(quiet) { + this.#state = kReaderDone; + const iterator = this.#iterator; + let result; + try { + const returnMethod = iterator.return; + if (returnMethod === undefined || returnMethod === null) return; + result = FunctionPrototypeCall(returnMethod, iterator); + } catch (error) { + if (quiet) return; + throw error; + } + if (!quiet && + ((typeof result !== 'object' && typeof result !== 'function') || + result === null)) { + throw new ERR_INVALID_RETURN_VALUE( + 'an object', 'iterator.return()', result); + } } } @@ -1184,19 +1361,19 @@ function* readSyncSource(source, context) { * Uint8Array, without an async generator layer for every batch (see * createAsyncSourceNormalizer()). * - * The source is read by the sync generator readSyncSource(), so for...of - * reads and closes it, and this iterator only adds the asynchronous - * normalization of values that need it. Operations on the generator mirror - * those on the async generator this replaces: return() and throw() are - * passed to it, after closing the value being normalized, and an error - * normalizing a value is thrown into it, so that for...of closes the source - * as for an error in the loop body. + * The source is read by a SyncSourceReader, which reads and closes it as + * for...of does, and this iterator only adds the asynchronous normalization + * of values that need it. Operations on the reader mirror those on the + * async generator this replaces: return() and throw() are passed to it, + * after closing the value being normalized, and an error normalizing a + * value is thrown into it, so that the source is closed as for an error in + * the body of a for...of loop. * @param {Iterable} source * @param {object} context * @returns {object} An object with next(), return() and throw(). */ function createSyncSourceNormalizer(source, context) { - const reader = readSyncSource(source, context); + const reader = new SyncSourceReader(source, context); let done = false; // A normalizeAsyncSourceValue() generator for the value being normalized. let valueBatches = null; @@ -1206,6 +1383,11 @@ function createSyncSourceNormalizer(source, context) { // until the tick after a yield or return (both await their operand), so // that a next(), return() or throw() made synchronously after this one is // queued behind it; do the same. + // TODO(@jasnell): This costs a microtask per batch. Running a call made + // while the previous one has finished synchronously right away would make + // for-await over from(syncSource) about 1.3x faster, at the cost of this + // async generator parity for calls made back to back without awaiting + // (see testFromSyncSourceQueuesBehindReturn). function finish(result) { PromisePrototypeThen(kResolvedPromise, release); return result; diff --git a/test/parallel/test-stream-iter-from-async.js b/test/parallel/test-stream-iter-from-async.js index cfbb46fa9b6..44befd8eb7e 100644 --- a/test/parallel/test-stream-iter-from-async.js +++ b/test/parallel/test-stream-iter-from-async.js @@ -724,6 +724,78 @@ async function testFromSyncSourceBatching() { ]); } +async function testFromSyncSourceLargeAndEmptyBatches() { + // Batches of more than 128 chunks are split, after the chunks collected + // before them; empty batches only flush collected chunks. Returning while + // a batch is split closes the source, returning after its last batch + // does not. + const large = Array.from({ length: 300 }, () => new Uint8Array(1)); + let log = []; + let iterator = from(createLoggedSyncSource(log, [ + Uint8Array.of(1), [], [], large, Uint8Array.of(2), + ]))[Symbol.asyncIterator](); + for (let i = 0; i < 6; i++) await settle(log, `result ${i}`, iterator.next()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'next 1', 'result 0: 1', 'next 2', 'next 3', + 'result 1: 128', 'result 2: 128', 'result 3: 44', 'next 4', 'next 5', + 'result 4: 1', 'result 5: done', + ]); + + log = []; + iterator = from(createLoggedSyncSource(log, [large]))[ + Symbol.asyncIterator](); + await settle(log, 'result', iterator.next()); + await settle(log, 'return', iterator.return()); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'result: 128', 'return', 'return: done', + ]); + + // A result whose value getter throws ends the iteration without closing + // the source. + log = []; + const source = { + [Symbol.iterator]() { + return { + next() { + log.push('next'); + return { done: false, get value() { throw new Error('getter'); } }; + }, + return() { + log.push('return'); + return { done: true }; + }, + }; + }, + }; + iterator = from(source)[Symbol.asyncIterator](); + await settle(log, 'result 0', iterator.next()); + await settle(log, 'result 1', iterator.next()); + assert.deepStrictEqual(log, ['next', 'result 0: getter', 'result 1: done']); + + // The next() method of the source's iterator is read once, as for...of + // reads it. + let reads = 0; + let count = 0; + const accessorSource = { + [Symbol.iterator]() { + return { + get next() { + reads++; + return () => (count < 3 ? + { done: false, value: [Uint8Array.of(count++)] } : + { done: true, value: undefined }); + }, + }; + }, + }; + const chunks = []; + for await (const batch of from(accessorSource)) chunks.push(...batch); + assert.deepStrictEqual(chunks, [ + Uint8Array.of(0), Uint8Array.of(1), Uint8Array.of(2), + ]); + assert.strictEqual(reads, 1); +} + async function testFromSyncSourceErrors() { // An error from the source does not close it. let log = []; @@ -900,6 +972,7 @@ Promise.all([ testFromReturnAndThrowBeforeStart(), testFromReturnAndThrowCloseSource(), testFromSyncSourceBatching(), + testFromSyncSourceLargeAndEmptyBatches(), testFromSyncSourceErrors(), testFromSyncSourceReturnAndThrow(), testFromSyncSourceQueuesBehindReturn(), From d6e4869b7ba2982e94ff0e5587a0fb549506025f Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 12:35:30 +0000 Subject: [PATCH 46/58] stream: read stream/iter share() sync sources at once A share() consumer that needed the next batch of the source always waited for an asynchronous read of it, also when the source was a sync iterable that from() can read synchronously, and every such read added a reaction to clear the consumer's pending read when it settled. When there is room in the buffer and no read of the source is pending, read a batch of a sync source synchronously, with from()'s kNextSyncBatch, as pipeTo() does: the consumer then gets it at once. Values that from() normalizes asynchronously are still read asynchronously. A read waiting for the source now clears itself from the consumer's pending read when it settles, without a reaction of its own; only next() calls queued behind another still add one. With two consumers of a share of a sync source, one 16-byte chunk per batch, reading is about 2.4 times faster, on par with tee() of a ReadableStream, with 60% less allocated per chunk. For an async source it is about 7% faster. Assisted-by: OpenCode --- lib/internal/streams/iter/share.js | 142 ++++++++++++++---- test/parallel/test-stream-iter-share-async.js | 52 +++++++ 2 files changed, 161 insertions(+), 33 deletions(-) diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index ba457e9023c..57e0885b2ea 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -30,6 +30,7 @@ const { fromSync, isAsyncIterable, isSyncIterable, + kNextSyncBatch, } = require('internal/streams/iter/from'); const { @@ -78,6 +79,15 @@ const kNoShareError = Symbol('kNoShareError'); const kPull = Symbol('kPull'); const kSetFactorySignal = Symbol('kSetFactorySignal'); +// Clear `pending` from the pendingNext of the consumer `state` when it +// settles, unless a next() has been queued behind it. +function clearPendingNext(state, pending) { + const onSettled = () => { + if (state.pendingNext === pending) state.pendingNext = null; + }; + PromisePrototypeThen(pending, onSettled, onSettled); +} + class ShareImpl { #source; #options; @@ -160,6 +170,8 @@ class ShareImpl { waiting: false, // Created by #readAfterPull() on first use. afterPull: undefined, + // The read #readAfterPull() returned, while it is pending. + waitingRead: null, // The last pending next() of the consumer, if it is waiting for the // source: the next one waits for it, as next() calls of an async // generator do. @@ -186,18 +198,19 @@ class ShareImpl { let result; try { result = self.#tryRead(state); + // A sync source can be read at once, when there is room. + while (result === kPull && + self.#bufferedBytes < self.#options.budget && + self.#pullSync()) { + result = self.#tryRead(state); + } } catch (error) { return PromiseReject(error); } if (result !== kPull) return PromiseResolve(result); return self.#readAsync(state); }; - const clearPending = (pending) => { - const onSettled = () => { - if (state.pendingNext === pending) state.pendingNext = null; - }; - PromisePrototypeThen(pending, onSettled, onSettled); - }; + const clearPending = (pending) => clearPendingNext(state, pending); return ObjectSetPrototypeOf({ next() { @@ -212,7 +225,9 @@ class ShareImpl { } state.pendingNext = next; markPromiseAsHandled(next); - clearPending(next); + // A read waiting for a pull of the source clears pendingNext + // itself (see #readAfterPull()). + if (next !== state.waitingRead) clearPending(next); return next; }, @@ -344,7 +359,9 @@ class ShareImpl { #readAsync(state) { state.waiting = true; if (this.#bufferedBytes < this.#options.budget) { - return this.#readAfterPull(state); + const read = this.#readAfterPull(state); + state.waitingRead = read; + return read; } return this.#readAsyncLoop(state); } @@ -386,14 +403,69 @@ class ShareImpl { // The common case of #readAsyncLoop(), without an async function: there // is room in the buffer, so read the source and then the consumer's batch. + // When the read settles with the result afterPull() returns, it is no + // longer pending: clear it from pendingNext, unless a next() has been + // queued behind it, without a reaction of its own. #readAfterPull(state) { state.afterPull ??= () => { - const result = this.#tryRead(state); - return result !== kPull ? result : this.#readAsyncLoop(state); + let result; + try { + result = this.#tryRead(state); + } catch (error) { + this.#clearWaitingRead(state); + throw error; + } + if (result === kPull) { + // Still pending: clear it when it settles. + const read = state.waitingRead; + state.waitingRead = null; + if (state.pendingNext === read) clearPendingNext(state, read); + return this.#readAsyncLoop(state); + } + this.#clearWaitingRead(state); + return result; }; return PromisePrototypeThen(this.#pullFromSource(false), state.afterPull); } + #clearWaitingRead(state) { + if (state.pendingNext === state.waitingRead) state.pendingNext = null; + state.waitingRead = null; + } + + // Read a batch of the source synchronously, if it can be read so (a sync + // source normalized by from()) and no read of it is pending. Returns + // whether the source was read: a batch buffered, or its end or error + // recorded. + #pullSync() { + if (this.#pulling || this.#sourceExhausted || this.#cancelled) { + return false; + } + try { + const iterator = this.#getSourceIterator(); + const nextSyncBatch = iterator[kNextSyncBatch]; + if (nextSyncBatch === undefined) return false; + const batch = FunctionPrototypeCall(nextSyncBatch, iterator); + if (batch === undefined) return false; + if (batch === null) { + this.#sourceExhausted = true; + } else { + this.#bufferBatch(batch); + } + } catch (error) { + this.#sourceError = error; + this.#sourceExhausted = true; + } + if (this.#sourceExhausted && this.#consumers.size === 0) { + this.#cleanupFactorySignal(); + } + for (let i = 0; i < this.#pullWaiters.length; i++) { + this.#pullWaiters[i](); + } + this.#pullWaiters = []; + return true; + } + async #waitForBufferSpace() { while (this.#bufferedBytes >= this.#options.budget) { if (this.#cancelled || @@ -461,29 +533,7 @@ class ShareImpl { let next; try { - if (!this.#sourceIterator) { - if (isAsyncIterable(this.#source)) { - this.#sourceIterator = - this.#source[SymbolAsyncIterator](); - } else if (isSyncIterable(this.#source)) { - const syncIterator = - this.#source[SymbolIterator](); - this.#sourceIterator = ObjectSetPrototypeOf({ - // Its result is passed to PromiseResolve(). - next() { - return syncIterator.next(); - }, - async return() { - return syncIterator.return?.() ?? - new IterResult(true, undefined); - }, - }, null); - } else { - throw new ERR_INVALID_ARG_TYPE( - 'source', ['AsyncIterable', 'Iterable'], this.#source); - } - } - next = this.#sourceIterator.next(); + next = this.#getSourceIterator().next(); } catch (error) { this.#onReadError(error); return promise; @@ -493,6 +543,32 @@ class ShareImpl { return promise; } + #getSourceIterator() { + if (!this.#sourceIterator) { + if (isAsyncIterable(this.#source)) { + this.#sourceIterator = + this.#source[SymbolAsyncIterator](); + } else if (isSyncIterable(this.#source)) { + const syncIterator = + this.#source[SymbolIterator](); + this.#sourceIterator = ObjectSetPrototypeOf({ + // Its result is passed to PromiseResolve(). + next() { + return syncIterator.next(); + }, + async return() { + return syncIterator.return?.() ?? + new IterResult(true, undefined); + }, + }, null); + } else { + throw new ERR_INVALID_ARG_TYPE( + 'source', ['AsyncIterable', 'Iterable'], this.#source); + } + } + return this.#sourceIterator; + } + // The handlers of a read of the source, created once per share. A read // ended by a cancellation is ignored. #onReadResult = (result) => { diff --git a/test/parallel/test-stream-iter-share-async.js b/test/parallel/test-stream-iter-share-async.js index 8ea06427543..11df7c9bfa5 100644 --- a/test/parallel/test-stream-iter-share-async.js +++ b/test/parallel/test-stream-iter-share-async.js @@ -507,6 +507,57 @@ async function testShareConsumerConcurrentNextCalls() { assert.strictEqual(dec.decode(r2.value[0]), 'second'); } +// A sync source is read as the consumers pull, including values that are +// normalized asynchronously, and its errors reach every consumer after the +// data before them. +async function testShareSyncSource() { + const reason = new Error('sync source boom'); + function* source() { + yield [Uint8Array.of(1)]; + yield Promise.resolve(Uint8Array.of(2)); + yield 'c'; + yield [Uint8Array.of(4)]; + throw reason; + } + const shared = share(source()); + const consumers = [shared.pull(), shared.pull()]; + const read = async (consumer) => { + const values = []; + try { + for await (const batch of consumer) values.push(...batch.map((c) => c[0])); + } catch (error) { + values.push(error); + } + return values; + }; + const results = await Promise.all(consumers.map(read)); + for (const values of results) { + assert.deepStrictEqual(values, [1, 2, 99, 4, reason]); + } + + // Concurrent next() calls get the batches in order. + function* numbers() { + for (let i = 0; i < 4; i++) yield [Uint8Array.of(i)]; + } + const it = share(numbers()).pull()[Symbol.asyncIterator](); + const all = await Promise.all([it.next(), it.next(), it.next(), it.next(), + it.next()]); + assert.deepStrictEqual(all.map((r) => (r.done ? 'done' : r.value[0][0])), + [0, 1, 2, 3, 'done']); + + // 'strict' backpressure rejects a consumer that would exceed the budget + // when another one lags. + function* big() { + for (let i = 0; i < 4; i++) yield [new Uint8Array(8)]; + } + const strict = share(big(), { budget: 16 }); + const fast = strict.pull()[Symbol.asyncIterator](); + strict.pull(); + await fast.next(); + await fast.next(); + await assert.rejects(fast.next(), { code: 'ERR_OUT_OF_RANGE' }); +} + // share() accepts string source directly (normalized via from()) async function testShareStringSource() { const shared = share('hello-share'); @@ -516,6 +567,7 @@ async function testShareStringSource() { Promise.all([ testBasicShare(), + testShareSyncSource(), testShareMultipleConsumers(), testShareConsumerCount(), testShareCancel(), From 3b21d30c85bd8e01b6f90d875285970880b649a8 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 12:40:44 +0000 Subject: [PATCH 47/58] stream: pump stream/iter Broadcast.from() sources with fewer layers Broadcast.from() read its source through yieldAbortable(), adding a promise, a race and an operation queue to every batch, and wrote each batch with the public writevSync(), converting the batch from() had already normalized as a Web IDL sequence and checking every chunk again. Read the source as pipeTo() does with a signal: one abort listener for the whole pump, with raceSignal(), and, for a sync source with a 'strict' or 'unbounded' policy, batches read synchronously with from()'s kNextSyncBatch while the writes succeed synchronously. The first batch is still read asynchronously, so that it is written after Broadcast.from() has returned and its consumers have been created; with 'drop-oldest' or 'drop-newest', whose writes always succeed, every batch is. Write batches to the writer without converting them again. As before, a cancellation or an abort closes the source without waiting for a pending read, and an error writing a batch closes it. With two consumers of Broadcast.from() of 16-byte chunks one per batch, reading a sync source is about 1.9 times faster, and an async source about 1.4 times faster. Assisted-by: OpenCode --- lib/internal/streams/iter/broadcast.js | 81 ++++++++++++++++--- .../test-stream-iter-broadcast-from.js | 63 +++++++++++++++ 2 files changed, 132 insertions(+), 12 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index e107927730c..06182e5870f 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -49,6 +49,7 @@ const { from, isAsyncIterable, isSyncIterable, + kNextSyncBatch, } = require('internal/streams/iter/from'); const { @@ -70,8 +71,8 @@ const { onSignalAbort, parsePullArgs, toWriterUint8Array, + raceSignal, validateBatchEntry, - yieldAbortable, validateBudget, } = require('internal/streams/iter/utils'); const { @@ -93,6 +94,10 @@ const kOnCancel = Symbol('kOnCancel'); const kPendingWriteRemoved = Symbol('kPendingWriteRemoved'); const kNoBroadcastError = Symbol('kNoBroadcastError'); const kSetFactorySignal = Symbol('kSetFactorySignal'); +// Write a batch of from() (already a Uint8Array[]), without converting it: +// synchronously, returning whether it was written, or asynchronously. +const kWriteBatchSync = Symbol('kWriteBatchSync'); +const kWriteBatch = Symbol('kWriteBatch'); function raceEndWithSignal(promise, signal) { if (!signal) return promise; @@ -721,6 +726,20 @@ class BroadcastWriter { return false; } + [kWriteBatchSync](chunks) { + if (this.#state !== 'open') return false; + const batch = createBatchEntry(chunks); + if (this.#broadcast[kWrite](batch)) { + this.#totalBytes += batch.byteLength; + return true; + } + return false; + } + + [kWriteBatch](chunks, signal) { + return this.#writeBatchSlow(createBatchEntry(chunks), signal); + } + end(options) { const signal = getWriterSignal(options); if (this.#state === 'errored') return PromiseReject(this.#error); @@ -955,20 +974,58 @@ const Broadcast = { }; result.broadcast[kOnCancel] = onCancel; + const w = result.writer; + // Batches of a sync source are read synchronously, as pipeTo() reads + // them, while the writes succeed synchronously. With a 'drop-oldest' or + // 'drop-newest' policy writes always do, so the whole source would be + // written before any consumer reads: read it asynchronously. + const policy = result.broadcast.backpressurePolicy; + const readSync = policy === 'strict' || policy === 'unbounded'; + // Read as for await...of does, stopping once the broadcast is + // cancelled or its signal aborts; see raceSignal(). + const abortState = { __proto__: null, aborted: false }; + const consume = async (iterator) => { + const nextSyncBatch = readSync ? iterator[kNextSyncBatch] : undefined; + // The first batch is read with next(), so that it is written after + // Broadcast.from() has returned and its consumers have been created. + let first = true; + for (;;) { + let chunks = nextSyncBatch !== undefined && !first ? + FunctionPrototypeCall(nextSyncBatch, iterator) : undefined; + first = false; + if (chunks === undefined) { + const result = await iterator.next(); + if (result.done || abortState.aborted) return; + chunks = result.value; + } else if (chunks === null) { + return; + } + try { + if (!w[kWriteBatchSync](chunks)) { + await w[kWriteBatch](chunks, signal); + } + } catch (error) { + // As for await closes the source when its body throws, ignoring + // errors from closing it. + if (!abortState.aborted) { + try { + await iterator.return?.(); + } catch { + // The error writing the batch is thrown. + } + } + throw error; + } + if (abortState.aborted) return; + } + }; + const pump = async () => { - const w = result.writer; try { + controller.signal.throwIfAborted(); if (isAsyncIterable(source)) { - for await (const chunks of yieldAbortable(source, controller.signal)) { - controller.signal.throwIfAborted(); - if (ArrayIsArray(chunks)) { - if (!w.writevSync(chunks)) { - await w.writev(chunks, signal ? { signal } : undefined); - } - } else if (!w.writeSync(chunks)) { - await w.write(chunks, signal ? { signal } : undefined); - } - } + await raceSignal(source[SymbolAsyncIterator](), controller.signal, + abortState, consume); } else if (isSyncIterable(source)) { for (const chunks of source) { controller.signal.throwIfAborted(); diff --git a/test/parallel/test-stream-iter-broadcast-from.js b/test/parallel/test-stream-iter-broadcast-from.js index 8bc2abde2e1..1d1049083d8 100644 --- a/test/parallel/test-stream-iter-broadcast-from.js +++ b/test/parallel/test-stream-iter-broadcast-from.js @@ -159,6 +159,67 @@ async function testBroadcastFromCancelWhileBlocked() { assert.strictEqual(sourceReturned, true); } +// A sync source is written as the broadcast has room, including values +// normalized asynchronously, and its error reaches the consumers after the +// data before it. Cancelling the broadcast while the pump waits for room +// closes the source. +async function testBroadcastFromSyncSource() { + const reason = new Error('sync source boom'); + function* source() { + yield [Uint8Array.of(1)]; + yield Promise.resolve(Uint8Array.of(2)); + yield 'c'; + yield [Uint8Array.of(4)]; + throw reason; + } + const { broadcast: bc } = Broadcast.from(source()); + const read = async (consumer) => { + const values = []; + try { + for await (const batch of consumer) values.push(...batch.map((c) => c[0])); + } catch (error) { + values.push(error); + } + return values; + }; + const results = await Promise.all([read(bc.push()), read(bc.push())]); + for (const values of results) { + assert.deepStrictEqual(values, [1, 2, 99, 4, reason]); + } + + let closed = false; + function* endless() { + try { + for (;;) yield [new Uint8Array(1024)]; + } finally { + closed = true; + } + } + const { broadcast: blocked } = Broadcast.from(endless(), { budget: 4096 }); + const consumer = blocked.push()[Symbol.asyncIterator](); + await consumer.next(); + await setImmediate(); + assert.strictEqual(closed, false); + blocked.cancel(); + await setImmediate(); + assert.strictEqual(closed, true); + assert.strictEqual((await consumer.next()).done, true); +} + +// With a 'drop-newest' policy, a sync source is still read one batch at a +// time while consumers read. +async function testBroadcastFromSyncSourceDropNewest() { + function* source() { + for (let i = 0; i < 8; i++) yield [Uint8Array.of(i)]; + } + const { broadcast: bc } = Broadcast.from(source(), { + budget: 2, backpressure: 'drop-newest', + }); + const values = []; + for await (const batch of bc.push()) values.push(...batch.map((c) => c[0])); + assert.deepStrictEqual(values, [0, 1, 2, 3, 4, 5, 6, 7]); +} + // ============================================================================= // Source error propagation via Broadcast.from() // ============================================================================= @@ -229,6 +290,8 @@ Promise.all([ testAbortSignal(), testAlreadyAbortedSignal(), testBroadcastFromCancelWhileBlocked(), + testBroadcastFromSyncSource(), + testBroadcastFromSyncSourceDropNewest(), testBroadcastFromSourceError(), testBroadcastProtocolReturnsBroadcast(), testBroadcastProtocolReturnsNull(), From 7942cc6fe8bfc838a6b0ac1a8354de3770d3f4c4 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 12:47:22 +0000 Subject: [PATCH 48/58] stream: create stream/iter from() value streams as class instances from() of a value that needs no normalization (a string, a buffer, a Uint8Array[]) returned an object literal with a null prototype, which V8 creates in dictionary mode, and each of its iterators was an object literal passed to ObjectSetPrototypeOf(), a runtime call. Together they were three quarters of the time to create and read such a stream. Use two classes, BatchSource and BatchIterator, whose prototypes do not inherit from Object.prototype either, as before. Creating a stream from a 16-byte chunk and reading it is about 6.5 times faster, and from a one-chunk array about 5.5 times faster. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 110 ++++++++++++++++++------------ 1 file changed, 65 insertions(+), 45 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index d32d2353bdc..0c962b4a8f9 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -1651,62 +1651,82 @@ function fromSync(input) { }; } -/** - * An async iterable yielding the chunks of `batch`, a Uint8Array[], in - * batches of at most FROM_BATCH_SIZE chunks, for from() of values that need - * no normalization. Each iterator does what an async generator yielding the - * batches would, without the generator object and the promises and frames - * of resuming it: next() gives the batches, then done; return() and throw() - * end the iteration, throw() rejecting with its argument. - * @param {Uint8Array[]} batch - * @returns {AsyncIterable} - */ -function createBatchSource(batch) { - return { - __proto__: null, - [SymbolAsyncIterator]() { - return createBatchIterator(batch); - }, - [kValidatedSource]: true, - }; -} +// BatchSource and BatchIterator are classes, rather than object literals +// with a null prototype, which V8 creates in dictionary mode: from() of a +// value is often called once per stream. Their prototypes do not inherit +// from Object.prototype either. +class BatchIterator { + #batch; + #index = 0; -function createBatchIterator(batch) { - let index = 0; + constructor(batch) { + this.#batch = batch; + } // The next batch, or null once done. - function nextBatch() { + #nextBatch() { + const batch = this.#batch; + const index = this.#index; if (index >= batch.length) { - index = batch.length; + this.#index = batch.length; return null; } if (index === 0 && batch.length <= FROM_BATCH_SIZE) { - index = batch.length; + this.#index = batch.length; return batch; } - const start = index; - index += FROM_BATCH_SIZE; - return ArrayPrototypeSlice(batch, start, index); + this.#index = index + FROM_BATCH_SIZE; + return ArrayPrototypeSlice(batch, index, index + FROM_BATCH_SIZE); } - return ObjectSetPrototypeOf({ - next() { - const value = nextBatch(); - return PromiseResolve(value === null ? - new IterResult(true, undefined) : new IterResult(false, value)); - }, - return(value) { - index = batch.length; - return PromiseResolve(new IterResult(true, value)); - }, - throw(error) { - index = batch.length; - return PromiseReject(error); - }, - [SymbolAsyncIterator]() { - return this; - }, - }, null); + next() { + const value = this.#nextBatch(); + return PromiseResolve(value === null ? + new IterResult(true, undefined) : new IterResult(false, value)); + } + + return(value) { + this.#index = this.#batch.length; + return PromiseResolve(new IterResult(true, value)); + } + + throw(error) { + this.#index = this.#batch.length; + return PromiseReject(error); + } + + [SymbolAsyncIterator]() { + return this; + } +} +ObjectSetPrototypeOf(BatchIterator.prototype, null); + +class BatchSource { + #batch; + + constructor(batch) { + this.#batch = batch; + } + + [SymbolAsyncIterator]() { + return new BatchIterator(this.#batch); + } +} +BatchSource.prototype[kValidatedSource] = true; +ObjectSetPrototypeOf(BatchSource.prototype, null); + +/** + * An async iterable yielding the chunks of `batch`, a Uint8Array[], in + * batches of at most FROM_BATCH_SIZE chunks, for from() of values that need + * no normalization. Each iterator does what an async generator yielding the + * batches would, without the generator object and the promises and frames + * of resuming it: next() gives the batches, then done; return() and throw() + * end the iteration, throw() rejecting with its argument. + * @param {Uint8Array[]} batch + * @returns {AsyncIterable} + */ +function createBatchSource(batch) { + return new BatchSource(batch); } /** From 7526d3d87a6e0030c4ec79c5e893ebb245d282b7 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 12:54:35 +0000 Subject: [PATCH 49/58] stream: normalize stream/iter fromSync() inputs without generators fromSync() returned null-prototype object literals, which V8 creates in dictionary mode, whose iterators were sync generators: one for values that need no normalization, and two layers, an outer generator delegating to normalizeSyncSource(), for iterables, resumed for every batch. Use classes, as from() does since the previous commits: SyncBatchSource for values, the sync counterpart of BatchSource, and SyncSource for iterables, whose iterator reads the source with the SyncSourceReader of from() and normalizes other values with normalizeSyncValue(). The iterators behave as the generators did: iterables can be iterated again, and return(), throw() and errors normalizing a value close the source as for...of does. normalizeSyncSource() is no longer used. Creating a sync stream from a value and reading it is about 18 times faster. Reading a sync source one chunk per batch with for...of is about 1.8 times faster, and piping it with pipeToSync() about 1.5 times faster. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 263 +++++++++++++------- test/parallel/test-stream-iter-from-sync.js | 83 ++++++ 2 files changed, 255 insertions(+), 91 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 0c962b4a8f9..0598859a189 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -395,60 +395,6 @@ function* yieldBoundedBatch(batch) { } } -/** - * Normalize a sync streamable source, yielding batches of Uint8Array. - * @param {Iterable} source - * @yields {Uint8Array[]} - */ -function* normalizeSyncSource(source) { - let batch = []; - - for (const value of source) { - // Fast path 1: value is already a Uint8Array[] batch - if (isUint8ArrayBatch(value)) { - if (batch.length > 0) { - yield batch; - batch = []; - } - if (value.length <= FROM_BATCH_SIZE) { - if (value.length !== 0) yield value; - } else { - yield* yieldBoundedBatch(value); - } - continue; - } - // Fast path 2: value is a single Uint8Array (very common) - if (isUint8Array(value)) { - batch[batch.length] = value; - if (batch.length === FROM_BATCH_SIZE) { - yield batch; - batch = []; - } - continue; - } - // Slow path: normalize the value - if (batch.length > 0) { - yield batch; - batch = []; - } - let valueBatch = []; - for (const chunk of normalizeSyncValue(value)) { - valueBatch[valueBatch.length] = chunk; - if (valueBatch.length === FROM_BATCH_SIZE) { - yield valueBatch; - valueBatch = []; - } - } - if (valueBatch.length > 0) { - yield valueBatch; - } - } - - if (batch.length > 0) { - yield batch; - } -} - function yieldNormalizationAbortable(source, context) { if (context === undefined) return source; return { @@ -1550,6 +1496,173 @@ async function* normalizeAsyncStreamableResult(result, context) { // Public API: from() and fromSync() // ============================================================================= +/** + * The sync iterator of a SyncBatchSource: what a sync generator yielding + * the batches of a BatchIterator would give, written out by hand. + */ +class SyncBatchIterator { + #batch; + #index = 0; + + constructor(batch) { + this.#batch = batch; + } + + next() { + const batch = this.#batch; + const index = this.#index; + if (index >= batch.length) { + this.#index = batch.length; + return new IterResult(true, undefined); + } + if (index === 0 && batch.length <= FROM_BATCH_SIZE) { + this.#index = batch.length; + return new IterResult(false, batch); + } + this.#index = index + FROM_BATCH_SIZE; + return new IterResult( + false, ArrayPrototypeSlice(batch, index, index + FROM_BATCH_SIZE)); + } + + return(value) { + this.#index = this.#batch.length; + return new IterResult(true, value); + } + + throw(error) { + this.#index = this.#batch.length; + throw error; + } + + [SymbolIterator]() { + return this; + } +} +ObjectSetPrototypeOf(SyncBatchIterator.prototype, null); + +/** + * A sync iterable yielding the chunks of `batch`, a Uint8Array[], in + * batches of at most FROM_BATCH_SIZE chunks, for fromSync() of values that + * need no normalization: the sync counterpart of BatchSource. + */ +class SyncBatchSource { + #batch; + + constructor(batch) { + this.#batch = batch; + } + + [SymbolIterator]() { + return new SyncBatchIterator(this.#batch); + } +} +ObjectSetPrototypeOf(SyncBatchSource.prototype, null); + +/** + * The sync iterator of a SyncSource: what a sync generator normalizing the + * source with for...of would yield, written out by hand. The source is read by a SyncSourceReader, which + * collects single chunks into batches and closes the source as for...of + * does; other values are normalized with normalizeSyncValue(), and their + * chunks returned in batches of up to FROM_BATCH_SIZE. An error normalizing + * a value is thrown into the reader, which closes the source, as for an + * error in the body of a for...of loop. + */ +class SyncSourceIterator { + #reader; + // The normalizeSyncValue() generator of the value being normalized. + #valueChunks = null; + + constructor(source) { + this.#reader = new SyncSourceReader(source, undefined); + } + + next() { + if (this.#valueChunks !== null) { + const batch = this.#readValue(); + if (batch !== null) return new IterResult(false, batch); + } + for (;;) { + const result = this.#reader.next(); + if (result.done) return result; + const value = result.value; + if (ArrayIsArray(value)) return result; + this.#valueChunks = normalizeSyncValue(value.value); + const batch = this.#readValue(); + if (batch !== null) return new IterResult(false, batch); + } + } + + // The next batch of the chunks of the value being normalized, or null + // once there are no more. + #readValue() { + const batch = []; + try { + for (;;) { + const result = this.#valueChunks.next(); + if (result.done) { + this.#valueChunks = null; + break; + } + batch[batch.length] = result.value; + if (batch.length === FROM_BATCH_SIZE) return batch; + } + } catch (error) { + this.#valueChunks = null; + this.#reader.throw(error); + } + return batch.length > 0 ? batch : null; + } + + return(value) { + const valueChunks = this.#valueChunks; + this.#valueChunks = null; + if (valueChunks !== null) { + try { + valueChunks.return(); + } catch (error) { + this.#reader.throw(error); + } + } + this.#reader.return(); + return new IterResult(true, value); + } + + throw(error) { + const valueChunks = this.#valueChunks; + this.#valueChunks = null; + if (valueChunks !== null) { + try { + valueChunks.return(); + } catch { + // The error thrown in takes precedence. + } + } + this.#reader.throw(error); + } + + [SymbolIterator]() { + return this; + } +} +ObjectSetPrototypeOf(SyncSourceIterator.prototype, null); + +/** + * A sync iterable normalizing a sync iterable source for fromSync() into + * batches of Uint8Array, without generators. + */ +class SyncSource { + #source; + + constructor(source) { + this.#source = source; + } + + [SymbolIterator]() { + return new SyncSourceIterator(this.#source); + } +} +ObjectSetPrototypeOf(SyncSource.prototype, null); + /** * Create a SyncByteStreamReadable from a ByteInput or SyncStreamable. * @param {string|ArrayBuffer|ArrayBufferView|Iterable} input @@ -1562,13 +1675,7 @@ function fromSync(input) { // Check for primitives first (ByteInput) if (isPrimitiveChunk(input)) { - const chunk = primitiveToUint8Array(input); - return { - __proto__: null, - *[SymbolIterator]() { - yield [chunk]; - }, - }; + return new SyncBatchSource([primitiveToUint8Array(input)]); } // Check toStreamable protocol (takes precedence over iteration protocols). @@ -1584,32 +1691,12 @@ function fromSync(input) { // data volume. Sub-batching keeps peak memory bounded while preserving // the throughput benefit of batched processing. if (ArrayIsArray(input)) { - if (input.length === 0) { - return { - __proto__: null, - *[SymbolIterator]() { - // Empty - yield nothing - }, - }; - } + // An empty array yields nothing. + if (input.length === 0) return new SyncBatchSource(input); // Check if it's an array of Uint8Array (common case) if (isUint8Array(input[0])) { const allUint8 = ArrayPrototypeEvery(input, isUint8Array); - if (allUint8) { - const batch = input; - return { - __proto__: null, - *[SymbolIterator]() { - if (batch.length <= FROM_BATCH_SIZE) { - yield batch; - } else { - for (let i = 0; i < batch.length; i += FROM_BATCH_SIZE) { - yield ArrayPrototypeSlice(batch, i, i + FROM_BATCH_SIZE); - } - } - }, - }; - } + if (allUint8) return new SyncBatchSource(input); } } @@ -1643,12 +1730,7 @@ function fromSync(input) { ); } - return { - __proto__: null, - *[SymbolIterator]() { - yield* normalizeSyncSource(input); - }, - }; + return new SyncSource(input); } // BatchSource and BatchIterator are classes, rather than object literals @@ -1819,7 +1901,6 @@ module.exports = { kNextUncancellable, normalizeAsyncSource, normalizeAsyncValue, - normalizeSyncSource, normalizeSyncValue, primitiveToUint8Array, }; diff --git a/test/parallel/test-stream-iter-from-sync.js b/test/parallel/test-stream-iter-from-sync.js index 133597914c4..f6b2b0a5f84 100644 --- a/test/parallel/test-stream-iter-from-sync.js +++ b/test/parallel/test-stream-iter-from-sync.js @@ -255,7 +255,90 @@ function testFromSyncFunctionWithToStreamable() { assert.throws(() => fromSync(() => {}), { code: 'ERR_INVALID_ARG_TYPE' }); } +// The iterators of fromSync() behave as generators do: values are batched +// and normalized, iterables can be iterated again, and return(), throw() +// and errors normalizing a value close the source as for...of does. +function testFromSyncIteratorProtocol() { + function loggedSource(log, values) { + return { + [Symbol.iterator]() { + let i = 0; + log.push('iterator'); + return { + next() { + log.push(`next ${i}`); + return i < values.length ? + { done: false, value: values[i++] } : + { done: true, value: undefined }; + }, + return() { + log.push('return'); + return { done: true, value: undefined }; + }, + }; + }, + }; + } + const lengths = (iterable) => Array.from(iterable, (batch) => batch.length); + + const chunks = Array.from({ length: 130 }, () => new Uint8Array(1)); + let log = []; + const source = fromSync(loggedSource(log, [ + ...chunks, ['ab', ['cd']], Uint8Array.of(1), [], + ])); + assert.deepStrictEqual(lengths(source), [128, 2, 2, 1]); + assert.strictEqual(log.at(-1), 'next 133'); + assert.deepStrictEqual(lengths(source), [128, 2, 2, 1]); + + // Values that need no normalization. + for (const value of ['x', Uint8Array.of(1), [Uint8Array.of(1)], []]) { + const iterable = fromSync(value); + assert.deepStrictEqual(lengths(iterable), lengths(iterable)); + const iterator = iterable[Symbol.iterator](); + assert.strictEqual(iterator[Symbol.iterator](), iterator); + assert.deepStrictEqual({ ...iterator.return('r') }, + { done: true, value: 'r' }); + assert.strictEqual(iterator.next().done, true); + assert.throws(() => iterable[Symbol.iterator]().throw(new Error('t')), + { message: 't' }); + } + assert.deepStrictEqual(lengths(fromSync(new Array(300).fill(chunks[0]))), + [128, 128, 44]); + + // return() while reading the source, or a value, closes the source; + // before reading it, or after it ended, does not. + for (const values of [[[Uint8Array.of(1)], [Uint8Array.of(2)]], + [new Array(200).fill('a')]]) { + log = []; + const iterator = fromSync(loggedSource(log, values))[Symbol.iterator](); + iterator.next(); + assert.deepStrictEqual({ ...iterator.return() }, + { done: true, value: undefined }); + assert.deepStrictEqual(log, ['iterator', 'next 0', 'return']); + assert.strictEqual(iterator.next().done, true); + } + log = []; + const unstarted = fromSync(loggedSource(log, ['a']))[Symbol.iterator](); + unstarted.return(); + assert.deepStrictEqual(log, []); + + // throw() and an error normalizing a value close the source, and the + // error is thrown. + log = []; + const thrown = fromSync(loggedSource(log, ['a', 'b']))[Symbol.iterator](); + thrown.next(); + assert.throws(() => thrown.throw(new Error('thrown')), { message: 'thrown' }); + assert.deepStrictEqual(log, ['iterator', 'next 0', 'return']); + log = []; + const invalid = fromSync(loggedSource(log, ['a', 42]))[Symbol.iterator](); + invalid.next(); + assert.throws(() => invalid.next(), { code: 'ERR_INVALID_ARG_TYPE' }); + assert.deepStrictEqual(log, ['iterator', 'next 0', 'next 1', 'return']); + assert.strictEqual(invalid.next().done, true); +} + Promise.all([ + testFromSyncIteratorProtocol(), testFromSyncString(), testFromSyncUint8Array(), testFromSyncArrayBuffer(), From ec892377befdbd492f5c228a4592ceb511c47a0e Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 13:01:27 +0000 Subject: [PATCH 50/58] stream: read a single stream/iter merge() source without a generator merge() of a single source read it with an async generator, a layer for every batch, on top of the abortable wrapper when a signal was given. Without a signal, read the source with MergeSourceIterator: calls to next() go to the source's iterator, which from() makes queue them as an async generator does, and return() and throw() wait for the last next() before closing the source, as the generator queued them. With a signal, return the iterator of the abortable wrapper, which already rejects a pending read when the signal aborts, closes the source and then ends. As the generator did, an abort before the first read now ends the iteration without opening the source. Iterating merge() of a single source of 16-byte chunks one per batch is about 1.7-2.2 times faster without a signal, and 1.5-1.7 times faster with one. Assisted-by: OpenCode --- lib/internal/streams/iter/consumers.js | 114 ++++++++++++++++-- lib/internal/streams/iter/utils.js | 11 +- .../test-stream-iter-consumers-merge.js | 72 +++++++++++ 3 files changed, 185 insertions(+), 12 deletions(-) diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index aa79795e93e..5177b8d9c74 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -17,8 +17,11 @@ const { ArrayPrototypeSlice, FunctionPrototypeCall, ObjectFreeze, + ObjectSetPrototypeOf, Promise, PromisePrototypeThen, + PromiseReject, + PromiseResolve, SafePromiseAllReturnVoid, SafeSet, Symbol, @@ -54,6 +57,7 @@ const { } = require('internal/streams/iter/from'); const { + IterResult, kNullOnceOption, concatBytes, getProtocolMethod, @@ -498,6 +502,84 @@ MergeEntry.prototype = ObjectFreeze({ __proto__: null }); * @param {...(AsyncIterable|object)} args * @returns {AsyncIterable} */ +/** + * The iterator of merge() of a single source without a signal: what an + * async generator doing `for await (const batch of source) yield batch;` + * gives, without its layer for every batch. Calls to next() are passed to + * the source's iterator, from from(), which queues them as an async + * generator does. return() and throw() wait for the last next() before + * closing the source, as the generator would queue them, and later calls + * wait for them in turn. The source is opened by the first next(). + */ +class MergeSourceIterator { + #source; + #iterator = null; + // The last next() of the source's iterator, and the promise of return() + // or throw() once one is called. + #last = null; + #closing = null; + + constructor(source) { + this.#source = source; + } + + next() { + if (this.#closing !== null) { + return PromisePrototypeThen(this.#closing, doneResult, doneResult); + } + try { + this.#iterator ??= this.#source[SymbolAsyncIterator](); + const next = PromiseResolve(this.#iterator.next()); + this.#last = next; + return next; + } catch (error) { + this.#closing = PromiseReject(error); + markPromiseAsHandled(this.#closing); + return this.#closing; + } + } + + return(value) { + return this.#close(() => new IterResult(true, value), false); + } + + throw(error) { + return this.#close(() => { throw error; }, true); + } + + // Close the source once the last next() has settled, if it was opened, + // then settle with `settle()`. Errors closing it are ignored if `quiet`, + // as for await ignores them when its body throws. + #close(settle, quiet) { + if (this.#closing !== null) { + return PromisePrototypeThen(this.#closing, settle, settle); + } + const iterator = this.#iterator; + if (iterator === null) { + this.#closing = PromiseResolve(); + return PromisePrototypeThen(this.#closing, settle); + } + const close = () => { + const closed = PromiseResolve(iterator.return?.()); + return quiet ? PromisePrototypeThen(closed, undefined, () => {}) : closed; + }; + const closing = PromisePrototypeThen(this.#last ?? PromiseResolve(), + close, close); + this.#closing = closing; + markPromiseAsHandled(closing); + return PromisePrototypeThen(closing, settle); + } + + [SymbolAsyncIterator]() { + return this; + } +} +ObjectSetPrototypeOf(MergeSourceIterator.prototype, null); + +function doneResult() { + return new IterResult(true, undefined); +} + function merge(...args) { let sources; let options; @@ -517,6 +599,28 @@ function merge(...args) { // Normalize each source via from() const normalized = ArrayPrototypeMap(sources, (source) => from(source)); + if (normalized.length === 1) { + // A single source is read without an async generator layer for every + // batch (see MergeSourceIterator). With a signal, the source is read through + // yieldAbortable(), which rejects a pending read when the signal aborts + // (also before the first read), closes the source, and then ends. An + // async iterable is made abortable before from(), so that an abort + // closes it even while from() is reading it. + const { signal } = options; + return { + __proto__: null, + [SymbolAsyncIterator]() { + if (signal === undefined) { + return new MergeSourceIterator(normalized[0]); + } + const source = isAsyncIterable(sources[0]) ? + from(yieldAbortable(sources[0], signal)) : + yieldAbortable(normalized[0], signal); + return source[SymbolAsyncIterator](); + }, + }; + } + return { __proto__: null, async *[SymbolAsyncIterator]() { @@ -526,16 +630,6 @@ function merge(...args) { if (normalized.length === 0) return; - if (normalized.length === 1) { - const source = signal !== undefined && isAsyncIterable(sources[0]) ? - from(yieldAbortable(sources[0], signal)) : - yieldAbortable(normalized[0], signal); - for await (const batch of source) { - yield batch; - } - return; - } - // Multiple sources - use a ready queue so that batches that settle // between consumer pulls are drained synchronously without an extra // async tick per batch. Each source has at most one pending .next() diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 7dd6a3fe172..7eb0da7dbfc 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -279,8 +279,8 @@ function yieldAbortable(source, signal) { * weakly so that the signal does not keep the iterator alive, and each * next() waits with a single promise that an abort rejects. As with the * generator: - * - the source is opened by the first next(), and calls made while one is in - * progress are queued; + * - the source is opened by the first next(), unless the signal has aborted + * by then, and calls made while one is in progress are queued; * - an abort before or while reading a value, or once it has been read, * rejects with the abort reason; * - unless the source has ended, an error (including an abort) closes it: @@ -390,6 +390,13 @@ function createAbortableIterator(source, signal) { return PromiseResolve(new IterResult(true, undefined)); } if (state === kStart) { + // An abort before the first read ends the iteration without opening + // the source. + if (signal.aborted) { + state = kDone; + operations.settled(); + return PromiseReject(signal.reason); + } try { iterator = source[SymbolAsyncIterator](); } catch (error) { diff --git a/test/parallel/test-stream-iter-consumers-merge.js b/test/parallel/test-stream-iter-consumers-merge.js index 5cbe773ac08..9e69d5f7fcb 100644 --- a/test/parallel/test-stream-iter-consumers-merge.js +++ b/test/parallel/test-stream-iter-consumers-merge.js @@ -461,7 +461,79 @@ async function testMergeMultiSourceBreakWithCleanupError() { ); } +// merge() of a single source behaves as an async generator reading it: +// an abort rejects the pending read, closes the source and ends the +// iteration; an abort before the first read does not open the source; +// return() is queued behind pending reads. +async function testMergeSingleSourceProtocol() { + const log = []; + function source(name, values, { hang = false } = {}) { + let i = 0; + return { + [Symbol.asyncIterator]() { + log.push(`${name} iterator`); + return { + next() { + log.push(`${name} next ${i}`); + if (hang && i === values.length) return new Promise(() => {}); + return Promise.resolve(i < values.length ? + { done: false, value: values[i++] } : { done: true }); + }, + return() { + log.push(`${name} return`); + return Promise.resolve({ done: true }); + }, + }; + }, + }; + } + + async function settle(label, promise) { + try { + const result = await promise; + log.push(`${label}: ${result.done ? 'done' : result.value[0][0]}`); + } catch (error) { + log.push(`${label}: ${error.name}`); + } + } + const chunk = (n) => [Uint8Array.of(n)]; + + const aborted = AbortSignal.abort(); + let it = merge(source('a', [chunk(1)]), { signal: aborted })[ + Symbol.asyncIterator](); + await settle('a 1', it.next()); + await settle('a 2', it.next()); + + const ac = new AbortController(); + it = merge(source('b', [chunk(1)], { hang: true }), { signal: ac.signal })[ + Symbol.asyncIterator](); + await settle('b 1', it.next()); + const pending = it.next(); + await setImmediate(); + ac.abort(); + await settle('b 2', pending); + await settle('b 3', it.next()); + + it = merge(source('c', [chunk(1), 'x', chunk(3)]))[Symbol.asyncIterator](); + const results = [it.next(), it.next(), it.return(), it.next()]; + for (let i = 0; i < results.length; i++) await settle(`c ${i}`, results[i]); + + it = merge(source('d', [chunk(1)]))[Symbol.asyncIterator](); + await settle('d return', it.return()); + await assert.rejects(it.throw(new Error('thrown')), { message: 'thrown' }); + + assert.deepStrictEqual(log, [ + 'a 1: AbortError', 'a 2: done', + 'b iterator', 'b next 0', 'b 1: 1', 'b next 1', 'b return', + 'b 2: AbortError', 'b 3: done', + 'c iterator', 'c next 0', 'c next 1', 'c 0: 1', 'c 1: 120', 'c return', + 'c 2: done', 'c 3: done', + 'd return: done', + ]); +} + Promise.all([ + testMergeSingleSourceProtocol(), testMergeTwoSources(), testMergeSingleSource(), testMergeEmpty(), From 6e43cb192eec7bb9b2e864aa927dca7ed32f1aec Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 13:56:50 +0000 Subject: [PATCH 51/58] stream: do not keep stream/iter from() sync sources busy for a tick The iterator of from() over a sync iterable stayed busy until the microtask after every result it produced synchronously, as an async generator stays busy until the tick after a yield, so that a call made synchronously after one was queued behind it. That cost a microtask per batch, also when nothing was ever queued. Settle each operation when its result is known instead, as the normalizer of async sources does: a call made after one that finished synchronously now runs at once, and calls queued while one was in progress, such as behind a value normalized asynchronously, still run a microtask later. Only calls made back to back without awaiting are affected: a second next() reads the source during the call, and a return() made after it no longer cancels it, as it did when it was still queued. The other iterators of from(), for values and async sources, already behaved this way. The queue's release() is no longer used. Iterating from() over a sync source yielding one-chunk batches is about 1.38 times faster, now on par with a ReadableStream, and with 64-chunk batches about 7% faster. Assisted-by: OpenCode --- lib/internal/streams/iter/from.js | 20 ++++---- lib/internal/streams/iter/utils.js | 8 +--- test/parallel/test-stream-iter-from-async.js | 50 ++++++++++++++++---- 3 files changed, 50 insertions(+), 28 deletions(-) diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 0598859a189..551a91158c5 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -1323,19 +1323,15 @@ function createSyncSourceNormalizer(source, context) { let done = false; // A normalizeAsyncSourceValue() generator for the value being normalized. let valueBatches = null; - const { run, settled, release, idle } = createOperationQueue(); - - // Results are often produced synchronously. An async generator stays busy - // until the tick after a yield or return (both await their operand), so - // that a next(), return() or throw() made synchronously after this one is - // queued behind it; do the same. - // TODO(@jasnell): This costs a microtask per batch. Running a call made - // while the previous one has finished synchronously right away would make - // for-await over from(syncSource) about 1.3x faster, at the cost of this - // async generator parity for calls made back to back without awaiting - // (see testFromSyncSourceQueuesBehindReturn). + const { run, settled, idle } = createOperationQueue(); + + // Results are often produced synchronously. Once one is, the next call + // runs at once, as for the async source normalizer; calls queued while + // an operation was in progress run a microtask later. (An async generator + // would stay busy until the tick after a yield, and queue a call made + // synchronously after one: that costs a microtask per batch.) function finish(result) { - PromisePrototypeThen(kResolvedPromise, release); + settled(); return result; } diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 7eb0da7dbfc..0191a4dc83d 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -604,7 +604,7 @@ class QueuedOperation { * Serializes the operations of a hand-written async iterator the way an * async generator queues its requests: an operation started while another * is in progress waits until it has settled. - * @returns {{ run: Function, settled: Function, release: Function }} + * @returns {{ run: Function, settled: Function, idle: Function }} */ function createOperationQueue() { let busy = false; @@ -639,12 +639,6 @@ function createOperationQueue() { PromisePrototypeThen(kResolvedPromise, drain); } }, - // Like settled(), but start the next queued operation now, as an async - // generator does once the await of a yield or return completes. - release() { - busy = false; - drain(); - }, // Whether no operation is running or queued. idle() { return !busy && (queue === null || queue.length === 0); diff --git a/test/parallel/test-stream-iter-from-async.js b/test/parallel/test-stream-iter-from-async.js index 44befd8eb7e..a6bd4a48692 100644 --- a/test/parallel/test-stream-iter-from-async.js +++ b/test/parallel/test-stream-iter-from-async.js @@ -856,20 +856,52 @@ async function testFromSyncSourceReturnAndThrow() { } } -async function testFromSyncSourceQueuesBehindReturn() { - // A next() made synchronously after another waits for it, and so sees a - // return() made synchronously after it as well. - const log = []; - const iterator = from(createLoggedSyncSource(log, [ +async function testFromSyncSourceCallOrder() { + // A next() made synchronously after another that finished synchronously + // runs at once, before a return() made after it. + let log = []; + let iterator = from(createLoggedSyncSource(log, [ [Uint8Array.of(1)], 'a', [Uint8Array.of(2)], ]))[Symbol.asyncIterator](); - const results = [iterator.next(), iterator.next(), iterator.return()]; + let results = [iterator.next(), iterator.next(), iterator.return()]; + for (let i = 0; i < results.length; i++) { + await settle(log, `result ${i}`, results[i]); + } + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'next 1', 'result 0: 1', 'result 1: 1', 'return', + 'result 2: done', + ]); + + // A call made while a value is normalized asynchronously is queued + // behind it, and so a return() made then cancels it. + log = []; + iterator = from(createLoggedSyncSource(log, [ + [Uint8Array.of(1)], Promise.resolve(Uint8Array.of(2)), [Uint8Array.of(3)], + ]))[Symbol.asyncIterator](); + results = [iterator.next(), iterator.next(), iterator.next(), + iterator.return()]; for (let i = 0; i < results.length; i++) { await settle(log, `result ${i}`, results[i]); } assert.deepStrictEqual(log, [ - 'iterator', 'next 0', 'next 1', 'return', 'result 0: 1', - 'result 1: AbortError', 'result 2: done', + 'iterator', 'next 0', 'next 1', 'result 0: 1', 'return', + 'result 1: AbortError', 'result 2: done', 'result 3: done', + ]); + + // A next() made synchronously after another reads the source at once. + log = []; + iterator = from(createLoggedSyncSource(log, [ + [Uint8Array.of(1)], [Uint8Array.of(2)], + ]))[Symbol.asyncIterator](); + const first = iterator.next(); + log.push('first called'); + const second = iterator.next(); + log.push('second called'); + await settle(log, 'first', first); + await settle(log, 'second', second); + assert.deepStrictEqual(log, [ + 'iterator', 'next 0', 'first called', 'next 1', 'second called', + 'first: 1', 'second: 1', ]); } @@ -975,5 +1007,5 @@ Promise.all([ testFromSyncSourceLargeAndEmptyBatches(), testFromSyncSourceErrors(), testFromSyncSourceReturnAndThrow(), - testFromSyncSourceQueuesBehindReturn(), + testFromSyncSourceCallOrder(), ]).then(common.mustCall()); From 8af1485cae3ccf00d30b1fd499919a21b29db0c4 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 14:45:19 +0000 Subject: [PATCH 52/58] stream: create stream/iter pending promises without a result object The promises stream/iter creates per read, per write and per batch when it has to wait (a pull() pipeline's pull, a from() async source's read, a push() or broadcast() consumer's read and pending write, a share() source read, drain waits, queued operations) were created with PromiseWithResolvers(), which also allocates an object to return the promise and its two functions in. Without pointer compression that is 299 bytes per call, against 214 for `new Promise(executor)` with an executor that is created once. newPendingPromise() / newResolveOnlyPromise() create the promise with a shared executor that stores the resolving functions in module slots, and takeResolve() / takeReject() take them from there right after. The executor runs synchronously, so nothing can run in between. Paths that create a promise once per stream or per call are unchanged. Allocation per chunk (sampling heap profile, 16-byte chunks): - for await over from(asyncSource): 1053 -> 967 bytes - pull() with one transform, async source: 1692 -> 1518 bytes - pull() with one transform, sync source: 917 -> 838 bytes Assisted-by: OpenCode --- lib/internal/streams/iter/broadcast.js | 19 +++++++-- lib/internal/streams/iter/classic.js | 11 +++++- lib/internal/streams/iter/from.js | 18 ++++++--- lib/internal/streams/iter/pull.js | 10 +++-- lib/internal/streams/iter/push.js | 15 +++++-- lib/internal/streams/iter/share.js | 12 ++++-- lib/internal/streams/iter/transform.js | 12 ++++-- lib/internal/streams/iter/utils.js | 54 +++++++++++++++++++++++++- 8 files changed, 123 insertions(+), 28 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index 06182e5870f..3d15dc272bb 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -74,6 +74,9 @@ const { raceSignal, validateBatchEntry, validateBudget, + newPendingPromise, + takeReject, + takeResolve, } = require('internal/streams/iter/utils'); const { converters, @@ -271,12 +274,16 @@ class BroadcastImpl { } if (state.resolve) { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); state.pending.push(new PendingRequest(resolve, reject)); return promise; } - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); state.resolve = resolve; state.reject = reject; self.#waiters.add(state); @@ -632,7 +639,9 @@ class BroadcastWriter { const canWrite = this.canWrite; if (canWrite === null) return null; if (canWrite) return PromiseResolve(true); - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); ArrayPrototypePush(this.#pendingDrains, new PendingRequest(resolve, reject)); return promise; } @@ -823,7 +832,9 @@ class BroadcastWriter { * @returns {Promise} */ #createPendingWrite(batch, signal) { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); const entry = new PendingWrite(batch, resolve, reject); this.#pendingWrites.push(entry); if (signal) { diff --git a/lib/internal/streams/iter/classic.js b/lib/internal/streams/iter/classic.js index 2b39bc92787..2fe90d58a46 100644 --- a/lib/internal/streams/iter/classic.js +++ b/lib/internal/streams/iter/classic.js @@ -69,6 +69,9 @@ const { onSignalAbort, validateBackpressure, toWriterUint8Array, + newPendingPromise, + takeReject, + takeResolve, } = require('internal/streams/iter/utils'); const { Buffer } = require('buffer'); @@ -773,7 +776,9 @@ function fromWritable(writable, options = kNullPrototype) { } function queueWrite(chunks, signal) { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); const entry = new QueuedWrite(chunks, resolve, reject); pendingWrites.push(entry); installDrainListener(); @@ -1049,7 +1054,9 @@ function fromWritable(writable, options = kNullPrototype) { if (pendingWrites.length === 0 && !isFull()) { return PromiseResolve(true); } - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); ArrayPrototypePush(drainWaiters, new PendingRequest(resolve, reject)); installDrainListener(); return promise; diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 551a91158c5..22821627ac8 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -21,7 +21,6 @@ const { PromisePrototypeThen, PromiseReject, PromiseResolve, - PromiseWithResolvers, Symbol, SymbolAsyncIterator, SymbolIterator, @@ -60,6 +59,9 @@ const { kActive, kDone, kResolvedPromise, + newPendingPromise, + takeReject, + takeResolve, kStart, createOperationQueue, getProtocolMethod, @@ -113,7 +115,9 @@ function throwIfNormalizationCancelled(context) { */ function waitForNormalization(value, context) { if (context === undefined) return value; - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); const onCancel = () => { if (context.resolve === onCancel) context.resolve = null; reject(context.reason); @@ -544,7 +548,9 @@ function yieldNormalizationAbortable(source, context) { if (context.cancelled) return PromiseReject(context.reason); reading = true; - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); let next; try { next = FunctionPrototypeCall(nextMethod, iterator); @@ -955,9 +961,9 @@ function createAsyncSourceNormalizer(source, context) { if (readable) { // One promise for the read, which a cancellation rejects, instead of // next()'s promise and a reaction to it. - const { promise, resolve, reject } = PromiseWithResolvers(); - resolveRead = resolve; - rejectRead = reject; + const promise = newPendingPromise(); + resolveRead = takeResolve(); + rejectRead = takeReject(); iterator[kRead](); return promise; } diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 2cc643eafb2..12a6b659b9e 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -17,7 +17,6 @@ const { PromisePrototypeThen, PromiseReject, PromiseResolve, - PromiseWithResolvers, Symbol, SymbolAsyncIterator, SymbolIterator, @@ -60,6 +59,9 @@ const { kDone, kNullOnceOption, kResolvedPromise, + newPendingPromise, + takeReject, + takeResolve, kStart, callWithByteView, checkFixedBatchChunk, @@ -1575,9 +1577,9 @@ function createAsyncPipeline(source, transforms, signal, } // The pull is pending from now on: the signal can abort while the // transforms or the source are called. - const { promise, resolve, reject } = PromiseWithResolvers(); - resolvePull = resolve; - rejectPull = reject; + const promise = newPendingPromise(); + resolvePull = takeResolve(); + rejectPull = takeReject(); let next; try { next = PromiseResolve(FunctionPrototypeCall(nextMethod, iterator)); diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index a5174ad7e55..559f15206ab 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -44,6 +44,9 @@ const { parsePullArgs, validateBatchEntry, validateBudget, + newPendingPromise, + takeReject, + takeResolve, } = require('internal/streams/iter/utils'); const { converters, @@ -294,7 +297,9 @@ class PushQueue { * @returns {Promise} */ #createPendingWrite(batch, signal) { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); const entry = new PendingWrite(batch, resolve, reject); this.#pendingWrites.push(entry); @@ -429,7 +434,9 @@ class PushQueue { * @returns {Promise} */ waitForDrain() { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); ArrayPrototypePush(this.#pendingDrains, new PendingRequest(resolve, reject)); return promise; } @@ -467,7 +474,9 @@ class PushQueue { throw this.#writerError; } - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); this.#pendingReads.push(new PendingRequest(resolve, reject)); return promise; } diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index 57e0885b2ea..2c593fb04c7 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -12,7 +12,6 @@ const { PromisePrototypeThen, PromiseReject, PromiseResolve, - PromiseWithResolvers, SafeSet, Symbol, SymbolAsyncIterator, @@ -49,6 +48,8 @@ const { splitBatchEntry, validateBatchEntry, validateBudget, + newResolveOnlyPromise, + takeResolve, } = require('internal/streams/iter/utils'); const { converters, @@ -480,7 +481,8 @@ class ShareImpl { 'buffered bytes', `< ${this.#options.budget}`, this.#bufferedBytes); case 'unbounded': { - const { promise, resolve } = PromiseWithResolvers(); + const promise = newResolveOnlyPromise(); + const resolve = takeResolve(); ArrayPrototypePush(this.#pullWaiters, resolve); await promise; break; @@ -512,7 +514,8 @@ class ShareImpl { !this.#cancelled && this.#sourceError === kNoShareError && !this.#sourceExhausted) { - const { promise, resolve } = PromiseWithResolvers(); + const promise = newResolveOnlyPromise(); + const resolve = takeResolve(); ArrayPrototypePush(this.#pullWaiters, resolve); await promise; } @@ -526,7 +529,8 @@ class ShareImpl { if (this.#pulling) return this.#pullDone; this.#pulling = true; - const { promise, resolve } = PromiseWithResolvers(); + const promise = newResolveOnlyPromise(); + const resolve = takeResolve(); this.#pullDone = promise; this.#resolvePull = resolve; this.#pullDiscard = discard; diff --git a/lib/internal/streams/iter/transform.js b/lib/internal/streams/iter/transform.js index e7239278500..da4258dada0 100644 --- a/lib/internal/streams/iter/transform.js +++ b/lib/internal/streams/iter/transform.js @@ -18,7 +18,6 @@ const { ObjectKeys, PromisePrototypeThen, PromiseResolve, - PromiseWithResolvers, StringPrototypeStartsWith, SymbolAsyncIterator, TypedArrayPrototypeFill, @@ -39,7 +38,12 @@ const { } = require('internal/errors'); const { isArrayBufferView, isAnyArrayBuffer } = require('internal/util/types'); const { kValidatedTransform } = require('internal/streams/iter/types'); -const { kNullOnceOption } = require('internal/streams/iter/utils'); +const { + kNullOnceOption, + newPendingPromise, + takeReject, + takeResolve, +} = require('internal/streams/iter/utils'); const { checkRangesOrGetDefault, kValidateObjectAllowArray, @@ -416,7 +420,9 @@ function makeZlibTransform(createHandleFn, processFlag, finishFlag) { signal.addEventListener('abort', onAbort, kNullOnceOption); function continueInputAsync() { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); resolveWrite = resolve; rejectWrite = reject; writeAvailOutBefore = chunkSize - outOffset; diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 0191a4dc83d..daa8f150a70 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -8,6 +8,7 @@ const { ObjectDefineProperty, ObjectFreeze, ObjectSetPrototypeOf, + Promise, PromisePrototypeThen, PromiseReject, PromiseResolve, @@ -62,6 +63,47 @@ const kNullOnceOption = ObjectFreeze({ __proto__: null, once: true }); // Cached resolved promise to avoid allocating a new one on every sync fast-path. const kResolvedPromise = PromiseResolve(); +// A pending promise and its resolving functions, for paths that create one +// per read or per batch. PromiseWithResolvers() also allocates an object to +// return the three in (about 85 of its 300 bytes without pointer +// compression). The executor runs synchronously, so the functions are taken +// from these slots right after the promise is created: +// const promise = newPendingPromise(); +// const resolve = takeResolve(); +// const reject = takeReject(); +let capturedResolve; +let capturedReject; +function captureResolvers(resolve, reject) { + capturedResolve = resolve; + capturedReject = reject; +} + +function newPendingPromise() { + return new Promise(captureResolvers); +} + +function takeResolve() { + const resolve = capturedResolve; + capturedResolve = undefined; + return resolve; +} + +function takeReject() { + const reject = capturedReject; + capturedReject = undefined; + return reject; +} + +// The same, for a promise that is only ever resolved: only takeResolve() +// follows it. +function captureResolve(resolve) { + capturedResolve = resolve; +} + +function newResolveOnlyPromise() { + return new Promise(captureResolve); +} + // Shared TextEncoder instance for string conversion. const encoder = new TextEncoder(); @@ -408,7 +450,9 @@ function createAbortableIterator(source, signal) { signal.addEventListener('abort', onAbort, { __proto__: null, [kWeakHandler]: self }); } - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); let next; try { signal.throwIfAborted(); @@ -625,7 +669,9 @@ function createOperationQueue() { // once its result is known, now or after the operations before it. run(method, arg) { if (busy || (queue !== null && queue.length !== 0)) { - const { promise, resolve, reject } = PromiseWithResolvers(); + const promise = newPendingPromise(); + const resolve = takeResolve(); + const reject = takeReject(); queue ??= new RingBuffer(); queue.push(new QueuedOperation(method, arg, resolve, reject)); return promise; @@ -992,6 +1038,10 @@ module.exports = { kNullOnceOption, kPushDefaultBudget, kResolvedPromise, + newPendingPromise, + newResolveOnlyPromise, + takeReject, + takeResolve, callWithByteView, concatBytes, createOperationQueue, From 610887712c94dd2ef92df67b94871d6442dcd5cf Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 14:51:48 +0000 Subject: [PATCH 53/58] stream: keep stream/iter broadcast waiters in reused lists Broadcast notified waiting consumers by swapping #waiters for a new SafeSet on every notification, so that a consumer re-waiting while being notified was not processed twice. That happens for every chunk that consumers wait for, and each Set operation allocates: the new Set, the for...of iterator, and clear() or deleting the last entries (120-150 bytes each). The waiters are now kept in two RingBuffers that are swapped back and forth; the notified list is drained with shift() and cleared, keeping its storage. A nested notification while the spare list is in use gets a new one. A consumer detaching removes itself with indexOf()/removeAt(), as it did with delete(); it is in the list at most once, as before. Broadcast.from() fan-out, async source, two consumers, 16-byte chunks: 2036 -> 1873 bytes allocated per chunk. Assisted-by: OpenCode --- lib/internal/streams/iter/broadcast.js | 42 +++++++++++---- ...st-stream-iter-broadcast-waiter-cleanup.js | 53 ++++++++++++++++++- 2 files changed, 85 insertions(+), 10 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index 3d15dc272bb..5aa21ef77f7 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -124,7 +124,12 @@ class BroadcastImpl { #buffer = new RingBuffer(); #bufferStart = 0; #consumers = new SafeSet(); - #waiters = new SafeSet(); // Consumers with pending resolve + // Consumers with a pending resolve, each at most once. A RingBuffer + // rather than a Set: adding, clearing and iterating a Set allocate, and + // this changes for every chunk that consumers wait for. + #waiters = new RingBuffer(4); + // The list #notifyConsumers() swaps in for #waiters (see there). + #spareWaiters = new RingBuffer(4); #ended = false; #error; #errored = false; @@ -224,7 +229,7 @@ class BroadcastImpl { function detach() { state.detached = true; - self.#waiters.delete(state); + self.#deleteWaiter(state); if (state.resolve) { state.resolve(new IterResult(true, undefined)); } @@ -286,7 +291,7 @@ class BroadcastImpl { const reject = takeReject(); state.resolve = resolve; state.reject = reject; - self.#waiters.add(state); + self.#waiters.push(state); return promise; }, @@ -526,11 +531,30 @@ class BroadcastImpl { #notifyConsumers() { const waiters = this.#waiters; - if (waiters.size === 0) return; + if (waiters.length === 0) return; // Swap out the waiters list so consumers that re-wait during - // resolve don't get processed twice in this cycle. - this.#waiters = new SafeSet(); - for (const consumer of waiters) { + // resolve don't get processed twice in this cycle. This runs for every + // write that consumers are waiting for, so the two lists are swapped + // back and forth rather than a new one created each time; a nested + // call, while the spare list is in use, gets a new one. + this.#waiters = this.#spareWaiters ?? new RingBuffer(4); + this.#spareWaiters = null; + try { + this.#notifyWaiters(waiters); + } finally { + waiters.clear(); + this.#spareWaiters = waiters; + } + } + + #deleteWaiter(consumer) { + const index = this.#waiters.indexOf(consumer); + if (index !== -1) this.#waiters.removeAt(index); + } + + #notifyWaiters(waiters) { + let consumer; + while ((consumer = waiters.shift()) !== undefined) { if (consumer.resolve) { const bufferIndex = consumer.cursor - this.#bufferStart; if (bufferIndex < this.#buffer.length) { @@ -549,11 +573,11 @@ class BroadcastImpl { if (consumer.detached && this.#deleteConsumer(consumer)) { this.#tryTrimBuffer(); } else if (this.#promotePending(consumer)) { - this.#waiters.add(consumer); + this.#waiters.push(consumer); } } else { // Still waiting -- put back - this.#waiters.add(consumer); + this.#waiters.push(consumer); } } } diff --git a/test/parallel/test-stream-iter-broadcast-waiter-cleanup.js b/test/parallel/test-stream-iter-broadcast-waiter-cleanup.js index 4e0f6dd4d47..3bb1c644324 100644 --- a/test/parallel/test-stream-iter-broadcast-waiter-cleanup.js +++ b/test/parallel/test-stream-iter-broadcast-waiter-cleanup.js @@ -31,4 +31,55 @@ async function testDetachedWaitersAreReleased() { shared.cancel(); } -testDetachedWaitersAreReleased().then(common.mustCall()); +// Consumers waiting for writes are notified once per write, across many +// writes: some wait with one next() at a time, one queues several next() +// calls, and one detaches while it is waiting. Every remaining consumer +// gets every chunk, in order. +async function testWaitersAcrossWrites() { + // A 4-byte budget, so that the writer also waits for the consumers. + const { writer, broadcast: shared } = broadcast({ budget: 4 }); + const n = 200; + const readAll = async (iterator) => { + const seen = []; + for (;;) { + const { done, value } = await iterator.next(); + if (done) return seen; + for (const chunk of value) seen.push(chunk[0]); + } + }; + const readQueued = async (iterator) => { + const seen = []; + for (;;) { + const results = await Promise.all( + [iterator.next(), iterator.next(), iterator.next()]); + for (const { done, value } of results) { + if (done) return seen; + for (const chunk of value) seen.push(chunk[0]); + } + } + }; + const a = readAll(shared.push()[Symbol.asyncIterator]()); + const b = readAll(shared.push()[Symbol.asyncIterator]()); + const c = readQueued(shared.push()[Symbol.asyncIterator]()); + for (let i = 0; i < n; i++) { + if (i === n / 2) { + // Joins at the current position, waits, and detaches while waiting. + const detaching = shared.push()[Symbol.asyncIterator](); + const pending = detaching.next(); + await detaching.return(); + await pending; + } + await writer.write(new Uint8Array([i % 256])); + if (i % 7 === 0) await new Promise(setImmediate); + } + await writer.end(); + const expected = Array.from({ length: n }, (_, i) => i % 256); + assert.deepStrictEqual(await a, expected); + assert.deepStrictEqual(await b, expected); + assert.deepStrictEqual(await c, expected); + assert.strictEqual(shared.consumerCount, 0); +} + +testDetachedWaitersAreReleased() + .then(testWaitersAcrossWrites) + .then(common.mustCall()); From 0fb5beba218234d3c6966be3d6c0cbf263aa2738 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 14:51:48 +0000 Subject: [PATCH 54/58] stream: create the stream/iter share() read reaction once per consumer share()'s #readAfterPull() created each consumer's afterPull() reaction lazily with `state.afterPull ??= () => {...}`. Because the arrow function captures `this` and `state`, V8 allocates its context on every call of #readAfterPull(), even when afterPull() already exists: about 65 bytes per read that waits for the source. afterPull() is now created by #createAfterPull(), so the context is only allocated when it is. share() fan-out, async source, two consumers, 16-byte chunks: 2065 -> 1944 bytes allocated per chunk. Assisted-by: OpenCode --- lib/internal/streams/iter/share.js | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index 2c593fb04c7..0b09a95b1b4 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -408,7 +408,15 @@ class ShareImpl { // longer pending: clear it from pendingNext, unless a next() has been // queued behind it, without a reaction of its own. #readAfterPull(state) { - state.afterPull ??= () => { + // afterPull() is created by a method of its own: if it were created + // here, V8 would allocate the context it captures on every call, even + // when it already exists. + state.afterPull ??= this.#createAfterPull(state); + return PromisePrototypeThen(this.#pullFromSource(false), state.afterPull); + } + + #createAfterPull(state) { + return () => { let result; try { result = this.#tryRead(state); @@ -426,7 +434,6 @@ class ShareImpl { this.#clearWaitingRead(state); return result; }; - return PromisePrototypeThen(this.#pullFromSource(false), state.afterPull); } #clearWaitingRead(state) { From 79f7b0c0de4c7b8a263cc4a863d67d0c9e6d695e Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 15:04:58 +0000 Subject: [PATCH 55/58] stream: read stream/iter from() sources in pipelines without a promise per read A pull() pipeline, pipeTo() with a signal and the consumers with a signal reject their own pending pull, pipe or read when they are aborted, and ignore the read of the source that is still pending then. Yet they read a from() async source with its cancellable next(), which gives every read a promise of its own so that a cancellation can reject it at once: about 230 bytes per batch, for a rejection nobody waits for. They now read such sources with kNextUncancellable, as pipeTo() without a signal already did, when nothing but stream/iter code waits for the reads: always for pipeTo() and the consumers, and for pull() pipelines without stateful transforms. A stateful transform's generator waits for the source in user code, so with one the reads stay cancellable and its finally block still runs at once. What makes this possible is that a cancellation now releases a pending kNextUncancellable read instead of waiting for it: the normalization ends, the source is closed at once (as for a cancelled next(), errors ignored), and the operations queued behind the read, such as the return() that cancelled it, run. The read settles when the source's next() does, rejecting with the cancellation reason, and is ignored. A pipeline stopped with return() or throw() while a pull is pending closes its (stateless) transform layers without waiting for them, since they wait for that read. Allocation per batch, 16-byte chunks, async source: - pipeTo() with a signal: 967 -> 738 bytes (as without a signal) - pull() with one transform: 1518 -> 1289 bytes - pipeTo() with a signal and one transform: 1829 -> 1601 bytes test-stream-iter-abort-pending-read.js checks, for pull(), pipeTo() and bytes(), with and without transforms and stopped by a signal, return() or throw() while the source's next() is pending, that the call rejects at once, the source is closed before that next() settles, and nothing happens when it does. It also passes before this change. Assisted-by: OpenCode --- lib/internal/streams/iter/consumers.js | 8 +- lib/internal/streams/iter/from.js | 64 +++++-- lib/internal/streams/iter/pull.js | 73 +++++++- .../test-stream-iter-abort-pending-read.js | 165 ++++++++++++++++++ 4 files changed, 287 insertions(+), 23 deletions(-) create mode 100644 test/parallel/test-stream-iter-abort-pending-read.js diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index 5177b8d9c74..acfcd55f55a 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -54,6 +54,7 @@ const { fromSync, isAsyncIterable, isSyncIterable, + kNextUncancellable, } = require('internal/streams/iter/from'); const { @@ -170,10 +171,15 @@ async function collectAsync(source, signal, limit) { // Read as for await...of does, stopping once the signal aborts; see // raceSignal(). + // A from() source is read with kNextUncancellable: on an abort, + // raceSignal() rejects and closes the source at once, and a read still + // pending is ignored. const state = { __proto__: null, aborted: false }; async function consume(iterator) { + const uncancellable = iterator[kNextUncancellable] !== undefined; for (;;) { - const result = await iterator.next(); + const result = await (uncancellable ? + iterator[kNextUncancellable]() : iterator.next()); if (result.done || state.aborted) return; try { recordBatch(result.value); diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 22821627ac8..269dc662d7b 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -152,8 +152,13 @@ function waitForNormalization(value, context) { const kNextSyncBatch = Symbol('kNextSyncBatch'); // The method of an iterator returned by from() for an async iterable that -// reads like next() when nothing can cancel the normalization while a read -// is pending, for pipeTo(): see createAsyncSourceNormalizer(). +// reads like next(), for internal callers that stop reading when they +// cancel the normalization (pipeTo(), pull() pipelines, consumers): a +// return() or throw() while a read is pending closes the source at once, +// as for next(), but does not settle the pending read early. That read +// settles when the source's next() does, rejecting with the cancellation +// reason, and the caller ignores it. It saves the read a promise of its +// own: see createAsyncSourceNormalizer(). const kNextUncancellable = Symbol('kNextUncancellable'); // Methods of the iterator returned by yieldNormalizationAbortable(), used by @@ -591,11 +596,12 @@ function yieldNormalizationAbortable(source, context) { }); return promise; }, - // Like next(), when nothing cancels the normalization while the read - // is pending, in two parts so that the caller handles the result in - // the same reaction: kReadUncancellable starts the read, returning a - // promise for the source's result, and the caller passes that result - // to kToIterResult, or calls kReadFailed if the promise rejects. + // Like next(), for a caller that ignores the read if it cancels the + // normalization while the read is pending (see kNextUncancellable), + // in two parts so that the caller handles the result in the same + // reaction: kReadUncancellable starts the read, returning a promise + // for the source's result, and the caller passes that result to + // kToIterResult, or calls kReadFailed if the promise rejects. [kReadUncancellable]() { if (completed) { return PromiseResolve(kDoneSourceResult); @@ -836,9 +842,11 @@ function createAsyncSourceNormalizer(source, context) { let boundedBatches = null; let valueBatches = null; // Whether the source is read with kNextUncancellable: set by the - // kNextUncancellable method, for callers that never cancel the - // normalization while a read is pending. + // kNextUncancellable method, for callers that ignore a read pending when + // they cancel the normalization. readReleased: whether a cancellation has + // released such a read (see onUncancellableCancelled()). let uncancellable = false; + let readReleased = false; // Whether the source is read with kRead, and the settling functions of // the read's promise while it is pending. let readable = false; @@ -915,6 +923,8 @@ function createAsyncSourceNormalizer(source, context) { // A kReadUncancellable result: handled as the result of the source's // next() is, in the same reaction. function onUncancellableResult(result) { + if (readReleased) throw context.reason; + context.resolve = null; let iterResult; try { iterResult = iterator[kToIterResult](result); @@ -925,10 +935,41 @@ function createAsyncSourceNormalizer(source, context) { } function onUncancellableError(error) { + if (readReleased) throw context.reason; + context.resolve = null; iterator[kReadFailed](); return fail(error); } + // context.resolve while a kNextUncancellable read is pending: the caller + // ignores that read, so it is released rather than waited for. The + // normalization ends, the source is closed at once, as for a cancelled + // kRead (errors ignored), and the operations queued behind the read, + // such as the return() that cancelled it, run now. The read settles when + // the source's next() does, rejecting with the cancellation reason. + function onUncancellableCancelled() { + context.resolve = null; + readReleased = true; + state = kDone; + markPromiseAsHandled(iterator.return()); + settled(); + } + + function readUncancellable() { + const cancelledBefore = context.cancelled; + const read = iterator[kReadUncancellable](); + if (!cancelledBefore) { + // Cancelled while the source's next() was called, or later. + if (context.cancelled) { + onUncancellableCancelled(); + } else { + context.resolve = onUncancellableCancelled; + } + } + return PromisePrototypeThen(read, onUncancellableResult, + onUncancellableError); + } + // The kRead handlers: the result is handled in the reaction to the // source's result, and settles the read's promise. function onRead(result) { @@ -954,10 +995,7 @@ function createAsyncSourceNormalizer(source, context) { } function pullSource() { - if (uncancellable) { - return PromisePrototypeThen(iterator[kReadUncancellable](), - onUncancellableResult, onUncancellableError); - } + if (uncancellable) return readUncancellable(); if (readable) { // One promise for the read, which a cancellation rejects, instead of // next()'s promise and a reaction to it. diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 12a6b659b9e..109d51c8168 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -1233,20 +1233,44 @@ class PipelineSource { #abort; #iterator = null; #closed = false; - - constructor(source, abort) { + // Whether a from() source is read with kNextUncancellable (see the + // constructor). + #uncancellable; + + // `uncancellable`: whether nothing but stream/iter code waits for the + // reads, so that a read pending when the pipeline aborts can be left to + // settle when the source's next() does. The pipeline rejects its own + // pending pull and ignores the read, and closing the source still calls + // its return() at once. This saves every read of a from() source a + // promise of its own. A stateful transform reads the source in user code + // (its generator waits for the read), so with one the reads stay + // cancellable, and an abort rejects them at once. + constructor(source, abort, uncancellable) { this.#source = source; this.#abort = abort; + this.#uncancellable = uncancellable; } [SymbolAsyncIterator]() { return this; } + // Whether the source is read with kNextUncancellable (known once it has + // been read). + get uncancellable() { + return this.#uncancellable && this.#iterator !== null; + } + next() { if (this.#abort.aborted) return PromiseReject(this.#abort.reason); if (this.#closed) return PromiseResolve(new IterResult(true, undefined)); - this.#iterator ??= this.#source[SymbolAsyncIterator](); + if (this.#iterator === null) { + this.#iterator = this.#source[SymbolAsyncIterator](); + if (this.#iterator[kNextUncancellable] === undefined) { + this.#uncancellable = false; + } + } + if (this.#uncancellable) return this.#iterator[kNextUncancellable](); return this.#iterator.next(); } @@ -1290,6 +1314,13 @@ class PipelineSource { } } +function hasTransformObject(transforms) { + for (let i = 0; i < transforms.length; i++) { + if (isTransformObject(transforms[i])) return true; + } + return false; +} + /** * Build the chain of transform layers of a pipeline. * @param {AsyncIterable} normalized @@ -1377,6 +1408,8 @@ function createAsyncPipeline(source, transforms, signal, // runs one pull at a time. let resolvePull = null; let rejectPull = null; + // See stop(). + let closeWithoutWaiting = false; const operations = createOperationQueue(); const self = ObjectSetPrototypeOf({ @@ -1437,6 +1470,14 @@ function createAsyncPipeline(source, transforms, signal, if (rejectPull !== null) { abortPipeline(reason, false); state = kActive; + // The pending pull is waiting for a read of the source that is now + // ignored, and with uncancellable reads nothing settles that read + // early: closing the transforms would wait for it, as a transform + // layer's return() waits for its pending next(). The transforms are + // all stateless then (see PipelineSource), so nothing but stream/iter + // code waits for them, and doReturn() / doThrow() close them without + // waiting. The source has already been closed. + closeWithoutWaiting = pipelineSource.uncancellable; } else { abort.abort(reason); } @@ -1542,7 +1583,8 @@ function createAsyncPipeline(source, transforms, signal, if (state === kStart) { state = kActive; abort = new PipelineAbort(); - pipelineSource = new PipelineSource(source, abort); + pipelineSource = new PipelineSource( + source, abort, !hasTransformObject(transforms)); if (signal !== undefined) { signal.addEventListener('abort', onSignalAbort, { __proto__: null, once: true, @@ -1600,6 +1642,11 @@ function createAsyncPipeline(source, transforms, signal, } state = kDone; removeSignalListener(); + if (closeWithoutWaiting) { + markPromiseAsHandled(closeAsyncIterator(iterator, false)); + operations.settled(); + return PromiseResolve(result); + } return PromisePrototypeThen( closeAsyncIterator(iterator, false), () => { operations.settled(); @@ -1618,6 +1665,11 @@ function createAsyncPipeline(source, transforms, signal, } state = kDone; removeSignalListener(); + if (closeWithoutWaiting) { + markPromiseAsHandled(closeAsyncIterator(iterator, true)); + operations.settled(); + return PromiseReject(error); + } return PromisePrototypeThen( closeAsyncIterator(iterator, true), () => { operations.settled(); @@ -2040,10 +2092,14 @@ async function pipeTo(source, ...args) { } // Write the batches of a source read with next(), as the for await...of - // loop below does, stopping once `signal` aborts. + // loop below does, stopping once `signal` aborts. A from() source is read + // with kNextUncancellable: on an abort, raceSignal() rejects and closes + // the source at once, and a read still pending is ignored. async function pipeSourceWithSignal(iterator) { + const uncancellable = iterator[kNextUncancellable] !== undefined; for (;;) { - const result = await iterator.next(); + const result = await (uncancellable ? + iterator[kNextUncancellable]() : iterator.next()); if (result.done || abortState.aborted) return; try { const p = writeBatch(result.value); @@ -2144,9 +2200,8 @@ async function pipeTo(source, ...args) { if (transforms.length === 0) { // Fast path: no transforms - iterate normalized source directly if (signal) { - // Batches are read synchronously when possible, as without a signal. - // Async reads go through next(), which return() can cancel while it - // is pending, unlike kNextUncancellable. + // Batches are read synchronously when possible, as without a signal, + // otherwise as pipeSourceWithSignal() says. const iterator = normalized[SymbolAsyncIterator](); await raceSignal(iterator, signal, abortState, iterator[kNextSyncBatch] !== undefined ? diff --git a/test/parallel/test-stream-iter-abort-pending-read.js b/test/parallel/test-stream-iter-abort-pending-read.js new file mode 100644 index 00000000000..87a051a17df --- /dev/null +++ b/test/parallel/test-stream-iter-abort-pending-read.js @@ -0,0 +1,165 @@ +// Flags: --experimental-stream-iter +'use strict'; + +// An abort while a from() source's next() is pending: the pull, pipe or +// consumer rejects at once with the abort reason, and the source is closed +// at once, before its pending next() settles. When that next() settles +// later, nothing more happens: no write, no batch delivered, and no +// unhandled rejection (which would fail the test). + +const common = require('../common'); +const assert = require('assert'); +const { bytes, pipeTo, pull, from } = require('stream/iter'); + +const tick = () => new Promise(setImmediate); + +// An async source whose first next() gives one batch and whose later +// next() calls stay pending until release(). +function pendingSource() { + const log = []; + let reads = 0; + let release = null; + const source = { + [Symbol.asyncIterator]() { return this; }, + next() { + reads++; + log.push('next'); + if (reads === 1) { + return Promise.resolve({ done: false, value: [new Uint8Array([1])] }); + } + return new Promise((resolve) => { + release = () => resolve({ done: false, value: [new Uint8Array([2])] }); + }); + }, + return() { + log.push('return'); + return Promise.resolve({ done: true, value: undefined }); + }, + }; + return { source, log, release: () => release() }; +} + +async function checkPendingRead(start, { stateful = false } = {}) { + const { source, log, release } = pendingSource(); + const ac = new AbortController(); + const reason = new Error('stop'); + const { done, readFirst, extra } = start(from(source), ac.signal, log); + await readFirst; + await tick(); + assert.deepStrictEqual(log, ['next', 'next']); + ac.abort(reason); + await assert.rejects(done, reason); + await tick(); + // Closed while its next() is pending. + assert.deepStrictEqual(log.slice(0, 3), ['next', 'next', 'return']); + if (stateful) assert.ok(log.includes('finally')); + release(); + await tick(); + await tick(); + assert.deepStrictEqual(extra(), []); + assert.strictEqual(log.filter((x) => x === 'return').length, 1); + assert.strictEqual(log.filter((x) => x === 'next').length, 2); +} + +// pull() with a signal, with and without a stateless transform. +async function testPull(transforms) { + await checkPendingRead((normalized, signal) => { + const it = pull(normalized, ...transforms, { signal })[Symbol.asyncIterator](); + const late = []; + const readFirst = it.next(); + const done = readFirst.then(() => it.next()).then((r) => late.push(r)); + return { done, readFirst, extra: () => late }; + }); +} + +// pull() without a signal, stopped with return() while a pull is pending. +async function testPullReturn() { + const { source, log, release } = pendingSource(); + const it = pull(from(source), (c) => c)[Symbol.asyncIterator](); + await it.next(); + const pending = it.next(); + await tick(); + const returned = it.return(); + await assert.rejects(pending, { name: 'AbortError' }); + await returned; + assert.deepStrictEqual(log, ['next', 'next', 'return']); + release(); + await tick(); + await tick(); + assert.deepStrictEqual(log, ['next', 'next', 'return']); +} + +// The same, stopped with throw(). +async function testPullThrow() { + const { source, log, release } = pendingSource(); + const it = pull(from(source), (c) => c)[Symbol.asyncIterator](); + await it.next(); + const pending = it.next(); + await tick(); + const error = new Error('thrown'); + const thrown = it.throw(error); + await assert.rejects(pending, error); + await assert.rejects(thrown, error); + assert.deepStrictEqual(log, ['next', 'next', 'return']); + release(); + await tick(); + await tick(); + assert.deepStrictEqual(log, ['next', 'next', 'return']); +} + +// A stateful transform's generator waits for the source in user code: it +// still sees the read fail at once, and its finally block runs before the +// source's next() settles. +async function testPullStateful() { + await checkPendingRead((normalized, signal, log) => { + const transform = { + async *transform(source) { + try { + for await (const batch of source) yield batch; + } finally { + log.push('finally'); + } + }, + }; + const it = pull(normalized, transform, { signal })[Symbol.asyncIterator](); + const late = []; + const readFirst = it.next(); + const done = readFirst.then(() => it.next()).then((r) => late.push(r)); + return { done, readFirst, extra: () => late }; + }, { stateful: true }); +} + +async function testPipeTo(transforms) { + await checkPendingRead((normalized, signal) => { + const written = []; + let first; + const readFirst = new Promise((resolve) => { first = resolve; }); + const writer = { + write(chunk) { written.push(chunk[0]); first(); }, + end() {}, + fail() {}, + }; + const done = pipeTo(normalized, ...transforms, writer, { signal }); + return { done, readFirst, extra: () => written.slice(1) }; + }); +} + +async function testBytes() { + await checkPendingRead((normalized, signal) => { + // bytes() gives nothing back before the end: the first read is done + // once the source has been read twice. + const done = bytes(normalized, { signal }); + return { done, readFirst: tick(), extra: () => [] }; + }); +} + +(async () => { + await testPull([]); + await testPull([(c) => c]); + await testPullReturn(); + await testPullThrow(); + await testPullStateful(); + await testPipeTo([]); + await testPipeTo([(c) => c]); + await testBytes(); +})().then(common.mustCall()); From a4627d98d1e1f7990338a79f2b19ccbabb09cd55 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 15:15:16 +0000 Subject: [PATCH 56/58] stream: buffer single-chunk stream/iter push() writes without a batch entry Every push() write allocated, before the chunk was buffered, a [chunk] array, a BatchEntry with a views array, and a FixedByteView; a read of several buffered writes then validated each into an array of its own and copied those into the batch it returns. That was most of what a write costs (192 of 223 bytes for a 16-byte chunk, without pointer compression). A single-chunk write() or writeSync() now buffers the chunk's FixedByteView itself, which has the byteLength a BatchEntry has; entries are told apart by `views`, which a FixedByteView lacks. writev() keeps its BatchEntry. A read validates the buffered views straight into the batch it returns, and a pending write is checked without collecting its chunks. The checks are the same as before. Allocation per write (16-byte chunks; the consumer reads the buffered writes as they come): - writeSync(), falling back to write(): 223 -> 70 bytes - await write(): 397 -> 246 bytes - strings (pooled encoding): 315 -> 185 bytes testBufferedEntriesReadTogether checks that single-chunk and batch writes read together are delivered in order, and that a view resized among them rejects the read. It also passes before this change. Assisted-by: OpenCode --- lib/internal/streams/iter/push.js | 67 +++++++++++++------ lib/internal/streams/iter/utils.js | 1 + .../test-stream-iter-resizable-buffers.js | 32 +++++++++ 3 files changed, 81 insertions(+), 19 deletions(-) diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index 559f15206ab..e4e2b84a244 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -42,8 +42,10 @@ const { convertChunks, getWriterSignal, parsePullArgs, + snapshotByteView, validateBatchEntry, validateBudget, + validateByteView, newPendingPromise, takeReject, takeResolve, @@ -193,11 +195,13 @@ class PushQueue { * Returns true if write completed, false if buffer is full. * @returns {boolean} */ - writeSync(chunks) { + // `entry` is a BatchEntry, or for a single chunk a FixedByteView (see + // entryOfChunk()). + writeSync(entry) { if (this.#writerState !== 'open') return false; if (this.#consumerState !== 'active') return false; - return this.#writeEntry(createBatchEntry(chunks)); + return this.#writeEntry(entry); } #writeEntry(entry) { @@ -249,7 +253,7 @@ class PushQueue { * failure. * @returns {Promise} */ - async writeAsync(chunks, signal) { + async writeAsync(entry, signal) { // Check writer state before signal (spec order: state, then signal) if (this.#writerState === 'closed') { throw new ERR_INVALID_STATE.TypeError('Writer is closed'); @@ -268,7 +272,6 @@ class PushQueue { // Check for pre-aborted signal (after state checks per spec) signal?.throwIfAborted(); - const entry = createBatchEntry(chunks); if (this.#writeEntry(entry)) { return; } @@ -509,16 +512,24 @@ class PushQueue { #drain() { try { if (this.#slots.length === 1) { - const result = validateBatchEntry(this.#slots.shift()); + const entry = this.#slots.shift(); + const result = entry.views === undefined ? + [validateByteView(entry)] : validateBatchEntry(entry); this.#bufferedBytes = 0; return result; } + // The chunks of every entry, validated straight into one batch. const result = []; for (let i = 0; i < this.#slots.length; i++) { - const batch = validateBatchEntry(this.#slots.get(i)); - for (let j = 0; j < batch.length; j++) { - result[result.length] = batch[j]; + const entry = this.#slots.get(i); + const views = entry.views; + if (views === undefined) { + result[result.length] = validateByteView(entry); + continue; + } + for (let j = 0; j < views.length; j++) { + result[result.length] = validateByteView(views[j]); } } this.#slots.clear(); @@ -587,7 +598,7 @@ class PushQueue { this.#bufferedBytes < this.#budget) { const pending = this.#pendingWrites.shift(); try { - validateBatchEntry(pending.batch); + checkEntry(pending.batch); this.#slots.push(pending.batch); this.#bufferedBytes += pending.batch.byteLength; this.#bytesWritten += pending.batch.byteLength; @@ -640,6 +651,26 @@ class PushQueue { } } +// The buffered entry of a single-chunk write: the chunk's FixedByteView, which +// has the byteLength a BatchEntry has, instead of a BatchEntry holding an +// array with it. Writing one chunk is the common case, and this saves the +// write the BatchEntry, its views array, and the [chunk] array it was made +// from. Entries are told apart by `views`, which a FixedByteView lacks. +function entryOfChunk(chunk) { + return snapshotByteView(chunk); +} + +// Check that the chunks of a buffered entry still have the byteLength they +// were accepted with, as validateBatchEntry() does, without collecting them. +function checkEntry(entry) { + const views = entry.views; + if (views === undefined) { + validateByteView(entry); + return; + } + for (let i = 0; i < views.length; i++) validateByteView(views[i]); +} + // ============================================================================= // PushWriter Implementation // ============================================================================= @@ -663,32 +694,30 @@ class PushWriter { } write(chunk, options) { - const bytes = toWriterUint8Array(chunk); + const entry = entryOfChunk(toWriterUint8Array(chunk)); const signal = getWriterSignal(options); if (!signal && this.#queue.canWriteSync()) { - this.#queue.writeSync([bytes]); + this.#queue.writeSync(entry); return kResolvedPromise; } - return this.#queue.writeAsync([bytes], signal); + return this.#queue.writeAsync(entry, signal); } writev(chunks, options) { - const bytes = convertChunks(chunks); + const entry = createBatchEntry(convertChunks(chunks)); const signal = getWriterSignal(options); - if (!signal && this.#queue.writeSync(bytes)) { + if (!signal && this.#queue.writeSync(entry)) { return kResolvedPromise; } - return this.#queue.writeAsync(bytes, signal); + return this.#queue.writeAsync(entry, signal); } writeSync(chunk) { - const bytes = toWriterUint8Array(chunk); - return this.#queue.writeSync([bytes]); + return this.#queue.writeSync(entryOfChunk(toWriterUint8Array(chunk))); } writevSync(chunks) { - const bytes = convertChunks(chunks); - return this.#queue.writeSync(bytes); + return this.#queue.writeSync(createBatchEntry(convertChunks(chunks))); } end(options) { diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index daa8f150a70..329e1d7f80f 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -1070,5 +1070,6 @@ module.exports = { validateRecordedChunks, throwByteViewChanged, validateByteView, + snapshotByteView, yieldAbortable, }; diff --git a/test/parallel/test-stream-iter-resizable-buffers.js b/test/parallel/test-stream-iter-resizable-buffers.js index 80375ffaf2d..783ea8b4c0e 100644 --- a/test/parallel/test-stream-iter-resizable-buffers.js +++ b/test/parallel/test-stream-iter-resizable-buffers.js @@ -42,6 +42,37 @@ async function testBufferedViewMutationRejected() { } } +// Several buffered writes, single chunks and batches, read as one batch: +// they are delivered in order, and a view resized after it was written +// among them rejects the read. +async function testBufferedEntriesReadTogether() { + { + const { writer, readable } = push(); + writer.writeSync(Uint8Array.of(1)); + writer.writevSync([Uint8Array.of(2), Uint8Array.of(3)]); + writer.writeSync(Uint8Array.of(4)); + writer.endSync(); + const batches = await array(readable); + assert.deepStrictEqual(batches.map((c) => c[0]), [1, 2, 3, 4]); + } + for (const single of [true, false]) { + const buffer = new ArrayBuffer(1, { maxByteLength: 2 }); + const { writer, readable } = push(); + writer.writeSync(Uint8Array.of(1)); + writer.writevSync([Uint8Array.of(2), Uint8Array.of(3)]); + if (single) { + writer.writeSync(new Uint8Array(buffer)); + } else { + writer.writevSync([Uint8Array.of(4), new Uint8Array(buffer)]); + } + buffer.resize(2); + await assert.rejects( + readable[Symbol.asyncIterator]().next(), + kResizeError, + ); + } +} + async function testDropOldestUsesAcceptedByteLength() { const buffer = new ArrayBuffer(16384, { maxByteLength: 16384 }); const { writer, broadcast: bc } = broadcast({ @@ -336,6 +367,7 @@ Promise.all([ testPipeRejectsDetachWithWriteOnly(), testFixedLengthViewOfResizedBuffer(), testBufferedViewMutationRejected(), + testBufferedEntriesReadTogether(), testDropOldestUsesAcceptedByteLength(), testPendingWritesRejectResizedViews(), testBroadcastRejectsResizedBufferedView(), From be801cd6bbed9d47f54c964a4951d9cd38ee0c58 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 15:18:42 +0000 Subject: [PATCH 57/58] stream: buffer single-chunk stream/iter broadcast() and share() batches lightly As for push() writes, broadcast() writes and the source batches share() buffers were each kept as a BatchEntry with an array of views, also for the common case of a single chunk, and broadcast's write() first wrapped its chunk in an array. createEntry() returns the chunk's FixedByteView for a single-chunk batch and a BatchEntry otherwise; validateBatchEntry() and splitBatchEntry() take either. broadcast() and share() buffer their batches with it, and push()'s writev() too. Allocation per chunk, two consumers, 16-byte chunks: - share(), sync source: 722 -> 626 bytes; async source: 1944 -> 1848 - Broadcast.from(), sync source: 868 -> 752 bytes; async: 1873 -> 1783 testBufferedBatchViewResizeRejected checks that a view resized in a buffered multi-chunk batch still rejects the read, for broadcast() and share(). It also passes before this change. Assisted-by: OpenCode --- lib/internal/streams/iter/broadcast.js | 17 ++++++------- lib/internal/streams/iter/push.js | 21 +++++----------- lib/internal/streams/iter/share.js | 6 ++--- lib/internal/streams/iter/utils.js | 24 +++++++++++++++---- .../test-stream-iter-resizable-buffers.js | 24 +++++++++++++++++++ 5 files changed, 62 insertions(+), 30 deletions(-) diff --git a/lib/internal/streams/iter/broadcast.js b/lib/internal/streams/iter/broadcast.js index 5aa21ef77f7..54a39f33c10 100644 --- a/lib/internal/streams/iter/broadcast.js +++ b/lib/internal/streams/iter/broadcast.js @@ -64,7 +64,8 @@ const { kNullOnceOption, kResolvedPromise, convertChunks, - createBatchEntry, + createEntry, + snapshotByteView, getProtocolMethod, getWriterSignal, getMinCursor, @@ -684,18 +685,18 @@ class BroadcastWriter { const signal = getWriterSignal(options); // Fast path: no signal, writer open, buffer has space if (this.#canUseWriteFastPath(signal)) { - const batch = createBatchEntry([converted]); + const batch = snapshotByteView(converted); this.#broadcast[kWrite](batch); this.#totalBytes += batch.byteLength; return kResolvedPromise; } - return this.#writeBatchSlow(createBatchEntry([converted]), signal); + return this.#writeBatchSlow(snapshotByteView(converted), signal); } writev(chunks, options) { const converted = convertChunks(chunks); const signal = getWriterSignal(options); - const batch = createBatchEntry(converted); + const batch = createEntry(converted); // Fast path: no signal, writer open, buffer has space if (this.#canUseWriteFastPath(signal)) { if (this.#state === 'open' && this.#broadcast[kWrite](batch)) { @@ -740,7 +741,7 @@ class BroadcastWriter { writeSync(chunk) { const converted = toWriterUint8Array(chunk); if (this.#state !== 'open') return false; - const batch = createBatchEntry([converted]); + const batch = snapshotByteView(converted); if (this.#broadcast[kWrite](batch)) { this.#totalBytes += batch.byteLength; return true; @@ -751,7 +752,7 @@ class BroadcastWriter { writevSync(chunks) { const converted = convertChunks(chunks); if (this.#state !== 'open') return false; - const batch = createBatchEntry(converted); + const batch = createEntry(converted); if (this.#broadcast[kWrite](batch)) { this.#totalBytes += batch.byteLength; return true; @@ -761,7 +762,7 @@ class BroadcastWriter { [kWriteBatchSync](chunks) { if (this.#state !== 'open') return false; - const batch = createBatchEntry(chunks); + const batch = createEntry(chunks); if (this.#broadcast[kWrite](batch)) { this.#totalBytes += batch.byteLength; return true; @@ -770,7 +771,7 @@ class BroadcastWriter { } [kWriteBatch](chunks, signal) { - return this.#writeBatchSlow(createBatchEntry(chunks), signal); + return this.#writeBatchSlow(createEntry(chunks), signal); } end(options) { diff --git a/lib/internal/streams/iter/push.js b/lib/internal/streams/iter/push.js index e4e2b84a244..66c66a0a6d2 100644 --- a/lib/internal/streams/iter/push.js +++ b/lib/internal/streams/iter/push.js @@ -36,7 +36,7 @@ const { kNullOnceOption, kPushDefaultBudget, kResolvedPromise, - createBatchEntry, + createEntry, onSignalAbort, toWriterUint8Array, convertChunks, @@ -196,7 +196,7 @@ class PushQueue { * @returns {boolean} */ // `entry` is a BatchEntry, or for a single chunk a FixedByteView (see - // entryOfChunk()). + // createEntry()). writeSync(entry) { if (this.#writerState !== 'open') return false; if (this.#consumerState !== 'active') return false; @@ -651,15 +651,6 @@ class PushQueue { } } -// The buffered entry of a single-chunk write: the chunk's FixedByteView, which -// has the byteLength a BatchEntry has, instead of a BatchEntry holding an -// array with it. Writing one chunk is the common case, and this saves the -// write the BatchEntry, its views array, and the [chunk] array it was made -// from. Entries are told apart by `views`, which a FixedByteView lacks. -function entryOfChunk(chunk) { - return snapshotByteView(chunk); -} - // Check that the chunks of a buffered entry still have the byteLength they // were accepted with, as validateBatchEntry() does, without collecting them. function checkEntry(entry) { @@ -694,7 +685,7 @@ class PushWriter { } write(chunk, options) { - const entry = entryOfChunk(toWriterUint8Array(chunk)); + const entry = snapshotByteView(toWriterUint8Array(chunk)); const signal = getWriterSignal(options); if (!signal && this.#queue.canWriteSync()) { this.#queue.writeSync(entry); @@ -704,7 +695,7 @@ class PushWriter { } writev(chunks, options) { - const entry = createBatchEntry(convertChunks(chunks)); + const entry = createEntry(convertChunks(chunks)); const signal = getWriterSignal(options); if (!signal && this.#queue.writeSync(entry)) { return kResolvedPromise; @@ -713,11 +704,11 @@ class PushWriter { } writeSync(chunk) { - return this.#queue.writeSync(entryOfChunk(toWriterUint8Array(chunk))); + return this.#queue.writeSync(snapshotByteView(toWriterUint8Array(chunk))); } writevSync(chunks) { - return this.#queue.writeSync(createBatchEntry(convertChunks(chunks))); + return this.#queue.writeSync(createEntry(convertChunks(chunks))); } end(options) { diff --git a/lib/internal/streams/iter/share.js b/lib/internal/streams/iter/share.js index 0b09a95b1b4..6496e548eb7 100644 --- a/lib/internal/streams/iter/share.js +++ b/lib/internal/streams/iter/share.js @@ -40,7 +40,7 @@ const { const { IterResult, kMultiConsumerDefaultBudget, - createBatchEntry, + createEntry, getProtocolMethod, getMinCursor, onSignalAbort, @@ -620,7 +620,7 @@ class ShareImpl { } #bufferBatch(batch) { - const entry = createBatchEntry(batch); + const entry = createEntry(batch); // 'drop-oldest' evicts whole entries. A single pulled batch can be much // larger than the budget (for example when from() combines many values // of a sync source), and evicting it would discard every chunk in it, @@ -948,7 +948,7 @@ class SyncShareImpl { } #bufferBatch(batch) { - const entry = createBatchEntry(batch); + const entry = createEntry(batch); // 'drop-oldest' evicts whole entries. A single pulled batch can be much // larger than the budget (for example when from() combines many values // of a sync source), and evicting it would discard every chunk in it, diff --git a/lib/internal/streams/iter/utils.js b/lib/internal/streams/iter/utils.js index 329e1d7f80f..01e4ea0ce67 100644 --- a/lib/internal/streams/iter/utils.js +++ b/lib/internal/streams/iter/utils.js @@ -789,6 +789,17 @@ function createBatchEntry(chunks) { return new BatchEntry(views, byteLength); } +// The entry a buffer (push(), broadcast(), share()) keeps for a written or +// read batch: for a single chunk, its FixedByteView, which has the +// byteLength a BatchEntry has, rather than a BatchEntry and an array of +// views. A batch of one chunk is the common case. The two are told apart +// by `views`, which a FixedByteView lacks; validateBatchEntry() and +// splitBatchEntry() take either. +function createEntry(chunks) { + if (chunks.length === 1) return snapshotByteView(chunks[0]); + return createBatchEntry(chunks); +} + /** * Split a batch entry into consecutive entries that are each smaller than * `limit` bytes, preserving chunk order. A single chunk of `limit` bytes or @@ -799,7 +810,9 @@ function createBatchEntry(chunks) { */ function splitBatchEntry(entry, limit) { const { views } = entry; - if (entry.byteLength < limit || views.length < 2) return undefined; + if (entry.byteLength < limit || views === undefined || views.length < 2) { + return undefined; + } const entries = []; let current = []; let byteLength = 0; @@ -818,9 +831,11 @@ function splitBatchEntry(entry, limit) { } function validateBatchEntry(entry) { - const chunks = new Array(entry.views.length); - for (let i = 0; i < entry.views.length; i++) { - chunks[i] = validateByteView(entry.views[i]); + const views = entry.views; + if (views === undefined) return [validateByteView(entry)]; + const chunks = new Array(views.length); + for (let i = 0; i < views.length; i++) { + chunks[i] = validateByteView(views[i]); } return chunks; } @@ -1048,6 +1063,7 @@ module.exports = { convertChunks, checkFixedBatchChunk, createBatchEntry, + createEntry, fixedBatchToEntry, snapshotFixedBatch, recordChunk, diff --git a/test/parallel/test-stream-iter-resizable-buffers.js b/test/parallel/test-stream-iter-resizable-buffers.js index 783ea8b4c0e..675e6949653 100644 --- a/test/parallel/test-stream-iter-resizable-buffers.js +++ b/test/parallel/test-stream-iter-resizable-buffers.js @@ -134,6 +134,29 @@ async function testBroadcastRejectsResizedBufferedView() { await assert.rejects(writer.end(), kResizeError); } +// The same, for a view buffered in a batch of several chunks. +async function testBufferedBatchViewResizeRejected() { + { + const buffer = new ArrayBuffer(1, { maxByteLength: 2 }); + const { writer, broadcast: bc } = broadcast(); + const iterator = bc.push()[Symbol.asyncIterator](); + assert.strictEqual( + writer.writevSync([Uint8Array.of(1), new Uint8Array(buffer)]), true); + buffer.resize(2); + await assert.rejects(iterator.next(), kResizeError); + await assert.rejects(writer.end(), kResizeError); + } + { + const buffer = new ArrayBuffer(1, { maxByteLength: 2 }); + const shared = share([[Uint8Array.of(1), new Uint8Array(buffer)]]); + const first = shared.pull()[Symbol.asyncIterator](); + const second = shared.pull()[Symbol.asyncIterator](); + assert.strictEqual((await first.next()).value.length, 2); + buffer.resize(2); + await assert.rejects(second.next(), kResizeError); + } +} + async function testShareRejectsResizedBufferedView() { const asyncBuffer = new ArrayBuffer(1, { maxByteLength: 2 }); const shared = share([[new Uint8Array(asyncBuffer)]]); @@ -372,6 +395,7 @@ Promise.all([ testPendingWritesRejectResizedViews(), testBroadcastRejectsResizedBufferedView(), testShareRejectsResizedBufferedView(), + testBufferedBatchViewResizeRejected(), testConsumersRejectResizedViews(), testConsumersRejectDetachedViews(), testPipeRejectsWriterResize(), From 00f64c671e86ff7d41a10c385f9877e8a65e9d28 Mon Sep 17 00:00:00 2001 From: James M Snell Date: Wed, 7 Oct 2026 15:32:36 +0000 Subject: [PATCH 58/58] stream: release only stream/iter reads that can be cancelled while pending Since b809ad3fc2e every kNextUncancellable read of a from() async source registered a hook that releases the read when the normalization is cancelled while it is pending. pipeTo() without a signal never cancels then, and paid for the hook on every read: 2-5% on async pipes to a write-only sink, measured in isolation. Releasing is now asked for with kNextUncancellable(true), by the callers that can cancel while a read is pending (pull() pipelines, pipeTo() with a signal, the consumers with a signal). pipeTo() without a signal reads as it did before b809ad3fc2e, with the same handlers. pipeTo() of an async source to a write-only sink, no signal, 16-byte chunks, isolated: 0.95-0.97x before this change, 0.98-1.01x after, against the parent of b809ad3fc2e. Assisted-by: OpenCode --- lib/internal/streams/iter/consumers.js | 2 +- lib/internal/streams/iter/from.js | 69 ++++++++++++++++---------- lib/internal/streams/iter/pull.js | 4 +- 3 files changed, 47 insertions(+), 28 deletions(-) diff --git a/lib/internal/streams/iter/consumers.js b/lib/internal/streams/iter/consumers.js index acfcd55f55a..35d0c2e6a14 100644 --- a/lib/internal/streams/iter/consumers.js +++ b/lib/internal/streams/iter/consumers.js @@ -179,7 +179,7 @@ async function collectAsync(source, signal, limit) { const uncancellable = iterator[kNextUncancellable] !== undefined; for (;;) { const result = await (uncancellable ? - iterator[kNextUncancellable]() : iterator.next()); + iterator[kNextUncancellable](true) : iterator.next()); if (result.done || state.aborted) return; try { recordBatch(result.value); diff --git a/lib/internal/streams/iter/from.js b/lib/internal/streams/iter/from.js index 269dc662d7b..c552d89dd61 100644 --- a/lib/internal/streams/iter/from.js +++ b/lib/internal/streams/iter/from.js @@ -153,12 +153,14 @@ const kNextSyncBatch = Symbol('kNextSyncBatch'); // The method of an iterator returned by from() for an async iterable that // reads like next(), for internal callers that stop reading when they -// cancel the normalization (pipeTo(), pull() pipelines, consumers): a -// return() or throw() while a read is pending closes the source at once, -// as for next(), but does not settle the pending read early. That read -// settles when the source's next() does, rejecting with the cancellation -// reason, and the caller ignores it. It saves the read a promise of its -// own: see createAsyncSourceNormalizer(). +// cancel the normalization (pipeTo(), pull() pipelines, consumers). It +// saves the read a promise of its own: see createAsyncSourceNormalizer(). +// With `releasable` true, a return() or throw() while a read is pending +// closes the source at once, as for next(), but does not settle the +// pending read early: the read settles when the source's next() does, +// rejecting with the cancellation reason, and the caller ignores it. +// Without it, the caller must not cancel while a read is pending (pipeTo() +// without a signal). The first call decides for all reads. const kNextUncancellable = Symbol('kNextUncancellable'); // Methods of the iterator returned by yieldNormalizationAbortable(), used by @@ -190,8 +192,8 @@ function createNormalizationIterator(createIterator) { cancelNormalization(context, error, true); return FunctionPrototypeCall(iterator.throw, iterator, error); }, - [kNextUncancellable]() { - return iterator[kNextUncancellable](); + [kNextUncancellable](releasable) { + return iterator[kNextUncancellable](releasable); }, [SymbolAsyncIterator]() { return this; @@ -843,9 +845,13 @@ function createAsyncSourceNormalizer(source, context) { let valueBatches = null; // Whether the source is read with kNextUncancellable: set by the // kNextUncancellable method, for callers that ignore a read pending when - // they cancel the normalization. readReleased: whether a cancellation has - // released such a read (see onUncancellableCancelled()). + // they cancel the normalization. releasableReads: whether the caller + // can cancel while a read is pending, so that a cancellation releases it + // (pipeTo() without a signal cannot, and its reads skip that). + // readReleased: whether a cancellation has released such a read (see + // onUncancellableCancelled()). let uncancellable = false; + let releasableReads = false; let readReleased = false; // Whether the source is read with kRead, and the settling functions of // the read's promise while it is pending. @@ -923,8 +929,6 @@ function createAsyncSourceNormalizer(source, context) { // A kReadUncancellable result: handled as the result of the source's // next() is, in the same reaction. function onUncancellableResult(result) { - if (readReleased) throw context.reason; - context.resolve = null; let iterResult; try { iterResult = iterator[kToIterResult](result); @@ -935,18 +939,29 @@ function createAsyncSourceNormalizer(source, context) { } function onUncancellableError(error) { - if (readReleased) throw context.reason; - context.resolve = null; iterator[kReadFailed](); return fail(error); } - // context.resolve while a kNextUncancellable read is pending: the caller - // ignores that read, so it is released rather than waited for. The - // normalization ends, the source is closed at once, as for a cancelled - // kRead (errors ignored), and the operations queued behind the read, - // such as the return() that cancelled it, run now. The read settles when - // the source's next() does, rejecting with the cancellation reason. + // The same, for a releasable read. + function onReleasableResult(result) { + if (readReleased) throw context.reason; + context.resolve = null; + return onUncancellableResult(result); + } + + function onReleasableError(error) { + if (readReleased) throw context.reason; + context.resolve = null; + return onUncancellableError(error); + } + + // context.resolve while a releasable kNextUncancellable read is pending: + // the caller ignores that read, so it is released rather than waited for. + // The normalization ends, the source is closed at once, as for a + // cancelled kRead (errors ignored), and the operations queued behind the + // read, such as the return() that cancelled it, run now. The read settles + // when the source's next() does, rejecting with the cancellation reason. function onUncancellableCancelled() { context.resolve = null; readReleased = true; @@ -955,7 +970,7 @@ function createAsyncSourceNormalizer(source, context) { settled(); } - function readUncancellable() { + function readReleasable() { const cancelledBefore = context.cancelled; const read = iterator[kReadUncancellable](); if (!cancelledBefore) { @@ -966,8 +981,7 @@ function createAsyncSourceNormalizer(source, context) { context.resolve = onUncancellableCancelled; } } - return PromisePrototypeThen(read, onUncancellableResult, - onUncancellableError); + return PromisePrototypeThen(read, onReleasableResult, onReleasableError); } // The kRead handlers: the result is handled in the reaction to the @@ -995,7 +1009,11 @@ function createAsyncSourceNormalizer(source, context) { } function pullSource() { - if (uncancellable) return readUncancellable(); + if (uncancellable) { + if (releasableReads) return readReleasable(); + return PromisePrototypeThen(iterator[kReadUncancellable](), + onUncancellableResult, onUncancellableError); + } if (readable) { // One promise for the read, which a cancellation rejects, instead of // next()'s promise and a reaction to it. @@ -1090,8 +1108,9 @@ function createAsyncSourceNormalizer(source, context) { next() { return run(doNext); }, return(value) { return run(doReturn, value); }, throw(error) { return run(doThrow, error); }, - [kNextUncancellable]() { + [kNextUncancellable](releasable = false) { uncancellable = true; + releasableReads = releasable; return run(doNext); }, }, null); diff --git a/lib/internal/streams/iter/pull.js b/lib/internal/streams/iter/pull.js index 109d51c8168..06c204e78c6 100644 --- a/lib/internal/streams/iter/pull.js +++ b/lib/internal/streams/iter/pull.js @@ -1270,7 +1270,7 @@ class PipelineSource { this.#uncancellable = false; } } - if (this.#uncancellable) return this.#iterator[kNextUncancellable](); + if (this.#uncancellable) return this.#iterator[kNextUncancellable](true); return this.#iterator.next(); } @@ -2099,7 +2099,7 @@ async function pipeTo(source, ...args) { const uncancellable = iterator[kNextUncancellable] !== undefined; for (;;) { const result = await (uncancellable ? - iterator[kNextUncancellable]() : iterator.next()); + iterator[kNextUncancellable](true) : iterator.next()); if (result.done || abortState.aborted) return; try { const p = writeBatch(result.value);