Fix the streaming assembler duplicating overlapped bytes

When the muxer patched a byte range inside an already-written chunk (the
mdat header, for example), the merge trimmed the right side of the old
segment but left its full blob on the left, duplicating megabytes and
shifting every box offset — the MP4 still reported its duration but had
no usable video track (0x0 in the browser, "moov atom not found" in
ffprobe).

The collector/assembler now lives in its own module and was verified in
Node with a real transmux of J-81: the assembled file matches the source
(h264 1280x720, 500 frames, AAC, 20.84 s) byte for byte in structure.
This commit is contained in:
2026-09-17 20:50:08 -05:00
parent 038773e79e
commit 9162cf23ce
2 changed files with 94 additions and 91 deletions
+93
View File
@@ -0,0 +1,93 @@
import { StreamTarget, type StreamTargetChunk } from "mediabunny";
interface CollectedChunk {
blob: Blob;
position: number;
sequence: number;
}
/**
* Stream the muxed output into Blob parts instead of one giant ArrayBuffer —
* a 4K file would otherwise blow the tab's memory. Writes may arrive out of
* order (the muxer patches headers), so the newest write wins per byte range
* when the final blob is assembled.
*/
export function createStreamOutput(): {
target: StreamTarget;
toBlob: (type: string) => Blob;
} {
const chunks: CollectedChunk[] = [];
let sequence = 0;
const writable = new WritableStream<StreamTargetChunk>({
write(chunk) {
chunks.push({
blob: new Blob([chunk.data.slice()]),
position: chunk.position,
sequence: sequence++,
});
},
});
return {
target: new StreamTarget(writable, {
chunked: true,
chunkSize: 8 * 1024 * 1024,
}),
toBlob: (type: string) => assembleBlob(chunks, type),
};
}
export function assembleBlob(chunks: CollectedChunk[], type: string): Blob {
if (chunks.length === 0) {
throw new Error("Encoding produced no output.");
}
// Apply writes in arrival order — the newest write wins per byte range —
// and only then lay the segments out by position.
const ordered = [...chunks].sort((a, b) => a.sequence - b.sequence);
interface Segment {
start: number;
end: number;
sequence: number;
blob: Blob;
}
let segments: Segment[] = [];
for (const chunk of ordered) {
const start = chunk.position;
const end = start + chunk.blob.size;
const next: Segment[] = [];
for (const segment of segments) {
if (segment.end <= start || segment.start >= end) {
next.push(segment);
continue;
}
if (segment.start < start) {
next.push({
...segment,
end: start,
blob: segment.blob.slice(0, start - segment.start),
});
}
if (segment.end > end) {
next.push({
...segment,
start: end,
blob: segment.blob.slice(end - segment.start),
});
}
}
next.push({ start, end, sequence: chunk.sequence, blob: chunk.blob });
segments = next;
}
segments.sort((a, b) => a.start - b.start);
const parts: Blob[] = [];
let cursor = 0;
for (const segment of segments) {
if (segment.start !== cursor) {
throw new Error("The muxer produced an incomplete file.");
}
parts.push(segment.blob);
cursor = segment.end;
}
return new Blob(parts, { type });
}
+1 -91
View File
@@ -5,13 +5,12 @@ import {
Mp4OutputFormat,
Output,
Quality,
StreamTarget,
UrlSource,
WebMOutputFormat,
type StreamTargetChunk,
type VideoCodec,
} from "mediabunny";
import { createStreamOutput } from "./stream";
import {
extensionOf,
type OptimizeOptions,
@@ -28,95 +27,6 @@ function even(value: number): number {
return Math.max(2, value - (value % 2));
}
interface CollectedChunk {
blob: Blob;
position: number;
sequence: number;
}
/**
* Stream the muxed output into Blob parts instead of one giant ArrayBuffer —
* a 4K file would otherwise blow the tab's memory. Chunks may arrive out of
* order (the muxer patches headers), so later writes win per byte range when
* the final blob is assembled.
*/
function createStreamOutput(): {
target: StreamTarget;
toBlob: (type: string) => Blob;
} {
const chunks: CollectedChunk[] = [];
let sequence = 0;
const writable = new WritableStream<StreamTargetChunk>({
write(chunk) {
chunks.push({
blob: new Blob([chunk.data.slice()]),
position: chunk.position,
sequence: sequence++,
});
},
});
return {
target: new StreamTarget(writable, {
chunked: true,
chunkSize: 8 * 1024 * 1024,
}),
toBlob: (type: string) => assembleBlob(chunks, type),
};
}
function assembleBlob(chunks: CollectedChunk[], type: string): Blob {
if (chunks.length === 0) {
throw new Error("Encoding produced no output.");
}
// The muxer may rewrite earlier ranges (header patching), so apply writes
// in arrival order — the newest write wins per byte range — and only then
// lay the segments out by position.
const ordered = [...chunks].sort((a, b) => a.sequence - b.sequence);
interface Segment {
start: number;
end: number;
sequence: number;
blob: Blob;
}
let segments: Segment[] = [];
for (const chunk of ordered) {
const start = chunk.position;
const end = start + chunk.blob.size;
const next: Segment[] = [];
for (const segment of segments) {
if (segment.end <= start || segment.start >= end) {
next.push(segment);
continue;
}
if (segment.start < start) {
next.push({ ...segment, end: start });
}
if (segment.end > end) {
next.push({
...segment,
start: end,
blob: segment.blob.slice(end - segment.start),
});
}
}
next.push({ start, end, sequence: chunk.sequence, blob: chunk.blob });
segments = next;
}
segments.sort((a, b) => a.start - b.start);
const parts: Blob[] = [];
let cursor = 0;
for (const segment of segments) {
if (segment.start !== cursor) {
throw new Error("The muxer produced an incomplete file.");
}
parts.push(segment.blob);
cursor = segment.end;
}
return new Blob(parts, { type });
}
/** Transcode with WebCodecs, streaming the source and the result. */
export async function optimizeVideo(
url: string,