From 773df34d9d3e8cbcae829ab27d3b4c77f3912642 Mon Sep 17 00:00:00 2001 From: Ralph Date: Sat, 20 Jun 2026 03:55:53 -0700 Subject: [PATCH] fix(fetch): copy raw bytes for binary Request bodies (#5483) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `new Request(url, { method, body })` read a Buffer/Uint8Array body at a +12-byte offset: the first 12 bytes were dropped and 12 NUL bytes appended (length preserved). String bodies were correct. This silently corrupted every POST/PUT with a binary body — e.g. the standard node:http↔app.fetch adapter builds `new Request(url, { body: Buffer.concat(chunks) })`, so hono got a mangled form body and `c.req.parseBody()` yielded empty fields. Root cause: `js_request_new` read the body via `string_from_header`, which takes data at the StringHeader data offset (20). A Buffer/Uint8Array is a BufferHeader/TypedArrayHeader (data at offset 8), so the byte length lined up but the data read landed 12 bytes in — the Request-side twin of #5435's Response zero-fill, manifesting as a shift instead. Fix: - Factor #5435's typed-array/buffer registry probe into `body_addr_buffer_bytes(addr)`, shared by the value-based `body_value_buffer_bytes` and the new pointer-based Request path. - `js_request_new` probes the registries first and copies the real bytes, falling back to a lossless StringHeader read for genuine string bodies. Both the static-literal and `js_request_new_from_init` paths funnel through `js_request_new`, so both are fixed. - Store `RequestRecord.body` as `Vec` so a binary body survives byte-for-byte through `arrayBuffer()`/`text()` (matches Node/#5435). Adds gap test test_gap_request_binary_body_5483.ts (string, Buffer, Uint8Array, and non-UTF-8 arrayBuffer bodies — matches Node byte-for-byte). Co-Authored-By: Claude Opus 4.8 (1M context) --- crates/perry-stdlib/src/fetch/dispatch.rs | 14 ++++++++++ crates/perry-stdlib/src/fetch/mod.rs | 7 +++-- crates/perry-stdlib/src/fetch/request_ctor.rs | 13 ++++++++- .../test_gap_request_binary_body_5483.ts | 28 +++++++++++++++++++ 4 files changed, 59 insertions(+), 3 deletions(-) create mode 100644 test-files/test_gap_request_binary_body_5483.ts diff --git a/crates/perry-stdlib/src/fetch/dispatch.rs b/crates/perry-stdlib/src/fetch/dispatch.rs index 2b7db86690..9f19ea9f04 100644 --- a/crates/perry-stdlib/src/fetch/dispatch.rs +++ b/crates/perry-stdlib/src/fetch/dispatch.rs @@ -116,6 +116,20 @@ pub(crate) unsafe fn body_value_buffer_bytes(value: f64) -> Option> { return None; } }; + body_addr_buffer_bytes(addr) +} + +/// Registry-probe core shared by `body_value_buffer_bytes` (which first decodes +/// a NaN-boxed body *value* to an address) and `js_request_new` (whose body +/// argument codegen already decoded to a raw heap address via +/// `js_get_string_pointer_unified`). Returns a copy of the raw bytes when `addr` +/// is a registered typed array / Buffer / ArrayBuffer, else `None` so the caller +/// falls back to a StringHeader read. A Buffer/Uint8Array body fed straight to +/// `string_from_header` read its byte length off the right field but its data +/// off the StringHeader data offset (20) instead of the buffer data offset (8), +/// shifting every binary body left by 12 bytes (#5483, the Request-side twin of +/// #5435's zero-fill). +pub(crate) unsafe fn body_addr_buffer_bytes(addr: usize) -> Option> { if addr < 0x1000 { return None; } diff --git a/crates/perry-stdlib/src/fetch/mod.rs b/crates/perry-stdlib/src/fetch/mod.rs index 2767ab8dbe..61082b4ab3 100644 --- a/crates/perry-stdlib/src/fetch/mod.rs +++ b/crates/perry-stdlib/src/fetch/mod.rs @@ -1120,7 +1120,10 @@ fn headers_from_header_map(headers: &reqwest::header::HeaderMap) -> HeadersStore struct RequestRecord { url: String, method: String, - body: Option, + /// Raw body bytes, stored verbatim so a binary (Buffer/Uint8Array) body + /// survives byte-for-byte through `arrayBuffer()`/`text()` (#5483). `text()` + /// / `json()` still decode lossily via `from_utf8_lossy`, matching Node. + body: Option>, body_used: bool, headers: HeadersStore, destination: String, @@ -1849,7 +1852,7 @@ fn consume_request_body(handle: f64) -> Result, &'static str> { return Err(BODY_ALREADY_USED_MESSAGE); } req.body_used = true; - Ok(body.into_bytes()) + Ok(body) } /// request.text() -> Promise. Mirrors `js_fetch_response_text`: the diff --git a/crates/perry-stdlib/src/fetch/request_ctor.rs b/crates/perry-stdlib/src/fetch/request_ctor.rs index dddf0aec62..85af77d9a4 100644 --- a/crates/perry-stdlib/src/fetch/request_ctor.rs +++ b/crates/perry-stdlib/src/fetch/request_ctor.rs @@ -37,7 +37,18 @@ pub unsafe extern "C" fn js_request_new( throw_fetch_type_error(&format!("'{raw_method}' HTTP method is unsupported.")); } let method = normalize_method(&raw_method); - let body = string_from_header(body_ptr); + // A Buffer / Uint8Array / typed-array / ArrayBuffer body reaches us as a + // BufferHeader/TypedArrayHeader pointer (codegen ran the value through + // `js_get_string_pointer_unified`), NOT a StringHeader — the same for both + // the static-literal path and `js_request_new_from_init`. Reading it via + // `string_from_header` took the byte length off the right field but the data + // off the StringHeader data offset (20) instead of the buffer data offset + // (8), shifting every binary body left by 12 bytes (#5483). Probe the + // typed-array/buffer registries first and copy the real bytes verbatim; a + // genuine string body falls through to the lossless StringHeader read so its + // UTF-8 bytes are preserved. + let body: Option> = dispatch::body_addr_buffer_bytes(body_ptr as usize) + .or_else(|| dispatch::body_bytes_from_header(body_ptr)); // GET/HEAD requests may not carry a body (WHATWG fetch). Refs #2643. if body.is_some() && (method == "GET" || method == "HEAD") { throw_fetch_type_error("Request with GET/HEAD method cannot have body."); diff --git a/test-files/test_gap_request_binary_body_5483.ts b/test-files/test_gap_request_binary_body_5483.ts new file mode 100644 index 0000000000..d904ad962e --- /dev/null +++ b/test-files/test_gap_request_binary_body_5483.ts @@ -0,0 +1,28 @@ +// Test new Request(url, { body }) with a Buffer/Uint8Array body round-trips +// without the +12-byte offset that dropped the first 12 bytes (#5483). +// Expected output: +// string 16: len=16 "ABCDEFGHIJKLMNOP" +// buffer 16: len=16 "ABCDEFGHIJKLMNOP" +// buffer 40: len=40 "0123456789012345678901234567890123456789" +// uint8 6 : len=6 "ABCDEF" +// arraybuffer bytes: [255,0,137,80,78,71] + +async function tb(label: string, body: any): Promise { + const req = new Request("http://h/x", { method: "POST", body }); + const t = await req.text(); + console.log(`${label}: len=${t.length} ${JSON.stringify(t)}`); +} + +async function main(): Promise { + await tb("string 16", "ABCDEFGHIJKLMNOP"); + await tb("buffer 16", Buffer.from("ABCDEFGHIJKLMNOP")); + await tb("buffer 40", Buffer.from("0123456789012345678901234567890123456789")); + await tb("uint8 6 ", new Uint8Array([65, 66, 67, 68, 69, 70])); // "ABCDEF" + + // Non-UTF-8 bytes must survive verbatim through arrayBuffer(). + const bin = new Uint8Array([255, 0, 137, 80, 78, 71]); + const req = new Request("http://h/x", { method: "POST", body: bin }); + const got = new Uint8Array(await req.arrayBuffer()); + console.log("arraybuffer bytes: " + JSON.stringify(Array.from(got))); +} +void main();