From 6c50dde28fbc62e6ec909e45cd3751f677d6f43d Mon Sep 17 00:00:00 2001 From: Mark007-R Date: Thu, 24 Sep 2026 19:58:55 +0530 Subject: [PATCH 1/2] fix(response): treat UTF-8 multi-byte characters as text in detection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A JSON response such as {"a":"测试"} served as application/json opens in the Raw format instead of JSON, while the same body with ASCII-only values opens as JSON. isLikelyText only counted printable ASCII and TAB/LF/CR as text and required more than 85% of the first 512 bytes to match, so every byte of a UTF-8 multi-byte character counted as binary. Short bodies with Chinese, Cyrillic, Japanese or emoji text fell under the threshold, detectContentTypeFromBase64 returned null, and useInitialResponseFormat treated that as "not ready", never mapping the content-type header to a format. Count complete, well-formed UTF-8 multi-byte sequences as text while walking the sample. Stray continuation bytes, invalid lead bytes and sequences cut off by the sample edge still count as non-text, so real binary data is still detected as binary. Fixes #9362 --- .../detectContentTypeFromBase64.spec.js | 74 +++++++++++++++++++ .../bruno-app/src/utils/response/index.js | 33 +++++++++ 2 files changed, 107 insertions(+) create mode 100644 packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js diff --git a/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js b/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js new file mode 100644 index 00000000000..65348b17c8d --- /dev/null +++ b/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js @@ -0,0 +1,74 @@ +import { detectContentTypeFromBase64 } from './index'; + +const toBase64 = (input) => Buffer.from(input).toString('base64'); + +describe('detectContentTypeFromBase64', () => { + describe('text detection', () => { + it('detects ASCII JSON as text', () => { + expect(detectContentTypeFromBase64(toBase64('{"a":"test"}'))).toBe('text/plain'); + expect(detectContentTypeFromBase64(toBase64('{\n "a": "test"\n}'))).toBe('text/plain'); + }); + + it('detects UTF-8 JSON with Chinese characters as text', () => { + expect(detectContentTypeFromBase64(toBase64('{"a":"测试"}'))).toBe('text/plain'); + expect(detectContentTypeFromBase64(toBase64('{"code":0,"msg":"操作成功","data":{"name":"张三"}}'))).toBe('text/plain'); + }); + + it('detects UTF-8 JSON with Cyrillic characters as text', () => { + expect(detectContentTypeFromBase64(toBase64('{"message":"Привет, мир"}'))).toBe('text/plain'); + }); + + it('detects UTF-8 text with Japanese characters and emoji as text', () => { + expect(detectContentTypeFromBase64(toBase64('{"greeting":"こんにちは世界","mood":"😀🎉"}'))).toBe('text/plain'); + }); + + it('detects a UTF-8 body longer than the 512-byte sample as text', () => { + const body = '{"a":"' + '测'.repeat(300) + '"}'; + // 6 ASCII bytes followed by 3-byte characters, so the sample is cut inside a character + expect(Buffer.byteLength(body)).toBeGreaterThan(512); + expect((512 - 6) % 3).not.toBe(0); + + expect(detectContentTypeFromBase64(toBase64(body))).toBe('text/plain'); + }); + }); + + describe('binary detection', () => { + it('detects binary formats by magic number', () => { + expect(detectContentTypeFromBase64(toBase64([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A]))).toBe('image/png'); + expect(detectContentTypeFromBase64(toBase64('%PDF-1.7\n'))).toBe('application/pdf'); + expect(detectContentTypeFromBase64(toBase64([0x1F, 0x8B, 0x08, 0x00, 0x00, 0x00]))).toBe('application/gzip'); + }); + + it('detects SVG content', () => { + expect(detectContentTypeFromBase64(toBase64(''))).toBe('image/svg+xml'); + expect(detectContentTypeFromBase64(toBase64('\n'))).toBe('image/svg+xml'); + }); + + it('returns null for bytes that are not valid UTF-8', () => { + const bytes = []; + for (let i = 0; i < 100; i++) { + bytes.push(0x80, 0x81, 0xFE, 0xFF, 0x00); + } + expect(detectContentTypeFromBase64(toBase64(bytes))).toBe(null); + }); + + it('returns null for UTF-8 lead bytes without continuation bytes', () => { + const bytes = []; + for (let i = 0; i < 256; i++) { + bytes.push(0xE6, 0x61); + } + expect(detectContentTypeFromBase64(toBase64(bytes))).toBe(null); + }); + + it('returns null for mostly binary data that contains a few UTF-8 characters', () => { + const buffer = Buffer.concat([Buffer.from('测试'), Buffer.alloc(200)]); + expect(detectContentTypeFromBase64(buffer.toString('base64'))).toBe(null); + }); + }); + + it('returns null for empty or missing input', () => { + expect(detectContentTypeFromBase64('')).toBe(null); + expect(detectContentTypeFromBase64(undefined)).toBe(null); + expect(detectContentTypeFromBase64(null)).toBe(null); + }); +}); diff --git a/packages/bruno-app/src/utils/response/index.js b/packages/bruno-app/src/utils/response/index.js index af980bd4cea..acf8a0a1972 100644 --- a/packages/bruno-app/src/utils/response/index.js +++ b/packages/bruno-app/src/utils/response/index.js @@ -69,6 +69,31 @@ export const escapeHtml = (text) => { .replace(/'/g, '''); }; +/** + * Returns the byte length of the UTF-8 multi-byte character starting at `index`, + * or 0 if the bytes there are not a complete multi-byte sequence before `end` + */ +const getUtf8SequenceLength = (buffer, index, end) => { + const byte = buffer[index]; + let length; + if (byte >= 0xC2 && byte <= 0xDF) { + length = 2; + } else if (byte >= 0xE0 && byte <= 0xEF) { + length = 3; + } else if (byte >= 0xF0 && byte <= 0xF4) { + length = 4; + } else { + return 0; + } + + if (index + length > end) return 0; + for (let j = 1; j < length; j++) { + // Continuation bytes are 10xxxxxx + if ((buffer[index + j] & 0xC0) !== 0x80) return 0; + } + return length; +}; + /** * Helper to detect if buffer contains text data */ @@ -85,6 +110,14 @@ const isLikelyText = (buffer) => { || byte === 0x0A // Line feed || byte === 0x0D) { // Carriage return textChars++; + continue; + } + + // Non-ASCII characters (e.g. Chinese, Cyrillic, emoji) encoded as UTF-8 are text too + const sequenceLength = getUtf8SequenceLength(buffer, i, sampleSize); + if (sequenceLength > 0) { + textChars += sequenceLength; + i += sequenceLength - 1; } } From ad3d004eb5380307f9d4ddf67cff471e5fce4a44 Mon Sep 17 00:00:00 2001 From: Mark007-R Date: Fri, 25 Sep 2026 17:01:08 +0530 Subject: [PATCH 2/2] fix(response): reject malformed UTF-8 sequences in text detection getUtf8SequenceLength only checked the lead byte and that the right number of continuation bytes followed it, so it also accepted overlong encodings (E0 80-9F, F0 80-8F), UTF-16 surrogates encoded as UTF-8 (ED A0-BF) and code points above U+10FFFF (F4 90-BF). Binary data made of those sequences could be reported as text. Check the second byte against the ranges Unicode allows after E0, ED, F0 and F4, so only well-formed sequences count as text. --- .../detectContentTypeFromBase64.spec.js | 18 ++++++++++++++++++ packages/bruno-app/src/utils/response/index.js | 12 +++++++++++- 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js b/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js index 65348b17c8d..35d5698185b 100644 --- a/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js +++ b/packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js @@ -1,6 +1,7 @@ import { detectContentTypeFromBase64 } from './index'; const toBase64 = (input) => Buffer.from(input).toString('base64'); +const repeatBytes = (sequence, times = 100) => Array.from({ length: times }, () => sequence).flat(); describe('detectContentTypeFromBase64', () => { describe('text detection', () => { @@ -30,6 +31,13 @@ describe('detectContentTypeFromBase64', () => { expect(detectContentTypeFromBase64(toBase64(body))).toBe('text/plain'); }); + + it('detects the first and last code points of the restricted UTF-8 ranges as text', () => { + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xE0, 0xA0, 0x80])))).toBe('text/plain'); // U+0800 + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xED, 0x9F, 0xBF])))).toBe('text/plain'); // U+D7FF + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF0, 0x90, 0x80, 0x80])))).toBe('text/plain'); // U+10000 + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF4, 0x8F, 0xBF, 0xBF])))).toBe('text/plain'); // U+10FFFF + }); }); describe('binary detection', () => { @@ -60,6 +68,16 @@ describe('detectContentTypeFromBase64', () => { expect(detectContentTypeFromBase64(toBase64(bytes))).toBe(null); }); + it('returns null for overlong, surrogate and out-of-range UTF-8 sequences', () => { + // Overlong 3- and 4-byte encodings + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xE0, 0x80, 0x80])))).toBe(null); + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF0, 0x80, 0x80, 0x80])))).toBe(null); + // A UTF-16 surrogate encoded as UTF-8 + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xED, 0xA0, 0x80])))).toBe(null); + // A code point above U+10FFFF + expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF4, 0x90, 0x80, 0x80])))).toBe(null); + }); + it('returns null for mostly binary data that contains a few UTF-8 characters', () => { const buffer = Buffer.concat([Buffer.from('测试'), Buffer.alloc(200)]); expect(detectContentTypeFromBase64(buffer.toString('base64'))).toBe(null); diff --git a/packages/bruno-app/src/utils/response/index.js b/packages/bruno-app/src/utils/response/index.js index acf8a0a1972..d3f9a0f9f06 100644 --- a/packages/bruno-app/src/utils/response/index.js +++ b/packages/bruno-app/src/utils/response/index.js @@ -71,7 +71,7 @@ export const escapeHtml = (text) => { /** * Returns the byte length of the UTF-8 multi-byte character starting at `index`, - * or 0 if the bytes there are not a complete multi-byte sequence before `end` + * or 0 if the bytes there are not a complete, well-formed multi-byte sequence before `end` */ const getUtf8SequenceLength = (buffer, index, end) => { const byte = buffer[index]; @@ -87,6 +87,16 @@ const getUtf8SequenceLength = (buffer, index, end) => { } if (index + length > end) return 0; + + // Reject overlong encodings, UTF-16 surrogates and code points above U+10FFFF + const secondByte = buffer[index + 1]; + if ((byte === 0xE0 && secondByte < 0xA0) + || (byte === 0xED && secondByte > 0x9F) + || (byte === 0xF0 && secondByte < 0x90) + || (byte === 0xF4 && secondByte > 0x8F)) { + return 0; + } + for (let j = 1; j < length; j++) { // Continuation bytes are 10xxxxxx if ((buffer[index + j] & 0xC0) !== 0x80) return 0;