-
Notifications
You must be signed in to change notification settings - Fork 2.9k
fix(response): treat UTF-8 multi-byte characters as text in detection #9367
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Open
Mark007-R
wants to merge
2
commits into
usebruno:main
Choose a base branch
from
Mark007-R:fix/response-utf8-text-detection-9362
base: main
Could not load branches
Branch not found: {{ refName }}
Loading
Could not load tags
Nothing to show
Loading
Are you sure you want to change the base?
Some commits from the old base branch may be removed from the timeline,
and old review comments may become outdated.
+135
−0
Open
Changes from all commits
Commits
File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
There are no files selected for viewing
92 changes: 92 additions & 0 deletions
92
packages/bruno-app/src/utils/response/detectContentTypeFromBase64.spec.js
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,92 @@ | ||
| import { detectContentTypeFromBase64 } from './index'; | ||
|
|
||
| const toBase64 = (input) => Buffer.from(input).toString('base64'); | ||
| const repeatBytes = (sequence, times = 100) => Array.from({ length: times }, () => sequence).flat(); | ||
|
|
||
| describe('detectContentTypeFromBase64', () => { | ||
| describe('text detection', () => { | ||
| it('detects ASCII JSON as text', () => { | ||
| expect(detectContentTypeFromBase64(toBase64('{"a":"test"}'))).toBe('text/plain'); | ||
| expect(detectContentTypeFromBase64(toBase64('{\n "a": "test"\n}'))).toBe('text/plain'); | ||
| }); | ||
|
|
||
| it('detects UTF-8 JSON with Chinese characters as text', () => { | ||
| expect(detectContentTypeFromBase64(toBase64('{"a":"测试"}'))).toBe('text/plain'); | ||
| expect(detectContentTypeFromBase64(toBase64('{"code":0,"msg":"操作成功","data":{"name":"张三"}}'))).toBe('text/plain'); | ||
| }); | ||
|
|
||
| it('detects UTF-8 JSON with Cyrillic characters as text', () => { | ||
| expect(detectContentTypeFromBase64(toBase64('{"message":"Привет, мир"}'))).toBe('text/plain'); | ||
| }); | ||
|
|
||
| it('detects UTF-8 text with Japanese characters and emoji as text', () => { | ||
| expect(detectContentTypeFromBase64(toBase64('{"greeting":"こんにちは世界","mood":"😀🎉"}'))).toBe('text/plain'); | ||
| }); | ||
|
|
||
| it('detects a UTF-8 body longer than the 512-byte sample as text', () => { | ||
| const body = '{"a":"' + '测'.repeat(300) + '"}'; | ||
| // 6 ASCII bytes followed by 3-byte characters, so the sample is cut inside a character | ||
| expect(Buffer.byteLength(body)).toBeGreaterThan(512); | ||
| expect((512 - 6) % 3).not.toBe(0); | ||
|
|
||
| expect(detectContentTypeFromBase64(toBase64(body))).toBe('text/plain'); | ||
| }); | ||
|
|
||
| it('detects the first and last code points of the restricted UTF-8 ranges as text', () => { | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xE0, 0xA0, 0x80])))).toBe('text/plain'); // U+0800 | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xED, 0x9F, 0xBF])))).toBe('text/plain'); // U+D7FF | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF0, 0x90, 0x80, 0x80])))).toBe('text/plain'); // U+10000 | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF4, 0x8F, 0xBF, 0xBF])))).toBe('text/plain'); // U+10FFFF | ||
| }); | ||
| }); | ||
|
|
||
| describe('binary detection', () => { | ||
| it('detects binary formats by magic number', () => { | ||
| expect(detectContentTypeFromBase64(toBase64([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A]))).toBe('image/png'); | ||
| expect(detectContentTypeFromBase64(toBase64('%PDF-1.7\n'))).toBe('application/pdf'); | ||
| expect(detectContentTypeFromBase64(toBase64([0x1F, 0x8B, 0x08, 0x00, 0x00, 0x00]))).toBe('application/gzip'); | ||
| }); | ||
|
|
||
| it('detects SVG content', () => { | ||
| expect(detectContentTypeFromBase64(toBase64('<svg xmlns="http://www.w3.org/2000/svg"></svg>'))).toBe('image/svg+xml'); | ||
| expect(detectContentTypeFromBase64(toBase64('<?xml version="1.0"?>\n<svg xmlns="http://www.w3.org/2000/svg"></svg>'))).toBe('image/svg+xml'); | ||
| }); | ||
|
|
||
| it('returns null for bytes that are not valid UTF-8', () => { | ||
| const bytes = []; | ||
| for (let i = 0; i < 100; i++) { | ||
| bytes.push(0x80, 0x81, 0xFE, 0xFF, 0x00); | ||
| } | ||
| expect(detectContentTypeFromBase64(toBase64(bytes))).toBe(null); | ||
| }); | ||
|
|
||
| it('returns null for UTF-8 lead bytes without continuation bytes', () => { | ||
| const bytes = []; | ||
| for (let i = 0; i < 256; i++) { | ||
| bytes.push(0xE6, 0x61); | ||
| } | ||
| expect(detectContentTypeFromBase64(toBase64(bytes))).toBe(null); | ||
| }); | ||
|
|
||
| it('returns null for overlong, surrogate and out-of-range UTF-8 sequences', () => { | ||
| // Overlong 3- and 4-byte encodings | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xE0, 0x80, 0x80])))).toBe(null); | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF0, 0x80, 0x80, 0x80])))).toBe(null); | ||
| // A UTF-16 surrogate encoded as UTF-8 | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xED, 0xA0, 0x80])))).toBe(null); | ||
| // A code point above U+10FFFF | ||
| expect(detectContentTypeFromBase64(toBase64(repeatBytes([0xF4, 0x90, 0x80, 0x80])))).toBe(null); | ||
| }); | ||
|
|
||
| it('returns null for mostly binary data that contains a few UTF-8 characters', () => { | ||
| const buffer = Buffer.concat([Buffer.from('测试'), Buffer.alloc(200)]); | ||
| expect(detectContentTypeFromBase64(buffer.toString('base64'))).toBe(null); | ||
| }); | ||
| }); | ||
|
|
||
| it('returns null for empty or missing input', () => { | ||
| expect(detectContentTypeFromBase64('')).toBe(null); | ||
| expect(detectContentTypeFromBase64(undefined)).toBe(null); | ||
| expect(detectContentTypeFromBase64(null)).toBe(null); | ||
| }); | ||
| }); |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
Uh oh!
There was an error while loading. Please reload this page.