From 617e11be2cde5800bc52d9dc5d73cc979af5a15a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 09:14:37 +0000 Subject: [PATCH 1/4] fix(shared): keep valid escapes decoded when percent-decoding malformed input MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `safeDecodeURIComponent` returned the whole input unchanged as soon as `decodeURIComponent` threw, so a single bad escape left every other escape encoded, e.g. `filename*=utf-8''%E4%B8%AD%ZZ.txt` parsed as `%E4%B8%AD%ZZ.txt`. Fall back to decoding each run of `%XX` escapes on its own and keep the runs that still fail as-is, like Hono's `tryDecode`. The filename above now parses as `中%ZZ.txt`. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014w3GkqAVAgbiC1X6JcXr4e --- packages/core/src/utils.test.ts | 4 ++++ packages/shared/src/uri.test.ts | 11 +++++++++-- packages/shared/src/uri.ts | 14 ++++++++++++-- 3 files changed, 25 insertions(+), 4 deletions(-) diff --git a/packages/core/src/utils.test.ts b/packages/core/src/utils.test.ts index 8780deda..8e44938a 100644 --- a/packages/core/src/utils.test.ts +++ b/packages/core/src/utils.test.ts @@ -108,6 +108,10 @@ it('getFilenameFromContentDisposition', () => { expect(getFilenameFromContentDisposition('inline; x="; filename*=utf-8\'\'evil.exe')).toEqual(undefined) expect(getFilenameFromContentDisposition('inline; filename="a\\"')).toEqual(undefined) + // malformed escapes in filename* are kept as-is, the valid ones are still decoded + expect(getFilenameFromContentDisposition('attachment; filename*=utf-8\'\'%E4%B8%AD%ZZ.txt')).toEqual('中%ZZ.txt') + expect(getFilenameFromContentDisposition('inline; filename*=%E2%82%AC%.txt')).toEqual('€%.txt') + // quoted filename* is tolerated expect(getFilenameFromContentDisposition('attachment; filename*="utf-8\'\'%E2%82%AC.txt"')).toEqual('€.txt') diff --git a/packages/shared/src/uri.test.ts b/packages/shared/src/uri.test.ts index dca337ac..2fdcd9b2 100644 --- a/packages/shared/src/uri.test.ts +++ b/packages/shared/src/uri.test.ts @@ -28,9 +28,16 @@ describe('safeDecodeURIComponent', () => { expect(safeDecodeURIComponent('')).toBe('') }) - it('returns malformed input unchanged instead of throwing', () => { - expect(safeDecodeURIComponent('invalid%20value%')).toBe('invalid%20value%') + it('keeps malformed escapes as-is and still decodes the valid ones instead of throwing', () => { + expect(safeDecodeURIComponent('invalid%20value%')).toBe('invalid value%') expect(safeDecodeURIComponent('%E0%A4%A')).toBe('%E0%A4%A') // Invalid UTF-8 sequence expect(safeDecodeURIComponent('%ZZ')).toBe('%ZZ') + expect(safeDecodeURIComponent('%E4%B8%AD%ZZ.txt')).toBe('中%ZZ.txt') + expect(safeDecodeURIComponent('%ZZ%e4%b8%ad%')).toBe('%ZZ中%') + expect(safeDecodeURIComponent('%FF-%E2%82%AC')).toBe('%FF-€') + // a run of escapes that is not valid UTF-8 is kept as a whole + expect(safeDecodeURIComponent('%E4%B8%AD%FF%')).toBe('%E4%B8%AD%FF%') + // decoded output is not decoded again + expect(safeDecodeURIComponent('%2541%ZZ')).toBe('%41%ZZ') }) }) diff --git a/packages/shared/src/uri.ts b/packages/shared/src/uri.ts index d0e2e675..60474832 100644 --- a/packages/shared/src/uri.ts +++ b/packages/shared/src/uri.ts @@ -1,4 +1,5 @@ const LONE_SURROGATE_REGEX = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + try { + // eslint-disable-next-line no-restricted-globals + return decodeURIComponent(escapes) + } + catch { + return escapes + } + }) } } From ea6010dbd25a402fa91abad854b7c47458019115 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 13:23:23 +0000 Subject: [PATCH 2/4] perf(shared): decode only valid UTF-8 escapes instead of catching decode errors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `safeDecodeURIComponent` tried the whole string, then on failure decoded each `%XX` run in its own try/catch. Every run that is not valid UTF-8 threw a `URIError`, so a crafted 16 KB Content-Disposition header such as `'%FF-'.repeat(4096)` took ~13 ms to parse (vs ~2 µs before the fallback). Match only RFC 3629 UTF-8 sequences spelled as escapes and decode each match. `decodeURIComponent` never throws on a match, so both try/catches go away: that input now takes ~12 µs and a malformed filename ~0.3 µs (was ~3 µs), well-formed input stays within ~1 µs. Valid sequences inside a run that is invalid as a whole are now decoded too: `%E4%B8%AD%FF` gives `中%FF` instead of staying encoded. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014w3GkqAVAgbiC1X6JcXr4e --- packages/shared/src/uri.test.ts | 22 ++++++++++++++++++---- packages/shared/src/uri.ts | 30 ++++++++++++------------------ 2 files changed, 30 insertions(+), 22 deletions(-) diff --git a/packages/shared/src/uri.test.ts b/packages/shared/src/uri.test.ts index 2fdcd9b2..a176282c 100644 --- a/packages/shared/src/uri.test.ts +++ b/packages/shared/src/uri.test.ts @@ -28,16 +28,30 @@ describe('safeDecodeURIComponent', () => { expect(safeDecodeURIComponent('')).toBe('') }) - it('keeps malformed escapes as-is and still decodes the valid ones instead of throwing', () => { + it('decodes every valid UTF-8 sequence', () => { + let chars = '' + for (let codePoint = 0; codePoint <= 0x10FFFF; codePoint++) { + if (codePoint < 0xD800 || codePoint > 0xDFFF) { + chars += String.fromCodePoint(codePoint) + } + } + + expect(safeDecodeURIComponent(safeEncodeURIComponent(chars))).toBe(chars) + }) + + it('keeps malformed and non-UTF-8 escapes as-is instead of throwing', () => { expect(safeDecodeURIComponent('invalid%20value%')).toBe('invalid value%') expect(safeDecodeURIComponent('%E0%A4%A')).toBe('%E0%A4%A') // Invalid UTF-8 sequence expect(safeDecodeURIComponent('%ZZ')).toBe('%ZZ') - expect(safeDecodeURIComponent('%E4%B8%AD%ZZ.txt')).toBe('中%ZZ.txt') expect(safeDecodeURIComponent('%ZZ%e4%b8%ad%')).toBe('%ZZ中%') expect(safeDecodeURIComponent('%FF-%E2%82%AC')).toBe('%FF-€') - // a run of escapes that is not valid UTF-8 is kept as a whole - expect(safeDecodeURIComponent('%E4%B8%AD%FF%')).toBe('%E4%B8%AD%FF%') + expect(safeDecodeURIComponent('%E4%B8%AD%FF%E4%B8')).toBe('中%FF%E4%B8') // decoded output is not decoded again expect(safeDecodeURIComponent('%2541%ZZ')).toBe('%41%ZZ') + + // overlong, surrogate, out-of-range and lone continuation sequences are not valid UTF-8 + for (const escapes of ['%C1%BF', '%E0%9F%BF', '%ED%A0%80', '%F0%8F%BF%BF', '%F4%90%80%80', '%F5%80%80%80', '%80']) { + expect(safeDecodeURIComponent(escapes)).toBe(escapes) + } }) }) diff --git a/packages/shared/src/uri.ts b/packages/shared/src/uri.ts index 60474832..5e64599a 100644 --- a/packages/shared/src/uri.ts +++ b/packages/shared/src/uri.ts @@ -1,5 +1,13 @@ const LONE_SURROGATE_REGEX = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { - try { - // eslint-disable-next-line no-restricted-globals - return decodeURIComponent(escapes) - } - catch { - return escapes - } - }) - } + // eslint-disable-next-line no-restricted-globals + return value.replace(UTF8_ESCAPES_REGEX, escapes => decodeURIComponent(escapes)) } From 336079bf0f4ec581157e119f88180b3c76b5fe9c Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 13:28:46 +0000 Subject: [PATCH 3/4] refactor(shared): decode each escape run in a try/catch, drop the UTF-8 regex Go back to catching `decodeURIComponent` errors per `%XX` run instead of matching an RFC 3629 grammar, which is harder to read and maintain. The whole-string attempt before the per-run fallback is dropped too: a valid UTF-8 sequence cannot span a non-escape character, so decoding run by run gives the same output, leaving one decode path and one try/catch. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014w3GkqAVAgbiC1X6JcXr4e --- packages/shared/src/uri.test.ts | 21 +++------------------ packages/shared/src/uri.ts | 23 +++++++++++------------ 2 files changed, 14 insertions(+), 30 deletions(-) diff --git a/packages/shared/src/uri.test.ts b/packages/shared/src/uri.test.ts index a176282c..34fef348 100644 --- a/packages/shared/src/uri.test.ts +++ b/packages/shared/src/uri.test.ts @@ -28,30 +28,15 @@ describe('safeDecodeURIComponent', () => { expect(safeDecodeURIComponent('')).toBe('') }) - it('decodes every valid UTF-8 sequence', () => { - let chars = '' - for (let codePoint = 0; codePoint <= 0x10FFFF; codePoint++) { - if (codePoint < 0xD800 || codePoint > 0xDFFF) { - chars += String.fromCodePoint(codePoint) - } - } - - expect(safeDecodeURIComponent(safeEncodeURIComponent(chars))).toBe(chars) - }) - - it('keeps malformed and non-UTF-8 escapes as-is instead of throwing', () => { + it('keeps malformed escapes as-is and still decodes the valid ones instead of throwing', () => { expect(safeDecodeURIComponent('invalid%20value%')).toBe('invalid value%') expect(safeDecodeURIComponent('%E0%A4%A')).toBe('%E0%A4%A') // Invalid UTF-8 sequence expect(safeDecodeURIComponent('%ZZ')).toBe('%ZZ') expect(safeDecodeURIComponent('%ZZ%e4%b8%ad%')).toBe('%ZZ中%') expect(safeDecodeURIComponent('%FF-%E2%82%AC')).toBe('%FF-€') - expect(safeDecodeURIComponent('%E4%B8%AD%FF%E4%B8')).toBe('中%FF%E4%B8') + // a run of escapes that is not valid UTF-8 is kept as a whole + expect(safeDecodeURIComponent('%E4%B8%AD%FF%')).toBe('%E4%B8%AD%FF%') // decoded output is not decoded again expect(safeDecodeURIComponent('%2541%ZZ')).toBe('%41%ZZ') - - // overlong, surrogate, out-of-range and lone continuation sequences are not valid UTF-8 - for (const escapes of ['%C1%BF', '%E0%9F%BF', '%ED%A0%80', '%F0%8F%BF%BF', '%F4%90%80%80', '%F5%80%80%80', '%80']) { - expect(safeDecodeURIComponent(escapes)).toBe(escapes) - } }) }) diff --git a/packages/shared/src/uri.ts b/packages/shared/src/uri.ts index 5e64599a..60d2b7b4 100644 --- a/packages/shared/src/uri.ts +++ b/packages/shared/src/uri.ts @@ -1,13 +1,5 @@ const LONE_SURROGATE_REGEX = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? decodeURIComponent(escapes)) + return value.replace(PERCENT_ESCAPES_REGEX, (escapes) => { + try { + // eslint-disable-next-line no-restricted-globals + return decodeURIComponent(escapes) + } + catch { + return escapes + } + }) } From 466e4ce7ce6f66de9e174adb087df2b733a689b1 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 13:32:02 +0000 Subject: [PATCH 4/4] perf(shared): try decoding the whole value before the per-run fallback Most values are valid, so one native `decodeURIComponent` call is about twice as fast as scanning and decoding run by run. Only malformed input pays for the throw and the per-run fallback, which keeps the valid runs decoded. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014w3GkqAVAgbiC1X6JcXr4e --- packages/shared/src/uri.ts | 24 +++++++++++++++--------- 1 file changed, 15 insertions(+), 9 deletions(-) diff --git a/packages/shared/src/uri.ts b/packages/shared/src/uri.ts index 60d2b7b4..cc72a6e0 100644 --- a/packages/shared/src/uri.ts +++ b/packages/shared/src/uri.ts @@ -23,13 +23,19 @@ export function safeDecodeURIComponent(value: string): string { return value } - return value.replace(PERCENT_ESCAPES_REGEX, (escapes) => { - try { - // eslint-disable-next-line no-restricted-globals - return decodeURIComponent(escapes) - } - catch { - return escapes - } - }) + try { + // eslint-disable-next-line no-restricted-globals + return decodeURIComponent(value) + } + catch { + return value.replace(PERCENT_ESCAPES_REGEX, (escapes) => { + try { + // eslint-disable-next-line no-restricted-globals + return decodeURIComponent(escapes) + } + catch { + return escapes + } + }) + } }