From c7226d702ecec5bf08380f2e419a4384779d6a64 Mon Sep 17 00:00:00 2001 From: devangpratap <115096812+devangpratap@users.noreply.github.com> Date: Mon, 21 Sep 2026 12:05:20 -0400 Subject: [PATCH] Do not drop an escaped U+FFFD in the shlex lexer write_escaped_ch compares the decoded rune against utf8.RuneError to detect invalid UTF-8, but a literal U+FFFD decodes to that same rune, so an escaped replacement character was silently deleted. The size returned by DecodeRuneInString already tells the two apart: invalid bytes decode with a size of one, a real U+FFFD with a size of three. Keep the rune when the size is greater than one. --- tools/utils/shlex/shlex.go | 3 ++- tools/utils/shlex/shlex_test.go | 5 +++++ 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/tools/utils/shlex/shlex.go b/tools/utils/shlex/shlex.go index cd1d5a7e8..d0d8c0bba 100644 --- a/tools/utils/shlex/shlex.go +++ b/tools/utils/shlex/shlex.go @@ -71,7 +71,8 @@ func (self *Lexer) write_escaped_ch() bool { ch, count := utf8.DecodeRuneInString(self.src[self.src_pos:]) if count > 0 { self.src_pos += count - if ch != utf8.RuneError { + // Invalid UTF-8 always decodes as (RuneError, 1), so a larger size means a real U+FFFD. + if ch != utf8.RuneError || count > 1 { self.buf.WriteRune(ch) } return true diff --git a/tools/utils/shlex/shlex_test.go b/tools/utils/shlex/shlex_test.go index 85064a761..3ef3293cf 100644 --- a/tools/utils/shlex/shlex_test.go +++ b/tools/utils/shlex/shlex_test.go @@ -81,6 +81,11 @@ func TestSplit(t *testing.T) { ``: nil, ` `: nil, " \tabc\n\t\r ": {{2, "abc"}}, + "a\\\xffb": {{0, "ab"}}, // an escaped invalid byte is still dropped + // An escaped U+FFFD is a real character, not a decoding failure, so it + // survives the escape the way it does unescaped or single quoted. + "a\\�b": {{0, "a�b"}}, + "\"a\\�b\"": {{0, "a�b"}}, } { if diff := cmp.Diff(expected, s(q)); diff != "" { t.Fatalf("Failed for string: %#v\n%s", q, diff)