Do not drop an escaped U+FFFD in the shlex lexer

write_escaped_ch compares the decoded rune against utf8.RuneError to
detect invalid UTF-8, but a literal U+FFFD decodes to that same rune, so
an escaped replacement character was silently deleted. The size returned
by DecodeRuneInString already tells the two apart: invalid bytes decode
with a size of one, a real U+FFFD with a size of three. Keep the rune
when the size is greater than one.
This commit is contained in:
devangpratap 2026-09-21 12:05:20 -04:00
parent d4d6db3cfd
commit c7226d702e
2 changed files with 7 additions and 1 deletions

View file

@ -71,7 +71,8 @@ func (self *Lexer) write_escaped_ch() bool {
ch, count := utf8.DecodeRuneInString(self.src[self.src_pos:])
if count > 0 {
self.src_pos += count
if ch != utf8.RuneError {
// Invalid UTF-8 always decodes as (RuneError, 1), so a larger size means a real U+FFFD.
if ch != utf8.RuneError || count > 1 {
self.buf.WriteRune(ch)
}
return true

View file

@ -81,6 +81,11 @@ func TestSplit(t *testing.T) {
``: nil,
` `: nil,
" \tabc\n\t\r ": {{2, "abc"}},
"a\\\xffb": {{0, "ab"}}, // an escaped invalid byte is still dropped
// An escaped U+FFFD is a real character, not a decoding failure, so it
// survives the escape the way it does unescaped or single quoted.
"a\\<5C>b": {{0, "a<>b"}},
"\"a\\<5C>b\"": {{0, "a<>b"}},
} {
if diff := cmp.Diff(expected, s(q)); diff != "" {
t.Fatalf("Failed for string: %#v\n%s", q, diff)