From cdc291aa64d912c1319c50568305b4733917e0d9 Mon Sep 17 00:00:00 2001 From: Junegunn Choi Date: Tue, 6 Oct 2026 08:34:00 +0900 Subject: [PATCH] Drop unrecognized escape sequences instead of typing them fzf consumed only the part of a sequence it recognized, and the rest was typed into the query. CTRL-A sent as \e[97;5u became "97;5u", and CTRL-F sent as \e[70;5u fired Home before typing ";5u". A BEL-terminated OSC reply also aborted fzf, because the BEL that followed the typed payload was read as CTRL-G. Frame CSI and SS3 by their parameter and final byte ranges, OSC, DCS and APC by their terminator. Drop the whole frame when the parser does not recognize it, or recognizes only a prefix of it, and never consume past it: the parser could count a key typed after it. rxvt ends keys with $, as in \e[7$ and \e[23$, so $ after plain digits is a final byte. Elsewhere it stays intermediate, as in DECRPM replies. Take the second-chance read only when the sequence at the start of the buffer is the only one and is unfinished. Otherwise it blocked until the next key, holding back what followed, CTRL-G included. Telling a terminal's sequence from an ALT key is the hard part, so these keep what the parser made of them: - An unterminated sequence, since that is how ALT-[, ALT-O, ALT-], ALT-P and ALT-_ arrive - A CSI or SS3 of four bytes or fewer. It could be an ALT key and typed text, and rxvt sends keys of that size fzf does not know, such as \e[3^ - A string sequence whose payload does not start like a reply: digits and ';' for OSC, '>|', '!|', '[01]$r' or '[01]+r' for DCS, 'G' and a key for APC - SOS and PM, which nothing sends Within a string, an ESC ends it and introduces a sequence of its own, and only OSC ends with BEL. The read loop does not wait for a string terminator: that held ALT-], ALT-P and ALT-_ for ESCDELAY and let typed bytes pile up until they looked like a sequence. --- CHANGELOG.md | 2 + src/tui/light.go | 193 ++++++++++++++++++++++++++--- src/tui/light_csi_test.go | 233 +++++++++++++++++++++++++++++++++++ src/tui/light_escape_test.go | 43 +++++++ 4 files changed, 451 insertions(+), 20 deletions(-) create mode 100644 src/tui/light_csi_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index dbf7615d..9ff8ef75 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ CHANGELOG 0.74.5 ------ - Fixed `--gap-line` cutting a grapheme cluster when filling the last cells of the line (#4920) +- Fixed an escape sequence fzf does not recognize being partly consumed, which leaked the rest into the query (#4926) + - A reply ending in BEL also aborted fzf, because the BEL was read as CTRL-G 0.74.4 ------ diff --git a/src/tui/light.go b/src/tui/light.go index 4f5ec4d5..f3643320 100644 --- a/src/tui/light.go +++ b/src/tui/light.go @@ -348,9 +348,119 @@ func getEnv(name string, defaultValue int) int { func csiContinues(b byte) bool { return b >= 0x20 && b <= 0x3f } func csiFinal(b byte) bool { return b >= 0x40 && b <= 0x7e } -// incompleteEscape reports whether the buffer ends in an escape sequence that -// has not been terminated yet. The read loop keeps waiting in that case, so the -// parser is never handed a fragment to guess at. +// csiEnd returns the length of the CSI or SS3 sequence at the start of the +// buffer, 0 if it has no final byte yet, or -1 if it is malformed. +func csiEnd(buffer []byte) int { + digits := true + for i := 2; i < len(buffer); i++ { + b := buffer[i] + // rxvt ends keys with $ after at least one digit, as in \e[7$ for + // SHIFT-HOME and \e[23$ for SHIFT-F11. Elsewhere $ is an intermediate byte. + if csiFinal(b) || b == '$' && digits && i > 2 { + return i + 1 + } + if !csiContinues(b) { + return -1 + } + digits = digits && b >= '0' && b <= '9' + } + return 0 +} + +// stringEnd returns the length of the string sequence (DCS, OSC or APC) at the +// start of the buffer, or 0 while nothing has ended it. A terminator is +// included, an ESC that introduces another sequence is not. +func stringEnd(buffer []byte) int { + if len(buffer) < 2 { + return 0 + } + // Every one of them ends with ST. BEL ends an OSC as well, because xterm has + // always allowed it, but stopping at a BEL inside a DCS or APC payload would + // frame only its first half and leave the rest to be typed into the query. + bel := buffer[1] == ']' + for i := 2; i < len(buffer); i++ { + switch buffer[i] { + case '\a': + if bel { + return i + 1 + } + case Esc.Byte(): + if i+1 == len(buffer) { + return 0 // ST may still be arriving + } + if buffer[i+1] == '\\' { + return i + 2 + } + // Any other ESC ends the string and introduces a sequence of its + // own, so frame only what precedes it and leave the ESC to be + // parsed again. Scanning past it would swallow that sequence too. + return i + } + } + return 0 +} + +// csiIntroducer reports whether the byte after ESC starts a sequence that is +// framed by a final byte. The parser routes CSI and SS3 through the same +// parameterized subcases, so both are framed by csiEnd. +func csiIntroducer(b byte) bool { + return b == '[' || b == 'O' +} + +// stringIntroducer reports whether the byte after ESC starts a string sequence. +// Only the three that terminals reply with: OSC for colors, title and clipboard, +// DCS for XTVERSION, DECRQSS and XTGETTCAP, APC for Kitty graphics. Nothing +// sends SOS or PM. +func stringIntroducer(b byte) bool { + switch b { + case 'P', ']', '_': + return true + } + return false +} + +// stringReply reports whether a framed string sequence starts the way a +// terminal's reply does. Its introducer is also ALT-], ALT-P or ALT-_, and typed +// text after one, ended by CTRL-G or another key, frames like a reply when it +// all arrives in one read. +func stringReply(buffer []byte) bool { + payload := buffer[2:] + switch buffer[1] { + case ']': // Ps ; ... + i := 0 + for i < len(payload) && payload[i] >= '0' && payload[i] <= '9' { + i++ + } + return i > 0 && i < len(payload) && payload[i] == ';' + case 'P': // >| or !| for the version, [01]$r or [01]+r for settings + if len(payload) < 2 { + return false + } + if payload[1] == '|' { + return payload[0] == '>' || payload[0] == '!' + } + return len(payload) >= 3 && (payload[0] == '0' || payload[0] == '1') && + (payload[1] == '$' || payload[1] == '+') && payload[2] == 'r' + case '_': // G, then key=value + return len(payload) >= 3 && payload[0] == 'G' && + (payload[1] >= 'a' && payload[1] <= 'z' || payload[1] >= 'A' && payload[1] <= 'Z') && + payload[2] == '=' + } + return false +} + +// stillArriving reports whether the CSI or SS3 sequence at the start of the +// buffer may still be completed by more input. It must be the only one in the buffer: +// whatever follows it, another sequence or a byte that ended it, means it is not +// waiting for anything. +func stillArriving(buffer []byte) bool { + return bytes.IndexByte(buffer[1:], Esc.Byte()) < 0 && incompleteEscape(buffer) +} + +// incompleteEscape reports whether the buffer ends in a CSI or SS3 sequence that +// has not been terminated yet. The read loop keeps waiting while it does, so a +// fragment reaches the parser only once that wait has run out. A string sequence +// is never waited for, because its introducer is also ALT-], ALT-P or ALT-_. func incompleteEscape(buffer []byte) bool { // Only the tail can hold a sequence still arriving. This runs once per byte // read, so scanning all of a large paste would make the read quadratic. @@ -362,21 +472,10 @@ func incompleteEscape(buffer []byte) bool { if start < 0 || len(tail)-start < 2 { return false } - switch tail[start+1] { - case '[': - for _, b := range tail[start+2:] { - if csiFinal(b) { - return false - } - if !csiContinues(b) { - return false // malformed, do not wait for a terminator - } - } - return true - case 'O': - return len(tail)-start < 3 - } - return false + // SS3 is judged like CSI: declaring it finished after one byte made the + // parser give up on a split \eO1;5A. A malformed sequence is not waited for, + // as nothing can complete it. + return csiIntroducer(tail[start+1]) && csiEnd(tail[start:]) == 0 } func (r *LightRenderer) getBytes(cancellable bool) ([]byte, getCharResult, error) { @@ -483,8 +582,10 @@ func (r *LightRenderer) GetChar(cancellable bool) Event { return Event{CtrlSlash, 0, nil} case Esc.Byte(): ev := r.escSequence(&sz) - // Second chance - if ev.Type == Invalid { + // Second chance, but only for an unfinished sequence. Re-reading + // otherwise blocks until the next keystroke, holding back whatever + // follows in the buffer. + if ev.Type == Invalid && stillArriving(r.buffer) { r.buffer, result, err = r.getBytes(true) if err != nil { return Event{Fatal, 0, nil} @@ -525,7 +626,41 @@ func (r *LightRenderer) setCancel(f func()) { r.mutex.Unlock() } +// escSequence parses an escape sequence, then checks a CSI or SS3 result +// against its frame. parseEscSequence can stop after a few bytes, and consuming +// only those would leave the rest to be typed into the query. Complete +// sequences it does not recognize are dropped there, unless they could be an +// ALT key followed by typed text. func (r *LightRenderer) escSequence(sz *int) Event { + ev := r.parseEscSequence(sz) + if len(r.buffer) < 3 || !csiIntroducer(r.buffer[1]) { + return ev + } + // A frame missing its final byte may still be arriving + end := csiEnd(r.buffer) + if end <= 0 { + return ev + } + // The parser can also set the size before checking the bytes and then give + // up, counting typed input past the frame, as in \e[3;1u followed by ~ + if ev.Type == Invalid && *sz > end { + *sz = end + } + // Four bytes or fewer could be ALT-[ or ALT-O and typed text, and rxvt sends + // keys of that size fzf does not know, such as \e[3^ for CTRL-DELETE + if end <= 4 { + return ev + } + // A key matched on a prefix, such as Home for \e[70;5u, or a sequence given + // up on partway + if end > *sz { + *sz = end + return Event{Invalid, 0, nil} + } + return ev +} + +func (r *LightRenderer) parseEscSequence(sz *int) Event { if len(r.buffer) < 2 { return Event{Esc, 0, nil} } @@ -987,6 +1122,24 @@ func (r *LightRenderer) escSequence(sz *int) Event { } // r.buffer[2] } // r.buffer[2] } // r.buffer[1] + // Nothing matched. A framed sequence is dropped whole: reading its + // introducer as an ALT-key below would type the rest into the query. + // Unterminated ones are left alone, as that is how ALT-[, ALT-O, ALT-], + // ALT-P and ALT-_ arrive. + if csiIntroducer(r.buffer[1]) { + // ALT-[ or ALT-O and one or two typed characters can look like a short + // sequence, so ask for at least two bytes before the final byte + if end := csiEnd(r.buffer); end > 4 { + *sz = end + return Event{Invalid, 0, nil} + } + } else if stringIntroducer(r.buffer[1]) { + if end := stringEnd(r.buffer); end > 0 && stringReply(r.buffer) { + *sz = end + return Event{Invalid, 0, nil} + } + } + rest := bytes.NewBuffer(r.buffer[1:]) c, size, err := rest.ReadRune() if err == nil { diff --git a/src/tui/light_csi_test.go b/src/tui/light_csi_test.go new file mode 100644 index 00000000..3eeb205e --- /dev/null +++ b/src/tui/light_csi_test.go @@ -0,0 +1,233 @@ +package tui + +import "testing" + +// An unrecognized escape sequence must be consumed whole, CSI and string +// sequences alike. Consuming only part of one leaves the rest to be read as +// input and typed into the query. +func TestUnknownEscapeSequence(t *testing.T) { + for _, c := range []struct { + sequence string + event EventType + size int + }{ + // Key encodings fzf does not implement + {"\x1b[97;5u", Invalid, 7}, + {"\x1b[70;5u", Invalid, 7}, // not Home, which its prefix \e[7 matches + {"\x1b[42;5u", Invalid, 7}, // nor End + {"\x1b[4;5~", Invalid, 6}, + {"\x1b[127;5u", Invalid, 8}, + {"\x1b[27;5;127~", Invalid, 11}, + {"\x1b[57441;1u", Invalid, 10}, + {"\x1b\x1b[97;5u", Invalid, 7}, // ALT prefixed, the first ESC is dropped + + // Replies to queries fzf did not send, or sent and stopped waiting for + {"\x1b[?1;2c", Invalid, 7}, + {"\x1b[>0;95;0c", Invalid, 10}, + + // Mouse report arriving while mouse input is off + {"\x1b[<0;1;1M", Invalid, 9}, + + // String sequences + {"\x1b]11;rgb:4a4a/4a4a/4a4a\x1b\\", Invalid, 25}, // background color reply + {"\x1b]0;a title\a", Invalid, 12}, // BEL terminated + {"\x1bP>|kitty(0.48.2)\x1b\\", Invalid, 19}, // XTVERSION reply + {"\x1b_Gi=1;OK\x1b\\", Invalid, 11}, // Kitty graphics reply + {"\x1bP1$r\a\x1b\\", Invalid, 8}, // BEL is payload in a DCS + {"\x1b]11;x\x1bX\x1b\\", Invalid, 6}, // ESC ends it, framing "\e]11;x" + + // SS3 is framed like CSI, since the parser routes both the same way + {"\x1bO9;5u", Invalid, 6}, + {"\x1bO1;5X", Invalid, 6}, // unknown final byte + {"\x1bO1;5A", CtrlUp, 6}, // recognized, unchanged + {"\x1bOP", F1, 3}, + + // Left alone: this is how ALT-[, ALT-O, ALT-], ALT-P and ALT-_ arrive + {"\x1b[a", Alt, 2}, + {"\x1b[9A", Alt, 2}, // ALT-[ and two characters can look like a CSI + {"\x1b[1m", Alt, 2}, + {"\x1b[ x", Alt, 2}, + {"\x1bOx", Alt, 2}, + {"\x1bO9A", Alt, 2}, + + // rxvt keys fzf does not know: dropped, not typed + {"\x1b[3^", Invalid, 4}, // CTRL-DELETE + {"\x1b[5^", Invalid, 4}, // CTRL-PAGEUP + {"\x1b[3@", Invalid, 4}, // CTRL-SHIFT-DELETE + {"\x1b[11^", Invalid, 5}, + + // rxvt ends keys with $ after digits, so what follows is typed + {"\x1b[7$x", Home, 4}, // SHIFT-HOME, then x + {"\x1b[7$1x", Home, 4}, + {"\x1b[3$x", Invalid, 4}, + {"\x1b[3$1x", Invalid, 4}, + {"\x1b[23$x", Invalid, 5}, // SHIFT-F11, then x + {"\x1b[24$1x", Invalid, 5}, + + // A parse that gives up past the frame consumes only the frame, so the + // typed ~ after these stays in the buffer + {"\x1b[3;1u~", Invalid, 6}, + {"\x1b[5;1u~", Invalid, 6}, + {"\x1b[2$~", Invalid, 4}, // rxvt SHIFT-INSERT + {"\x1b[2^~", Invalid, 4}, // rxvt CTRL-INSERT + {"\x1b[2A~", Invalid, 4}, + + // Elsewhere $ and other intermediate bytes do not end a sequence + {"\x1b[4;0;10;20;0&w", Invalid, 15}, // not End, which \e[4 matches + {"\x1b[4;2$y", Invalid, 7}, + {"\x1b[2 q", Invalid, 5}, + {"\x1b]abc", Alt, 2}, + {"\x1b]11;rgb:", Alt, 2}, // terminator has not arrived + {"\x1b]\x1b]", Alt, 2}, // a second sequence must not swallow the ALT key + {"\x1b]\x1b[A", Alt, 2}, + {"\x1bP\x1bP", Alt, 2}, + {"\x1b_\x1b_", Alt, 2}, + {"\x1b]\x1b\\", Alt, 2}, // ALT-] then ALT-backslash, not an empty OSC + {"\x1b]\a", Alt, 2}, // ALT-] then CTRL-G, which must still abort + + // Typed text after the introducer is not a reply, even when a + // terminator follows in the same read + {"\x1b]a\a", Alt, 2}, + {"\x1b]1\a", Alt, 2}, + {"\x1b]a\x1b[A", Alt, 2}, + {"\x1bPx\x1b[A", Alt, 2}, + {"\x1b_ab\x1bb", Alt, 2}, + + // SOS and PM are not framed + {"\x1bXsos\x1b\\", Alt, 2}, + {"\x1b^status\x1b\\", Alt, 2}, + + // Left alone: no final byte yet, so the sequence may still be arriving + {"\x1b[", Invalid, 2}, + + // Recognized sequences keep their existing parsing + {"\x1b[1;5A", CtrlUp, 6}, + {"\x1b[3;5~", CtrlDelete, 6}, + {"\x1b[2~", Insert, 4}, + {"\x1b[200~", BracketedPasteBegin, 6}, + {"\x1b[Z", ShiftTab, 3}, + {"\x1bOA", Up, 3}, + {"\x1b[12;34R", Invalid, 8}, + {"\x1b[?2004;2$y", Invalid, 11}, + } { + r := &LightRenderer{buffer: []byte(c.sequence)} + sz := 1 + event := r.escSequence(&sz) + if event.Type != c.event { + t.Errorf("escSequence(%q) = %s, want %s", + c.sequence, event.Type.String(), c.event.String()) + } + if sz != c.size { + t.Errorf("escSequence(%q) consumed %d bytes, want %d", c.sequence, sz, c.size) + } + } +} + +func TestStringEnd(t *testing.T) { + for _, c := range []struct { + buffer string + want int + }{ + // Terminated + {"\x1b]0;t\a", 6}, + {"\x1b]0;t\x1b\\", 7}, + {"\x1b_G\x1b\\", 5}, + {"\x1bP1$r\a\x1b\\", 8}, // BEL is payload in a DCS, ST ends it + + // BEL terminates an OSC only + {"\x1bP\a", 0}, + {"\x1b_Gi=1\a", 0}, + + // An ESC ends the string and introduces a sequence of its own + {"\x1b]foo\x1bX\x1b\\", 5}, + {"\x1b]foo\x1bP", 5}, + {"\x1b]\x1b]", 2}, + + // Nothing has ended it yet + {"\x1b]0;t", 0}, + {"\x1b]0;t\x1b", 0}, // ST half arrived + {"\x1b]foo\x1b", 0}, + {"\x1b]", 0}, + {"\x1b", 0}, // shorter than an introducer + + } { + if got := stringEnd([]byte(c.buffer)); got != c.want { + t.Errorf("stringEnd(%q) = %d, want %d", c.buffer, got, c.want) + } + } +} + +// A dropped sequence followed by ALT-[ is not waiting for anything, so the +// bytes in between must not be held back until the next keystroke +func TestStillArriving(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1b[97;5u\a\x1b[", false}, + {"\x1b[97;5u\x1b[", false}, + {"\x1b[3;\a\x1b[", false}, // the BEL ended it, the trailing \e[ is another one + {"\x1b[1;\a", false}, // the BEL has already ended it + {"\x1b[1;", true}, + {"\x1b[", true}, + {"\x1bO1", true}, + {"\x1b[<0;1", true}, + } { + if got := stillArriving([]byte(c.buffer)); got != c.want { + t.Errorf("stillArriving(%q) = %v, want %v", c.buffer, got, c.want) + } + } +} + +func TestStringReply(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1b]11;rgb:1/2/3\a", true}, + {"\x1b]0;title\a", true}, + {"\x1b]52;c;YWJj\x1b\\", true}, + {"\x1bP>|kitty(0.48.2)\x1b\\", true}, + {"\x1bP!|0\x1b\\", true}, + {"\x1bP1$r0m\x1b\\", true}, + {"\x1bP0+r\x1b\\", true}, + {"\x1b_Gi=1;OK\x1b\\", true}, + + // What typed text after ALT-], ALT-P or ALT-_ looks like + {"\x1b]\a", false}, + {"\x1b]a\a", false}, + {"\x1b]1\a", false}, + {"\x1b];\a", false}, + {"\x1bPx\x1b\\", false}, + {"\x1bP1\x1b\\", false}, + {"\x1b_ab\x1b\\", false}, + {"\x1b_G\x1b\\", false}, + } { + if got := stringReply([]byte(c.buffer)); got != c.want { + t.Errorf("stringReply(%q) = %v, want %v", c.buffer, got, c.want) + } + } +} + +func TestCsiEnd(t *testing.T) { + for _, c := range []struct { + buffer string + want int + }{ + {"\x1b[97;5u", 7}, + {"\x1b[A", 3}, + {"\x1b[<0;1;1M", 9}, + {"\x1b[?2004;2$y", 11}, + {"\x1b[7$x", 4}, // rxvt ends keys with $ after digits + {"\x1b[23$", 5}, + {"\x1b[4;2$y", 7}, // but $ is intermediate elsewhere + {"\x1b[97;5", 0}, // no final byte + {"\x1b[", 0}, // no final byte + {"\x1b[1\x01A", -1}, // malformed, do not frame it + {"\x1b[\a", -1}, + } { + if got := csiEnd([]byte(c.buffer)); got != c.want { + t.Errorf("csiEnd(%q) = %d, want %d", c.buffer, got, c.want) + } + } +} diff --git a/src/tui/light_escape_test.go b/src/tui/light_escape_test.go index f83d3ac4..d405c043 100644 --- a/src/tui/light_escape_test.go +++ b/src/tui/light_escape_test.go @@ -51,3 +51,46 @@ func TestIncompleteEscape(t *testing.T) { } } } + +// A string sequence is not waited for. Its introducer is also ALT-], ALT-P or +// ALT-_, and holding the read open on those both delayed the key and let typed +// bytes accumulate into something that looked like a sequence. +func TestIncompleteStringEscape(t *testing.T) { + for _, buffer := range []string{ + "\x1b]11;rgb:", + "\x1bP>|kitty", + "\x1b_Gi=1", + "\x1b]0;t\x1b", + "\x1b]0;title\a", + "ab\x1b]11;rgb:", + } { + if incompleteEscape([]byte(buffer)) { + t.Errorf("incompleteEscape(%q) = true, want false", buffer) + } + } +} + +// SS3 reaches the same parameterized subcases as CSI in the parser, so it has to +// be judged the same way here. Declaring it finished after one byte made the +// parser give up on a split \eO1;5A and type its tail into the query. +func TestIncompleteSS3(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1bO", true}, + {"\x1bO1", true}, + {"\x1bO1;", true}, + {"\x1bO1;5", true}, + {"\x1bOA", false}, + {"\x1bOP", false}, + {"\x1bO1;5A", false}, + {"\x1b[23$", false}, // rxvt SHIFT-F11 is complete + {"\x1b[7$", false}, + {"\x1b[4;2$", true}, // $ after a ; is intermediate + } { + if got := incompleteEscape([]byte(c.buffer)); got != c.want { + t.Errorf("incompleteEscape(%q) = %v, want %v", c.buffer, got, c.want) + } + } +}