From d2a99bb929a7b05cb47c73fde309c383e1a52a2a Mon Sep 17 00:00:00 2001 From: Junegunn Choi Date: Mon, 5 Oct 2026 09:25:28 +0900 Subject: [PATCH] Drop unrecognized escape sequences instead of typing them fzf consumed only the part of a sequence it recognized, and the rest was typed into the query. CTRL-A sent as \e[97;5u became "97;5u", and CTRL-F sent as \e[70;5u fired Home before typing ";5u". A BEL-terminated OSC reply also aborted fzf, because the BEL that followed the typed payload was read as CTRL-G. Frame CSI and SS3 by their parameter and final byte ranges, OSC, DCS and APC by their terminator. Drop the whole frame when the parser does not recognize it, or recognizes only a prefix of it. Take the second-chance read only when the sequence at the start of the buffer is the only one and is unfinished. Otherwise it blocked until the next key, holding back what followed, CTRL-G included. Telling a terminal's sequence from an ALT key is the hard part, so these keep what the parser made of them: - An unterminated sequence, since that is how ALT-[, ALT-O, ALT-], ALT-P and ALT-_ arrive - A CSI or SS3 of four bytes or fewer. It could be an ALT key and typed text, and rxvt sends keys of that size fzf does not know, such as \e[3^ - A final byte after an intermediate $. rxvt ends keys with $, as in \e[7$ and \e[23$, so the byte after it is the next key - A string sequence whose payload does not start like a reply: digits and ';' for OSC, '>|', '!|', '[01]$r' or '[01]+r' for DCS, 'G' and a key for APC - SOS and PM, which nothing sends Within a string, an ESC ends it and introduces a sequence of its own, and only OSC ends with BEL. The read loop does not wait for a string terminator: that held ALT-], ALT-P and ALT-_ for ESCDELAY and let typed bytes pile up until they looked like a sequence. --- CHANGELOG.md | 2 + src/tui/light.go | 172 ++++++++++++++++++++++++++-- src/tui/light_csi_test.go | 214 +++++++++++++++++++++++++++++++++++ src/tui/light_escape_test.go | 40 +++++++ 4 files changed, 420 insertions(+), 8 deletions(-) create mode 100644 src/tui/light_csi_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index dbf7615d..9ff8ef75 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ CHANGELOG 0.74.5 ------ - Fixed `--gap-line` cutting a grapheme cluster when filling the last cells of the line (#4920) +- Fixed an escape sequence fzf does not recognize being partly consumed, which leaked the rest into the query (#4926) + - A reply ending in BEL also aborted fzf, because the BEL was read as CTRL-G 0.74.4 ------ diff --git a/src/tui/light.go b/src/tui/light.go index 4f5ec4d5..bf7aea98 100644 --- a/src/tui/light.go +++ b/src/tui/light.go @@ -348,9 +348,115 @@ func getEnv(name string, defaultValue int) int { func csiContinues(b byte) bool { return b >= 0x20 && b <= 0x3f } func csiFinal(b byte) bool { return b >= 0x40 && b <= 0x7e } +func csiIntermediate(b byte) bool { return b >= 0x20 && b <= 0x2f } + +// csiEnd returns the length of the CSI sequence at the start of the buffer, or +// 0 if it has no final byte yet or is malformed. +func csiEnd(buffer []byte) int { + for i := 2; i < len(buffer); i++ { + if csiFinal(buffer[i]) { + return i + 1 + } + if !csiContinues(buffer[i]) { + return 0 + } + } + return 0 +} + +// stringEnd returns the length of the string sequence (DCS, OSC or APC) at the +// start of the buffer, up to and including whatever ended it, or 0 while nothing +// has. +func stringEnd(buffer []byte) int { + if len(buffer) < 2 { + return 0 + } + // Every one of them ends with ST. BEL ends an OSC as well, because xterm has + // always allowed it, but stopping at a BEL inside a DCS or APC payload would + // frame only its first half and leave the rest to be typed into the query. + bel := buffer[1] == ']' + for i := 2; i < len(buffer); i++ { + switch buffer[i] { + case '\a': + if bel { + return i + 1 + } + case Esc.Byte(): + if i+1 == len(buffer) { + return 0 // ST may still be arriving + } + if buffer[i+1] == '\\' { + return i + 2 + } + // Any other ESC ends the string and introduces a sequence of its + // own, so frame only what precedes it and leave the ESC to be + // parsed again. Scanning past it would swallow that sequence too. + return i + } + } + return 0 +} + +// csiIntroducer reports whether the byte after ESC starts a sequence that is +// framed by a final byte. The parser routes CSI and SS3 through the same +// parameterized subcases, so both are framed by csiEnd. +func csiIntroducer(b byte) bool { + return b == '[' || b == 'O' +} + +// stringIntroducer reports whether the byte after ESC starts a string sequence. +// Only the three that terminals reply with: OSC for colors, title and clipboard, +// DCS for XTVERSION, DECRQSS and XTGETTCAP, APC for Kitty graphics. Nothing +// sends SOS or PM. +func stringIntroducer(b byte) bool { + switch b { + case 'P', ']', '_': + return true + } + return false +} + +// stringReply reports whether a framed string sequence starts the way a +// terminal's reply does. Its introducer is also ALT-], ALT-P or ALT-_, and typed +// text after one, ended by CTRL-G or another key, frames like a reply when it +// all arrives in one read. +func stringReply(buffer []byte) bool { + payload := buffer[2:] + switch buffer[1] { + case ']': // Ps ; ... + i := 0 + for i < len(payload) && payload[i] >= '0' && payload[i] <= '9' { + i++ + } + return i > 0 && i < len(payload) && payload[i] == ';' + case 'P': // >| or !| for the version, [01]$r or [01]+r for settings + if len(payload) < 2 { + return false + } + if payload[1] == '|' { + return payload[0] == '>' || payload[0] == '!' + } + return len(payload) >= 3 && (payload[0] == '0' || payload[0] == '1') && + (payload[1] == '$' || payload[1] == '+') && payload[2] == 'r' + case '_': // G, then key=value + return len(payload) >= 3 && payload[0] == 'G' && + (payload[1] >= 'a' && payload[1] <= 'z' || payload[1] >= 'A' && payload[1] <= 'Z') && + payload[2] == '=' + } + return false +} + +// stillArriving reports whether the sequence at the start of the buffer may +// still be completed by more input. It must be the only one in the buffer: +// whatever follows it, another sequence or a byte that ended it, means it is not +// waiting for anything. +func stillArriving(buffer []byte) bool { + return bytes.IndexByte(buffer[1:], Esc.Byte()) < 0 && incompleteEscape(buffer) +} + // incompleteEscape reports whether the buffer ends in an escape sequence that -// has not been terminated yet. The read loop keeps waiting in that case, so the -// parser is never handed a fragment to guess at. +// has not been terminated yet. The read loop keeps waiting while it does, so a +// fragment reaches the parser only once that wait has run out. func incompleteEscape(buffer []byte) bool { // Only the tail can hold a sequence still arriving. This runs once per byte // read, so scanning all of a large paste would make the read quadratic. @@ -362,8 +468,9 @@ func incompleteEscape(buffer []byte) bool { if start < 0 || len(tail)-start < 2 { return false } - switch tail[start+1] { - case '[': + // Declaring SS3 finished after one byte made the parser give up on a split + // \eO1;5A and type its tail into the query. + if csiIntroducer(tail[start+1]) { for _, b := range tail[start+2:] { if csiFinal(b) { return false @@ -373,8 +480,6 @@ func incompleteEscape(buffer []byte) bool { } } return true - case 'O': - return len(tail)-start < 3 } return false } @@ -483,8 +588,10 @@ func (r *LightRenderer) GetChar(cancellable bool) Event { return Event{CtrlSlash, 0, nil} case Esc.Byte(): ev := r.escSequence(&sz) - // Second chance - if ev.Type == Invalid { + // Second chance, but only for an unfinished sequence. Re-reading + // otherwise blocks until the next keystroke, holding back whatever + // follows in the buffer. + if ev.Type == Invalid && stillArriving(r.buffer) { r.buffer, result, err = r.getBytes(true) if err != nil { return Event{Fatal, 0, nil} @@ -525,7 +632,38 @@ func (r *LightRenderer) setCancel(f func()) { r.mutex.Unlock() } +// escSequence parses an escape sequence, then checks a CSI or SS3 result +// against its frame. parseEscSequence can stop after a few bytes, and consuming +// only those would leave the rest to be typed into the query. Complete +// sequences it does not recognize are dropped there, unless they could be an +// ALT key followed by typed text. func (r *LightRenderer) escSequence(sz *int) Event { + ev := r.parseEscSequence(sz) + if len(r.buffer) < 3 || !csiIntroducer(r.buffer[1]) { + return ev + } + // A frame missing its final byte may still be arriving + end := csiEnd(r.buffer) + if end == 0 { + return ev + } + // Four bytes or fewer could be ALT-[ or ALT-O and typed text, and rxvt sends + // keys of that size fzf does not know, such as \e[3^ for CTRL-DELETE + if end <= 4 { + return ev + } + // A key matched on a prefix, such as Home for \e[70;5u, or a sequence given + // up on partway. rxvt ends keys with the intermediate byte $, as in \e[7$ + // for SHIFT-HOME and \e[23$ for SHIFT-F11, so when one precedes the final + // byte, the final byte is the next key and is not widened over. + if end > *sz && !csiIntermediate(r.buffer[end-2]) { + *sz = end + return Event{Invalid, 0, nil} + } + return ev +} + +func (r *LightRenderer) parseEscSequence(sz *int) Event { if len(r.buffer) < 2 { return Event{Esc, 0, nil} } @@ -987,6 +1125,24 @@ func (r *LightRenderer) escSequence(sz *int) Event { } // r.buffer[2] } // r.buffer[2] } // r.buffer[1] + // Nothing matched. A framed sequence is dropped whole: reading its + // introducer as an ALT-key below would type the rest into the query. + // Unterminated ones are left alone, as that is how ALT-[, ALT-O, ALT-], + // ALT-P and ALT-_ arrive. + if csiIntroducer(r.buffer[1]) { + // ALT-[ or ALT-O and one or two typed characters can look like a short + // sequence, so ask for more than one parameter byte before dropping it + if end := csiEnd(r.buffer); end > 4 { + *sz = end + return Event{Invalid, 0, nil} + } + } else if stringIntroducer(r.buffer[1]) { + if end := stringEnd(r.buffer); end > 0 && stringReply(r.buffer) { + *sz = end + return Event{Invalid, 0, nil} + } + } + rest := bytes.NewBuffer(r.buffer[1:]) c, size, err := rest.ReadRune() if err == nil { diff --git a/src/tui/light_csi_test.go b/src/tui/light_csi_test.go new file mode 100644 index 00000000..ec23db3c --- /dev/null +++ b/src/tui/light_csi_test.go @@ -0,0 +1,214 @@ +package tui + +import "testing" + +// An unrecognized escape sequence must be consumed whole, CSI and string +// sequences alike. Consuming only part of one leaves the rest to be read as +// input and typed into the query. +func TestUnknownEscapeSequence(t *testing.T) { + for _, c := range []struct { + sequence string + event EventType + size int + }{ + // Key encodings fzf does not implement + {"\x1b[97;5u", Invalid, 7}, + {"\x1b[70;5u", Invalid, 7}, // not Home, which its prefix \e[7 matches + {"\x1b[42;5u", Invalid, 7}, // nor End + {"\x1b[4;5~", Invalid, 6}, + {"\x1b[127;5u", Invalid, 8}, + {"\x1b[27;5;127~", Invalid, 11}, + {"\x1b[57441;1u", Invalid, 10}, + {"\x1b\x1b[97;5u", Invalid, 7}, // ALT prefixed, the first ESC is dropped + + // Replies to queries fzf did not send, or sent and stopped waiting for + {"\x1b[?1;2c", Invalid, 7}, + {"\x1b[>0;95;0c", Invalid, 10}, + + // Mouse report arriving while mouse input is off + {"\x1b[<0;1;1M", Invalid, 9}, + + // String sequences + {"\x1b]11;rgb:4a4a/4a4a/4a4a\x1b\\", Invalid, 25}, // background color reply + {"\x1b]0;a title\a", Invalid, 12}, // BEL terminated + {"\x1bP>|kitty(0.48.2)\x1b\\", Invalid, 19}, // XTVERSION reply + {"\x1b_Gi=1;OK\x1b\\", Invalid, 11}, // Kitty graphics reply + {"\x1bP1$r\a\x1b\\", Invalid, 8}, // BEL is payload in a DCS + {"\x1b]11;x\x1bX\x1b\\", Invalid, 6}, // ESC ends it, framing "\e]11;x" + + // SS3 is framed like CSI, since the parser routes both the same way + {"\x1bO9;5u", Invalid, 6}, + {"\x1bO1;5X", Invalid, 6}, // unknown final byte + {"\x1bO1;5A", CtrlUp, 6}, // recognized, unchanged + {"\x1bOP", F1, 3}, + + // Left alone: this is how ALT-[, ALT-O, ALT-], ALT-P and ALT-_ arrive + {"\x1b[a", Alt, 2}, + {"\x1b[9A", Alt, 2}, // ALT-[ and two characters can look like a CSI + {"\x1b[1m", Alt, 2}, + {"\x1b[ x", Alt, 2}, + {"\x1bOx", Alt, 2}, + {"\x1bO9A", Alt, 2}, + + // rxvt keys fzf does not know: dropped, not typed + {"\x1b[3^", Invalid, 4}, // CTRL-DELETE + {"\x1b[5^", Invalid, 4}, // CTRL-PAGEUP + {"\x1b[3@", Invalid, 4}, // CTRL-SHIFT-DELETE + {"\x1b[11^", Invalid, 5}, + + // rxvt ends keys with $, so a key typed after one is not part of it + {"\x1b[7$x", Home, 4}, // SHIFT-HOME, then x + {"\x1b[3$x", Invalid, 4}, + {"\x1b[23$x", Invalid, 4}, // SHIFT-F11, then x + {"\x1b[24$x", Invalid, 4}, // SHIFT-F12, then x + {"\x1b]abc", Alt, 2}, + {"\x1b]11;rgb:", Alt, 2}, // terminator has not arrived + {"\x1b]\x1b]", Alt, 2}, // a second sequence must not swallow the ALT key + {"\x1b]\x1b[A", Alt, 2}, + {"\x1bP\x1bP", Alt, 2}, + {"\x1b_\x1b_", Alt, 2}, + {"\x1b]\x1b\\", Alt, 2}, // ALT-] then ALT-backslash, not an empty OSC + {"\x1b]\a", Alt, 2}, // ALT-] then CTRL-G, which must still abort + + // Typed text after the introducer is not a reply, even when a + // terminator follows in the same read + {"\x1b]a\a", Alt, 2}, + {"\x1b]1\a", Alt, 2}, + {"\x1b]a\x1b[A", Alt, 2}, + {"\x1bPx\x1b[A", Alt, 2}, + {"\x1b_ab\x1bb", Alt, 2}, + + // SOS and PM are not framed + {"\x1bXsos\x1b\\", Alt, 2}, + {"\x1b^status\x1b\\", Alt, 2}, + + // Left alone: no final byte yet, so the sequence may still be arriving + {"\x1b[", Invalid, 2}, + + // Recognized sequences keep their existing parsing + {"\x1b[1;5A", CtrlUp, 6}, + {"\x1b[3;5~", CtrlDelete, 6}, + {"\x1b[2~", Insert, 4}, + {"\x1b[200~", BracketedPasteBegin, 6}, + {"\x1b[Z", ShiftTab, 3}, + {"\x1bOA", Up, 3}, + {"\x1b[12;34R", Invalid, 8}, + {"\x1b[?2004;2$y", Invalid, 11}, + } { + r := &LightRenderer{buffer: []byte(c.sequence)} + sz := 1 + event := r.escSequence(&sz) + if event.Type != c.event { + t.Errorf("escSequence(%q) = %s, want %s", + c.sequence, event.Type.String(), c.event.String()) + } + if sz != c.size { + t.Errorf("escSequence(%q) consumed %d bytes, want %d", c.sequence, sz, c.size) + } + } +} + +func TestStringEnd(t *testing.T) { + for _, c := range []struct { + buffer string + want int + }{ + // Terminated + {"\x1b]0;t\a", 6}, + {"\x1b]0;t\x1b\\", 7}, + {"\x1b_G\x1b\\", 5}, + {"\x1bP1$r\a\x1b\\", 8}, // BEL is payload in a DCS, ST ends it + + // BEL terminates an OSC only + {"\x1bP\a", 0}, + {"\x1b_Gi=1\a", 0}, + + // An ESC ends the string and introduces a sequence of its own + {"\x1b]foo\x1bX\x1b\\", 5}, + {"\x1b]foo\x1bP", 5}, + {"\x1b]\x1b]", 2}, + + // Nothing has ended it yet + {"\x1b]0;t", 0}, + {"\x1b]0;t\x1b", 0}, // ST half arrived + {"\x1b]foo\x1b", 0}, + {"\x1b]", 0}, + {"\x1b", 0}, // shorter than an introducer + + } { + if got := stringEnd([]byte(c.buffer)); got != c.want { + t.Errorf("stringEnd(%q) = %d, want %d", c.buffer, got, c.want) + } + } +} + +// A dropped sequence followed by ALT-[ is not waiting for anything, so the +// bytes in between must not be held back until the next keystroke +func TestStillArriving(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1b[97;5u\a\x1b[", false}, + {"\x1b[97;5u\x1b[", false}, + {"\x1b[3;\a\x1b[", false}, // the BEL ended it, the trailing \e[ is another one + {"\x1b[1;\a", false}, // the BEL has already ended it + {"\x1b[1;", true}, + {"\x1b[", true}, + {"\x1bO1", true}, + {"\x1b[<0;1", true}, + } { + if got := stillArriving([]byte(c.buffer)); got != c.want { + t.Errorf("stillArriving(%q) = %v, want %v", c.buffer, got, c.want) + } + } +} + +func TestStringReply(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1b]11;rgb:1/2/3\a", true}, + {"\x1b]0;title\a", true}, + {"\x1b]52;c;YWJj\x1b\\", true}, + {"\x1bP>|kitty(0.48.2)\x1b\\", true}, + {"\x1bP!|0\x1b\\", true}, + {"\x1bP1$r0m\x1b\\", true}, + {"\x1bP0+r\x1b\\", true}, + {"\x1b_Gi=1;OK\x1b\\", true}, + + // What typed text after ALT-], ALT-P or ALT-_ looks like + {"\x1b]\a", false}, + {"\x1b]a\a", false}, + {"\x1b]1\a", false}, + {"\x1b];\a", false}, + {"\x1bPx\x1b\\", false}, + {"\x1bP1\x1b\\", false}, + {"\x1b_ab\x1b\\", false}, + {"\x1b_G\x1b\\", false}, + } { + if got := stringReply([]byte(c.buffer)); got != c.want { + t.Errorf("stringReply(%q) = %v, want %v", c.buffer, got, c.want) + } + } +} + +func TestCsiEnd(t *testing.T) { + for _, c := range []struct { + buffer string + want int + }{ + {"\x1b[97;5u", 7}, + {"\x1b[A", 3}, + {"\x1b[<0;1;1M", 9}, + {"\x1b[?2004;2$y", 11}, + {"\x1b[97;5", 0}, // no final byte + {"\x1b[", 0}, // no final byte + {"\x1b[1\x01A", 0}, // malformed, do not frame it + } { + if got := csiEnd([]byte(c.buffer)); got != c.want { + t.Errorf("csiEnd(%q) = %d, want %d", c.buffer, got, c.want) + } + } +} diff --git a/src/tui/light_escape_test.go b/src/tui/light_escape_test.go index f83d3ac4..db44f5e0 100644 --- a/src/tui/light_escape_test.go +++ b/src/tui/light_escape_test.go @@ -51,3 +51,43 @@ func TestIncompleteEscape(t *testing.T) { } } } + +// A string sequence is not waited for. Its introducer is also ALT-], ALT-P or +// ALT-_, and holding the read open on those both delayed the key and let typed +// bytes accumulate into something that looked like a sequence. +func TestIncompleteStringEscape(t *testing.T) { + for _, buffer := range []string{ + "\x1b]11;rgb:", + "\x1bP>|kitty", + "\x1b_Gi=1", + "\x1b]0;t\x1b", + "\x1b]0;title\a", + "ab\x1b]11;rgb:", + } { + if incompleteEscape([]byte(buffer)) { + t.Errorf("incompleteEscape(%q) = true, want false", buffer) + } + } +} + +// SS3 reaches the same parameterized subcases as CSI in the parser, so it has to +// be judged the same way here. Declaring it finished after one byte made the +// parser give up on a split \eO1;5A and type its tail into the query. +func TestIncompleteSS3(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1bO", true}, + {"\x1bO1", true}, + {"\x1bO1;", true}, + {"\x1bO1;5", true}, + {"\x1bOA", false}, + {"\x1bOP", false}, + {"\x1bO1;5A", false}, + } { + if got := incompleteEscape([]byte(c.buffer)); got != c.want { + t.Errorf("incompleteEscape(%q) = %v, want %v", c.buffer, got, c.want) + } + } +}