From 55a69985d15639659d9bef5e44132e1842f5a3b2 Mon Sep 17 00:00:00 2001 From: Junegunn Choi Date: Wed, 30 Sep 2026 21:31:27 +0900 Subject: [PATCH] Drop unrecognized escape sequences instead of typing them fzf consumed only the part of a sequence it recognized, and the rest was typed into the query. CTRL-A sent as \e[97;5u became "97;5u". A BEL-terminated OSC reply also aborted fzf, because the BEL that followed the typed payload was read as CTRL-G. Frame CSI and SS3 by their parameter and final byte ranges, OSC, DCS and APC by their terminator, then drop the whole sequence when nothing matches it. The second-chance read now runs only while a sequence is unfinished, since after a complete one it blocked until the next keystroke. Telling a terminal's sequence from an ALT key is the hard part, so several shapes are deliberately left alone: - An unterminated sequence, since that is how ALT-[, ALT-O, ALT-], ALT-P and ALT-_ arrive - A CSI or SS3 short enough to be an ALT key with one or two characters typed after it - A string sequence whose payload does not start like a reply: digits and ';' for OSC, '>|', '!|', '[01]$r' or '[01]+r' for DCS, 'G' and a key for APC. Otherwise typed text after ALT-], ALT-P or ALT-_, ended by CTRL-G or another key in the same read, framed like a reply and lost the abort - SOS and PM, which nothing sends Within a string, an ESC ends it and introduces a sequence of its own, so scanning on to a later ST would swallow that one. Only OSC ends with BEL. DCS and APC end with ST, and stopping at a BEL in their payload framed just its first half. The read loop does not wait for a string terminator. Waiting held ALT-], ALT-P and ALT-_ for ESCDELAY and let typed bytes pile into the buffer until they looked like a sequence. --- CHANGELOG.md | 2 + src/tui/light.go | 148 +++++++++++++++++++++++++++-- src/tui/light_csi_test.go | 177 +++++++++++++++++++++++++++++++++++ src/tui/light_escape_test.go | 40 ++++++++ 4 files changed, 359 insertions(+), 8 deletions(-) create mode 100644 src/tui/light_csi_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index dbf7615d..9ff8ef75 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ CHANGELOG 0.74.5 ------ - Fixed `--gap-line` cutting a grapheme cluster when filling the last cells of the line (#4920) +- Fixed an escape sequence fzf does not recognize being partly consumed, which leaked the rest into the query (#4926) + - A reply ending in BEL also aborted fzf, because the BEL was read as CTRL-G 0.74.4 ------ diff --git a/src/tui/light.go b/src/tui/light.go index 4f5ec4d5..bd1a4399 100644 --- a/src/tui/light.go +++ b/src/tui/light.go @@ -348,9 +348,105 @@ func getEnv(name string, defaultValue int) int { func csiContinues(b byte) bool { return b >= 0x20 && b <= 0x3f } func csiFinal(b byte) bool { return b >= 0x40 && b <= 0x7e } +// csiEnd returns the length of the CSI sequence at the start of the buffer, or +// 0 if it has no final byte yet or is malformed. +func csiEnd(buffer []byte) int { + for i := 2; i < len(buffer); i++ { + if csiFinal(buffer[i]) { + return i + 1 + } + if !csiContinues(buffer[i]) { + return 0 + } + } + return 0 +} + +// stringEnd returns the length of the string sequence (DCS, OSC or APC) at the +// start of the buffer, up to and including whatever ended it, or 0 while nothing +// has. +func stringEnd(buffer []byte) int { + if len(buffer) < 2 { + return 0 + } + // Every one of them ends with ST. BEL ends an OSC as well, because xterm has + // always allowed it, but stopping at a BEL inside a DCS or APC payload would + // frame only its first half and leave the rest to be typed into the query. + bel := buffer[1] == ']' + for i := 2; i < len(buffer); i++ { + switch buffer[i] { + case '\a': + if bel { + return i + 1 + } + case Esc.Byte(): + if i+1 == len(buffer) { + return 0 // ST may still be arriving + } + if buffer[i+1] == '\\' { + return i + 2 + } + // Any other ESC ends the string and introduces a sequence of its + // own, so frame only what precedes it and leave the ESC to be + // parsed again. Scanning past it would swallow that sequence too. + return i + } + } + return 0 +} + +// csiIntroducer reports whether the byte after ESC starts a sequence that is +// framed by a final byte. The parser routes CSI and SS3 through the same +// parameterized subcases, so both are framed by csiEnd. +func csiIntroducer(b byte) bool { + return b == '[' || b == 'O' +} + +// stringIntroducer reports whether the byte after ESC starts a string sequence. +// Only the three that terminals reply with: OSC for colors, title and clipboard, +// DCS for XTVERSION, DECRQSS and XTGETTCAP, APC for Kitty graphics. Nothing +// sends SOS or PM. +func stringIntroducer(b byte) bool { + switch b { + case 'P', ']', '_': + return true + } + return false +} + +// stringReply reports whether a framed string sequence starts the way a +// terminal's reply does. Its introducer is also ALT-], ALT-P or ALT-_, and typed +// text after one, ended by CTRL-G or another key, frames like a reply when it +// all arrives in one read. +func stringReply(buffer []byte) bool { + payload := buffer[2:] + switch buffer[1] { + case ']': // Ps ; ... + i := 0 + for i < len(payload) && payload[i] >= '0' && payload[i] <= '9' { + i++ + } + return i > 0 && i < len(payload) && payload[i] == ';' + case 'P': // >| or !| for the version, [01]$r or [01]+r for settings + if len(payload) < 2 { + return false + } + if payload[1] == '|' { + return payload[0] == '>' || payload[0] == '!' + } + return len(payload) >= 3 && (payload[0] == '0' || payload[0] == '1') && + (payload[1] == '$' || payload[1] == '+') && payload[2] == 'r' + case '_': // G, then key=value + return len(payload) >= 3 && payload[0] == 'G' && + (payload[1] >= 'a' && payload[1] <= 'z' || payload[1] >= 'A' && payload[1] <= 'Z') && + payload[2] == '=' + } + return false +} + // incompleteEscape reports whether the buffer ends in an escape sequence that -// has not been terminated yet. The read loop keeps waiting in that case, so the -// parser is never handed a fragment to guess at. +// has not been terminated yet. The read loop keeps waiting while it does, so a +// fragment reaches the parser only once that wait has run out. func incompleteEscape(buffer []byte) bool { // Only the tail can hold a sequence still arriving. This runs once per byte // read, so scanning all of a large paste would make the read quadratic. @@ -362,8 +458,9 @@ func incompleteEscape(buffer []byte) bool { if start < 0 || len(tail)-start < 2 { return false } - switch tail[start+1] { - case '[': + // Declaring SS3 finished after one byte made the parser give up on a split + // \eO1;5A and type its tail into the query. + if csiIntroducer(tail[start+1]) { for _, b := range tail[start+2:] { if csiFinal(b) { return false @@ -373,8 +470,6 @@ func incompleteEscape(buffer []byte) bool { } } return true - case 'O': - return len(tail)-start < 3 } return false } @@ -483,8 +578,10 @@ func (r *LightRenderer) GetChar(cancellable bool) Event { return Event{CtrlSlash, 0, nil} case Esc.Byte(): ev := r.escSequence(&sz) - // Second chance - if ev.Type == Invalid { + // Second chance, but only while the buffer ends in an unfinished + // sequence. Re-reading otherwise blocks until the next keystroke, + // holding back whatever follows in the buffer. + if ev.Type == Invalid && incompleteEscape(r.buffer) { r.buffer, result, err = r.getBytes(true) if err != nil { return Event{Fatal, 0, nil} @@ -525,7 +622,24 @@ func (r *LightRenderer) setCancel(f func()) { r.mutex.Unlock() } +// escSequence parses an escape sequence, widening a CSI or SS3 sequence that +// parseEscSequence recognized the start of but gave up on partway. Consuming +// only the part that parsed would leave the rest to be read as input and typed +// into the query. Complete sequences it does not recognize are dropped there. func (r *LightRenderer) escSequence(sz *int) Event { + ev := r.parseEscSequence(sz) + if ev.Type != Invalid || len(r.buffer) < 3 || !csiIntroducer(r.buffer[1]) { + return ev + } + // Only a framed sequence is dropped. One still missing its final byte may + // yet be arriving, and the caller gives it another chance. + if end := csiEnd(r.buffer); end > *sz { + *sz = end + } + return ev +} + +func (r *LightRenderer) parseEscSequence(sz *int) Event { if len(r.buffer) < 2 { return Event{Esc, 0, nil} } @@ -987,6 +1101,24 @@ func (r *LightRenderer) escSequence(sz *int) Event { } // r.buffer[2] } // r.buffer[2] } // r.buffer[1] + // Nothing matched. A framed sequence is dropped whole: reading its + // introducer as an ALT-key below would type the rest into the query. + // Unterminated ones are left alone, as that is how ALT-[, ALT-O, ALT-], + // ALT-P and ALT-_ arrive. + if csiIntroducer(r.buffer[1]) { + // ALT-[ or ALT-O and one or two typed characters can look like a short + // sequence, so ask for more than one parameter byte before dropping it + if end := csiEnd(r.buffer); end > 4 { + *sz = end + return Event{Invalid, 0, nil} + } + } else if stringIntroducer(r.buffer[1]) { + if end := stringEnd(r.buffer); end > 0 && stringReply(r.buffer) { + *sz = end + return Event{Invalid, 0, nil} + } + } + rest := bytes.NewBuffer(r.buffer[1:]) c, size, err := rest.ReadRune() if err == nil { diff --git a/src/tui/light_csi_test.go b/src/tui/light_csi_test.go new file mode 100644 index 00000000..b702bb29 --- /dev/null +++ b/src/tui/light_csi_test.go @@ -0,0 +1,177 @@ +package tui + +import "testing" + +// An unrecognized escape sequence must be consumed whole, CSI and string +// sequences alike. Consuming only part of one leaves the rest to be read as +// input and typed into the query. +func TestUnknownEscapeSequence(t *testing.T) { + for _, c := range []struct { + sequence string + event EventType + size int + }{ + // Key encodings fzf does not implement + {"\x1b[97;5u", Invalid, 7}, + {"\x1b[127;5u", Invalid, 8}, + {"\x1b[27;5;127~", Invalid, 11}, + {"\x1b[57441;1u", Invalid, 10}, + {"\x1b\x1b[97;5u", Invalid, 7}, // ALT prefixed, the first ESC is dropped + + // Replies to queries fzf did not send, or sent and stopped waiting for + {"\x1b[?1;2c", Invalid, 7}, + {"\x1b[>0;95;0c", Invalid, 10}, + + // Mouse report arriving while mouse input is off + {"\x1b[<0;1;1M", Invalid, 9}, + + // String sequences + {"\x1b]11;rgb:4a4a/4a4a/4a4a\x1b\\", Invalid, 25}, // background color reply + {"\x1b]0;a title\a", Invalid, 12}, // BEL terminated + {"\x1bP>|kitty(0.48.2)\x1b\\", Invalid, 19}, // XTVERSION reply + {"\x1b_Gi=1;OK\x1b\\", Invalid, 11}, // Kitty graphics reply + {"\x1bP1$r\a\x1b\\", Invalid, 8}, // BEL is payload in a DCS + {"\x1b]11;x\x1bX\x1b\\", Invalid, 6}, // ESC ends it, framing "\e]11;x" + + // SS3 is framed like CSI, since the parser routes both the same way + {"\x1bO9;5u", Invalid, 6}, + {"\x1bO1;5X", Invalid, 6}, // unknown final byte + {"\x1bO1;5A", CtrlUp, 6}, // recognized, unchanged + {"\x1bOP", F1, 3}, + + // Left alone: this is how ALT-[, ALT-O, ALT-], ALT-P and ALT-_ arrive + {"\x1b[a", Alt, 2}, + {"\x1b[9A", Alt, 2}, // ALT-[ and two characters can look like a CSI + {"\x1b[1m", Alt, 2}, + {"\x1b[ x", Alt, 2}, + {"\x1bOx", Alt, 2}, + {"\x1bO9A", Alt, 2}, + {"\x1b]abc", Alt, 2}, + {"\x1b]11;rgb:", Alt, 2}, // terminator has not arrived + {"\x1b]\x1b]", Alt, 2}, // a second sequence must not swallow the ALT key + {"\x1b]\x1b[A", Alt, 2}, + {"\x1bP\x1bP", Alt, 2}, + {"\x1b_\x1b_", Alt, 2}, + {"\x1b]\x1b\\", Alt, 2}, // ALT-] then ALT-backslash, not an empty OSC + {"\x1b]\a", Alt, 2}, // ALT-] then CTRL-G, which must still abort + + // Typed text after the introducer is not a reply, even when a + // terminator follows in the same read + {"\x1b]a\a", Alt, 2}, + {"\x1b]1\a", Alt, 2}, + {"\x1b]a\x1b[A", Alt, 2}, + {"\x1bPx\x1b[A", Alt, 2}, + {"\x1b_ab\x1bb", Alt, 2}, + + // SOS and PM are not framed + {"\x1bXsos\x1b\\", Alt, 2}, + {"\x1b^status\x1b\\", Alt, 2}, + + // Left alone: no final byte yet, so the sequence may still be arriving + {"\x1b[", Invalid, 2}, + + // Recognized sequences keep their existing parsing + {"\x1b[1;5A", CtrlUp, 6}, + {"\x1b[3;5~", CtrlDelete, 6}, + {"\x1b[2~", Insert, 4}, + {"\x1b[200~", BracketedPasteBegin, 6}, + {"\x1b[Z", ShiftTab, 3}, + {"\x1bOA", Up, 3}, + {"\x1b[12;34R", Invalid, 8}, + {"\x1b[?2004;2$y", Invalid, 11}, + } { + r := &LightRenderer{buffer: []byte(c.sequence)} + sz := 1 + event := r.escSequence(&sz) + if event.Type != c.event { + t.Errorf("escSequence(%q) = %s, want %s", + c.sequence, event.Type.String(), c.event.String()) + } + if sz != c.size { + t.Errorf("escSequence(%q) consumed %d bytes, want %d", c.sequence, sz, c.size) + } + } +} + +func TestStringEnd(t *testing.T) { + for _, c := range []struct { + buffer string + want int + }{ + // Terminated + {"\x1b]0;t\a", 6}, + {"\x1b]0;t\x1b\\", 7}, + {"\x1b_G\x1b\\", 5}, + {"\x1bP1$r\a\x1b\\", 8}, // BEL is payload in a DCS, ST ends it + + // BEL terminates an OSC only + {"\x1bP\a", 0}, + {"\x1b_Gi=1\a", 0}, + + // An ESC ends the string and introduces a sequence of its own + {"\x1b]foo\x1bX\x1b\\", 5}, + {"\x1b]foo\x1bP", 5}, + {"\x1b]\x1b]", 2}, + + // Nothing has ended it yet + {"\x1b]0;t", 0}, + {"\x1b]0;t\x1b", 0}, // ST half arrived + {"\x1b]foo\x1b", 0}, + {"\x1b]", 0}, + {"\x1b", 0}, // shorter than an introducer + + } { + if got := stringEnd([]byte(c.buffer)); got != c.want { + t.Errorf("stringEnd(%q) = %d, want %d", c.buffer, got, c.want) + } + } +} + +func TestStringReply(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1b]11;rgb:1/2/3\a", true}, + {"\x1b]0;title\a", true}, + {"\x1b]52;c;YWJj\x1b\\", true}, + {"\x1bP>|kitty(0.48.2)\x1b\\", true}, + {"\x1bP!|0\x1b\\", true}, + {"\x1bP1$r0m\x1b\\", true}, + {"\x1bP0+r\x1b\\", true}, + {"\x1b_Gi=1;OK\x1b\\", true}, + + // What typed text after ALT-], ALT-P or ALT-_ looks like + {"\x1b]\a", false}, + {"\x1b]a\a", false}, + {"\x1b]1\a", false}, + {"\x1b];\a", false}, + {"\x1bPx\x1b\\", false}, + {"\x1bP1\x1b\\", false}, + {"\x1b_ab\x1b\\", false}, + {"\x1b_G\x1b\\", false}, + } { + if got := stringReply([]byte(c.buffer)); got != c.want { + t.Errorf("stringReply(%q) = %v, want %v", c.buffer, got, c.want) + } + } +} + +func TestCsiEnd(t *testing.T) { + for _, c := range []struct { + buffer string + want int + }{ + {"\x1b[97;5u", 7}, + {"\x1b[A", 3}, + {"\x1b[<0;1;1M", 9}, + {"\x1b[?2004;2$y", 11}, + {"\x1b[97;5", 0}, // no final byte + {"\x1b[", 0}, // no final byte + {"\x1b[1\x01A", 0}, // malformed, do not frame it + } { + if got := csiEnd([]byte(c.buffer)); got != c.want { + t.Errorf("csiEnd(%q) = %d, want %d", c.buffer, got, c.want) + } + } +} diff --git a/src/tui/light_escape_test.go b/src/tui/light_escape_test.go index f83d3ac4..db44f5e0 100644 --- a/src/tui/light_escape_test.go +++ b/src/tui/light_escape_test.go @@ -51,3 +51,43 @@ func TestIncompleteEscape(t *testing.T) { } } } + +// A string sequence is not waited for. Its introducer is also ALT-], ALT-P or +// ALT-_, and holding the read open on those both delayed the key and let typed +// bytes accumulate into something that looked like a sequence. +func TestIncompleteStringEscape(t *testing.T) { + for _, buffer := range []string{ + "\x1b]11;rgb:", + "\x1bP>|kitty", + "\x1b_Gi=1", + "\x1b]0;t\x1b", + "\x1b]0;title\a", + "ab\x1b]11;rgb:", + } { + if incompleteEscape([]byte(buffer)) { + t.Errorf("incompleteEscape(%q) = true, want false", buffer) + } + } +} + +// SS3 reaches the same parameterized subcases as CSI in the parser, so it has to +// be judged the same way here. Declaring it finished after one byte made the +// parser give up on a split \eO1;5A and type its tail into the query. +func TestIncompleteSS3(t *testing.T) { + for _, c := range []struct { + buffer string + want bool + }{ + {"\x1bO", true}, + {"\x1bO1", true}, + {"\x1bO1;", true}, + {"\x1bO1;5", true}, + {"\x1bOA", false}, + {"\x1bOP", false}, + {"\x1bO1;5A", false}, + } { + if got := incompleteEscape([]byte(c.buffer)); got != c.want { + t.Errorf("incompleteEscape(%q) = %v, want %v", c.buffer, got, c.want) + } + } +}