diff --git a/crates/neomacs-tui-tests/tests/issue_470.rs b/crates/neomacs-tui-tests/tests/issue_470.rs new file mode 100644 index 0000000000..2a53a17eb1 --- /dev/null +++ b/crates/neomacs-tui-tests/tests/issue_470.rs @@ -0,0 +1,77 @@ +#![cfg(unix)] +//! Issue #470: a Unicode dot in the mode line must not degrade to a literal +//! `\267` escape after initially rendering correctly. +//! +//! GNU displays a raw eight-bit character as `\` plus its octal byte +//! (src/xdisp.c:8649-8662, `%03o` of CHAR_TO_BYTE8), so a literal `\267` on +//! screen is the signature of the mode-line text having been re-derived with +//! the multibyte U+00B7 turned into a raw byte. The regression drives real +//! PTYs: boot a GNU/neomacs pair, install a mode line containing both a +//! literal U+00B7 and `(string 183)`, then force repeated mode-line +//! re-evaluations and require the dot to stay a dot on both sides across +//! every redisplay. + +use crate::support::{ + boot_pair, eval_expression, settle_session, use_backend_only_vc_mode_line, + use_deterministic_emacs_version, +}; +use neomacs_tui_tests::TuiSession; +use std::time::Duration; + +const INSTALL_DOT_MODE_LINE: &str = + "(setq mode-line-format (list \"A\" \"·\" \"B\" (string 183) \"C\"))"; + +fn boot_pair_plain() -> (TuiSession, TuiSession) { + let (mut gnu, mut neo) = boot_pair(""); + settle_session(&mut gnu); + settle_session(&mut neo); + // Strip the VC segment so both mode lines are stable text. + use_backend_only_vc_mode_line(&mut gnu, &mut neo); + use_deterministic_emacs_version(&mut gnu, &mut neo); + (gnu, neo) +} + +/// Install the dot mode line and force N full mode-line re-evaluations with +/// redisplays on both sessions. +fn install_and_force_updates(gnu: &mut TuiSession, neo: &mut TuiSession, rounds: usize) { + eval_expression(gnu, neo, INSTALL_DOT_MODE_LINE); + for _ in 0..rounds { + eval_expression( + gnu, + neo, + "(progn (force-mode-line-update t) (redisplay) (sit-for 0))", + ); + std::thread::sleep(Duration::from_millis(100)); + } +} + +#[test] +fn mode_line_unicode_dot_stays_a_dot_across_forced_updates() { + let (mut gnu, mut neo) = boot_pair_plain(); + install_and_force_updates(&mut gnu, &mut neo, 5); + + // The reporter saw a CORRECT first frame and a later escaped one, so the + // assertion must hold after arbitrarily many re-evaluations, not only on + // the first render. Probed on real GNU 31.1 (PTY, LC_ALL=C.UTF-8): the + // installed mode line renders `A·B·C` -- BOTH the literal "·" and the + // `(string 183)` element show the dot glyph, because GNU's `concat` makes + // the integer character 183 a MULTIBYTE string holding U+00B7 (princ + // emits c2 b7 for both), never a raw byte. A `\267` on screen is + // therefore the raw-byte degradation signature on either element. + let gnu_screen = gnu.screen().contents(); + let neo_screen = neo.screen().contents(); + assert!( + gnu_screen.contains('·'), + "GNU lost the dot after forced updates:\n{gnu_screen}" + ); + assert!( + neo_screen.contains('·'), + "neomacs degraded the dot after forced updates:\n{neo_screen}" + ); + // Exact parity: neomacs must match GNU's `A·B·C` -- neither element may + // degrade to its octal escape. + assert_eq!( + gnu_screen, neo_screen, + "mode lines must render identically after forced updates" + ); +} diff --git a/crates/neomacs-tui-tests/tests/tui.rs b/crates/neomacs-tui-tests/tests/tui.rs index 32b27d77c4..649681c804 100644 --- a/crates/neomacs-tui-tests/tests/tui.rs +++ b/crates/neomacs-tui-tests/tests/tui.rs @@ -61,6 +61,8 @@ mod issue_445; mod issue_445_ibuffer_filter_groups; #[path = "issue_446_align_to_hscroll.rs"] mod issue_446_align_to_hscroll; +#[path = "issue_470.rs"] +mod issue_470; #[path = "mark_region_fill.rs"] mod mark_region_fill; #[path = "menu_bar.rs"] diff --git a/crates/neovm-core/src/emacs_core/display/xdisp/mod.rs b/crates/neovm-core/src/emacs_core/display/xdisp/mod.rs index 99937e333c..dcd75af539 100644 --- a/crates/neovm-core/src/emacs_core/display/xdisp/mod.rs +++ b/crates/neovm-core/src/emacs_core/display/xdisp/mod.rs @@ -2958,6 +2958,12 @@ struct ModeLineRendered { /// them, and the legacy storage-String round-trip that used to bridge the /// gap has been retired (issue #131). text: Vec, + /// GNU string identity: multibyte iff any FORMAT INPUT was multibyte (the + /// `concat' rule), never derived from the accumulated content. Deriving + /// it from the codes instead re-encoded every U+0080..U+00FF character + /// whose code fits in a byte as a unibyte raw byte, which then displays + /// as the octal escape `\NNN` -- issue #470's "dot turns into `\267`". + multibyte: bool, text_props: TextPropertyTable, source_spans: Vec, min_width_transitions: Vec, @@ -2984,7 +2990,11 @@ struct ModeLineMinWidthTransition { /// Decode a `LispString` into its sequence of Emacs character codes. Multibyte /// strings are scanned one Emacs character at a time (eight-bit characters -/// surface as `0x3FFF00+`); unibyte strings yield one code per raw byte. +/// surface as `0x3FFF00+`); a unibyte string's high byte IS a raw byte, so it +/// surfaces as its byte8 character (`0x3FFF00+`), never as a plain Unicode +/// code -- GNU's `BYTE8_TO_CHAR` (src/character.h). Byte identity then +/// survives any later multibyte promotion, exactly as GNU's `concat` keeps a +/// unibyte operand's raw bytes raw (`str_to_multibyte`). fn mode_line_string_char_codes(string: &crate::heap_types::LispString) -> Vec { let bytes = string.as_bytes(); if string.is_multibyte() { @@ -2997,13 +3007,18 @@ fn mode_line_string_char_codes(string: &crate::heap_types::LispString) -> Vec) -> Self { + let text = text.into(); + let multibyte = text.chars().any(|c| !c.is_ascii()); Self { - text: text.into().chars().map(|c| c as u32).collect(), + text: text.chars().map(|c| c as u32).collect(), + multibyte, text_props: TextPropertyTable::new(), source_spans: Vec::new(), min_width_transitions: Vec::new(), @@ -3081,6 +3099,7 @@ impl ModeLineRendered { fn append_rendered(&mut self, other: &Self) { let char_offset = self.char_len(); + self.multibyte |= other.multibyte; self.text.extend_from_slice(&other.text); self.append_properties(&other.text_props, char_offset); for span in &other.source_spans { @@ -3147,6 +3166,7 @@ impl ModeLineRendered { match value.as_lisp_string() { Some(string) => { let char_offset = self.char_len(); + self.multibyte |= string.is_multibyte(); self.text.extend(mode_line_string_char_codes(string)); self.record_source_span(char_offset, self.char_len(), *value, 0); if let Some(props) = get_string_text_properties_table_for_value(*value) { @@ -3158,6 +3178,7 @@ impl ModeLineRendered { return; }; let char_offset = self.char_len(); + self.multibyte |= text.chars().any(|c| !c.is_ascii()); self.text.extend(text.chars().map(|c| c as u32)); self.record_source_span(char_offset, self.char_len(), *value, 0); } @@ -3168,6 +3189,7 @@ impl ModeLineRendered { if value.is_string() { self.append_string_value_preserving_props(value); } else if let Some(ch) = value.as_char() { + self.multibyte |= !ch.is_ascii(); self.text.push(ch as u32); } } @@ -3184,6 +3206,10 @@ impl ModeLineRendered { match value.as_lisp_string() { Some(string) => { let char_offset = self.char_len(); + // GNU `substring' semantics: the slice keeps the SOURCE + // string's multibyte flag even when the taken range is pure + // ASCII. + self.multibyte |= string.is_multibyte(); self.text.extend( mode_line_string_char_codes(string) .into_iter() @@ -3203,6 +3229,7 @@ impl ModeLineRendered { return; }; let char_offset = self.char_len(); + self.multibyte |= text.chars().any(|c| !c.is_ascii()); self.text.extend( text.chars() .skip(start_char) @@ -3223,6 +3250,7 @@ impl ModeLineRendered { } fn push_plain_char(&mut self, ch: char) { + self.multibyte |= !ch.is_ascii(); self.text.push(ch as u32); } @@ -3233,6 +3261,9 @@ impl ModeLineRendered { fn slice_chars(&self, precision: usize) -> Self { Self { gc_roots: None, + // A truncation, not a re-derivation: keep the source identity, + // like GNU `substring'. + multibyte: self.multibyte, text: self.text.iter().take(precision).copied().collect(), text_props: self .text_props @@ -3480,10 +3511,14 @@ impl ModeLineRendered { fn into_display_output(mut self, face_spec: ModeLineFaceSpec) -> ModeLineDisplayOutput { self.realize_display_min_width_transitions(); - // A multibyte result iff any accumulated character exceeds a single - // byte; otherwise every code fits in a unibyte byte. Mirrors the old - // storage path's `decode_storage_char_codes_auto(..).any(> 0xFF)`. - let multibyte = self.text.iter().any(|&code| code > 0xFF); + // GNU string identity follows the format INPUTS (the `concat' rule: + // multibyte iff any argument is multibyte), never the content. The + // old content heuristic (`any(code > 0xFF)`) re-encoded every + // U+0080..U+00FF character -- whose code fits in one byte -- as a + // unibyte raw byte, and a raw byte displays as the octal escape + // `\NNN' (src/xdisp.c:8649-8662): issue #470's "dot turns into + // `\267`" between mode-line re-evaluations. + let multibyte = self.multibyte; if face_spec.no_props { return ModeLineDisplayOutput { value: Value::heap_string(mode_line_lisp_string_from_codes(&self.text, multibyte)), @@ -9389,6 +9424,10 @@ mod tests; #[path = "tests/mode_line_gc_roots.rs"] mod mode_line_gc_roots; +#[cfg(test)] +#[path = "tests/mode_line_multibyte_identity.rs"] +mod mode_line_multibyte_identity; + #[cfg(test)] #[path = "tests/mode_line_incremental_roots.rs"] mod mode_line_incremental_roots; diff --git a/crates/neovm-core/src/emacs_core/display/xdisp/tests/mode_line_multibyte_identity.rs b/crates/neovm-core/src/emacs_core/display/xdisp/tests/mode_line_multibyte_identity.rs new file mode 100644 index 0000000000..2af836ebc2 --- /dev/null +++ b/crates/neovm-core/src/emacs_core/display/xdisp/tests/mode_line_multibyte_identity.rs @@ -0,0 +1,127 @@ +//! Issue #470: the mode-line result's multibyte identity must come from the +//! FORMAT INPUTS, never from whether the accumulated characters happen to fit +//! in one byte. +//! +//! `into_display_output' used to derive `multibyte` as "any code > 0xFF", so +//! a mode line whose only non-ASCII character is a Latin-1-supplement +//! character (U+0080..U+00FF, e.g. U+00B7 MIDDLE DOT) was re-encoded as a +//! UNIBYTE string of raw bytes -- and a raw byte displays as GNU's octal +//! escape (`\267`, src/xdisp.c:8649-8662 `"%03o"` of CHAR_TO_BYTE8). That is +//! exactly the reported "dot renders, then turns into a literal `\267`". +//! GNU never re-encodes: a string's multibyte flag follows its inputs +//! (the `concat' rule -- multibyte iff any argument is multibyte), so U+00B7 +//! stays U+00B7 across every mode-line re-evaluation. + +use super::*; +use crate::emacs_core::Context; + +fn interactive() -> Context { + let mut eval = Context::new(); + eval.set_variable("noninteractive", Value::NIL); + eval +} + +fn render(eval: &mut Context, elements: Vec) -> ModeLineDisplayOutput { + let buffer_id = eval.buffers.current_buffer().expect("current buffer").id; + let frame_id = eval + .frames + .create_frame("mode-line-multibyte-identity", 800, 600, buffer_id); + let window_id = eval.frames.get(frame_id).expect("frame").selected_window; + format_mode_line_for_display_with_sources( + eval, + Value::list(elements), + Value::make_window(window_id.0), + Value::make_buffer(buffer_id), + 80, + ) +} + +/// `(multibyte, bytes)` of the rendered output string. +fn output_identity(output: &ModeLineDisplayOutput) -> (bool, Vec) { + let value = output.value(); + let string = value + .as_lisp_string() + .expect("mode-line output is a string"); + (string.is_multibyte(), string.as_bytes().to_vec()) +} + +#[test] +fn mode_line_middle_dot_stays_multibyte_like_its_input() { + crate::test_utils::init_test_tracing(); + let mut eval = interactive(); + let output = render( + &mut eval, + vec![Value::string("A"), Value::string("·"), Value::string("B")], + ); + let (multibyte, bytes) = output_identity(&output); + assert!( + multibyte, + "a mode line built from multibyte inputs must stay multibyte; \ + got unibyte bytes {bytes:?} -- U+00B7 degrades to a raw byte and \ + displays as \\267 (issue #470)" + ); + assert_eq!(bytes, b"A\xC2\xB7B".to_vec()); +} + +#[test] +fn mode_line_latin1_supplement_band_is_multibyte_not_raw_bytes() { + // The whole degrading band U+0080..U+00FF: every character in it has a + // code that fits in one byte, which is exactly what the old content + // heuristic got wrong. + crate::test_utils::init_test_tracing(); + let mut eval = interactive(); + let band: String = (0x80u32..=0xFF).flat_map(char::from_u32).collect(); + let output = render(&mut eval, vec![Value::string(&band)]); + let (multibyte, bytes) = output_identity(&output); + assert!(multibyte); + assert_eq!(bytes, band.as_bytes().to_vec()); +} + +#[test] +fn mode_line_re_derivation_keeps_the_multibyte_identity() { + // The issue's observed transition: a later mode-line re-evaluation must + // not change the identity of the first result. Feed the rendered output + // back through the walker and require byte-identical, still-multibyte + // output. + crate::test_utils::init_test_tracing(); + let mut eval = interactive(); + let first = render( + &mut eval, + vec![Value::string("A"), Value::string("·"), Value::string("B")], + ); + let second = render(&mut eval, vec![first.value()]); + let (multibyte, bytes) = output_identity(&second); + assert!(multibyte, "re-derived mode line lost multibyteness"); + assert_eq!(bytes, b"A\xC2\xB7B".to_vec()); +} + +#[test] +fn mode_line_unibyte_raw_byte_input_stays_unibyte_like_gnu() { + // GNU parity control: a UNIBYTE input with a raw byte keeps unibyte + // identity (and therefore displays as the raw-byte escape `\267`, + // exactly as GNU displays a unibyte string's raw byte). The fix must + // not "repair" raw bytes that GNU also leaves raw. + crate::test_utils::init_test_tracing(); + let mut eval = interactive(); + let raw = Value::heap_string(crate::heap_types::LispString::from_unibyte(vec![0xB7])); + let output = render(&mut eval, vec![Value::string("A"), raw, Value::string("B")]); + let (multibyte, bytes) = output_identity(&output); + assert!(!multibyte, "unibyte input must not become multibyte"); + assert_eq!(bytes, b"A\xB7B".to_vec()); +} + +#[test] +fn mode_line_multibyte_input_with_high_codepoints_stays_multibyte() { + // Regression guard above the degrading band: U+2014 EM DASH already + // forced multibyteness under the old content heuristic (> 0xFF); the + // input-driven rule keeps it. + crate::test_utils::init_test_tracing(); + let mut eval = interactive(); + let output = render( + &mut eval, + vec![Value::string("A"), Value::string("—"), Value::string("B")], + ); + let (multibyte, bytes) = output_identity(&output); + assert!(multibyte); + assert_eq!(bytes, "A—B".as_bytes().to_vec()); +} diff --git a/crates/neovm-oracle-tests/tests/oracle_mode_line_flow.rs b/crates/neovm-oracle-tests/tests/oracle_mode_line_flow.rs index c85fb207f5..e777d7c03a 100644 --- a/crates/neovm-oracle-tests/tests/oracle_mode_line_flow.rs +++ b/crates/neovm-oracle-tests/tests/oracle_mode_line_flow.rs @@ -93,6 +93,38 @@ fn format_mode_line_eval_throw_reaches_outer_catch() { ); } +#[test] +fn format_mode_line_multibyte_identity_follows_inputs() { + // Issue #470: a mode line whose non-ASCII content lives entirely in + // U+0080..U+00FF (code fits in one byte) must still produce a MULTIBYTE + // result string, because GNU string identity follows the inputs (the + // `concat' rule), not the content. Deriving the flag from the codes + // re-encodes the dot as a unibyte raw byte, which GNU's redisplay then + // shows as the octal escape `\267` (src/xdisp.c:8649-8662). + return_if_neovm_enable_oracle_proptest_not_set!(); + assert_active_mode_line_parity( + r#"(let ((noninteractive nil)) + (let ((result (format-mode-line '("A" "·" "B") 0))) + (list (multibyte-string-p result) result)))"#, + expect_test::expect![[r#""OK (t \"A·B\")""#]], + ); +} + +#[test] +fn format_mode_line_multibyte_identity_survives_re_derivation() { + // The issue's observed transition: a LATER mode-line evaluation must not + // change the identity the first one produced. Feed the rendered string + // back in as a mode-line element and require multibyte identity again. + return_if_neovm_enable_oracle_proptest_not_set!(); + assert_active_mode_line_parity( + r#"(let ((noninteractive nil)) + (let* ((first (format-mode-line '("A" "·" "B") 0)) + (second (format-mode-line (list first) 0))) + (list (multibyte-string-p second) second)))"#, + expect_test::expect![[r#""OK (t \"A·B\")""#]], + ); +} + #[test] fn format_mode_line_eval_signal_is_logged_inside_outer_condition_case() { return_if_neovm_enable_oracle_proptest_not_set!();