From 45a56af642d1813fbdf3e9ff00d76999738bb5d7 Mon Sep 17 00:00:00 2001 From: Nucleic Date: Sat, 18 Jul 2026 15:33:08 -0700 Subject: [PATCH] Merge nucleic/olive-ember-seal-q7vk into dev --- brush-builtins/src/read.rs | 108 +++++++++++++++++++++++++++++++------ 1 file changed, 92 insertions(+), 16 deletions(-) diff --git a/brush-builtins/src/read.rs b/brush-builtins/src/read.rs index b239203..c30d7a0 100644 --- a/brush-builtins/src/read.rs +++ b/brush-builtins/src/read.rs @@ -298,12 +298,18 @@ struct InputReader { deadline: Option, /// Single-byte read buffer. /// - /// TODO(utf-8): This only handles ASCII correctly. Multi-byte UTF-8 characters - /// will be read as separate bytes and incorrectly interpreted. To fix this, - /// we would need to buffer up to 4 bytes and decode incrementally using - /// `std::str::from_utf8`. Note that bash's `-n` counts bytes, not Unicode - /// codepoints, so the fix needs to preserve that behavior. + /// nash(D2): multi-byte UTF-8 sequences are buffered and decoded incrementally in + /// `read_event` (up to 4 bytes via `std::str::from_utf8`); invalid sequences degrade + /// to per-byte Latin-1 chars (the pre-fix behavior for every byte). bash's `-n` + /// counts bytes, which is preserved because the char-limit check measures + /// `line.len()` — the accumulated String's UTF-8 byte length. buffer: [u8; 1], + /// nash(D2): a byte consumed while expecting a UTF-8 continuation that turned out to + /// start the next character; re-consumed before the stream is read again. + pushback: Option, + /// nash(D2): decoded-but-undelivered chars from an invalid sequence flushed as + /// Latin-1; drained before the stream is read again. + pending: VecDeque, /// Terminal mode guard - kept alive for RAII cleanup on drop. /// The guard restores original terminal settings when dropped, even though /// we don't access the field directly after construction. @@ -338,6 +344,8 @@ impl InputReader { input, deadline: timeout.map(|t| Instant::now() + t), buffer: [0; 1], + pushback: None, + pending: VecDeque::new(), _term_mode: term_mode, } } @@ -348,10 +356,40 @@ impl InputReader { brush_core::sys::poll::poll_for_input(&self.input, Duration::ZERO).unwrap_or(false) } + /// nash(D2): reads one byte, honoring the pushback slot and the read deadline. + /// Returns `Ok(None)` on EOF; timeout is reported via `TimedOut` in errors' stead by + /// the caller checking the deadline, so this only polls when a deadline exists. + fn read_byte(&mut self) -> Result, brush_core::Error> { + if let Some(b) = self.pushback.take() { + return Ok(Some(b)); + } + let n = self.input.read(&mut self.buffer)?; + if n == 0 { + return Ok(None); + } + Ok(Some(self.buffer[0])) + } + /// Reads the next input event, handling timeout and control characters. + /// + /// nash(D2): decodes UTF-8 incrementally instead of casting each byte to `char` + /// (which was a Latin-1 decode and re-encoded input bytes — divergence D2 in + /// shell/corpus/DIVERGENCES.md, found when ca-certificates' postinst mangled a + /// UTF-8 filename under narOS). Valid multi-byte sequences round-trip + /// byte-identically through the accumulated String; invalid sequences degrade to + /// the pre-fix per-byte Latin-1 mapping (a String cannot hold raw invalid bytes — + /// full byte-fidelity for invalid UTF-8 would need Vec plumbing shell-wide). fn read_event(&mut self) -> Result { - // Check timeout before attempting read. - if let Some(deadline) = self.deadline { + // Deliver chars flushed from a previous invalid sequence first. + if let Some(ch) = self.pending.pop_front() { + return Ok(InputEvent::Char(ch)); + } + + // Check timeout before attempting read (skipped when a pushback byte is + // waiting — it was already read from the stream). + if self.pushback.is_none() + && let Some(deadline) = self.deadline + { let remaining = deadline.saturating_duration_since(Instant::now()); if remaining.is_zero() { return Ok(InputEvent::Timeout); @@ -365,19 +403,57 @@ impl InputReader { } } - let n = self.input.read(&mut self.buffer)?; - if n == 0 { + let Some(b0) = self.read_byte()? else { return Ok(InputEvent::Eof); + }; + + if b0 < 0x80 { + let ch = b0 as char; + // Map control characters to events. + return Ok(match ch { + CTRL_C => InputEvent::CtrlC, + CTRL_D => InputEvent::CtrlD, + _ => InputEvent::Char(ch), + }); } - let ch = self.buffer[0] as char; + // Multi-byte lead? 0xC2..=0xF4 begin valid sequences of the indicated length; + // anything else (stray continuation, overlong lead) is invalid outright. + let need = match b0 { + 0xC2..=0xDF => 2, + 0xE0..=0xEF => 3, + 0xF0..=0xF4 => 4, + _ => 1, + }; - // Map control characters to events. - Ok(match ch { - CTRL_C => InputEvent::CtrlC, - CTRL_D => InputEvent::CtrlD, - _ => InputEvent::Char(ch), - }) + let mut seq = vec![b0]; + while seq.len() < need { + // Continuation bytes come from the same producer as the lead byte; a + // blocking read here matches `read`'s existing wait-for-delimiter behavior + // (a non-continuation byte — e.g. the eventual newline — unblocks us and is + // pushed back). + match self.read_byte()? { + Some(b) if (b & 0xC0) == 0x80 => seq.push(b), + Some(b) => { + self.pushback = Some(b); + break; + } + None => break, + } + } + + if seq.len() == need + && let Ok(s) = std::str::from_utf8(&seq) + && let Some(ch) = s.chars().next() + { + return Ok(InputEvent::Char(ch)); + } + + // Invalid/incomplete sequence: per-byte Latin-1, first byte now, rest queued. + let mut bytes = seq.into_iter().map(|b| b as char); + let first = bytes.next().unwrap_or('\u{FFFD}'); + self.pending.extend(bytes); + Ok(InputEvent::Char(first)) } }