Merge nucleic/olive-ember-seal-q7vk into dev
This commit is contained in:
+90
-14
@@ -298,12 +298,18 @@ struct InputReader {
|
||||
deadline: Option<Instant>,
|
||||
/// Single-byte read buffer.
|
||||
///
|
||||
/// TODO(utf-8): This only handles ASCII correctly. Multi-byte UTF-8 characters
|
||||
/// will be read as separate bytes and incorrectly interpreted. To fix this,
|
||||
/// we would need to buffer up to 4 bytes and decode incrementally using
|
||||
/// `std::str::from_utf8`. Note that bash's `-n` counts bytes, not Unicode
|
||||
/// codepoints, so the fix needs to preserve that behavior.
|
||||
/// nash(D2): multi-byte UTF-8 sequences are buffered and decoded incrementally in
|
||||
/// `read_event` (up to 4 bytes via `std::str::from_utf8`); invalid sequences degrade
|
||||
/// to per-byte Latin-1 chars (the pre-fix behavior for every byte). bash's `-n`
|
||||
/// counts bytes, which is preserved because the char-limit check measures
|
||||
/// `line.len()` — the accumulated String's UTF-8 byte length.
|
||||
buffer: [u8; 1],
|
||||
/// nash(D2): a byte consumed while expecting a UTF-8 continuation that turned out to
|
||||
/// start the next character; re-consumed before the stream is read again.
|
||||
pushback: Option<u8>,
|
||||
/// nash(D2): decoded-but-undelivered chars from an invalid sequence flushed as
|
||||
/// Latin-1; drained before the stream is read again.
|
||||
pending: VecDeque<char>,
|
||||
/// Terminal mode guard - kept alive for RAII cleanup on drop.
|
||||
/// The guard restores original terminal settings when dropped, even though
|
||||
/// we don't access the field directly after construction.
|
||||
@@ -338,6 +344,8 @@ impl InputReader {
|
||||
input,
|
||||
deadline: timeout.map(|t| Instant::now() + t),
|
||||
buffer: [0; 1],
|
||||
pushback: None,
|
||||
pending: VecDeque::new(),
|
||||
_term_mode: term_mode,
|
||||
}
|
||||
}
|
||||
@@ -348,10 +356,40 @@ impl InputReader {
|
||||
brush_core::sys::poll::poll_for_input(&self.input, Duration::ZERO).unwrap_or(false)
|
||||
}
|
||||
|
||||
/// nash(D2): reads one byte, honoring the pushback slot and the read deadline.
|
||||
/// Returns `Ok(None)` on EOF; timeout is reported via `TimedOut` in errors' stead by
|
||||
/// the caller checking the deadline, so this only polls when a deadline exists.
|
||||
fn read_byte(&mut self) -> Result<Option<u8>, brush_core::Error> {
|
||||
if let Some(b) = self.pushback.take() {
|
||||
return Ok(Some(b));
|
||||
}
|
||||
let n = self.input.read(&mut self.buffer)?;
|
||||
if n == 0 {
|
||||
return Ok(None);
|
||||
}
|
||||
Ok(Some(self.buffer[0]))
|
||||
}
|
||||
|
||||
/// Reads the next input event, handling timeout and control characters.
|
||||
///
|
||||
/// nash(D2): decodes UTF-8 incrementally instead of casting each byte to `char`
|
||||
/// (which was a Latin-1 decode and re-encoded input bytes — divergence D2 in
|
||||
/// shell/corpus/DIVERGENCES.md, found when ca-certificates' postinst mangled a
|
||||
/// UTF-8 filename under narOS). Valid multi-byte sequences round-trip
|
||||
/// byte-identically through the accumulated String; invalid sequences degrade to
|
||||
/// the pre-fix per-byte Latin-1 mapping (a String cannot hold raw invalid bytes —
|
||||
/// full byte-fidelity for invalid UTF-8 would need Vec<u8> plumbing shell-wide).
|
||||
fn read_event(&mut self) -> Result<InputEvent, brush_core::Error> {
|
||||
// Check timeout before attempting read.
|
||||
if let Some(deadline) = self.deadline {
|
||||
// Deliver chars flushed from a previous invalid sequence first.
|
||||
if let Some(ch) = self.pending.pop_front() {
|
||||
return Ok(InputEvent::Char(ch));
|
||||
}
|
||||
|
||||
// Check timeout before attempting read (skipped when a pushback byte is
|
||||
// waiting — it was already read from the stream).
|
||||
if self.pushback.is_none()
|
||||
&& let Some(deadline) = self.deadline
|
||||
{
|
||||
let remaining = deadline.saturating_duration_since(Instant::now());
|
||||
if remaining.is_zero() {
|
||||
return Ok(InputEvent::Timeout);
|
||||
@@ -365,19 +403,57 @@ impl InputReader {
|
||||
}
|
||||
}
|
||||
|
||||
let n = self.input.read(&mut self.buffer)?;
|
||||
if n == 0 {
|
||||
let Some(b0) = self.read_byte()? else {
|
||||
return Ok(InputEvent::Eof);
|
||||
}
|
||||
|
||||
let ch = self.buffer[0] as char;
|
||||
};
|
||||
|
||||
if b0 < 0x80 {
|
||||
let ch = b0 as char;
|
||||
// Map control characters to events.
|
||||
Ok(match ch {
|
||||
return Ok(match ch {
|
||||
CTRL_C => InputEvent::CtrlC,
|
||||
CTRL_D => InputEvent::CtrlD,
|
||||
_ => InputEvent::Char(ch),
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
// Multi-byte lead? 0xC2..=0xF4 begin valid sequences of the indicated length;
|
||||
// anything else (stray continuation, overlong lead) is invalid outright.
|
||||
let need = match b0 {
|
||||
0xC2..=0xDF => 2,
|
||||
0xE0..=0xEF => 3,
|
||||
0xF0..=0xF4 => 4,
|
||||
_ => 1,
|
||||
};
|
||||
|
||||
let mut seq = vec![b0];
|
||||
while seq.len() < need {
|
||||
// Continuation bytes come from the same producer as the lead byte; a
|
||||
// blocking read here matches `read`'s existing wait-for-delimiter behavior
|
||||
// (a non-continuation byte — e.g. the eventual newline — unblocks us and is
|
||||
// pushed back).
|
||||
match self.read_byte()? {
|
||||
Some(b) if (b & 0xC0) == 0x80 => seq.push(b),
|
||||
Some(b) => {
|
||||
self.pushback = Some(b);
|
||||
break;
|
||||
}
|
||||
None => break,
|
||||
}
|
||||
}
|
||||
|
||||
if seq.len() == need
|
||||
&& let Ok(s) = std::str::from_utf8(&seq)
|
||||
&& let Some(ch) = s.chars().next()
|
||||
{
|
||||
return Ok(InputEvent::Char(ch));
|
||||
}
|
||||
|
||||
// Invalid/incomplete sequence: per-byte Latin-1, first byte now, rest queued.
|
||||
let mut bytes = seq.into_iter().map(|b| b as char);
|
||||
let first = bytes.next().unwrap_or('\u{FFFD}');
|
||||
self.pending.extend(bytes);
|
||||
Ok(InputEvent::Char(first))
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user