fix(fs): only split UTF-16 lines on aligned line feeds (#3631)

In plugins/fs/src/commands.rs read_until_bytes / LinesBytes, the line
feed sequence was matched at any byte offset, so UTF-16LE "ਗĀ"
(17 0a 00 01) or UTF-16BE "Āਗ" (01 00 0a 17) split mid code unit and every
following line was decoded misaligned.

Line feed and carriage return sequences now only match on a multiple of
their length from the start of the line. Unit tests with both examples
(they fail without the fix).
This commit is contained in:
Lucas Fernandes Nogueira authored and GitHub committed 2026-10-09 13:04:09 -03:00
1 parent dcc8cb032b
commit 5c144e7466
2 files changed
+44 -3

No files matched your search

+6
View File
@@ -0,0 +1,6 @@
---
fs: patch
fs-js: patch
---
Fixed `readTextFileLines` with a UTF-16 encoding splitting lines on a `0x0A 0x00` (LE) or `0x00 0x0A` (BE) byte sequence that is not aligned on a code unit, e.g. `ਗĀ`, which garbled every following line.
+38 -3
View File
@@ -1733,9 +1733,9 @@ impl<B: BufRead> Iterator for LinesBytes<B> {
Ok(0) => None,
Ok(_n) => {
// Remove '\n' or '\r\n'
if buf.ends_with(&self.lf_bytes) {
if ends_with_aligned(&buf, &self.lf_bytes) {
buf.truncate(buf.len() - self.lf_bytes.len());
if buf.ends_with(&self.cr_bytes) {
if ends_with_aligned(&buf, &self.cr_bytes) {
buf.truncate(buf.len() - self.cr_bytes.len());
}
}
@@ -1760,17 +1760,25 @@ fn read_until_bytes(
return r.read_until(last_byte, buf);
}
let start = buf.len();
let mut total_n = 0;
loop {
let n = r.read_until(last_byte, buf)?;
total_n += n;
if n == 0 || buf.ends_with(bytes) {
// for multi-byte code units (UTF-16), only a sequence aligned on a code unit boundary
// is a line feed: `0x0A 0x00` can also be the end of `U+xx0A` followed by `U+00xx`
if n == 0 || ends_with_aligned(&buf[start..], bytes) {
return Ok(total_n);
}
}
}
/// Whether `buf` ends with `bytes`, starting on a multiple of `bytes.len()`.
fn ends_with_aligned(buf: &[u8], bytes: &[u8]) -> bool {
buf.ends_with(bytes) && buf.len() % bytes.len() == 0
}
struct StdLinesResource(Mutex<LinesBytes<BufReader<File>>>);
impl StdLinesResource {
@@ -1994,5 +2002,32 @@ mod test {
assert_eq!(lines.next().map(Result::unwrap), Some(utf16("line 4")));
assert!(lines.next().is_none());
}
// UTF-16 with a line feed byte sequence at an odd offset
{
fn utf16le(text: &str) -> Vec<u8> {
text.encode_utf16().flat_map(|u| u.to_le_bytes()).collect()
}
fn utf16be(text: &str) -> Vec<u8> {
text.encode_utf16().flat_map(|u| u.to_be_bytes()).collect()
}
// "ਗĀ" is `17 0a 00 01` in UTF-16LE
let bytes = utf16le("ਗĀ\nnext\r\nlast\n");
let mut lines =
LinesBytes::new(BufReader::new(&bytes[..]), utf16le("\n"), utf16le("\r"));
assert_eq!(lines.next().map(Result::unwrap), Some(utf16le("ਗĀ")));
assert_eq!(lines.next().map(Result::unwrap), Some(utf16le("next")));
assert_eq!(lines.next().map(Result::unwrap), Some(utf16le("last")));
assert!(lines.next().is_none());
// "Āਗ" is `01 00 0a 17` in UTF-16BE
let bytes = utf16be("Āਗ\nnext");
let mut lines =
LinesBytes::new(BufReader::new(&bytes[..]), utf16be("\n"), utf16be("\r"));
assert_eq!(lines.next().map(Result::unwrap), Some(utf16be("Āਗ")));
assert_eq!(lines.next().map(Result::unwrap), Some(utf16be("next")));
assert!(lines.next().is_none());
}
}
}