mirror of
https://github.com/tauri-apps/plugins-workspace.git
synced 2026-10-10 22:38:44 +02:00
fix(fs): only split UTF-16 lines on aligned line feeds (#3631)
In plugins/fs/src/commands.rs read_until_bytes / LinesBytes, the line feed sequence was matched at any byte offset, so UTF-16LE "ਗĀ" (17 0a 00 01) or UTF-16BE "Āਗ" (01 00 0a 17) split mid code unit and every following line was decoded misaligned. Line feed and carriage return sequences now only match on a multiple of their length from the start of the line. Unit tests with both examples (they fail without the fix).
This commit is contained in:
1 parent
dcc8cb032b
commit
5c144e7466
2 files changed
+44
-3
No files matched your search
@@ -0,0 +1,6 @@
|
||||
---
|
||||
fs: patch
|
||||
fs-js: patch
|
||||
---
|
||||
|
||||
Fixed `readTextFileLines` with a UTF-16 encoding splitting lines on a `0x0A 0x00` (LE) or `0x00 0x0A` (BE) byte sequence that is not aligned on a code unit, e.g. `ਗĀ`, which garbled every following line.
|
||||
@@ -1733,9 +1733,9 @@ impl<B: BufRead> Iterator for LinesBytes<B> {
|
||||
Ok(0) => None,
|
||||
Ok(_n) => {
|
||||
// Remove '\n' or '\r\n'
|
||||
if buf.ends_with(&self.lf_bytes) {
|
||||
if ends_with_aligned(&buf, &self.lf_bytes) {
|
||||
buf.truncate(buf.len() - self.lf_bytes.len());
|
||||
if buf.ends_with(&self.cr_bytes) {
|
||||
if ends_with_aligned(&buf, &self.cr_bytes) {
|
||||
buf.truncate(buf.len() - self.cr_bytes.len());
|
||||
}
|
||||
}
|
||||
@@ -1760,17 +1760,25 @@ fn read_until_bytes(
|
||||
return r.read_until(last_byte, buf);
|
||||
}
|
||||
|
||||
let start = buf.len();
|
||||
let mut total_n = 0;
|
||||
loop {
|
||||
let n = r.read_until(last_byte, buf)?;
|
||||
total_n += n;
|
||||
|
||||
if n == 0 || buf.ends_with(bytes) {
|
||||
// for multi-byte code units (UTF-16), only a sequence aligned on a code unit boundary
|
||||
// is a line feed: `0x0A 0x00` can also be the end of `U+xx0A` followed by `U+00xx`
|
||||
if n == 0 || ends_with_aligned(&buf[start..], bytes) {
|
||||
return Ok(total_n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `buf` ends with `bytes`, starting on a multiple of `bytes.len()`.
|
||||
fn ends_with_aligned(buf: &[u8], bytes: &[u8]) -> bool {
|
||||
buf.ends_with(bytes) && buf.len() % bytes.len() == 0
|
||||
}
|
||||
|
||||
struct StdLinesResource(Mutex<LinesBytes<BufReader<File>>>);
|
||||
|
||||
impl StdLinesResource {
|
||||
@@ -1994,5 +2002,32 @@ mod test {
|
||||
assert_eq!(lines.next().map(Result::unwrap), Some(utf16("line 4")));
|
||||
assert!(lines.next().is_none());
|
||||
}
|
||||
|
||||
// UTF-16 with a line feed byte sequence at an odd offset
|
||||
{
|
||||
fn utf16le(text: &str) -> Vec<u8> {
|
||||
text.encode_utf16().flat_map(|u| u.to_le_bytes()).collect()
|
||||
}
|
||||
fn utf16be(text: &str) -> Vec<u8> {
|
||||
text.encode_utf16().flat_map(|u| u.to_be_bytes()).collect()
|
||||
}
|
||||
|
||||
// "ਗĀ" is `17 0a 00 01` in UTF-16LE
|
||||
let bytes = utf16le("ਗĀ\nnext\r\nlast\n");
|
||||
let mut lines =
|
||||
LinesBytes::new(BufReader::new(&bytes[..]), utf16le("\n"), utf16le("\r"));
|
||||
assert_eq!(lines.next().map(Result::unwrap), Some(utf16le("ਗĀ")));
|
||||
assert_eq!(lines.next().map(Result::unwrap), Some(utf16le("next")));
|
||||
assert_eq!(lines.next().map(Result::unwrap), Some(utf16le("last")));
|
||||
assert!(lines.next().is_none());
|
||||
|
||||
// "Āਗ" is `01 00 0a 17` in UTF-16BE
|
||||
let bytes = utf16be("Āਗ\nnext");
|
||||
let mut lines =
|
||||
LinesBytes::new(BufReader::new(&bytes[..]), utf16be("\n"), utf16be("\r"));
|
||||
assert_eq!(lines.next().map(Result::unwrap), Some(utf16be("Āਗ")));
|
||||
assert_eq!(lines.next().map(Result::unwrap), Some(utf16be("next")));
|
||||
assert!(lines.next().is_none());
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user