Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion crates/host-core/src/tools/hashline/apply.rs
Original file line number Diff line number Diff line change
Expand Up @@ -604,6 +604,7 @@ pub fn apply_edit(
text: next_text.clone(),
ending: live.ending,
bom: live.bom,
utf16: live.utf16,
};
Ok((
next_file,
Expand All @@ -618,7 +619,7 @@ pub fn apply_edit(
}

pub fn encode_success(file: &NormalizedFile) -> Vec<u8> {
encode_bytes(&file.text, file.ending, file.bom)
encode_bytes(&file.text, file.ending, file.bom, file.utf16)
}

pub fn record_post_write(
Expand Down
57 changes: 55 additions & 2 deletions crates/host-core/src/tools/hashline/tag.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,8 @@
use sha2::{Digest, Sha256};

const UTF8_BOM: &[u8] = &[0xEF, 0xBB, 0xBF];
const UTF16LE_BOM: &[u8] = &[0xFF, 0xFE];
const UTF16BE_BOM: &[u8] = &[0xFE, 0xFF];

#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum LineEnding {
Expand All @@ -27,10 +29,25 @@ pub struct NormalizedFile {
/// LF-normalized text, BOM stripped, original trailing whitespace kept.
pub text: String,
pub ending: LineEnding,
/// Whether a UTF-8 BOM was present.
pub bom: bool,
/// Byte order for BOM-marked UTF-16: true for little endian, false for big endian.
pub utf16: Option<bool>,
}

pub fn decode_bytes(bytes: &[u8]) -> (bool, String) {
if bytes.starts_with(UTF16LE_BOM) || bytes.starts_with(UTF16BE_BOM) {
let little_endian = bytes.starts_with(UTF16LE_BOM);
let body = &bytes[2..];
let units = body.chunks_exact(2).map(|pair| {
if little_endian {
u16::from_le_bytes([pair[0], pair[1]])
} else {
u16::from_be_bytes([pair[0], pair[1]])
}
});
return (false, String::from_utf16_lossy(&units.collect::<Vec<_>>()));
}
let bom = bytes.starts_with(UTF8_BOM);
let rest = if bom { &bytes[3..] } else { bytes };
(bom, String::from_utf8_lossy(rest).into_owned())
Expand All @@ -52,11 +69,19 @@ pub fn to_lf(text: &str) -> String {

pub fn normalize_file(bytes: &[u8]) -> NormalizedFile {
let (bom, decoded) = decode_bytes(bytes);
let utf16 = if bytes.starts_with(UTF16LE_BOM) {
Some(true)
} else if bytes.starts_with(UTF16BE_BOM) {
Some(false)
} else {
None
};
let ending = detect_ending(&decoded);
NormalizedFile {
text: to_lf(&decoded),
ending,
bom,
utf16,
}
}

Expand Down Expand Up @@ -122,11 +147,26 @@ pub fn tag_of_lf_text(lf_text: &str) -> String {
tag_of_hash_input(&hash_input(lf_text))
}

pub fn encode_bytes(lf_text: &str, ending: LineEnding, bom: bool) -> Vec<u8> {
pub fn encode_bytes(lf_text: &str, ending: LineEnding, bom: bool, utf16: Option<bool>) -> Vec<u8> {
let body = match ending {
LineEnding::Lf => lf_text.to_string(),
LineEnding::Crlf => lf_text.replace('\n', "\r\n"),
};
if let Some(little_endian) = utf16 {
let mut out = if little_endian {
UTF16LE_BOM.to_vec()
} else {
UTF16BE_BOM.to_vec()
};
for unit in body.encode_utf16() {
out.extend(if little_endian {
unit.to_le_bytes()
} else {
unit.to_be_bytes()
});
}
return out;
}
if !bom {
return body.into_bytes();
}
Expand All @@ -137,6 +177,19 @@ pub fn encode_bytes(lf_text: &str, ending: LineEnding, bom: bool) -> Vec<u8> {
}

pub fn looks_binary_bytes(bytes: &[u8]) -> bool {
if bytes.starts_with(UTF16LE_BOM) || bytes.starts_with(UTF16BE_BOM) {
let little_endian = bytes.starts_with(UTF16LE_BOM);
let body = &bytes[2..];
return !body.len().is_multiple_of(2)
|| std::char::decode_utf16(body.chunks_exact(2).map(|pair| {
if little_endian {
u16::from_le_bytes([pair[0], pair[1]])
} else {
u16::from_be_bytes([pair[0], pair[1]])
}
}))
.any(|character| character.is_err() || character.is_ok_and(|c| c == '\0'));
}
let sample = &bytes[..bytes.len().min(4096)];
if sample.contains(&0) {
return true;
Expand Down Expand Up @@ -230,7 +283,7 @@ mod tests {
assert!(file.bom);
assert_eq!(file.text, "hi\n");
assert_eq!(tag_of_lf_text(&file.text), tag_of_lf_text("hi\n"));
assert_eq!(encode_bytes("hi\n", LineEnding::Lf, true), raw);
assert_eq!(encode_bytes("hi\n", LineEnding::Lf, true, None), raw);
}

#[test]
Expand Down
103 changes: 103 additions & 0 deletions crates/host-core/src/tools/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3531,6 +3531,109 @@ mod tests {
}
}

#[tokio::test]
async fn read_powershell_utf16le_log() {
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("build.log");
let mut log = vec![0xff, 0xfe];
log.extend(
"build passed\r\nnext line\r\n"
.encode_utf16()
.flat_map(u16::to_le_bytes),
);
std::fs::write(&path, log).unwrap();

let result = execute_tool(
Some(dir.path()),
None,
"Read",
&serde_json::json!({ "path": "build.log" }),
5_000,
)
.await;
assert!(
result.ok,
"UTF-16LE log should be readable: {:?}",
result.content
);
assert!(result.content["content"]
.as_str()
.unwrap()
.contains("build passed"));
assert!(result.content["content"]
.as_str()
.unwrap()
.contains("2:next line"));

let tag = result.content["tag"].as_str().unwrap();
let edit = execute_tool(
Some(dir.path()),
None,
"Edit",
&serde_json::json!({
"path": "build.log",
"tag": tag,
"ops": "PUT 2.=2:\n+final line\n"
}),
5_000,
)
.await;
assert!(edit.ok, "UTF-16LE edit failed: {:?}", edit.content);
let written = std::fs::read(path).unwrap();
let mut expected = vec![0xff, 0xfe];
expected.extend(
"build passed\r\nfinal line\r\n"
.encode_utf16()
.flat_map(u16::to_le_bytes),
);
assert_eq!(written, expected);
assert_eq!(
hashline::normalize_file(&written).text,
"build passed\nfinal line\n"
);
}

#[tokio::test]
async fn read_and_edit_utf16be_chinese_text() {
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("status.log");
let mut log = vec![0xfe, 0xff];
log.extend("开始\n完成\n".encode_utf16().flat_map(u16::to_be_bytes));
std::fs::write(&path, log).unwrap();

let read = execute_tool(
Some(dir.path()),
None,
"Read",
&serde_json::json!({ "path": "status.log" }),
5_000,
)
.await;
assert!(read.ok, "UTF-16BE Read failed: {:?}", read.content);
let content = read.content["content"].as_str().unwrap();
assert!(content.contains("1:开始"));
assert!(content.contains("2:完成"));

let edit = execute_tool(
Some(dir.path()),
None,
"Edit",
&serde_json::json!({
"path": "status.log",
"tag": read.content["tag"],
"ops": "PUT 2.=2:\n+已完成\n"
}),
5_000,
)
.await;
assert!(edit.ok, "UTF-16BE Edit failed: {:?}", edit.content);
let written = std::fs::read(path).unwrap();
let mut expected = vec![0xfe, 0xff];
expected.extend("开始\n已完成\n".encode_utf16().flat_map(u16::to_be_bytes));
assert_eq!(written, expected);
assert_eq!(hashline::normalize_file(&written).text, "开始\n已完成\n");
}

#[tokio::test]
async fn read_returns_image_blocks_for_image_files() {
// Minimal real signatures so the sniffing branch is exercised.
Expand Down
5 changes: 4 additions & 1 deletion docs/spec/03-runtime/18-line-anchored-edit-contract.md
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,10 @@ Three properties are enforced, in this order, before any byte is written:

Before hashing, and before any line is addressed, file text is normalized:

1. A leading UTF-8 BOM is stripped and retained for restoration on write.
1. A leading UTF-8 or UTF-16LE/BE BOM is decoded and retained for restoration
on write. BOM-marked UTF-16 text is checked after decoding, so the zero bytes
in ordinary PowerShell logs do not cause a binary-file rejection. Edit
preserves the original byte order, BOM, and line endings.
2. Line endings are detected and normalized to `LF`; the dominant original
ending is retained for restoration on write.
3. Trailing `[ \t\r]` is removed from every line, including the last.
Expand Down
10 changes: 10 additions & 0 deletions docs/spec/06-delivery/04-e2e-test-plan.md
Original file line number Diff line number Diff line change
Expand Up @@ -16927,3 +16927,13 @@ host-created files. The full app's file-preview viewer is covered separately.
image model selection and provider configuration remain available.
- Coverage: recent-models.test.mjs, recent-model-flow.test.mjs,
default-model-picker.test.mjs, and scripts/e2e-composer-model-selection.mjs.


## BOM-marked UTF-16 text tools

- Create a UTF-16LE PowerShell build log with a BOM and CRLF, then ask the agent
to Read it. The tool and the next model request contain readable log lines.
- Edit a displayed line. The original BOM, endian and CRLF bytes are preserved.
- Repeat with UTF-16BE Chinese text. Ordinary binary files remain rejected.
- Automated coverage: `read_powershell_utf16le_log`,
`read_and_edit_utf16be_chinese_text`, and the existing binary/CRLF tool tests.
Loading