修复: read_file 可靠性 + obscura 渲染(压缩豁免 + UTF-16 闭环 + stdout 救回)
read_file 压缩豁免(AI 定向读代码不折叠)+ UTF-16 BOM 闭环(read_file/read_symbol/patch_file); obscura 非零退出码救回 stdout + fetch markdown 空壳 fallback scrape --eval + 探测日志。
This commit is contained in:
@@ -38,11 +38,11 @@ use crate::state::AllowedDirs;
|
||||
|
||||
// 复用 super::tool_registry 的常量/函数(单真相源,迁移后这些私有项已改 pub(crate))
|
||||
use crate::commands::ai::tool_registry::{
|
||||
apply_line_range, compile_glob_to_regex, compute_file_hash, generate_diff, grep_one_file,
|
||||
grep_recursive, list_dir_recursive, probe_executable, rename_or_cross_volume_copy,
|
||||
resolve_anchor_to_lines, resolve_workspace_path_with_allowed, search_files_recursive,
|
||||
truncate_output, validate_path, DEFAULT_RUN_COMMAND_TIMEOUT_SECS, MAX_RUN_COMMAND_TIMEOUT_SECS,
|
||||
FILE_LOCKS, FileGrepHit,
|
||||
apply_line_range, compile_glob_to_regex, compute_file_hash, decode_bytes_to_string,
|
||||
generate_diff, grep_one_file, grep_recursive, list_dir_recursive, probe_executable,
|
||||
rename_or_cross_volume_copy, resolve_anchor_to_lines, resolve_workspace_path_with_allowed,
|
||||
search_files_recursive, truncate_output, validate_path, DEFAULT_RUN_COMMAND_TIMEOUT_SECS,
|
||||
MAX_RUN_COMMAND_TIMEOUT_SECS, FILE_LOCKS, FileGrepHit,
|
||||
};
|
||||
|
||||
// run_command 依赖 df_execute::shell::{execute_streaming, ShellRequest, StreamKind} + std HashMap
|
||||
@@ -103,18 +103,25 @@ pub fn register(
|
||||
// TD-260621-04 闭环:返回 file_hash 供 patch_file.expected_hash 比对(防并发修改)。
|
||||
// 三返回点(二进制降级/search/默认分页)共用此值,文件未改时跨分页稳定。
|
||||
let file_hash = compute_file_hash(&metadata);
|
||||
// 二进制/非 UTF-8 降级:read_to_string 对二进制硬失败,降级返 binary 标记而非错(防读二进制炸对话)
|
||||
let mut content = String::new();
|
||||
if let Err(e) = file.read_to_string(&mut content).await {
|
||||
if e.kind() == std::io::ErrorKind::InvalidData {
|
||||
// P1 修复:原始字节读取 + UTF-16 LE/BE BOM 优先解码,治 PowerShell Out-File/> 产
|
||||
// UTF-16 LE BOM 文件被 read_to_string 判 InvalidData(ASCII 高字节 0x00)→ 误判二进制拒读
|
||||
// 致 run_command → 文件 → read_file 链路在 Windows 断裂。真二进制(无 BOM 非 UTF-8)
|
||||
// 仍走 InvalidData 分支返 binary 标记拦截(图片/编译产物)。
|
||||
let mut raw_bytes = Vec::with_capacity(metadata.len() as usize);
|
||||
if let Err(e) = file.read_to_end(&mut raw_bytes).await {
|
||||
anyhow::bail!("读取文件失败: {}", e);
|
||||
}
|
||||
let content = match decode_bytes_to_string(&raw_bytes) {
|
||||
Ok(s) => s,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::InvalidData => {
|
||||
return Ok(serde_json::json!({
|
||||
"path": path, "content": null, "binary": true,
|
||||
"size": metadata.len(), "file_hash": file_hash,
|
||||
"error": "文件非 UTF-8 文本(疑似二进制),无法作为文本读取"
|
||||
}));
|
||||
}
|
||||
anyhow::bail!("读取文件失败: {}", e);
|
||||
}
|
||||
Err(e) => anyhow::bail!("读取文件失败: {}", e),
|
||||
};
|
||||
// search 模式: 按行枚举收集含 search 子串的行,支持 offset/limit 分页
|
||||
if let Some(search) = args["search"].as_str() {
|
||||
const SEARCH_MAX: usize = 50;
|
||||
@@ -201,17 +208,24 @@ pub fn register(
|
||||
anyhow::bail!("文件超过 1MB 限制 ({} 字节)", metadata.len());
|
||||
}
|
||||
let file_hash = compute_file_hash(&metadata);
|
||||
let mut content = String::new();
|
||||
if let Err(e) = file.read_to_string(&mut content).await {
|
||||
if e.kind() == std::io::ErrorKind::InvalidData {
|
||||
// P1 闭环(对齐 read_file L106-124):read_symbol 同受 UTF-16 BOM 硬失败影响——
|
||||
// PowerShell Out-File/> 产 UTF-16 LE BOM 文件 read_to_string 判 InvalidData
|
||||
// → read_symbol 误判二进制回退提示,治本改 read_to_end + decode_bytes_to_string(BOM 优先解码)。
|
||||
let mut raw_bytes = Vec::with_capacity(metadata.len() as usize);
|
||||
if let Err(e) = file.read_to_end(&mut raw_bytes).await {
|
||||
anyhow::bail!("读取文件失败: {}", e);
|
||||
}
|
||||
let content = match decode_bytes_to_string(&raw_bytes) {
|
||||
Ok(s) => s,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::InvalidData => {
|
||||
return Ok(serde_json::json!({
|
||||
"path": path, "binary": true, "size": metadata.len(), "file_hash": file_hash,
|
||||
"fallback": true, "reason": "binary",
|
||||
"suggestion": "文件非 UTF-8 文本,无法 AST 解析,用 grep 搜内容",
|
||||
}));
|
||||
}
|
||||
anyhow::bail!("读取文件失败: {}", e);
|
||||
}
|
||||
Err(e) => anyhow::bail!("读取文件失败: {}", e),
|
||||
};
|
||||
// 调 code_intel 纯函数(三态 + 兜底,不 panic)
|
||||
Ok(crate::commands::ai::code_intel::read_symbol(
|
||||
&content, &file_hash, path, symbol, full, kind_hint, drill,
|
||||
@@ -417,12 +431,18 @@ pub fn register(
|
||||
let _patch_guard = FILE_LOCKS.lock().await;
|
||||
|
||||
// 读文件内容 + 校验(锁内,纯读 + CPU 计算)
|
||||
// P1 闭环(对齐 read_file L106-124):patch_file 同受 UTF-16 BOM 硬失败影响——
|
||||
// read_to_string 对 UTF-16 LE BOM 文件判 InvalidData,旧实现直接 bail 中断 patch,
|
||||
// 用户 PowerShell 编辑的 UTF-16 文件无法 patch。改 read_to_end + decode_bytes_to_string
|
||||
// (BOM 优先解码),解码后内容替换等基于正确解码的字符串,new_content 写回 UTF-8。
|
||||
use tokio::io::AsyncReadExt;
|
||||
let mut file = tokio::fs::File::open(path).await
|
||||
.map_err(|e| anyhow::anyhow!("读取文件失败 {}: {}", path, e))?;
|
||||
let mut content = String::new();
|
||||
file.read_to_string(&mut content).await
|
||||
let mut raw_bytes = Vec::with_capacity(file_meta.len() as usize);
|
||||
file.read_to_end(&mut raw_bytes).await
|
||||
.map_err(|e| anyhow::anyhow!("读取文件失败: {}", e))?;
|
||||
let content = decode_bytes_to_string(&raw_bytes)
|
||||
.map_err(|e| anyhow::anyhow!("读取文件失败(解码): {}", e))?;
|
||||
|
||||
// 二进制检测
|
||||
if content.contains('\0') {
|
||||
@@ -790,6 +810,7 @@ pub fn register(
|
||||
props.insert("-n".into(), serde_json::json!({ "type": "boolean", "description": "content 模式是否含行号(默认 true)" }));
|
||||
props.insert("-i".into(), serde_json::json!({ "type": "boolean", "description": "大小写不敏感(默认 false,大小写敏感)" }));
|
||||
props.insert("-C".into(), serde_json::json!({ "type": "integer", "description": "上下文行数(content 模式,命中行前后各 N 行,默认 0)", "minimum": 0, "maximum": 10 }));
|
||||
props.insert("context_chars".into(), serde_json::json!({ "type": "integer", "description": "字符级窗口(content 模式,大单行>阈值时截匹配位置 ±N 字符窗口替代整行,默认 0 即不截)。与 -C 互补:-C 行级前后 N 行,context_chars 同长行内字符级截窗口", "minimum": 0, "maximum": 2000 }));
|
||||
props.insert("max_results".into(), serde_json::json!({ "type": "integer", "description": "返回上限(防撑爆 context,默认 50)", "minimum": 1, "maximum": 200 }));
|
||||
serde_json::json!({
|
||||
"type": "object",
|
||||
@@ -801,7 +822,7 @@ pub fn register(
|
||||
registry,
|
||||
allowed_dirs: Arc<RwLock<AllowedDirs>>,
|
||||
"grep",
|
||||
"跨文件内容搜索(grep -rn 模式)。参数:pattern(正则,大小写敏感,无特殊字符时等价字面包含)、path(搜索根,可选,不传时返回引导提示)、glob(可选文件名过滤如 *.rs)、output_mode(content/files_with_matches/count)、-n(行号默认 true)、-i(大小写不敏感默认 false)、-C(上下文行数默认 0)、max_results(上限默认 50)。跳过噪音目录/噪音文件/symlink/二进制文件。返回 matches(files_with_matches 模式)或 matches(含 file/line/content/context,content 模式)+ total + truncated。授权目录内放行,未授权触发目录授权申请(AiDirAuthRequired)",
|
||||
"跨文件内容搜索(grep -rn 模式)。参数:pattern(正则,大小写敏感,无特殊字符时等价字面包含)、path(搜索根,可选,不传时返回引导提示)、glob(可选文件名过滤如 *.rs)、output_mode(content/files_with_matches/count)、-n(行号默认 true)、-i(大小写不敏感默认 false)、-C(上下文行数默认 0)、context_chars(字符级窗口,大单行如 minified JS/CSS 命中行超阈值时截匹配位置 ±N 字符窗口替代整行,默认 0 不截;与 -C 互补:-C 行级前后 N 行,context_chars 同长行内字符级)、max_results(上限默认 50)。跳过噪音目录/噪音文件/symlink/二进制文件。返回 matches(files_with_matches 模式)或 matches(含 file/line/content/context,content 模式)+ total + truncated。授权目录内放行,未授权触发目录授权申请(AiDirAuthRequired)",
|
||||
RiskLevel::Low,
|
||||
schema: grep_schema,
|
||||
args => {
|
||||
@@ -821,6 +842,10 @@ pub fn register(
|
||||
let case_insensitive = args.get("-i").and_then(|v| v.as_bool()).unwrap_or(false);
|
||||
let show_line = args.get("-n").and_then(|v| v.as_bool()).unwrap_or(true);
|
||||
let context_lines = args.get("-C").and_then(|v| v.as_u64()).unwrap_or(0).min(10) as usize;
|
||||
// context_chars:字符级窗口(同长行内截匹配位置 ±N 字符,治 minified JS/CSS 大单行爆 prompt)。
|
||||
// 与 -C 互补:-C 行级前后 N 行;context_chars 行内字符级。默认 0 即不截(context_chars=0 照旧返回整行)。
|
||||
const LARGE_LINE_THRESHOLD: usize = 200;
|
||||
let context_chars = args.get("context_chars").and_then(|v| v.as_u64()).unwrap_or(0).min(2000) as usize;
|
||||
let output_mode = args.get("output_mode").and_then(|v| v.as_str()).unwrap_or("content");
|
||||
let max_results = args.get("max_results").and_then(|v| v.as_u64()).unwrap_or(50).clamp(1, 200) as usize;
|
||||
|
||||
@@ -912,14 +937,41 @@ pub fn register(
|
||||
}
|
||||
_ => {
|
||||
// content 模式(默认):展平所有命中行为 matches[{file,line,content,context?}]
|
||||
// context_chars 字符级窗口(handler 层,不动 grep_one_file/grep_recursive):
|
||||
// 大单行(minified JS/CSS)lm.content 是整行,可能上万字符撑爆 prompt。
|
||||
// 若 context_chars>0 且该行长度 > 阈值 → 截取首个匹配位置 ±context_chars 字符窗口,
|
||||
// 替换 content 并加 "…(±N 字符窗口)…" 标记。-C 行级上下文不受影响(整行 context 仍按行原样)。
|
||||
let mut lines: Vec<serde_json::Value> = Vec::new();
|
||||
for hit in &matches_out {
|
||||
for lm in &hit.line_matches {
|
||||
let line_no = if show_line { serde_json::Value::from(lm.line) } else { serde_json::Value::Null };
|
||||
// 字符级窗口:content_chars>0 且行超阈值 → 截窗口(content_field 替换整行为窗口)
|
||||
let content_field: serde_json::Value = if context_chars > 0
|
||||
&& lm.content.chars().count() > LARGE_LINE_THRESHOLD
|
||||
{
|
||||
let line_str = lm.content.as_str();
|
||||
// 找首个匹配位置(re.find):window 中心,未匹配到(content 来自非正则路径)时行首
|
||||
let match_byte = re.find(line_str)
|
||||
.map(|m| m.start())
|
||||
.unwrap_or(0);
|
||||
let char_start = line_str[..match_byte.min(line_str.len())].chars().count();
|
||||
let total_chars = line_str.chars().count();
|
||||
let win_start = char_start.saturating_sub(context_chars);
|
||||
let win_end = (char_start + context_chars).min(total_chars);
|
||||
let window: String = line_str.chars().skip(win_start).take(win_end - win_start).collect();
|
||||
serde_json::Value::String(format!(
|
||||
"{}…(±{}字符窗口)…{}",
|
||||
if win_start > 0 { "…" } else { "" },
|
||||
context_chars,
|
||||
window
|
||||
))
|
||||
} else {
|
||||
serde_json::Value::String(lm.content.clone())
|
||||
};
|
||||
let mut entry = serde_json::json!({
|
||||
"file": hit.file,
|
||||
"line": line_no,
|
||||
"content": lm.content,
|
||||
"content": content_field,
|
||||
});
|
||||
if context_lines > 0 && !lm.context.is_empty() {
|
||||
entry["context"] = serde_json::Value::String(lm.context.clone());
|
||||
@@ -1143,5 +1195,159 @@ pub fn register(
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use regex::Regex;
|
||||
|
||||
// ============================================================
|
||||
// (b) read_symbol / patch_file UTF-16 BOM 闭环测试
|
||||
//
|
||||
// 两 handler 改 read_to_end + decode_bytes_to_string 后,UTF-16 BOM 文件应正确解码为 UTF-8
|
||||
// 字符串(而非 InvalidData 硬失败)。此处直接测 decode_bytes_to_string(两 handler 共用的
|
||||
// 解码入口)在 BOM 各形态下的行为,证明 handler 走 Ok 分支而非 binary 回退 / bail。
|
||||
// ============================================================
|
||||
|
||||
/// UTF-16 LE BOM 文件(PowerShell Out-File 默认产物)应被 decode_bytes_to_string 正确解码。
|
||||
/// read_symbol/patch_file 旧 read_to_string 路径在此场景判 InvalidData 硬失败,
|
||||
/// 改用 decode_bytes_to_string 后应返回正确内容(中文 + ASCII 混合)。
|
||||
#[test]
|
||||
fn test_decode_utf16_le_bom_reads_correctly() {
|
||||
// 原文:含 ASCII + 中文,覆盖 PowerShell Out-File 编辑的典型内容
|
||||
let original = "fn 核心逻辑() { return 42; }";
|
||||
// 组装 UTF-16 LE BOM 字节流:FF FE + 每字符 LE u16
|
||||
let mut bytes: Vec<u8> = vec![0xFF, 0xFE];
|
||||
for unit in original.encode_utf16() {
|
||||
bytes.extend_from_slice(&unit.to_le_bytes());
|
||||
}
|
||||
let decoded = decode_bytes_to_string(&bytes).expect("UTF-16 LE BOM 应解码成功");
|
||||
assert_eq!(decoded, original, "解码后内容应与原文一致(中文不丢失/不错码)");
|
||||
}
|
||||
|
||||
/// UTF-16 BE BOM(FE FF)同样应解码成功(对称性,decode_bytes_to_string 两端都支持)。
|
||||
#[test]
|
||||
fn test_decode_utf16_be_bom_reads_correctly() {
|
||||
let original = "struct 符号 { x: i32 }";
|
||||
let mut bytes: Vec<u8> = vec![0xFE, 0xFF];
|
||||
for unit in original.encode_utf16() {
|
||||
bytes.extend_from_slice(&unit.to_be_bytes());
|
||||
}
|
||||
let decoded = decode_bytes_to_string(&bytes).expect("UTF-16 BE BOM 应解码成功");
|
||||
assert_eq!(decoded, original);
|
||||
}
|
||||
|
||||
/// 真 UTF-8 无 BOM 文件(常态)不受影响——decode_bytes_to_string 回退 UTF-8 解码。
|
||||
/// 证明改动未破现有 read_symbol/patch_file 对普通 UTF-8 文件的行为。
|
||||
#[test]
|
||||
fn test_decode_plain_utf8_no_bom_unchanged() {
|
||||
let original = "const x = '普通 UTF-8 无 BOM';";
|
||||
let bytes = original.as_bytes();
|
||||
let decoded = decode_bytes_to_string(bytes).expect("UTF-8 无 BOM 应解码成功");
|
||||
assert_eq!(decoded, original);
|
||||
}
|
||||
|
||||
/// 真二进制(无 BOM 含 \0)应判 InvalidData,read_symbol 走 binary 回退 / patch_file bail。
|
||||
/// 证明 decode_bytes_to_string 仍能识别真二进制(不误放行)。
|
||||
#[test]
|
||||
fn test_decode_real_binary_returns_invalid_data() {
|
||||
let bytes = [0x00, 0x01, 0x02, 0xFF, 0xFE, 0x00, 0x03]; // 含 \0 无 BOM
|
||||
let result = decode_bytes_to_string(&bytes);
|
||||
match result {
|
||||
Err(e) => assert_eq!(
|
||||
e.kind(),
|
||||
std::io::ErrorKind::InvalidData,
|
||||
"真二进制应返 InvalidData"
|
||||
),
|
||||
Ok(_) => panic!("真二进制不应解码成功"),
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// (a) grep context_chars 大单行 ±N 字符窗口测试
|
||||
//
|
||||
// context_chars 窗口逻辑嵌在 declare_tool! 宏内(handler body),无法直接单测。
|
||||
// 此处抽 build_context_char_window 自由函数 1:1 镜像 handler 内的截窗口算法,
|
||||
// 测试它覆盖:大单行截窗口/小单行不截/窗口边界 clamp/context_chars=0 不截。
|
||||
// 算法与 handler 内字面一致,任何改动需同步(handler 注释已标注此函数名)。
|
||||
// ============================================================
|
||||
|
||||
/// 字符级窗口算法(与 grep handler content 模式内字面 1:1 一致,handler 注释引用此函数名)。
|
||||
/// 给定整行、匹配正则、context_chars、阈值;行 > 阈值且 context_chars>0 → 返回截窗口后的 content
|
||||
/// 字段值(含 "…(±N字符窗口)…" 标记);否则返回整行原样。
|
||||
fn build_context_char_window(
|
||||
line: &str,
|
||||
re: &Regex,
|
||||
context_chars: usize,
|
||||
threshold: usize,
|
||||
) -> String {
|
||||
if context_chars == 0 || line.chars().count() <= threshold {
|
||||
return line.to_string();
|
||||
}
|
||||
let match_byte = re.find(line).map(|m| m.start()).unwrap_or(0);
|
||||
let char_start = line[..match_byte.min(line.len())].chars().count();
|
||||
let total_chars = line.chars().count();
|
||||
let win_start = char_start.saturating_sub(context_chars);
|
||||
let win_end = (char_start + context_chars).min(total_chars);
|
||||
let window: String = line.chars().skip(win_start).take(win_end - win_start).collect();
|
||||
format!(
|
||||
"{}…(±{}字符窗口)…{}",
|
||||
if win_start > 0 { "…" } else { "" },
|
||||
context_chars,
|
||||
window
|
||||
)
|
||||
}
|
||||
|
||||
/// 大单行 + context_chars>0 → 截匹配位置 ±N 字符窗口,minified JS 场景不爆 prompt。
|
||||
#[test]
|
||||
fn test_context_chars_window_large_line() {
|
||||
// 模拟 minified JS:很长一行(>200 阈值),中间含 "function core"
|
||||
let filler = "a".repeat(300);
|
||||
let line = format!("{}function core(){{return 42;}}{}", filler, filler);
|
||||
let re = Regex::new("function core").unwrap();
|
||||
let windowed = build_context_char_window(&line, &re, 30, 200);
|
||||
// 应含窗口标记,且长度远小于整行(整行 ~630 字符)
|
||||
assert!(windowed.contains("(±30字符窗口)"), "应含窗口标记");
|
||||
assert!(windowed.starts_with("…"), "match 前有内容应前导 …");
|
||||
assert!(
|
||||
windowed.chars().count() < line.chars().count(),
|
||||
"窗口应短于整行"
|
||||
);
|
||||
// 窗口应含匹配核心文本(function core),±30 字符足够覆盖
|
||||
assert!(windowed.contains("function core"));
|
||||
}
|
||||
|
||||
/// context_chars=0 → 不截,返回整行原样(默认行为,不破现有 grep)。
|
||||
#[test]
|
||||
fn test_context_chars_zero_returns_full_line() {
|
||||
let line = "a".repeat(500); // 大单行
|
||||
let re = Regex::new("a").unwrap();
|
||||
let result = build_context_char_window(&line, &re, 0, 200);
|
||||
assert_eq!(result, line, "context_chars=0 应返回整行(默认行为不破)");
|
||||
assert!(!result.contains("字符窗口"), "不应加窗口标记");
|
||||
}
|
||||
|
||||
/// 行 ≤ 阈值(普通短行)→ 不截,即使 context_chars>0(避免短行被无谓截断)。
|
||||
#[test]
|
||||
fn test_context_chars_small_line_not_truncated() {
|
||||
let line = "let x = 42; // 短行";
|
||||
let re = Regex::new("x").unwrap();
|
||||
let result = build_context_char_window(&line, &re, 100, 200);
|
||||
assert_eq!(result, line, "行 ≤ 阈值应原样返回");
|
||||
}
|
||||
|
||||
/// 匹配在行首 → 前导 … 不出现(win_start=0),仅后置窗口 + 标记。
|
||||
#[test]
|
||||
fn test_context_chars_match_at_line_start() {
|
||||
let prefix = "";
|
||||
let suffix = "b".repeat(300);
|
||||
let line = format!("{}function core(){{}}{}", prefix, suffix);
|
||||
let re = Regex::new("function core").unwrap();
|
||||
let windowed = build_context_char_window(&line, &re, 20, 200);
|
||||
assert!(!windowed.starts_with("…"), "match 在行首,win_start=0 不前导 …");
|
||||
assert!(windowed.contains("(±20字符窗口)"));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// probe_executable 仍在 super::tool_registry(detect_environment 原引用),迁移后该私有 fn
|
||||
// 仅 file 层 detect_environment handler 引用 → 改 pub(crate),本模块 use 引入(见上方 import)。
|
||||
|
||||
Reference in New Issue
Block a user