feat: ship lime 1.6.1 rollout

This commit is contained in:
coso
2026-04-09 18:47:33 +08:00
parent db3bfae4ff
commit 182c0d464a
152 changed files with 19877 additions and 2413 deletions
+19 -19
View File
@@ -5101,7 +5101,7 @@ dependencies = [
[[package]]
name = "lime"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"anyhow",
"arboard",
@@ -5206,7 +5206,7 @@ dependencies = [
[[package]]
name = "lime-agent"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"anyhow",
"aster-core",
@@ -5235,7 +5235,7 @@ dependencies = [
[[package]]
name = "lime-browser-runtime"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"chrono",
"futures",
@@ -5252,7 +5252,7 @@ dependencies = [
[[package]]
name = "lime-cli"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"clap",
"lime-core",
@@ -5264,7 +5264,7 @@ dependencies = [
[[package]]
name = "lime-config"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"async-trait",
"lime-core",
@@ -5280,7 +5280,7 @@ dependencies = [
[[package]]
name = "lime-core"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"aster-models",
"async-trait",
@@ -5320,7 +5320,7 @@ dependencies = [
[[package]]
name = "lime-credential"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"axum 0.7.9",
"base64 0.22.1",
@@ -5355,7 +5355,7 @@ dependencies = [
[[package]]
name = "lime-gateway"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"aes",
"axum 0.7.9",
@@ -5385,7 +5385,7 @@ dependencies = [
[[package]]
name = "lime-infra"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"chrono",
"dashmap 5.5.3",
@@ -5405,7 +5405,7 @@ dependencies = [
[[package]]
name = "lime-mcp"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"async-trait",
"dirs 5.0.1",
@@ -5421,7 +5421,7 @@ dependencies = [
[[package]]
name = "lime-media-runtime"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"axum 0.7.9",
"chrono",
@@ -5452,7 +5452,7 @@ dependencies = [
[[package]]
name = "lime-processor"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"async-trait",
"lime-core",
@@ -5471,7 +5471,7 @@ dependencies = [
[[package]]
name = "lime-providers"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"anyhow",
"async-stream",
@@ -5526,7 +5526,7 @@ dependencies = [
[[package]]
name = "lime-server"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"aster-core",
"async-stream",
@@ -5571,7 +5571,7 @@ dependencies = [
[[package]]
name = "lime-server-utils"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"axum 0.7.9",
"futures",
@@ -5586,7 +5586,7 @@ dependencies = [
[[package]]
name = "lime-services"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"anyhow",
"aster-core",
@@ -5628,7 +5628,7 @@ dependencies = [
[[package]]
name = "lime-skills"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"async-trait",
"dirs 5.0.1",
@@ -5646,7 +5646,7 @@ dependencies = [
[[package]]
name = "lime-terminal"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"async-trait",
"base64 0.22.1",
@@ -5673,7 +5673,7 @@ dependencies = [
[[package]]
name = "lime-websocket"
version = "1.6.0"
version = "1.6.1"
dependencies = [
"axum 0.7.9",
"chrono",
+2 -2
View File
@@ -4,7 +4,7 @@ exclude = ["crates/aster", "crates/aster-models", "crates/aster-rust"]
resolver = "2"
[workspace.package]
version = "1.6.0"
version = "1.6.1"
edition = "2021"
authors = ["coso"]
repository = "https://github.com/aiclientproxy/lime"
@@ -189,7 +189,7 @@ version = "2.4"
[package]
name = "lime"
version = "1.6.0"
version = "1.6.1"
description = "AI API Proxy Desktop App"
authors = ["you"]
edition = "2021"
@@ -31,7 +31,7 @@ pub const TOOL_GUIDELINES: &str = r#"# 工具使用策略
### 搜索工具
- **Glob**: 使用 glob 模式搜索文件路径
- **Grep**: 使用正则表达式搜索文件内容
- **ToolSearch**: 搜索当前会话可用工具,尤其是 extension / MCP / 延迟加载工具
- **ToolSearch**: 只用于搜索 deferred 的 extension / MCP 工具;使用精确工具名,例如 `select:Read,Edit,Grep` 或 `select:mcp__playwright__browser_click`。如果 Read / Write / Edit / Glob / Grep / Bash / WebFetch / WebSearch 已经在当前工具面中可见,不要再用 ToolSearch 去找它们,也不要把 `read_file`、`write_file`、`edit_file`、`system` 之类别名继续丢给 ToolSearch
- **ListMcpResourcesTool / ReadMcpResourceTool**: 浏览和读取 MCP 资源
### 系统工具
@@ -349,6 +349,21 @@ pub struct StreamReplyExecution {
pub attempts_summary: String,
}
fn build_empty_final_reply_fallback(
diagnostics: &StreamEventDiagnostics,
emitted_any: bool,
) -> Option<String> {
if !emitted_any {
return None;
}
if diagnostics.tool_start_count > 0 || diagnostics.tool_end_count > 0 {
return Some("本轮执行已完成,详细过程与产物已保留在当前对话中。".to_string());
}
Some("本轮执行已结束,过程记录已保留在当前对话中。".to_string())
}
#[derive(Debug, Default)]
struct AutoCompactionProjectionState;
@@ -1614,6 +1629,23 @@ where
emitted_any,
});
}
if let Some(fallback_text) =
build_empty_final_reply_fallback(&diagnostics, emitted_any)
{
tracing::warn!(
"[AsterAgent][ReplyPolicy] empty final text downgraded to synthesized fallback: emitted_any={}, tool_starts={}, tool_ends={}, attempts={}",
emitted_any,
diagnostics.tool_start_count,
diagnostics.tool_end_count,
web_search_tracker.format_attempts()
);
return Ok(StreamReplyExecution {
text_output: fallback_text,
event_errors,
emitted_any,
attempts_summary: web_search_tracker.format_attempts(),
});
}
return Err(ReplyAttemptError {
message: format!(
"已完成当前回合的工具执行,但模型未输出最终答复。\n尝试记录: {}",
@@ -1832,6 +1864,27 @@ mod tests {
assert!(err.contains("尝试记录"));
}
#[test]
fn empty_final_reply_with_tool_events_should_use_fallback_text() {
let diagnostics = StreamEventDiagnostics {
tool_start_count: 1,
tool_end_count: 1,
..Default::default()
};
assert_eq!(
build_empty_final_reply_fallback(&diagnostics, true).as_deref(),
Some("本轮执行已完成,详细过程与产物已保留在当前对话中。")
);
}
#[test]
fn empty_final_reply_without_any_emission_should_still_error() {
let diagnostics = StreamEventDiagnostics::default();
assert_eq!(build_empty_final_reply_fallback(&diagnostics, false), None);
}
#[test]
fn detects_news_expansion_for_daily_news_summary_requests() {
assert!(message_suggests_news_expansion("帮我汇总3月13日国际新闻"));
File diff suppressed because it is too large Load Diff
@@ -884,6 +884,10 @@ impl TurnItemRuntimeProjector {
}
fn project_message(&mut self, message: &Message) -> Vec<AgentEvent> {
if !message.is_user_visible() {
return Vec::new();
}
message
.content
.iter()
@@ -3752,7 +3756,9 @@ impl Agent {
if let Some(final_output_tool) = self.final_output_tool.lock().await.as_ref() {
if final_output_tool.final_output.is_none() {
warn!("Final output tool has not been called yet. Continuing agent loop.");
let message = Message::user().with_text(FINAL_OUTPUT_CONTINUATION_MESSAGE);
let message = Message::user()
.with_text(FINAL_OUTPUT_CONTINUATION_MESSAGE)
.agent_only();
messages_to_add.push(message.clone());
yield AgentEvent::Message(message);
} else {
@@ -4506,6 +4512,25 @@ mod tests {
);
}
#[test]
fn test_project_message_skips_agent_only_text_message() {
let turn = TurnRuntime::new(
"turn-hidden",
"session-1",
"thread-1",
Some("隐藏内部提示".to_string()),
None,
);
let mut projector = TurnItemRuntimeProjector::new(&turn);
let message = Message::user()
.with_text("internal continuation")
.agent_only();
let events = projector.project_agent_event(&AgentEvent::Message(message));
assert!(events.is_empty(), "agent-only 消息不应再投影到用户可见事件流");
}
#[test]
fn test_project_tool_response_emits_file_artifact_runtime_item() {
let turn = TurnRuntime::new(
@@ -6,7 +6,7 @@ use std::borrow::Cow;
pub const FINAL_OUTPUT_TOOL_NAME: &str = "StructuredOutput";
pub const FINAL_OUTPUT_CONTINUATION_MESSAGE: &str =
"You MUST call the `StructuredOutput` tool NOW with the structured final output for the user.";
"Do not use ToolSearch, WebSearch, or any other search tool. `StructuredOutput` is already available in the current tool list. Call `StructuredOutput` NOW with the final JSON object for the user.";
#[derive(Debug)]
pub struct FinalOutputTool {
@@ -16,7 +16,9 @@ pub const TOOL_GUIDELINES: &str = r#"# Tool usage policy
- Use specialized tools instead of bash commands when possible, as this provides a better user experience.
- NEVER use bash echo or other command-line tools to communicate thoughts, explanations, or instructions to the user.
- Use TaskCreate, TaskList, TaskGet, and TaskUpdate to track progress on multi-step work.
- Use ToolSearch to discover deferred extension tools, and use `select:<tool_name>` when you need to load a specific deferred tool into the active tool surface.
- Use ToolSearch only for deferred extension/MCP tools, and use exact names such as `select:Read,Edit,Grep` or `select:mcp__playwright__browser_click` when you need to load or confirm a tool. Do not use ToolSearch for already-visible native tools like Read, Write, Edit, Glob, Grep, Bash, WebFetch, or WebSearch.
- Do not search for native tools via aliases like `read_file`, `write_file`, `edit_file`, or `system`; call the actual tool names directly.
- If ToolSearch returns no matches, do not keep retrying with synonyms. Either call the already-visible native tools directly or explain that the deferred capability is unavailable.
- Use Config when the user asks to inspect or update supported runtime settings such as model selection or permission mode.
- Use Sleep instead of `Bash(sleep ...)` when you intentionally need to wait.
- Only use host-injected delegation tools when the tool schema explicitly exposes them."#;
@@ -9,8 +9,10 @@ problems using the tools in these extensions, and can interact with multiple at
If the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional
extensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the
extension_name. You should only enable extensions found from the search_available_extensions tool.
Use ToolSearch to discover deferred extension tools, and use `select:<tool_name>` when you need to load a specific
deferred tool into the active tool surface.
Use ToolSearch to discover deferred extension tools, and use exact names such as `select:Read,Edit,Grep` or
`select:mcp__playwright__browser_click` when you need to load a specific deferred tool into the active tool surface.
Do not keep retrying ToolSearch with synonyms like `read_file`, `write_file`, `edit_file`, or `system`. If ToolSearch
returns no matches, call already-visible native tools directly or report that the deferred capability is unavailable.
If Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load
new ones.
@@ -17,8 +17,10 @@ problems using the tools in these extensions, and can interact with multiple at
If the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional
extensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the
extension_name. You should only enable extensions found from the search_available_extensions tool.
Use ToolSearch to discover deferred extension tools, and use `select:<tool_name>` when you need to load a specific
deferred tool into the active tool surface.
Use ToolSearch to discover deferred extension tools, and use exact names such as `select:Read,Edit,Grep` or
`select:mcp__playwright__browser_click` when you need to load a specific deferred tool into the active tool surface.
Do not keep retrying ToolSearch with synonyms like `read_file`, `write_file`, `edit_file`, or `system`. If ToolSearch
returns no matches, call already-visible native tools directly or report that the deferred capability is unavailable.
If Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load
new ones.
@@ -30,7 +30,8 @@ Extensions allow other applications to provide context to aster. Extensions conn
You are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.
If the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional extensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the extension_name. You should only enable extensions found from the search_available_extensions tool.
Use ToolSearch to discover deferred extension tools, and use `select:<tool_name>` when you need to load a specific deferred tool into the active tool surface.
Use ToolSearch to discover deferred extension tools, and use exact names such as `select:Read,Edit,Grep` or `select:mcp__playwright__browser_click` when you need to load a specific deferred tool into the active tool surface.
Do not keep retrying ToolSearch with synonyms like `read_file`, `write_file`, `edit_file`, or `system`. If ToolSearch returns no matches, call already-visible native tools directly or report that the deferred capability is unavailable.
If Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load new ones.
{% if (extensions is defined) and extensions %}
@@ -15,6 +15,7 @@ use std::path::{Path, PathBuf};
use async_trait::async_trait;
use base64::{engine::general_purpose::STANDARD as BASE64, Engine};
use serde::{Deserialize, Serialize};
use serde_json::json;
use tracing::debug;
use super::{compute_content_hash, FileReadRecord, SharedFileReadHistory};
@@ -843,12 +844,12 @@ impl Tool for ReadTool {
}
fn description(&self) -> &str {
"Enhanced multimodal file reader with intelligent analysis capabilities. \
Supports text files (with syntax highlighting and language detection), \
images (with metadata and AI analysis hints), PDF files (with document processing), \
"Multimodal file reader. \
Text files are returned as direct line-numbered content by default, \
with an optional enhanced analysis mode when explicitly requested. \
Also supports images (with metadata and AI analysis hints), PDF files (with document processing), \
SVG files (with vector graphics analysis), and Jupyter notebooks (with computational analysis). \
Automatically detects file type and provides structured information optimized for AI processing. \
Aligned with the current multimodal file understanding surface."
Optimized for reliable file reading in agent workflows."
}
fn input_schema(&self) -> serde_json::Value {
@@ -868,6 +869,11 @@ impl Tool for ReadTool {
"type": "integer",
"description": "End line number (1-indexed, inclusive, for text files only)",
"minimum": 1
},
"text_output_mode": {
"type": "string",
"enum": ["plain", "enhanced"],
"description": "For text files only. `plain` returns direct line-numbered content and is the default. `enhanced` adds file-analysis headers and hints."
}
},
"required": ["path"]
@@ -923,11 +929,16 @@ impl Tool for ReadTool {
// Enhanced text file reading with intelligent analysis
let range = self.extract_line_range(&params);
let content = self.read_text_enhanced(path, range, context).await?;
let text_output_mode = self.extract_text_output_mode(&params);
let content = match text_output_mode {
TextOutputMode::Plain => self.read_text(path, range, context).await?,
TextOutputMode::Enhanced => self.read_text_enhanced(path, range, context).await?,
};
Ok(ToolResult::success(content)
.with_metadata("file_type", serde_json::json!("text"))
.with_metadata("analysis_type", serde_json::json!("enhanced_textual")))
.with_metadata("analysis_type", serde_json::json!("textual"))
.with_metadata("text_output_mode", json!(text_output_mode.as_str())))
}
async fn check_permissions(
@@ -977,6 +988,17 @@ impl ReadTool {
}
}
fn extract_text_output_mode(&self, params: &serde_json::Value) -> TextOutputMode {
match params
.get("text_output_mode")
.and_then(|value| value.as_str())
.unwrap_or("plain")
{
"enhanced" => TextOutputMode::Enhanced,
_ => TextOutputMode::Plain,
}
}
/// Read a text file with enhanced analysis capabilities
///
/// 增强版实现,对齐当前文本读取能力:
@@ -1281,6 +1303,21 @@ impl ReadTool {
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum TextOutputMode {
Plain,
Enhanced,
}
impl TextOutputMode {
fn as_str(self) -> &'static str {
match self {
Self::Plain => "plain",
Self::Enhanced => "enhanced",
}
}
}
// =============================================================================
// Unit Tests
// =============================================================================
@@ -1499,11 +1536,17 @@ mod tests {
let result = tool.execute(params, &context).await.unwrap();
assert!(result.is_success());
assert!(result.output.unwrap().contains("Hello, World!"));
let output = result.output.unwrap();
assert!(output.contains("1 | Hello, World!"));
assert!(!output.contains("[Enhanced Text Analysis:"));
assert_eq!(
result.metadata.get("file_type"),
Some(&serde_json::json!("text"))
);
assert_eq!(
result.metadata.get("text_output_mode"),
Some(&serde_json::json!("plain"))
);
}
#[tokio::test]
@@ -1535,6 +1578,31 @@ mod tests {
assert!(!output.contains("Line 5"));
}
#[tokio::test]
async fn test_tool_execute_text_with_enhanced_mode() {
let temp_dir = TempDir::new().unwrap();
let file_path = temp_dir.path().join("test.md");
fs::write(&file_path, "# Title\n\nBody").unwrap();
let tool = create_read_tool();
let context = create_test_context(temp_dir.path());
let params = serde_json::json!({
"path": file_path.to_str().unwrap(),
"text_output_mode": "enhanced"
});
let result = tool.execute(params, &context).await.unwrap();
assert!(result.is_success());
let output = result.output.unwrap();
assert!(output.contains("[Enhanced Text Analysis:"));
assert!(output.contains("File Content:"));
assert_eq!(
result.metadata.get("text_output_mode"),
Some(&serde_json::json!("enhanced"))
);
}
#[tokio::test]
async fn test_tool_execute_missing_path() {
let temp_dir = TempDir::new().unwrap();
@@ -18,6 +18,24 @@ use crate::agents::ExtensionManager;
const TOOL_SEARCH_TOOL_NAME: &str = "ToolSearch";
const TOOL_SURFACE_UPDATED_KEY: &str = "tool_surface_updated";
const BUILTIN_VISIBLE_NATIVE_TOOLS: &[(&str, &str)] = &[
("Read", "Read file contents with line-aware output."),
("Write", "Create or overwrite files."),
("Edit", "Apply targeted edits to existing files."),
("Glob", "Find files by path pattern."),
("Grep", "Search file contents by pattern."),
("Bash", "Run shell commands in the workspace."),
("WebFetch", "Fetch and read a specific URL."),
("WebSearch", "Search the web for current information."),
(
"StructuredOutput",
"Return the final JSON answer for the current turn without re-searching tools.",
),
(
"AskUserQuestion",
"Ask the user for clarification or missing information.",
),
];
#[derive(Debug, Clone, Deserialize)]
struct ToolSearchInput {
@@ -31,6 +49,8 @@ struct ToolSearchOutput {
matches: Vec<String>,
query: String,
total_deferred_tools: usize,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
notes: Vec<String>,
}
#[derive(Debug, Clone, PartialEq, Eq)]
@@ -95,6 +115,7 @@ async fn collect_tool_search_state(
.into_iter()
.map(|tool| tool.name.to_string())
.collect::<HashSet<_>>();
let mut visible_names = visible_names;
let mut all_searchable = Vec::with_capacity(all_tools.len());
let mut deferred_searchable = Vec::new();
@@ -112,6 +133,8 @@ async fn collect_tool_search_state(
all_searchable.push(searchable);
}
append_builtin_visible_native_tools(&mut all_searchable, &mut visible_names);
Ok(ToolSearchState {
all_tools: all_searchable,
deferred_tools: deferred_searchable,
@@ -119,6 +142,25 @@ async fn collect_tool_search_state(
})
}
fn append_builtin_visible_native_tools(
all_tools: &mut Vec<SearchableTool>,
visible_names: &mut HashSet<String>,
) {
for (name, description) in BUILTIN_VISIBLE_NATIVE_TOOLS {
visible_names.insert((*name).to_string());
if all_tools
.iter()
.any(|tool| tool.name.eq_ignore_ascii_case(name))
{
continue;
}
all_tools.push(SearchableTool {
name: (*name).to_string(),
description: (*description).to_string(),
});
}
}
fn parse_select_query(query: &str) -> Option<Vec<String>> {
let prefix = "select:";
let actual_prefix = query.get(..prefix.len())?;
@@ -147,16 +189,24 @@ fn resolve_selected_tools(
for requested_name in requested {
let maybe_match = deferred_tools
.iter()
.find(|tool| tool.name.eq_ignore_ascii_case(requested_name))
.filter_map(|tool| {
select_match_rank(&tool.name, requested_name).map(|rank| (rank, &tool.name))
})
.max_by(|left, right| left.0.cmp(&right.0).then_with(|| right.1.cmp(left.1)))
.map(|(_, name)| name)
.or_else(|| {
all_tools
.iter()
.find(|tool| tool.name.eq_ignore_ascii_case(requested_name))
.filter_map(|tool| {
select_match_rank(&tool.name, requested_name).map(|rank| (rank, &tool.name))
})
.max_by(|left, right| left.0.cmp(&right.0).then_with(|| right.1.cmp(left.1)))
.map(|(_, name)| name)
});
if let Some(tool) = maybe_match {
if !found.iter().any(|existing| existing == &tool.name) {
found.push(tool.name.clone());
if !found.iter().any(|existing| existing == tool) {
found.push(tool.clone());
}
}
}
@@ -172,6 +222,69 @@ fn tool_search_lookup_key(value: &str) -> String {
.collect()
}
fn native_tool_search_aliases(name: &str) -> &'static [&'static str] {
match tool_search_lookup_key(name).as_str() {
"read" | "readtool" => &[
"read_file",
"read file",
"open file",
"workspace file",
"project file",
],
"write" | "writetool" => &[
"write_file",
"write file",
"create_file",
"create file",
"save file",
"workspace file",
"project file",
],
"edit" | "edittool" => &[
"edit_file",
"edit file",
"modify file",
"patch file",
"workspace file",
"project file",
],
"glob" | "globtool" => &[
"find_files",
"find files",
"file_search",
"list files",
"path search",
],
"grep" | "greptool" => &[
"search_files",
"search files",
"search in files",
"content search",
"text search",
],
"bash" | "bashtool" => &[
"system",
"shell",
"terminal",
"run command",
"command execution",
],
"webfetch" | "webfetchtool" => &["fetch url", "fetch page", "read url", "web reader"],
"websearch" | "websearchtool" => &["search web", "internet search", "web search"],
"structuredoutput" | "syntheticoutputtool" => &[
"structured output",
"final output",
"final output tool",
"final response",
],
"askuserquestion" | "askuserquestiontool" => {
&["request_user_input", "ask user", "user input"]
}
"toolsearch" | "toolsearchtool" => &["tool lookup", "search tools", "find tool"],
_ => &[],
}
}
fn parse_tool_name(name: &str) -> ParsedToolName {
let is_mcp = name.starts_with("mcp__");
let normalized = if is_mcp {
@@ -269,11 +382,85 @@ fn tool_search_exact_match(name: &str, query: &str) -> bool {
}
let parsed = parse_tool_name(name);
parsed.inner_name.as_deref().is_some_and(|inner_name| {
if parsed.inner_name.as_deref().is_some_and(|inner_name| {
inner_name == query_lower || tool_search_lookup_key(inner_name) == query_key
}) {
return true;
}
let query_parts = split_identifier_parts(&query_lower);
if parsed.inner_name.as_deref().is_some_and(|inner_name| {
let inner_parts = split_identifier_parts(inner_name);
inner_parts.len() >= 2
&& (suffix_identifier_match(&inner_parts, &query_parts)
|| suffix_identifier_match(&query_parts, &inner_parts))
}) {
return true;
}
native_tool_search_aliases(name).iter().any(|alias| {
alias.eq_ignore_ascii_case(&query_lower)
|| (!query_key.is_empty() && tool_search_lookup_key(alias) == query_key)
})
}
fn suffix_identifier_match(parts: &[String], suffix: &[String]) -> bool {
!suffix.is_empty()
&& suffix.len() <= parts.len()
&& parts[(parts.len() - suffix.len())..]
.iter()
.zip(suffix.iter())
.all(|(left, right)| left == right)
}
fn select_match_rank(name: &str, query: &str) -> Option<i32> {
let query_lower = query.trim().to_ascii_lowercase();
if query_lower.is_empty() {
return None;
}
if name.eq_ignore_ascii_case(&query_lower) {
return Some(500);
}
let query_key = tool_search_lookup_key(&query_lower);
if !query_key.is_empty() && tool_search_lookup_key(name) == query_key {
return Some(450);
}
let parsed = parse_tool_name(name);
if let Some(inner_name) = parsed.inner_name.as_deref() {
if inner_name == query_lower {
return Some(420);
}
if !query_key.is_empty() && tool_search_lookup_key(inner_name) == query_key {
return Some(400);
}
let query_parts = split_identifier_parts(&query_lower);
let inner_parts = split_identifier_parts(inner_name);
let looks_like_identifier =
query_lower.contains('_') || query_lower.contains('-') || query_lower.contains(' ');
if looks_like_identifier
&& inner_parts.len() >= 2
&& query_parts.len() >= 2
&& (suffix_identifier_match(&inner_parts, &query_parts)
|| suffix_identifier_match(&query_parts, &inner_parts))
{
return Some(360);
}
}
if native_tool_search_aliases(name).iter().any(|alias| {
alias.eq_ignore_ascii_case(&query_lower)
|| (!query_key.is_empty() && tool_search_lookup_key(alias) == query_key)
}) {
return Some(350);
}
None
}
fn score_searchable_tool(
tool: &SearchableTool,
scoring_terms: &[&str],
@@ -343,37 +530,54 @@ fn score_query_match(
}
}
let query_terms = query_lower
let raw_query_terms = query_lower
.split_whitespace()
.filter(|term| !term.is_empty())
.collect::<Vec<_>>();
if query_terms.is_empty() {
if raw_query_terms.is_empty() {
return Vec::new();
}
let mut required_terms = Vec::new();
let mut optional_terms = Vec::new();
for term in &query_terms {
if let Some(required_term) = term.strip_prefix('+') {
if !required_term.is_empty() {
required_terms.push(required_term);
continue;
}
for term in &raw_query_terms {
let (required, normalized_term) = if let Some(required_term) = term.strip_prefix('+') {
(true, required_term)
} else {
(false, *term)
};
if normalized_term.is_empty() {
continue;
}
optional_terms.push(*term);
let split_terms = split_identifier_parts(normalized_term);
let target = if required {
&mut required_terms
} else {
&mut optional_terms
};
if split_terms.is_empty() {
target.push(normalized_term.to_string());
} else {
target.extend(split_terms);
}
}
if required_terms.is_empty() && optional_terms.is_empty() {
return Vec::new();
}
let scoring_terms = if required_terms.is_empty() {
query_terms.clone()
optional_terms.clone()
} else {
required_terms
.iter()
.copied()
.chain(optional_terms.iter().copied())
.cloned()
.chain(optional_terms.iter().cloned())
.collect::<Vec<_>>()
};
let term_patterns = compile_term_patterns(&scoring_terms);
let scoring_term_refs = scoring_terms.iter().map(String::as_str).collect::<Vec<_>>();
let term_patterns = compile_term_patterns(&scoring_term_refs);
let mut scored = Vec::new();
for tool in deferred_tools {
@@ -381,7 +585,7 @@ fn score_query_match(
let lower_description = tool.description.to_lowercase();
let required_matches = required_terms.iter().all(|term| {
let Some(pattern) = term_patterns.get(*term) else {
let Some(pattern) = term_patterns.get(term.as_str()) else {
return false;
};
parsed
@@ -395,7 +599,7 @@ fn score_query_match(
continue;
}
let score = score_searchable_tool(tool, &scoring_terms, &term_patterns);
let score = score_searchable_tool(tool, &scoring_term_refs, &term_patterns);
if score == 0 {
continue;
@@ -414,6 +618,29 @@ fn pretty_json<T: Serialize>(value: &T) -> Result<String, ToolError> {
})
}
fn build_tool_search_notes(query: &str, matches: &[String]) -> Vec<String> {
if !matches.is_empty() {
return Vec::new();
}
let trimmed = query.trim();
if trimmed.is_empty() {
return Vec::new();
}
let mut notes = Vec::new();
if parse_select_query(trimmed).is_some() {
notes.push(
"未命中任何工具。不要继续用同义词反复重试;优先直接调用当前已可见的原生工具,如 Read / Write / Edit / Glob / Grep / Bash / WebFetch / WebSearch / StructuredOutput。".to_string(),
);
} else {
notes.push(
"未命中任何 deferred 工具。若你需要文件、命令、网页或最终答复能力,请直接调用当前已可见的 Read / Write / Edit / Glob / Grep / Bash / WebFetch / WebSearch / StructuredOutput,而不是继续用 ToolSearch 改写同义词。".to_string(),
);
}
notes
}
fn build_tool_search_result(
output: &ToolSearchOutput,
tool_surface_updated: bool,
@@ -423,6 +650,7 @@ fn build_tool_search_result(
.with_metadata("matches", json!(&output.matches))
.with_metadata("query", json!(&output.query))
.with_metadata("total_deferred_tools", json!(output.total_deferred_tools))
.with_metadata("notes", json!(&output.notes))
.with_metadata(TOOL_SURFACE_UPDATED_KEY, json!(tool_surface_updated)))
}
@@ -433,7 +661,7 @@ impl Tool for ToolSearchTool {
}
fn description(&self) -> &str {
"Searches deferred tools by keyword or select:<tool_a,tool_b> so their schemas can be loaded into the active tool surface."
"Fetches full schema definitions for deferred extension/MCP tools so they can be called. Use select:<tool_name> for direct selection, or keywords like \"browser click\" / \"+playwright click\". Do not use ToolSearch for already-visible native tools such as Read, Write, Edit, Glob, Grep, or StructuredOutput."
}
fn input_schema(&self) -> serde_json::Value {
@@ -442,7 +670,7 @@ impl Tool for ToolSearchTool {
"properties": {
"query": {
"type": "string",
"description": "Query to find deferred tools. Use select:<tool_name> for direct selection, or keywords to search."
"description": "Query to find deferred extension/MCP tools. Use select:<tool_name>[,<tool_name>] for direct selection, or keywords like browser click / +playwright click. Do not use this for already-visible native tools such as Read/Write/Edit/Glob/Grep/StructuredOutput."
},
"max_results": {
"type": "number",
@@ -482,9 +710,20 @@ impl Tool for ToolSearchTool {
let mut tool_surface_updated = false;
let mut total_deferred_tools = state_before.deferred_tools.len();
if !matches.is_empty() {
let deferred_match_names = matches
.iter()
.filter(|name| {
state_before
.deferred_tools
.iter()
.any(|tool| tool.name.eq_ignore_ascii_case(name))
})
.cloned()
.collect::<Vec<_>>();
if !deferred_match_names.is_empty() {
extension_manager
.load_deferred_tools(&matches)
.load_deferred_tools(&deferred_match_names)
.await
.map_err(map_extension_error)?;
@@ -495,6 +734,7 @@ impl Tool for ToolSearchTool {
return build_tool_search_result(
&ToolSearchOutput {
notes: build_tool_search_notes(query, &matches),
matches,
query: query.to_string(),
total_deferred_tools,
@@ -511,6 +751,7 @@ impl Tool for ToolSearchTool {
build_tool_search_result(
&ToolSearchOutput {
notes: build_tool_search_notes(query, &matches),
matches,
query: query.to_string(),
total_deferred_tools: state_before.deferred_tools.len(),
@@ -568,6 +809,62 @@ mod tests {
assert_eq!(matches, vec!["beta__tool".to_string()]);
}
#[test]
fn test_resolve_selected_tools_matches_inner_prefixed_tool_name() {
let deferred = vec![searchable(
"mcp__lime-browser__browser_file_upload",
"upload a file through the browser",
)];
let all = deferred.clone();
let matches = resolve_selected_tools(&["browser_file_upload".to_string()], &deferred, &all);
assert_eq!(
matches,
vec!["mcp__lime-browser__browser_file_upload".to_string()]
);
}
#[test]
fn test_resolve_selected_tools_matches_native_alias_for_visible_tool() {
let deferred = Vec::new();
let all = vec![searchable("Read", "read a file")];
let matches = resolve_selected_tools(&["read_file".to_string()], &deferred, &all);
assert_eq!(matches, vec!["Read".to_string()]);
}
#[test]
fn test_resolve_selected_tools_matches_system_alias_for_bash() {
let deferred = Vec::new();
let all = vec![searchable("Bash", "run shell commands")];
let matches = resolve_selected_tools(&["system".to_string()], &deferred, &all);
assert_eq!(matches, vec!["Bash".to_string()]);
}
#[test]
fn test_resolve_selected_tools_normalizes_server_prefix_variants() {
let deferred = vec![searchable(
"mcp__lime-browser__browser_file_upload",
"upload a file through the browser",
)];
let all = deferred.clone();
let matches = resolve_selected_tools(
&["mcp__lime_browser__browser_file_upload".to_string()],
&deferred,
&all,
);
assert_eq!(
matches,
vec!["mcp__lime-browser__browser_file_upload".to_string()]
);
}
#[test]
fn test_score_query_match_prefers_deferred_tool_keyword_hits() {
let deferred = vec![
@@ -613,6 +910,22 @@ mod tests {
assert_eq!(matches, vec!["BrowserClick".to_string()]);
}
#[test]
fn test_score_query_match_splits_identifier_queries() {
let deferred = vec![searchable(
"mcp__lime-browser__workspace_read_file",
"read a file from the browser workspace",
)];
let all = deferred.clone();
let matches = score_query_match("read_file", &deferred, &all);
assert_eq!(
matches,
vec!["mcp__lime-browser__workspace_read_file".to_string()]
);
}
#[test]
fn test_score_query_match_supports_prefixed_tool_prefix_queries() {
let deferred = vec![
@@ -639,6 +952,24 @@ mod tests {
"browser_click"
));
assert!(tool_search_exact_match("BrowserClick", "browser_click"));
assert!(tool_search_exact_match("Read", "read_file"));
assert!(tool_search_exact_match("Bash", "system"));
}
#[test]
fn test_tool_search_exact_match_does_not_match_generic_tool_suffix() {
assert!(!tool_search_exact_match("alpha__tool", "beta__tool"));
}
#[test]
fn test_select_match_rank_supports_server_prefixed_identifier_variant() {
assert_eq!(
select_match_rank(
"mcp__playwright__browser_file_upload",
"playwright_browser_file_upload"
),
Some(360)
);
}
#[test]
@@ -660,6 +991,7 @@ mod tests {
matches: vec!["alpha__tool".to_string()],
query: "select:alpha__tool".to_string(),
total_deferred_tools: 3,
notes: Vec::new(),
};
let result = build_tool_search_result(&output, true).unwrap();
@@ -669,4 +1001,48 @@ mod tests {
Some(&json!(true))
);
}
#[test]
fn test_build_tool_search_notes_warns_against_retry_loops() {
let notes = build_tool_search_notes("select:unknown_tool", &[]);
assert_eq!(notes.len(), 1);
assert!(notes[0].contains("不要继续用同义词反复重试"));
}
#[test]
fn test_append_builtin_visible_native_tools_adds_core_native_tools_once() {
let mut all_tools = vec![searchable("Read", "existing read tool")];
let mut visible_names = HashSet::new();
append_builtin_visible_native_tools(&mut all_tools, &mut visible_names);
assert!(visible_names.contains("Read"));
assert!(visible_names.contains("Write"));
assert!(visible_names.contains("StructuredOutput"));
assert_eq!(
all_tools
.iter()
.filter(|tool| tool.name.eq_ignore_ascii_case("Read"))
.count(),
1
);
assert!(all_tools
.iter()
.any(|tool| tool.name.eq_ignore_ascii_case("Write")));
assert!(all_tools
.iter()
.any(|tool| tool.name.eq_ignore_ascii_case("StructuredOutput")));
}
#[test]
fn test_score_query_match_can_resolve_structured_output_alias() {
let matches = score_query_match(
"final output tool",
&[],
&[searchable("StructuredOutput", "return the final JSON answer")],
);
assert_eq!(matches, vec!["StructuredOutput".to_string()]);
}
}
+85 -14
View File
@@ -212,6 +212,15 @@ fn native_tool_search_aliases(name: &str) -> &'static [&'static str] {
"content search",
"text search",
],
"bash" | "bashtool" => &[
"system",
"shell",
"terminal",
"run command",
"command execution",
],
"webfetch" | "webfetchtool" => &["fetch url", "fetch page", "read url", "web reader"],
"websearch" | "websearchtool" => &["search web", "internet search", "web search"],
"askuserquestion" | "askuserquestiontool" => {
&["request_user_input", "ask user", "user input"]
}
@@ -296,6 +305,15 @@ fn compile_tool_search_term_patterns(terms: &[&str]) -> HashMap<String, Regex> {
patterns
}
fn suffix_identifier_match(parts: &[String], suffix: &[String]) -> bool {
!suffix.is_empty()
&& suffix.len() <= parts.len()
&& parts[(parts.len() - suffix.len())..]
.iter()
.zip(suffix.iter())
.all(|(left, right)| left == right)
}
pub fn tool_search_exact_match(name: &str, query: &str) -> bool {
let query_lower = query.trim().to_ascii_lowercase();
if query_lower.is_empty() {
@@ -318,6 +336,16 @@ pub fn tool_search_exact_match(name: &str, query: &str) -> bool {
return true;
}
let query_parts = split_tool_search_identifier(&query_lower);
if parsed.inner_name.as_deref().is_some_and(|inner_name| {
let inner_parts = split_tool_search_identifier(inner_name);
inner_parts.len() >= 2
&& (suffix_identifier_match(&inner_parts, &query_parts)
|| suffix_identifier_match(&query_parts, &inner_parts))
}) {
return true;
}
native_tool_search_aliases(name).iter().any(|alias| {
alias.eq_ignore_ascii_case(&query_lower)
|| (!query_key.is_empty() && tool_search_lookup_key(alias) == query_key)
@@ -339,36 +367,54 @@ pub fn score_tool_match(name: &str, description: &str, tags: &[String], query: &
return 160;
}
let query_terms = query
let raw_query_terms = query
.split_whitespace()
.filter(|term| !term.is_empty())
.collect::<Vec<_>>();
if query_terms.is_empty() {
if raw_query_terms.is_empty() {
return 0;
}
let mut required_terms = Vec::new();
let mut optional_terms = Vec::new();
for term in &query_terms {
if let Some(required_term) = term.strip_prefix('+') {
if !required_term.is_empty() {
required_terms.push(required_term);
continue;
}
for term in &raw_query_terms {
let (required, normalized_term) = if let Some(required_term) = term.strip_prefix('+') {
(true, required_term)
} else {
(false, *term)
};
if normalized_term.is_empty() {
continue;
}
optional_terms.push(*term);
let split_terms = split_tool_search_identifier(normalized_term);
let target = if required {
&mut required_terms
} else {
&mut optional_terms
};
if split_terms.is_empty() {
target.push(normalized_term.to_string());
} else {
target.extend(split_terms);
}
}
if required_terms.is_empty() && optional_terms.is_empty() {
return 0;
}
let scoring_terms = if required_terms.is_empty() {
query_terms.clone()
optional_terms.clone()
} else {
required_terms
.iter()
.copied()
.chain(optional_terms.iter().copied())
.cloned()
.chain(optional_terms.iter().cloned())
.collect::<Vec<_>>()
};
let term_patterns = compile_tool_search_term_patterns(&scoring_terms);
let scoring_term_refs = scoring_terms.iter().map(String::as_str).collect::<Vec<_>>();
let term_patterns = compile_tool_search_term_patterns(&scoring_term_refs);
let parsed = parse_tool_search_name(name);
let description_lc = description.to_ascii_lowercase();
let aliases = native_tool_search_aliases(name)
@@ -385,7 +431,8 @@ pub fn score_tool_match(name: &str, description: &str, tags: &[String], query: &
.collect::<Vec<_>>();
let required_matches = required_terms.iter().all(|term| {
let Some(pattern) = term_patterns.get(*term) else {
let term = term.as_str();
let Some(pattern) = term_patterns.get(term) else {
return false;
};
parsed
@@ -408,6 +455,7 @@ pub fn score_tool_match(name: &str, description: &str, tags: &[String], query: &
let mut score = 0;
for term in scoring_terms {
let term = term.as_str();
let Some(pattern) = term_patterns.get(term) else {
continue;
};
@@ -786,6 +834,29 @@ mod tests {
"mcp__playwright__browser_click",
"browser_click"
));
assert!(tool_search_exact_match("Bash", "system"));
}
#[test]
fn test_tool_search_exact_match_does_not_match_generic_tool_suffix() {
assert!(!tool_search_exact_match("alpha__tool", "beta__tool"));
}
#[test]
fn test_score_tool_match_splits_identifier_queries() {
let exact = score_tool_match(
"mcp__playwright__browser_click",
"Click inside browser",
&["browser".to_string()],
"browser_click",
);
let partial = score_tool_match(
"mcp__playwright__browser_hover",
"Hover inside browser",
&["browser".to_string()],
"browser_click",
);
assert!(exact > partial);
}
#[test]
@@ -1,11 +1,40 @@
async (args, helpers) => {
const ARTICLE_ROOT_SELECTOR = '[data-testid="twitterArticleReadView"]';
const ARTICLE_TITLE_SELECTOR = '[data-testid="twitter-article-title"]';
const ARTICLE_CONTENT_SELECTOR =
'[data-testid="longformRichTextComponent"] [data-contents="true"], [data-testid="longformRichTextComponent"]';
const ARTICLE_CONTENT_SELECTORS = [
'[data-testid="twitterArticleRichTextView"]',
'[data-testid="longformRichTextComponent"]',
];
const ARTICLE_CONTENT_SELECTOR = ARTICLE_CONTENT_SELECTORS.join(", ");
const IMAGE_SELECTOR = '[data-testid="tweetPhoto"] img';
const CODE_BLOCK_SELECTOR =
'[data-testid="markdown-code-block"], [data-testid="prism-code-block"], pre';
const IMAGE_BLOCK_SELECTOR =
'[data-testid="tweetPhoto"], figure, [role="img"], [style*="background-image"]';
const CODE_BLOCK_SELECTOR = [
'[data-testid="markdown-code-block"]',
'[data-testid="prism-code-block"]',
'[data-testid*="code"]',
'[data-testid*="Code"]',
'.prism-code',
'.react-syntax-highlighter-line-number',
'[class*="syntax-highlighter"]',
'[class*="code-block"]',
'[class*="CodeBlock"]',
'[class*="language-"]',
'pre',
].join(", ");
const IGNORED_BLOCK_SELECTOR = [
ARTICLE_TITLE_SELECTOR,
"script",
"style",
"noscript",
"svg",
"button",
"[role='button']",
"time",
"[data-testid='UserAvatar-Container']",
"[data-testid='User-Name']",
"[data-testid='socialContext']",
].join(", ");
function normalizeText(value) {
return String(value || "")
@@ -23,8 +52,17 @@ async (args, helpers) => {
.trim();
}
function normalizeCodeText(value) {
return String(value || "")
.replace(/\r\n/g, "\n")
.replace(/\u00a0/g, " ")
.replace(/\n{3,}/g, "\n\n")
.replace(/^\n+/, "")
.replace(/\n+$/, "");
}
function escapeMarkdownText(value) {
return String(value || "").replace(/([\\`*_{}\[\]()#+\-.!|>])/g, "\\$1");
return String(value || "").replace(/([\\`*_{}\[\]()#+\-!|>])/g, "\\$1");
}
function normalizeArticleUrl(rawUrl) {
@@ -58,6 +96,9 @@ async (args, helpers) => {
}
function normalizeImageUrl(rawUrl) {
if (!normalizeText(rawUrl)) {
return "";
}
try {
const url = new URL(rawUrl, location.href);
if (url.hostname.includes("pbs.twimg.com")) {
@@ -90,11 +131,19 @@ async (args, helpers) => {
const pictureSources = picture
? Array.from(picture.querySelectorAll("source"))
: [];
const candidate = [
imageElement.currentSrc,
const explicitSourceCandidates = [
imageElement.getAttribute("src"),
imageElement.getAttribute("data-src"),
imageElement.getAttribute("data-image-url"),
imageElement.getAttribute("srcset"),
...pictureSources.map((source) => source.getAttribute("srcset")),
].filter((value) => normalizeText(value));
const candidate = [
explicitSourceCandidates.length > 0 ||
normalizeText(imageElement.currentSrc) !== location.href
? imageElement.currentSrc
: "",
...explicitSourceCandidates,
resolveSrcsetUrl(imageElement.getAttribute("srcset")),
...pictureSources.map((source) =>
resolveSrcsetUrl(source.getAttribute("srcset")),
@@ -148,8 +197,7 @@ async (args, helpers) => {
};
}
function registerImage(state, imageElement) {
const rawUrl = resolveImageSource(imageElement);
function registerImageUrl(state, rawUrl, altText, suggestedFileName) {
const normalizedUrl = normalizeImageUrl(rawUrl);
if (!normalizedUrl) {
return "";
@@ -159,16 +207,319 @@ async (args, helpers) => {
state.imageUrls.add(normalizedUrl);
state.images.push({
url: normalizedUrl,
alt: normalizeText(imageElement.getAttribute("alt")) || undefined,
suggested_file_name: resolveSuggestedImageName(
normalizedUrl,
state.images.length,
),
alt: normalizeText(altText) || undefined,
suggested_file_name:
normalizeText(suggestedFileName) ||
resolveSuggestedImageName(normalizedUrl, state.images.length),
});
}
const altText = normalizeText(imageElement.getAttribute("alt")) || "插图";
return `![${escapeMarkdownText(altText)}](${normalizedUrl})`;
const markdownAltText = normalizeText(altText) || "插图";
return `![${escapeMarkdownText(markdownAltText)}](${normalizedUrl})`;
}
function registerImage(state, imageElement) {
const rawUrl = resolveImageSource(imageElement);
return registerImageUrl(
state,
rawUrl,
imageElement.getAttribute("alt"),
undefined,
);
}
function extractBackgroundImageUrl(value) {
const matches = Array.from(
String(value || "").matchAll(/url\((['"]?)(.*?)\1\)/g),
);
for (const match of matches) {
const candidate = normalizeImageUrl(match[2]);
if (candidate) {
return candidate;
}
}
return "";
}
function resolveBackgroundImageSource(element) {
const inlineStyle = element.getAttribute("style") || "";
const inlineUrl = extractBackgroundImageUrl(inlineStyle);
if (inlineUrl) {
return inlineUrl;
}
if (typeof getComputedStyle === "function") {
const computedUrl = extractBackgroundImageUrl(
getComputedStyle(element).backgroundImage,
);
if (computedUrl) {
return computedUrl;
}
}
return "";
}
function resolveElementAltText(element) {
return (
normalizeText(element.getAttribute("aria-label")) ||
normalizeText(element.getAttribute("title")) ||
normalizeText(element.getAttribute("data-alt")) ||
normalizeText(element.getAttribute("alt")) ||
""
);
}
function resolveElementDimension(element, property) {
const rect =
typeof element.getBoundingClientRect === "function"
? element.getBoundingClientRect()
: null;
const rectValue = property === "width" ? rect?.width : rect?.height;
if (Number.isFinite(rectValue) && rectValue > 0) {
return rectValue;
}
const attributeValue = Number(element.getAttribute(property));
if (Number.isFinite(attributeValue) && attributeValue > 0) {
return attributeValue;
}
const styleValue = Number.parseFloat(
property === "width" ? element.style?.width : element.style?.height,
);
if (Number.isFinite(styleValue) && styleValue > 0) {
return styleValue;
}
return 0;
}
function isLikelyContentImageElement(element) {
const width = resolveElementDimension(element, "width");
const height = resolveElementDimension(element, "height");
if (width === 0 && height === 0) {
return true;
}
return width >= 48 && height >= 48;
}
function registerBackgroundImage(state, element) {
const rawUrl = resolveBackgroundImageSource(element);
if (!rawUrl || !isLikelyContentImageElement(element)) {
return "";
}
return registerImageUrl(
state,
rawUrl,
resolveElementAltText(element),
undefined,
);
}
function collectImageMarkdown(element, state) {
const markdown = [];
const seenUrls = new Set();
const pushMarkdown = (value) => {
if (!value) {
return;
}
const urlMatch = value.match(/\]\((.+)\)$/);
const dedupeKey = urlMatch?.[1] || value;
if (seenUrls.has(dedupeKey)) {
return;
}
seenUrls.add(dedupeKey);
markdown.push(value);
};
if (element.tagName === "IMG") {
pushMarkdown(registerImage(state, element));
}
Array.from(element.querySelectorAll("img")).forEach((image) => {
pushMarkdown(registerImage(state, image));
});
const backgroundCandidates = [
element,
...Array.from(
element.querySelectorAll("[role='img'], [style*='background-image']"),
),
];
backgroundCandidates.forEach((candidate) => {
pushMarkdown(registerBackgroundImage(state, candidate));
});
return markdown;
}
function hasNestedSerializableContent(element) {
if (!(element instanceof Element)) {
return false;
}
return Boolean(
element.querySelector(
[
"img",
"[role='img']",
"[style*='background-image']",
"pre",
"[data-language]",
"[data-testid*='code']",
"[data-testid*='Code']",
".prism-code",
"[class*='syntax-highlighter']",
"[class*='code-block']",
"[class*='CodeBlock']",
"[class*='language-']",
].join(", "),
),
);
}
function shouldIgnoreElement(element) {
if (!(element instanceof Element)) {
return false;
}
if (
element.matches?.(ARTICLE_CONTENT_SELECTOR) ||
element.matches?.(IMAGE_SELECTOR) ||
element.matches?.(IMAGE_BLOCK_SELECTOR) ||
element.matches?.(CODE_BLOCK_SELECTOR) ||
element.tagName === "IMG" ||
hasNestedSerializableContent(element)
) {
return false;
}
return element.matches(IGNORED_BLOCK_SELECTOR);
}
function countCodeLikeLines(lines) {
return lines.filter((line) =>
/^\s*(#\s*[\w./-]+|---$|name:|description:|metadata:|Step \d+:|class |def |function |const |let |var |async |await |if |for |while |return |import |from |export |\{|\}|\[|\]|<\/?[a-z])/i.test(
line,
),
).length;
}
function looksLikeCodeBlock(element) {
if (!(element instanceof Element)) {
return false;
}
if (element.matches(CODE_BLOCK_SELECTOR)) {
return true;
}
const dataTestId = normalizeText(element.getAttribute("data-testid"));
if (/code|syntax|prism/i.test(dataTestId)) {
return true;
}
const className =
typeof element.className === "string"
? element.className
: element.getAttribute("class") || "";
if (/prism|syntax-highlighter|code-block|CodeBlock|highlight/i.test(className)) {
return true;
}
if (element.querySelector("pre, code, [data-language], [class*='language-']")) {
return true;
}
const rawText = normalizeCodeText(element.innerText || element.textContent || "");
if (!rawText) {
return false;
}
const lines = rawText.split("\n").filter((line) => line.trim().length > 0);
if (lines.length < 3) {
return false;
}
const codeLikeLines = countCodeLikeLines(lines);
const style = typeof getComputedStyle === "function" ? getComputedStyle(element) : null;
const fontFamily = normalizeText(style?.fontFamily);
const whiteSpace = normalizeText(style?.whiteSpace);
const looksMonospace = /mono|courier|menlo|monaco|consolas/i.test(fontFamily);
const preservesWhitespace = /pre|break-spaces/i.test(whiteSpace);
return preservesWhitespace || looksMonospace || codeLikeLines >= 2;
}
function serializeCodeBlock(element) {
const codeText = normalizeCodeText(element.innerText || element.textContent || "");
if (!codeText) {
return [];
}
const language = resolveCodeLanguage(element);
return [`\`\`\`${language}\n${codeText}\n\`\`\``];
}
function isSerializableArticleSegment(element) {
if (!(element instanceof Element)) {
return false;
}
if (shouldIgnoreElement(element) || element.matches?.(ARTICLE_TITLE_SELECTOR)) {
return false;
}
return (
element.matches?.(ARTICLE_CONTENT_SELECTOR) ||
element.matches?.(IMAGE_SELECTOR) ||
element.matches?.(IMAGE_BLOCK_SELECTOR) ||
looksLikeCodeBlock(element)
);
}
function collectArticleSegments(articleRoot, preferredRoot) {
const preferredRoots = [
...(preferredRoot ? [preferredRoot] : []),
...Array.from(articleRoot.querySelectorAll(ARTICLE_CONTENT_SELECTOR)).filter(
(element) => element !== preferredRoot,
),
];
const rawCandidates = Array.from(articleRoot.querySelectorAll("*")).filter(
(element) =>
isSerializableArticleSegment(element) &&
!preferredRoots.includes(element),
);
const orderedCandidates = [...preferredRoots, ...rawCandidates];
const segments = [];
for (const candidate of orderedCandidates) {
if (
segments.some(
(existing) =>
existing === candidate ||
existing.contains(candidate) ||
candidate.contains(existing),
)
) {
continue;
}
segments.push(candidate);
}
return segments.length > 0 ? segments : [preferredRoot || articleRoot];
}
function resolveArticleContentRoot(articleRoot) {
for (const selector of ARTICLE_CONTENT_SELECTORS) {
const matched = articleRoot.querySelector(selector);
if (matched) {
return matched;
}
}
return null;
}
function collectInlineMarkdown(node, state) {
@@ -189,6 +540,17 @@ async (args, helpers) => {
return registerImage(state, element);
}
if (element.matches?.(IMAGE_BLOCK_SELECTOR)) {
return collectImageMarkdown(element, state).join("\n\n");
}
if (
!hasMeaningfulText(element) &&
element.querySelector?.("img, [role='img'], [style*='background-image']")
) {
return collectImageMarkdown(element, state).join("\n\n");
}
if (element.matches?.(CODE_BLOCK_SELECTOR)) {
return "";
}
@@ -233,6 +595,21 @@ async (args, helpers) => {
return normalizeText(element.textContent || "").length > 0;
}
function hasRenderableMedia(element) {
if (!(element instanceof Element)) {
return false;
}
return (
element.matches?.(IMAGE_SELECTOR) ||
element.matches?.(IMAGE_BLOCK_SELECTOR) ||
element.tagName === "IMG" ||
Boolean(
element.querySelector("img, [role='img'], [style*='background-image']"),
)
);
}
function serializeList(element, state, depth = 0) {
const items = Array.from(element.children)
.filter((child) => child.tagName === "LI")
@@ -272,26 +649,16 @@ async (args, helpers) => {
return [];
}
if (element.matches(ARTICLE_TITLE_SELECTOR)) {
if (shouldIgnoreElement(element)) {
return [];
}
if (element.matches(CODE_BLOCK_SELECTOR)) {
const codeText = normalizeMultilineText(element.innerText || element.textContent || "");
if (!codeText) {
return [];
}
const language = resolveCodeLanguage(element);
return [`\`\`\`${language}\n${codeText}\n\`\`\``];
if (looksLikeCodeBlock(element)) {
return serializeCodeBlock(element);
}
if (element.matches(IMAGE_SELECTOR) || element.matches('[data-testid="tweetPhoto"], figure')) {
const images = element.matches(IMAGE_SELECTOR)
? [element]
: Array.from(element.querySelectorAll("img"));
return images
.map((image) => registerImage(state, image))
.filter(Boolean);
if (element.matches(IMAGE_SELECTOR) || element.matches(IMAGE_BLOCK_SELECTOR)) {
return collectImageMarkdown(element, state);
}
if (element.matches("ul,ol")) {
@@ -321,12 +688,15 @@ async (args, helpers) => {
if (child.matches(CODE_BLOCK_SELECTOR)) {
return true;
}
if (child.matches(IMAGE_SELECTOR) || child.matches('[data-testid="tweetPhoto"], figure')) {
if (child.matches(IMAGE_SELECTOR) || child.matches(IMAGE_BLOCK_SELECTOR)) {
return true;
}
if (child.matches("ul,ol,blockquote,h1,h2,h3,h4,h5,h6,hr")) {
return true;
}
if (looksLikeCodeBlock(child)) {
return true;
}
return child.children.length > 0 && !["A", "SPAN", "EM", "STRONG", "I", "B", "CODE"].includes(child.tagName);
});
@@ -334,7 +704,7 @@ async (args, helpers) => {
return directBlockChildren.flatMap((child) => serializeBlock(child, state));
}
if (!hasMeaningfulText(element) && !element.querySelector("img")) {
if (!hasMeaningfulText(element) && !hasRenderableMedia(element)) {
return [];
}
@@ -356,6 +726,170 @@ async (args, helpers) => {
return fallback;
}
function serializeArticleSegments(segments, state) {
const blocks = segments
.flatMap((segment) => {
if (segment.matches?.(ARTICLE_CONTENT_SELECTOR)) {
const serialized = serializeArticle(segment, state);
return serialized ? [serialized] : [];
}
return serializeBlock(segment, state);
})
.map((block) => normalizeMultilineText(block))
.filter(Boolean);
return blocks.join("\n\n");
}
function collectPotentialCodeBlockElements(root) {
const resolveNodeDepth = (element) => {
let depth = 0;
let current = element.parentElement;
while (current) {
depth += 1;
current = current.parentElement;
}
return depth;
};
const candidates = Array.from(root.querySelectorAll("*"))
.filter((element) => !shouldIgnoreElement(element) && looksLikeCodeBlock(element))
.sort((left, right) => {
const leftDepth = resolveNodeDepth(left);
const rightDepth = resolveNodeDepth(right);
return rightDepth - leftDepth;
});
const uniqueCandidates = [];
for (const candidate of candidates) {
if (
uniqueCandidates.some(
(existing) =>
existing === candidate ||
existing.contains(candidate) ||
candidate.contains(existing),
)
) {
continue;
}
uniqueCandidates.push(candidate);
}
return uniqueCandidates;
}
function collectFallbackCodeBlocks(root) {
const uniqueCandidates = collectPotentialCodeBlockElements(root);
return uniqueCandidates.flatMap((candidate) => serializeCodeBlock(candidate));
}
function collectPotentialMediaCarriers(root) {
if (!(root instanceof Element)) {
return [];
}
const carriers = [];
const seen = new Set();
const push = (element) => {
if (!(element instanceof Element) || seen.has(element)) {
return;
}
if (shouldIgnoreElement(element)) {
return;
}
seen.add(element);
carriers.push(element);
};
if (
root.matches?.(IMAGE_SELECTOR) ||
root.matches?.(IMAGE_BLOCK_SELECTOR) ||
root.tagName === "IMG"
) {
push(root);
}
Array.from(
root.querySelectorAll("img, [data-testid='tweetPhoto'], figure, [role='img'], [style*='background-image']"),
).forEach((element) => push(element));
return carriers;
}
function appendMissingBlocks(markdown, blocks) {
const normalizedMarkdown = normalizeMultilineText(markdown);
const appendedBlocks = [];
const appendedSet = new Set();
blocks.forEach((block) => {
const normalizedBlock = normalizeMultilineText(block);
if (!normalizedBlock) {
return;
}
if (
normalizedMarkdown.includes(normalizedBlock) ||
appendedSet.has(normalizedBlock)
) {
return;
}
appendedSet.add(normalizedBlock);
appendedBlocks.push(normalizedBlock);
});
if (appendedBlocks.length === 0) {
return normalizedMarkdown;
}
return [normalizedMarkdown, ...appendedBlocks].filter(Boolean).join("\n\n");
}
function countExtractedCodeBlocks(markdown) {
return Array.from(String(markdown || "").matchAll(/^```/gm)).length;
}
function buildWarmupSnapshot(articleRoot) {
return {
articleBlocks: articleRoot.querySelectorAll(ARTICLE_CONTENT_SELECTOR).length,
codeBlocks: collectPotentialCodeBlockElements(articleRoot).length,
mediaCarriers: collectPotentialMediaCarriers(articleRoot).length,
textLength: normalizeText(articleRoot.textContent || "").length,
};
}
async function warmupArticleContent(articleRoot) {
if (!(articleRoot instanceof Element)) {
return;
}
const initialScrollY = Math.round(window.scrollY);
await helpers.waitForDomStable?.({
root: articleRoot,
stableMs: 400,
timeoutMs: 2500,
});
await helpers.scrollUntilSettled?.({
root: articleRoot,
maxScrolls: 12,
delayMs: 350,
settleRounds: 2,
getSnapshot: () => buildWarmupSnapshot(articleRoot),
});
await helpers.waitForDomStable?.({
root: articleRoot,
stableMs: 500,
timeoutMs: 3000,
});
await helpers.waitForImagesReady?.(articleRoot, {
timeoutMs: 3200,
intervalMs: 200,
stableRounds: 2,
});
if (Math.round(window.scrollY) !== initialScrollY) {
window.scrollTo(0, initialScrollY);
await helpers.sleep?.(80);
}
}
const requestedUrl = String(args.url || "").trim();
if (requestedUrl && !normalizeArticleUrl(requestedUrl)) {
return {
@@ -385,11 +919,17 @@ async (args, helpers) => {
};
}
await warmupArticleContent(articleRoot);
const title =
normalizeText(articleRoot.querySelector(ARTICLE_TITLE_SELECTOR)?.textContent) ||
normalizeText(document.title.replace(/\s+on X.*$/i, ""));
const contentRoot = articleRoot.querySelector(ARTICLE_CONTENT_SELECTOR);
if (!contentRoot) {
const richTextRoot = await helpers.waitFor(
() => resolveArticleContentRoot(articleRoot),
5000,
250,
);
if (!richTextRoot) {
return {
ok: false,
error_code: helpers.looksLikeLoginWall() ? "auth_required" : "adapter_runtime_error",
@@ -398,7 +938,12 @@ async (args, helpers) => {
}
const state = createState();
const markdown = serializeArticle(contentRoot, state);
const articleSegments = collectArticleSegments(articleRoot, richTextRoot);
const fallbackCodeBlocks = collectFallbackCodeBlocks(articleRoot);
const potentialMediaCarriers = collectPotentialMediaCarriers(articleRoot);
let markdown = serializeArticleSegments(articleSegments, state);
markdown = appendMissingBlocks(markdown, fallbackCodeBlocks);
markdown = appendMissingBlocks(markdown, collectImageMarkdown(articleRoot, state));
if (!markdown) {
return {
ok: false,
@@ -407,6 +952,22 @@ async (args, helpers) => {
};
}
if (potentialMediaCarriers.length > 0 && state.images.length === 0) {
return {
ok: false,
error_code: "adapter_runtime_error",
error_message: "X 长文中的图片资源尚未完全加载,请稍后重试导出。",
};
}
if (fallbackCodeBlocks.length > 0 && countExtractedCodeBlocks(markdown) === 0) {
return {
ok: false,
error_code: "adapter_runtime_error",
error_message: "X 长文中的代码示例尚未完全加载,请稍后重试导出。",
};
}
const publishedAt =
articleRoot.querySelector("time[datetime]")?.getAttribute("datetime") ||
document.querySelector("time[datetime]")?.getAttribute("datetime") ||
@@ -0,0 +1,396 @@
// @vitest-environment jsdom
import fs from "node:fs";
import path from "node:path";
import { beforeEach, describe, expect, it } from "vitest";
const runnerSource = fs.readFileSync(
path.resolve(
process.cwd(),
"src-tauri/resources/site-adapters/bundled/scripts/x-article-export.js",
),
"utf8",
);
const xArticleExportRunner = Function(
'"use strict"; return (' +
runnerSource.trim().replace(/;\s*$/, "") +
");",
)() as (args: Record<string, unknown>, helpers: Record<string, unknown>) => Promise<{
ok: boolean;
data?: {
markdown?: string;
images?: Array<{ url?: string }>;
};
}>;
function createHelpers(
overrides: Partial<Record<string, (...args: unknown[]) => unknown>> = {},
) {
return {
sleep: async () => undefined,
absoluteUrl(value: string) {
try {
return new URL(value, "https://x.com/GoogleCloudTech/article/2033953579824758855").toString();
} catch {
return value;
}
},
async waitFor<T>(resolver: () => T) {
const value = resolver();
if (!value) {
throw new Error("waitFor 在测试夹具中未命中元素");
}
return value;
},
looksLikeLoginWall() {
return false;
},
waitForDomStable: async () => true,
scrollUntilSettled: async () => ({
scrolls: 0,
scrollY: 0,
scrollHeight: 0,
stableCount: 0,
}),
waitForImagesReady: async () => ({
total: 0,
ready: 0,
}),
...overrides,
};
}
describe("x/article-export adapter", () => {
beforeEach(() => {
document.body.innerHTML = "";
document.title = "GoogleCloudTech on X";
window.history.replaceState(
{},
"",
"/GoogleCloudTech/article/2033953579824758855",
);
});
it("能从更宽的文章容器中提取代码块和背景图图片", async () => {
document.body.innerHTML = `
<div data-testid="twitterArticleReadView">
<div class="article-body">
<div data-testid="twitter-article-title">5 Agent Skill design patterns every ADK developer should know</div>
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Intro paragraph before examples.</span></div>
</div>
</div>
<div
class="prism-code language-markdown"
data-testid="rich-code-block"
data-language="markdown"
style="white-space: pre-wrap; font-family: Menlo, monospace;"
># skills/api-expert/SKILL.md
---
name: api-expert
description: FastAPI development best practices and conventions.</div>
<figure
aria-label="Skill diagram"
style="width: 640px; height: 360px; background-image: url('https://pbs.twimg.com/media/skill-diagram?format=png&name=small');"
></figure>
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Outro paragraph after examples.</span></div>
</div>
</div>
</div>
</div>
`;
const result = await xArticleExportRunner(
{
url: "https://x.com/GoogleCloudTech/article/2033953579824758855",
},
createHelpers(),
);
expect(result.ok).toBe(true);
expect(result.data?.markdown).toContain("Intro paragraph before examples.");
expect(result.data?.markdown).toContain("```markdown");
expect(result.data?.markdown).toContain("# skills/api-expert/SKILL.md");
expect(result.data?.markdown).toContain("name: api-expert");
expect(result.data?.markdown).toContain("![Skill diagram]");
expect(result.data?.markdown).toContain("Outro paragraph after examples.");
expect(result.data?.images).toHaveLength(1);
expect(result.data?.images?.[0]?.url).toContain(
"https://pbs.twimg.com/media/skill-diagram?format=png&name=orig",
);
});
it("能按整篇文章顺序提取分散在不同包装层里的正文、代码块和图片", async () => {
document.body.innerHTML = `
<div data-testid="twitterArticleReadView">
<section class="title-wrap">
<div data-testid="twitter-article-title">Wrapped article export</div>
</section>
<section class="intro-wrap">
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Intro paragraph before the rich blocks.</span></div>
</div>
</div>
</section>
<section class="diagram-wrap">
<figure>
<img
alt="Pattern 1 diagram"
src="https://pbs.twimg.com/media/pattern-1-diagram?format=jpg&name=small"
/>
</figure>
</section>
<section class="code-wrap">
<div
class="rich-block-shell"
style="white-space: pre-wrap; font-family: Menlo, monospace;"
># skills/api-expert/SKILL.md
---
name: api-expert
description: FastAPI development best practices and conventions.
metadata:
pattern: tool-wrapper</div>
</section>
<section class="outro-wrap">
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Outro paragraph after the rich blocks.</span></div>
</div>
</div>
</section>
</div>
`;
const result = await xArticleExportRunner(
{
url: "https://x.com/GoogleCloudTech/article/2033953579824758855",
},
createHelpers(),
);
expect(result.ok).toBe(true);
expect(result.data?.markdown).toContain("Intro paragraph before the rich blocks.");
expect(result.data?.markdown).toContain("![Pattern 1 diagram]");
expect(result.data?.markdown).toContain("```");
expect(result.data?.markdown).toContain("# skills/api-expert/SKILL.md");
expect(result.data?.markdown).toContain("pattern: tool-wrapper");
expect(result.data?.markdown).toContain("Outro paragraph after the rich blocks.");
expect(result.data?.images).toHaveLength(1);
expect(result.data?.images?.[0]?.url).toContain(
"https://pbs.twimg.com/media/pattern-1-diagram?format=jpg&name=orig",
);
});
it("不会把包着真实内容的交互容器误判成噪音节点", async () => {
document.body.innerHTML = `
<div data-testid="twitterArticleReadView">
<div data-testid="twitter-article-title">Interactive wrapper export</div>
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Paragraph before interactive content.</span></div>
<div role="button" aria-label="Open example image">
<a href="https://x.com/GoogleCloudTech/status/1/photo/1">
<div
aria-label="Pattern overview"
style="width: 640px; height: 360px; background-image: url('https://pbs.twimg.com/media/pattern-overview?format=png&name=small');"
></div>
</a>
</div>
<div role="button" aria-label="Expand code sample">
<div
class="code-shell"
data-language="markdown"
style="white-space: pre-wrap; font-family: Menlo, monospace;"
># skills/report-generator/SKILL.md
---
name: report-generator
description: Generates structured technical reports in Markdown.</div>
</div>
<div><span>Paragraph after interactive content.</span></div>
</div>
</div>
</div>
`;
const result = await xArticleExportRunner(
{
url: "https://x.com/GoogleCloudTech/article/2033953579824758855",
},
createHelpers(),
);
expect(result.ok).toBe(true);
expect(result.data?.markdown).toContain("Paragraph before interactive content.");
expect(result.data?.markdown).toContain("![Pattern overview]");
expect(result.data?.markdown).toContain("```markdown");
expect(result.data?.markdown).toContain("# skills/report-generator/SKILL.md");
expect(result.data?.markdown).toContain("Paragraph after interactive content.");
expect(result.data?.images).toHaveLength(1);
expect(result.data?.images?.[0]?.url).toContain(
"https://pbs.twimg.com/media/pattern-overview?format=png&name=orig",
);
});
it("会在滚动预热后提取延迟挂载的技能代码块和图片", async () => {
document.body.innerHTML = `
<div data-testid="twitterArticleReadView">
<div data-testid="twitter-article-title">Warmup export</div>
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Paragraph before lazy content.</span></div>
<div id="lazy-slot"></div>
<div><span>Paragraph after lazy content.</span></div>
</div>
</div>
</div>
`;
let warmupCalled = false;
const result = await xArticleExportRunner(
{
url: "https://x.com/GoogleCloudTech/article/2033953579824758855",
},
createHelpers({
async scrollUntilSettled() {
warmupCalled = true;
const lazySlot = document.querySelector("#lazy-slot");
if (lazySlot) {
lazySlot.innerHTML = `
<div
class="lazy-code-shell"
data-language="markdown"
style="white-space: pre-wrap; font-family: Menlo, monospace;"
># skills/doc-pipeline/SKILL.md
---
name: doc-pipeline
description: Generates API documentation from source.</div>
<div
role="img"
aria-label="Pipeline diagram"
style="width: 640px; height: 360px; background-image: url('https://pbs.twimg.com/media/pipeline-diagram?format=png&name=small');"
></div>
`;
}
return {
scrolls: 3,
scrollY: 1200,
scrollHeight: 2400,
stableCount: 2,
};
},
}),
);
expect(warmupCalled).toBe(true);
expect(result.ok).toBe(true);
expect(result.data?.markdown).toContain("Paragraph before lazy content.");
expect(result.data?.markdown).toContain("```markdown");
expect(result.data?.markdown).toContain("# skills/doc-pipeline/SKILL.md");
expect(result.data?.markdown).toContain("![Pipeline diagram]");
expect(result.data?.images).toHaveLength(1);
});
it("能兼容真实 X 长文常见的 twitterArticleRichTextView 结构并保留封面图", async () => {
document.body.innerHTML = `
<div data-testid="twitterArticleReadView">
<div data-testid="twitter-article-title">Real X article DOM export</div>
<div data-testid="tweetPhoto">
<img
alt="Article cover"
src="https://pbs.twimg.com/media/article-cover?format=jpg&name=small"
/>
</div>
<div data-testid="twitterArticleRichTextView">
<div data-block="true">
<span data-text="true">Intro paragraph from the real DOM layout.</span>
</div>
<section data-block="true">
<div data-testid="tweetPhoto">
<img
alt="Skill overview"
src="https://pbs.twimg.com/media/skill-overview?format=png&name=small"
/>
</div>
</section>
<section data-block="true">
<div data-testid="markdown-code-block">
<code># skills/api-expert/SKILL.md
---
name: api-expert
description: FastAPI development best practices and conventions.
metadata:
pattern: tool-wrapper</code>
</div>
</section>
<div data-block="true">
<span data-text="true">Outro paragraph from the real DOM layout.</span>
</div>
</div>
</div>
`;
const result = await xArticleExportRunner(
{
url: "https://x.com/GoogleCloudTech/article/2033953579824758855",
},
createHelpers(),
);
expect(result.ok).toBe(true);
expect(result.data?.markdown).toContain(
"Intro paragraph from the real DOM layout.",
);
expect(result.data?.markdown).toContain("![Article cover]");
expect(result.data?.markdown).toContain("![Skill overview]");
expect(result.data?.markdown).toContain("```");
expect(result.data?.markdown).toContain("# skills/api-expert/SKILL.md");
expect(result.data?.markdown).toContain("pattern: tool-wrapper");
expect(result.data?.markdown).toContain(
"Outro paragraph from the real DOM layout.",
);
expect(result.data?.images).toHaveLength(2);
const imageUrls = (result.data?.images || []).map((image) => image.url || "");
expect(imageUrls).toContain(
"https://pbs.twimg.com/media/article-cover?format=jpg&name=orig",
);
expect(imageUrls).toContain(
"https://pbs.twimg.com/media/skill-overview?format=png&name=orig",
);
});
it("检测到媒体容器但资源未就绪时应返回错误而不是假成功", async () => {
document.body.innerHTML = `
<div data-testid="twitterArticleReadView">
<div data-testid="twitter-article-title">Incomplete media export</div>
<div data-testid="longformRichTextComponent">
<div data-contents="true">
<div><span>Paragraph before incomplete image.</span></div>
<figure>
<img alt="Broken image" />
</figure>
</div>
</div>
</div>
`;
const result = await xArticleExportRunner(
{
url: "https://x.com/GoogleCloudTech/article/2033953579824758855",
},
createHelpers(),
);
expect(result.ok).toBe(false);
expect(result).toMatchObject({
error_code: "adapter_runtime_error",
});
expect(String((result as { error_message?: string }).error_message || "")).toContain(
"图片资源尚未完全加载",
);
});
});
@@ -6739,4 +6739,34 @@ mod tests {
.is_some());
assert!(tools.iter().all(|tool| tool["name"] != "admin_secret"));
}
#[tokio::test]
async fn test_tool_search_bridge_registration_replaces_legacy_tool_search() {
let registry = Arc::new(tokio::sync::RwLock::new(aster::tools::ToolRegistry::new()));
let mut guard = registry.write().await;
guard.register(Box::new(aster::tools::ToolSearchTool::new(
std::sync::Weak::new(),
)));
assert!(guard
.get("ToolSearch")
.expect("legacy ToolSearch should exist")
.description()
.contains("Fetches full schema definitions"));
super::tool_runtime::search_bridge::register_tool_search_tool_to_registry(
&mut guard,
registry.clone(),
None,
);
let tool = guard
.get("ToolSearch")
.expect("bridge ToolSearch should replace legacy implementation");
assert!(tool.description().contains("统一搜索当前会话工具面"));
assert!(tool
.input_schema()["properties"]
.get("caller")
.is_some());
}
}
@@ -114,6 +114,27 @@ impl ToolSearchBridgeTool {
status.extension_name,
)
}
fn build_tool_search_notes(query: &str, hit_count: usize) -> Vec<String> {
if hit_count > 0 {
return Vec::new();
}
let trimmed = query.trim();
if trimmed.is_empty() {
return Vec::new();
}
if Self::parse_select_query(trimmed).is_some() {
return vec![
"未命中任何工具。不要继续改写同义词反复重试;如果需要文件、命令或网页原生能力,请直接调用当前可见的 Read / Write / Edit / Glob / Grep / Bash / WebFetch / WebSearch。".to_string(),
];
}
vec![
"未命中任何工具。优先直接调用当前可见的原生工具,或补充更明确的产品域关键词;不要继续用 ToolSearch 反复改写同义词。".to_string(),
]
}
}
#[async_trait]
@@ -336,10 +357,12 @@ impl Tool for ToolSearchBridgeTool {
.take(limit)
.map(|(_, item)| item)
.collect::<Vec<_>>();
let notes = Self::build_tool_search_notes(&raw_query, result.len());
let text = serde_json::to_string_pretty(&serde_json::json!({
"query": raw_query,
"caller": caller,
"count": result.len(),
"notes": notes,
"tools": result
}))
.map_err(|e| {
@@ -355,9 +378,8 @@ pub(super) fn register_tool_search_tool_to_registry(
registry_arc: Arc<tokio::sync::RwLock<aster::tools::ToolRegistry>>,
extension_manager: Option<Arc<aster::agents::extension_manager::ExtensionManager>>,
) {
if registry.contains(TOOL_SEARCH_TOOL_NAME) {
return;
}
// Lime runtime 里的 ToolSearch 事实源是 bridge 实现。
// 这里始终重新注册,确保旧 aster ToolSearch 不会抢占当前 surface。
registry.register(Box::new(ToolSearchBridgeTool::new(
registry_arc,
extension_manager,
@@ -328,6 +328,12 @@ enum SiteAdapterTransportRoute {
const ADAPTER_HELPERS_SCRIPT: &str = r#"
const helpers = {
sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
resolveRoot: (value) => {
if (value && typeof value === "object" && "nodeType" in value) {
return value;
}
return document.body;
},
text: (value) => {
const text = value?.textContent || "";
return text.replace(/\s+/g, " ").trim();
@@ -361,6 +367,205 @@ const helpers = {
}
return null;
},
waitForDomStable: async (options = {}) => {
const root = helpers.resolveRoot(options.root);
const stableMs = Math.max(100, Number(options.stableMs) || 500);
const timeoutMs = Math.max(stableMs, Number(options.timeoutMs) || 2500);
if (!root || typeof MutationObserver === "undefined") {
await helpers.sleep(Math.min(timeoutMs, stableMs));
return true;
}
return await new Promise((resolve) => {
let finished = false;
let settleTimer = null;
let timeoutTimer = null;
const finish = () => {
if (finished) {
return;
}
finished = true;
if (settleTimer) clearTimeout(settleTimer);
if (timeoutTimer) clearTimeout(timeoutTimer);
observer.disconnect();
resolve(true);
};
const schedule = () => {
if (settleTimer) clearTimeout(settleTimer);
settleTimer = setTimeout(finish, stableMs);
};
const observer = new MutationObserver(() => {
schedule();
});
observer.observe(root, {
childList: true,
subtree: true,
attributes: true,
characterData: true,
});
timeoutTimer = setTimeout(finish, timeoutMs);
schedule();
});
},
scrollUntilSettled: async (options = {}) => {
const root = helpers.resolveRoot(options.root);
const maxScrolls = Math.max(1, Number(options.maxScrolls) || 8);
const delayMs = Math.max(80, Number(options.delayMs) || 350);
const settleRounds = Math.max(1, Number(options.settleRounds) || 2);
const viewportFactor = Math.max(
0.2,
Number(options.viewportFactor) || 0.9,
);
const getSnapshot =
typeof options.getSnapshot === "function" ? options.getSnapshot : () => ({});
const scrollingElement =
document.scrollingElement || document.documentElement || document.body;
let lastSignature = "";
let lastScrollY = Math.round(window.scrollY);
let stableCount = 0;
let scrolls = 0;
while (scrolls < maxScrolls) {
const stepPx = Math.max(
320,
Number(options.stepPx) || Math.round(window.innerHeight * viewportFactor),
);
window.scrollBy(0, stepPx);
await helpers.sleep(delayMs);
await helpers.waitForDomStable({
root,
stableMs: Math.min(600, delayMs),
timeoutMs: Math.max(1200, delayMs * 4),
});
let snapshot = {};
try {
snapshot = getSnapshot() || {};
} catch {
snapshot = {};
}
const signature = JSON.stringify({
scrollY: Math.round(window.scrollY),
scrollHeight: Math.round(scrollingElement?.scrollHeight || 0),
...snapshot,
});
if (signature === lastSignature || Math.round(window.scrollY) === lastScrollY) {
stableCount += 1;
} else {
stableCount = 0;
}
lastSignature = signature;
lastScrollY = Math.round(window.scrollY);
scrolls += 1;
if (stableCount >= settleRounds) {
break;
}
}
return {
scrolls,
scrollY: Math.round(window.scrollY),
scrollHeight: Math.round(scrollingElement?.scrollHeight || 0),
stableCount,
};
},
waitForImagesReady: async (rootOrOptions, maybeOptions = {}) => {
const root =
rootOrOptions && typeof rootOrOptions === "object" && "nodeType" in rootOrOptions
? rootOrOptions
: helpers.resolveRoot(maybeOptions.root);
const options =
rootOrOptions && typeof rootOrOptions === "object" && "nodeType" in rootOrOptions
? maybeOptions
: rootOrOptions || {};
const timeoutMs = Math.max(200, Number(options.timeoutMs) || 3000);
const intervalMs = Math.max(80, Number(options.intervalMs) || 200);
const stableRounds = Math.max(1, Number(options.stableRounds) || 2);
const collectCandidates = () => {
const target = helpers.resolveRoot(root);
const entries = [];
const seen = new Set();
const push = (element) => {
if (!element || seen.has(element)) {
return;
}
seen.add(element);
entries.push(element);
};
if (target?.matches?.("img, [role='img'], [style*='background-image']")) {
push(target);
}
target
?.querySelectorAll?.("img, [role='img'], [style*='background-image']")
?.forEach?.((element) => push(element));
const details = entries.map((element) => {
const inlineStyle = element.getAttribute?.("style") || "";
const computedStyle =
typeof getComputedStyle === "function"
? getComputedStyle(element).backgroundImage || ""
: "";
const bgMatch = String(`${inlineStyle};${computedStyle}`).match(
/url\((['"]?)(.*?)\1\)/,
);
const source =
element.currentSrc ||
element.getAttribute?.("src") ||
element.getAttribute?.("data-src") ||
element.getAttribute?.("data-image-url") ||
element.getAttribute?.("srcset") ||
bgMatch?.[2] ||
"";
return String(source || "").trim();
});
return {
total: details.length,
ready: details.filter(Boolean).length,
};
};
const startedAt = Date.now();
let previousSignature = "";
let stableCount = 0;
let lastSnapshot = collectCandidates();
while (Date.now() - startedAt < timeoutMs) {
lastSnapshot = collectCandidates();
const signature = JSON.stringify(lastSnapshot);
if (lastSnapshot.total > 0 && lastSnapshot.ready >= lastSnapshot.total) {
return lastSnapshot;
}
if (signature === previousSignature) {
stableCount += 1;
} else {
stableCount = 0;
}
previousSignature = signature;
if (stableCount >= stableRounds) {
return lastSnapshot;
}
await helpers.sleep(intervalMs);
}
return lastSnapshot;
},
take: (items, limit) => items.slice(0, Math.max(1, limit)),
number: (value, fallbackValue) => {
const parsed = Number(value);
+1 -1
View File
@@ -1,7 +1,7 @@
{
"$schema": "https://schema.tauri.app/config/2",
"productName": "Lime",
"version": "1.6.0",
"version": "1.6.1",
"identifier": "com.lime.app",
"build": {
"beforeDevCommand": "npm run dev:web-bridge",
+1 -1
View File
@@ -1,7 +1,7 @@
{
"$schema": "https://schema.tauri.app/config/2",
"productName": "Lime",
"version": "1.6.0",
"version": "1.6.1",
"identifier": "com.lime.app",
"build": {
"beforeDevCommand": "npm run dev",