|
| 1 | +//! web 采集命令面(v0.20.4 / REQ-303 阶段 1 内核)。 |
| 2 | +//! |
| 3 | +//! @ai-context: URL 采集 = 课堂助手动线「URL 采集」→ 本地抽取管线(ureq 静态 |
| 4 | +//! 直取 + 轻量规则转 MD)→ kind='web' 会话 + web_session_pages |
| 5 | +//! 页面(正文整篇初稿/元数据/raw_html 降级附件);转笔记复用 |
| 6 | +//! 会话↔笔记通道(session_to_note 对 kind=web 走 web 专用管线: |
| 7 | +//! 正文 MD 直落 + properties 元数据 + 标题锚点回链)。 |
| 8 | +//! @ai-context: 失败语义(Foresight 兜底链):网络/解析失败返回明确错误且 |
| 9 | +//! 不产生半成品会话;正文抽取过少 → extracted_ok=0 保留 raw_html |
| 10 | +//! 附件可再处理。SPA/登录墙缺口由阶段 2 扩展覆盖。 |
| 11 | +//! @ai-context: 安全:仅 http/https、5MB 上限、UA 标识、15s 总超时;URL 只作 |
| 12 | +//! 出站读,不落执行上下文。 |
| 13 | +
|
| 14 | +use serde::Serialize; |
| 15 | +use std::time::Duration; |
| 16 | +use tauri::State; |
| 17 | + |
| 18 | +use crate::commands::AppState; |
| 19 | +use crate::db_web::WebPage; |
| 20 | +use crate::types::NewNote; |
| 21 | +use crate::web_capture::extract_page; |
| 22 | + |
| 23 | +/// 抓取体量上限(防超大响应拖垮内存/耗时)。 |
| 24 | +const FETCH_MAX_BYTES: usize = 5 * 1024 * 1024; |
| 25 | +/// 单次抓取超时(网络兜底——不可达快速报错)。 |
| 26 | +const FETCH_TIMEOUT: Duration = Duration::from_secs(15); |
| 27 | +/// 出站 UA(常见静态站基本放行)。 |
| 28 | +const UA: &str = "EntropyDecrease/0.20.4 (+https://github.com/Aparencia/Entropydecrease)"; |
| 29 | + |
| 30 | +/// 采集结果视图。 |
| 31 | +#[derive(Debug, Clone, Serialize)] |
| 32 | +#[serde(rename_all = "camelCase")] |
| 33 | +pub struct WebCaptureView { |
| 34 | + pub session_id: i64, |
| 35 | + pub title: String, |
| 36 | + pub site: Option<String>, |
| 37 | + pub author: Option<String>, |
| 38 | + pub chars: usize, |
| 39 | + /// 正文抽取是否成功(false=已保留 raw_html 附件可再处理) |
| 40 | + pub extracted_ok: bool, |
| 41 | +} |
| 42 | + |
| 43 | +/// URL 采集(async + spawn_blocking——网络/IO 不占 UI 线程)。 |
| 44 | +#[tauri::command] |
| 45 | +pub async fn web_capture_url( |
| 46 | + state: State<'_, AppState>, |
| 47 | + url: String, |
| 48 | +) -> Result<WebCaptureView, String> { |
| 49 | + let url = url.trim().to_string(); |
| 50 | + if !(url.starts_with("https://") || url.starts_with("http://")) || url.len() > 2048 { |
| 51 | + return Err("仅支持 http(s):// URL".to_string()); |
| 52 | + } |
| 53 | + let st: AppState = (*state).clone(); |
| 54 | + tauri::async_runtime::spawn_blocking(move || capture_inner(&st, &url)) |
| 55 | + .await |
| 56 | + .map_err(|e| format!("任务调度失败: {}", e))? |
| 57 | +} |
| 58 | + |
| 59 | +fn capture_inner(st: &AppState, url: &str) -> Result<WebCaptureView, String> { |
| 60 | + let agent = ureq::AgentBuilder::new() |
| 61 | + .timeout(FETCH_TIMEOUT) |
| 62 | + .redirects(5) |
| 63 | + .user_agent(UA) |
| 64 | + .build(); |
| 65 | + let response = agent |
| 66 | + .get(url) |
| 67 | + .call() |
| 68 | + .map_err(|e| format!("抓取失败(离线/站点拒绝?): {}", e))?; |
| 69 | + // 分块读 + 显式上限(ureq trait object 无 take——手写护栏) |
| 70 | + use std::io::Read; |
| 71 | + let mut reader = response.into_reader(); |
| 72 | + let mut body: Vec<u8> = Vec::new(); |
| 73 | + let mut chunk = [0u8; 8192]; |
| 74 | + loop { |
| 75 | + let n = reader |
| 76 | + .read(&mut chunk) |
| 77 | + .map_err(|e| format!("读取响应失败: {}", e))?; |
| 78 | + if n == 0 { |
| 79 | + break; |
| 80 | + } |
| 81 | + body.extend_from_slice(&chunk[..n]); |
| 82 | + if body.len() > FETCH_MAX_BYTES { |
| 83 | + return Err("页面超过 5MB 上限(已放弃,防内存拖垮)".to_string()); |
| 84 | + } |
| 85 | + } |
| 86 | + let html = String::from_utf8_lossy(&body).into_owned(); |
| 87 | + let page = extract_page(&html); |
| 88 | + let now = crate::db::unix_seconds(); |
| 89 | + // kind=web 会话(finished——无采集过程,直接可用) |
| 90 | + let session = st |
| 91 | + .db |
| 92 | + .create_session(&crate::types::NewSession { |
| 93 | + title: if page.title.trim().is_empty() { |
| 94 | + host_of(url).unwrap_or_else(|| "网页".to_string()) |
| 95 | + } else { |
| 96 | + page.title.trim().chars().take(100).collect() |
| 97 | + }, |
| 98 | + source_window: Some(url.to_string()), |
| 99 | + profile: None, |
| 100 | + kind: Some("web".to_string()), |
| 101 | + }) |
| 102 | + .map_err(|e| e.to_string())?; |
| 103 | + st.db |
| 104 | + .insert_web_page(&WebPage { |
| 105 | + session_id: session.id, |
| 106 | + url: url.to_string(), |
| 107 | + site: page.site.clone(), |
| 108 | + author: page.author.clone(), |
| 109 | + published: page.published.clone(), |
| 110 | + markdown: if page.ok { page.markdown.clone() } else { String::new() }, |
| 111 | + raw_html: if page.ok { None } else { Some(html.clone()) }, |
| 112 | + extracted_ok: page.ok, |
| 113 | + fetched_at: now, |
| 114 | + }) |
| 115 | + .map_err(|e| e.to_string())?; |
| 116 | + // 会话域广播(列表即时可见) |
| 117 | + crate::notify::emit_changed(&st.app, crate::notify::DataDomain::Sessions); |
| 118 | + Ok(WebCaptureView { |
| 119 | + session_id: session.id, |
| 120 | + title: session.title, |
| 121 | + site: page.site, |
| 122 | + author: page.author, |
| 123 | + chars: page.markdown.chars().count(), |
| 124 | + extracted_ok: page.ok, |
| 125 | + }) |
| 126 | +} |
| 127 | + |
| 128 | +fn host_of(url: &str) -> Option<String> { |
| 129 | + let rest = url.split("://").nth(1)?; |
| 130 | + Some( |
| 131 | + rest.split(['/', '?', '#']) |
| 132 | + .next() |
| 133 | + .unwrap_or("") |
| 134 | + .chars() |
| 135 | + .take(100) |
| 136 | + .collect(), |
| 137 | + ) |
| 138 | +} |
| 139 | + |
| 140 | +/// web 会话页面读取(详情展示/回链跳转数据源)。 |
| 141 | +#[tauri::command] |
| 142 | +pub fn web_page_get(state: State<'_, AppState>, session_id: i64) -> Result<Option<WebPage>, String> { |
| 143 | + state.db.get_web_page(session_id).map_err(|e| e.to_string()) |
| 144 | +} |
| 145 | + |
| 146 | +/// web → 笔记核心(正文 MD 直落 + properties 元数据 + 来源回链); |
| 147 | +/// session_to_note 对 kind=web 分支调用(commands_session_note 接线点)。 |
| 148 | +pub(crate) fn web_session_to_note_core(db: &crate::db::Db, session_id: i64) -> Result<crate::types::Note, String> { |
| 149 | + let session = db |
| 150 | + .get_session(session_id) |
| 151 | + .map_err(|e| e.to_string())? |
| 152 | + .ok_or_else(|| "会话不存在".to_string())?; |
| 153 | + let page = db |
| 154 | + .get_web_page(session_id) |
| 155 | + .map_err(|e| e.to_string())? |
| 156 | + .ok_or_else(|| "web 页面不存在(kind=web 会话必有页面行)".to_string())?; |
| 157 | + if !page.extracted_ok && page.markdown.trim().is_empty() { |
| 158 | + return Err("该页正文抽取失败(已保留原 HTML 附件)——可稍后在扩展/快照路径重试,暂无法转笔记".to_string()); |
| 159 | + } |
| 160 | + let props = serde_json::json!({ |
| 161 | + "url": page.url, |
| 162 | + "site": page.site, |
| 163 | + "author": page.author, |
| 164 | + "published": page.published, |
| 165 | + "fetchedAt": page.fetched_at, |
| 166 | + "type": "web-article" |
| 167 | + }) |
| 168 | + .to_string(); |
| 169 | + let note = db |
| 170 | + .create_note(&NewNote { |
| 171 | + title: session.title.clone(), |
| 172 | + content: page.markdown.clone(), |
| 173 | + source: "web".to_string(), |
| 174 | + session_id: Some(session_id), |
| 175 | + rule_version: None, |
| 176 | + purify_stats: None, |
| 177 | + tags: None, |
| 178 | + properties: Some(props), |
| 179 | + group_id: None, |
| 180 | + }) |
| 181 | + .map_err(|e| e.to_string())?; |
| 182 | + Ok(note) |
| 183 | +} |
| 184 | + |
| 185 | +#[cfg(test)] |
| 186 | +#[path = "commands_web_tests.rs"] |
| 187 | +mod tests; |
0 commit comments