Skip to content

Commit 1f0bace

Browse files
committed
feat(web): REQ-305 整页快照静态内联档(自研规避 AGPL)
1 parent faec963 commit 1f0bace

5 files changed

Lines changed: 339 additions & 0 deletions

File tree

‎app/src-tauri/src/commands_web.rs‎

Lines changed: 82 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -173,3 +173,85 @@ pub(crate) fn web_session_to_note_core(db: &crate::db::Db, session_id: i64) -> R
173173
#[cfg(test)]
174174
#[path = "commands_web_tests.rs"]
175175
mod tests;
176+
177+
// ── v0.20.4(REQ-305)整页快照(静态内联档——自研规避 SingleFile AGPL)──
178+
179+
/// 快照预算(资源数量/单张/总量护栏——防拖垮与超大文件)。
180+
const SNAPSHOT_MAX_ASSETS: usize = 40;
181+
const SNAPSHOT_ASSET_CAP: usize = 2 * 1024 * 1024;
182+
const SNAPSHOT_TOTAL_CAP: usize = 24 * 1024 * 1024;
183+
184+
fn fetch_bytes(url: &str, cap: usize) -> Option<Vec<u8>> {
185+
let agent = ureq::AgentBuilder::new()
186+
.timeout(std::time::Duration::from_secs(10))
187+
.redirects(4)
188+
.user_agent(UA)
189+
.build();
190+
let resp = agent.get(url).call().ok()?;
191+
use std::io::Read;
192+
let mut reader = resp.into_reader();
193+
let mut body = Vec::new();
194+
let mut chunk = [0u8; 8192];
195+
loop {
196+
let n = reader.read(&mut chunk).ok()?;
197+
if n == 0 {
198+
break;
199+
}
200+
body.extend_from_slice(&chunk[..n]);
201+
if body.len() > cap {
202+
return None;
203+
}
204+
}
205+
Some(body)
206+
}
207+
208+
/// 快照导出到用户选择路径(save 对话框授权;.html 白名单;快照=自研静态内联)。
209+
#[tauri::command]
210+
pub async fn web_snapshot_export(
211+
state: State<'_, AppState>,
212+
session_id: i64,
213+
path: String,
214+
) -> Result<serde_json::Value, String> {
215+
let p = std::path::Path::new(&path);
216+
if p.extension().and_then(|e| e.to_str()).map(|e| e.to_ascii_lowercase()).as_deref() != Some("html") {
217+
return Err("仅支持 .html 快照文件".to_string());
218+
}
219+
let st: AppState = (*state).clone();
220+
tauri::async_runtime::spawn_blocking(move || {
221+
let page = st
222+
.db
223+
.get_web_page(session_id)
224+
.map_err(|e| e.to_string())?
225+
.ok_or_else(|| "web 页面不存在".to_string())?;
226+
// HTML 源:raw_html(抽取失败保留)> 原文重抓(静态页兜底)
227+
let html = match page.raw_html.clone() {
228+
Some(h) => h,
229+
None => {
230+
let bytes = fetch_bytes(&page.url, SNAPSHOT_TOTAL_CAP)
231+
.ok_or_else(|| "原文重抓失败(离线/站点拒绝?)".to_string())?;
232+
String::from_utf8_lossy(&bytes).into_owned()
233+
}
234+
};
235+
let mut assets = 0usize;
236+
let mut total_bytes = 0usize;
237+
let mut resolver = |url: &str| -> Option<String> {
238+
if assets >= SNAPSHOT_MAX_ASSETS {
239+
return None;
240+
}
241+
let bytes = fetch_bytes(url, SNAPSHOT_ASSET_CAP)?;
242+
assets += 1;
243+
total_bytes += bytes.len();
244+
if total_bytes > SNAPSHOT_TOTAL_CAP {
245+
return None;
246+
}
247+
use base64::Engine as _;
248+
Some(base64::engine::general_purpose::STANDARD.encode(bytes))
249+
};
250+
let snap = crate::web_snapshot::inline_html(&page.url, &html, &mut resolver);
251+
let chars = snap.chars().count();
252+
std::fs::write(&path, snap.as_bytes()).map_err(|e| format!("快照写入失败: {}", e))?;
253+
Ok(serde_json::json!({ "chars": chars, "assets": assets }))
254+
})
255+
.await
256+
.map_err(|e| format!("任务调度失败: {}", e))?
257+
}

‎app/src-tauri/src/lib.rs‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -87,6 +87,7 @@ mod db_web;
8787
mod commands_web;
8888
mod web_inbox;
8989
mod commands_web_inbox;
90+
mod web_snapshot;
9091
mod asr_clean;
9192
mod asr_confusion;
9293
mod asr_dedupe;
@@ -976,6 +977,8 @@ pub fn run() {
976977
// v0.20.4(REQ-303):web 采集阶段 1——URL 采集/页面读取
977978
commands_web::web_capture_url,
978979
commands_web::web_page_get,
980+
// v0.20.4(REQ-305):整页快照(静态内联档→用户文件)
981+
commands_web::web_snapshot_export,
979982
// v0.20.4(REQ-304):扩展收件服务——起停/状态
980983
commands_web_inbox::web_inbox_start,
981984
commands_web_inbox::web_inbox_status,

‎app/src-tauri/src/web_snapshot.rs‎

Lines changed: 177 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,177 @@
1+
//! 整页静态快照(v0.20.4 / REQ-305——自研 core 内联,规避 SingleFile AGPL)。
2+
//!
3+
//! @ai-context: 渲染型整页 DOM+子资源内联需要隐藏 wry 窗口(tauri WebView2
4+
//! 无 MHTML API——已否决,Foresight §二);本模块提供**静态降级档**
5+
//! (monolith CC0 同思路自研:纯函数规则内联 <link>/<img>/样式块,
6+
//! 外链脚本剔除防快照 XSS——快照是可离线查看的文档存档,不执行
7+
//! 原文 JS)。渲染档(wry eval 注入)与截图兜底链登记后置。
8+
//! @ai-context: 纯函数 + 注入 resolver(测试传假数据;生产=ureq 拉取),
9+
//! 相对 URL 按页面 base 解析;资源数量/单个体量/总预算护栏。
10+
11+
/// 资源解析器:绝对 URL → data URI(None=拉取失败,保留原引用降级)。
12+
pub type Resolver<'a> = &'a mut dyn FnMut(&str) -> Option<String>;
13+
14+
/// 相对引用按 base 解析(纯函数;仅 http(s)/同页锚/相对路径)。
15+
pub fn resolve_url(base: &str, href: &str) -> Option<String> {
16+
let href = href.trim();
17+
if href.is_empty() {
18+
return None;
19+
}
20+
if href.starts_with("data:") || href.starts_with('#') {
21+
return Some(href.to_string());
22+
}
23+
if href.starts_with("http://") || href.starts_with("https://") {
24+
return Some(href.to_string());
25+
}
26+
if href.starts_with("//") {
27+
return Some(format!("https:{}", href)); // 协议相对
28+
}
29+
let scheme_split = base.find("://")?;
30+
let scheme_end = scheme_split + 3;
31+
let rest = &base[scheme_end..];
32+
let authority = rest.split('/').next().unwrap_or("");
33+
if href.starts_with('/') {
34+
return Some(format!("{}://{}{}", &base[..scheme_split], authority, href));
35+
}
36+
// 目录相对:取 base 目录(最后一个 / 之前,保留 authority)
37+
let dir = match rest.rfind('/') {
38+
Some(i) => rest[..=i].to_string(),
39+
None => format!("{}/", rest),
40+
};
41+
Some(format!("{}://{}{}", &base[..scheme_split], dir, href))
42+
}
43+
44+
/// 内联单资源占位(style/img 空档——防空串替换破坏结构)。
45+
fn data_or_keep(
46+
kind: &str,
47+
resolved: Option<String>,
48+
resolver: &mut dyn FnMut(&str) -> Option<String>,
49+
) -> Option<String> {
50+
let url = resolved?;
51+
let data = resolver(&url)?;
52+
let mime = kind_mime(kind, &url);
53+
Some(format!("data:{};base64,{}", mime, data))
54+
}
55+
56+
fn kind_mime(kind: &str, url: &str) -> String {
57+
if kind == "style" {
58+
return "text/css".to_string();
59+
}
60+
let lower = url.to_ascii_lowercase();
61+
if lower.contains(".png") || lower.contains(".apng") {
62+
"image/png".to_string()
63+
} else if lower.contains(".webp") {
64+
"image/webp".to_string()
65+
} else if lower.contains(".svg") {
66+
"image/svg+xml".to_string()
67+
} else if lower.contains(".gif") {
68+
"image/gif".to_string()
69+
} else {
70+
"image/jpeg".to_string()
71+
}
72+
}
73+
74+
/// HTML → 静态内联快照(纯函数;返回内联后 HTML)。
75+
///
76+
/// @ai-context: 处理 <link rel=stylesheet href>、<img src>、外链 <script src>
77+
/// (剔除并注释——不执行原文脚本,快照 XSS 面归零);
78+
/// 行内 <style>/<script> 保持原文(行内 script 属原文内容,
79+
/// 快照只存不开——由打开方语境保证)。
80+
pub fn inline_html(base: &str, html: &str, resolver: Resolver) -> String {
81+
let mut out = String::new();
82+
let mut rest = html;
83+
while let Some(pos) = rest.find('<') {
84+
out.push_str(&rest[..pos]);
85+
rest = &rest[pos..];
86+
let end = rest.find('>').map(|e| e + 1).unwrap_or(rest.len());
87+
let tag = &rest[..end];
88+
rest = &rest[end..];
89+
let lower = tag.to_ascii_lowercase();
90+
if lower.starts_with("<link") {
91+
if let Some(href) = extract_attr(tag, "href") {
92+
if lower.contains("stylesheet") {
93+
if let Some(data) = data_or_keep("style", resolve_url(base, &href), &mut *resolver) {
94+
out.push_str(&format!("<style data-inlined=\"{}\">{}</style>", escape_attr(&href), data));
95+
continue;
96+
}
97+
}
98+
}
99+
out.push_str(tag);
100+
} else if lower.starts_with("<img") {
101+
if let Some(src) = extract_attr(tag, "src") {
102+
let data = data_or_keep("img", resolve_url(base, &src), &mut *resolver);
103+
if let Some(d) = data {
104+
let replaced = replace_attr(tag, "src", &d);
105+
out.push_str(&replaced);
106+
continue;
107+
}
108+
}
109+
out.push_str(tag);
110+
} else if lower.starts_with("<script") && tag.contains("src=") {
111+
// 外链脚本剔除(防快照执行第三方 JS——只存档不执行),连同闭合标签
112+
out.push_str("<!-- entropy-snapshot: external script removed -->");
113+
let lower_rest = rest.to_ascii_lowercase();
114+
if let Some(close_idx) = lower_rest.find("</script") {
115+
if let Some(gt) = rest[close_idx..].find('>') {
116+
rest = &rest[close_idx + gt + 1..];
117+
continue;
118+
}
119+
}
120+
// 无闭合标签的畸形脚本:继续正常扫描
121+
} else {
122+
out.push_str(tag);
123+
}
124+
}
125+
out.push_str(rest);
126+
out
127+
}
128+
129+
fn extract_attr(tag: &str, name: &str) -> Option<String> {
130+
let mut search = tag;
131+
while let Some(p) = search.find(name) {
132+
let after = &search[p + name.len()..];
133+
let after = after.trim_start();
134+
if let Some(after_eq) = after.strip_prefix('=') {
135+
let v = after_eq.trim_start();
136+
let value = if let Some(q) = v.strip_prefix('"') {
137+
q.split('"').next().unwrap_or("").to_string()
138+
} else if let Some(q) = v.strip_prefix('\'') {
139+
q.split('\'').next().unwrap_or("").to_string()
140+
} else {
141+
v.split_whitespace().next().unwrap_or("").to_string()
142+
};
143+
return Some(value);
144+
}
145+
search = after;
146+
}
147+
None
148+
}
149+
150+
fn replace_attr(tag: &str, name: &str, new_value: &str) -> String {
151+
if let Some(p) = tag.find(name) {
152+
let after = &tag[p + name.len()..];
153+
let after = after.trim_start();
154+
if let Some(after_eq) = after.strip_prefix('=') {
155+
let v = after_eq.trim_start();
156+
if let Some(first) = v.chars().next() {
157+
if first == '"' || first == '\'' {
158+
let rest = &v[first.len_utf8()..];
159+
if let Some(end) = rest.find(first) {
160+
let head = format!("{}{}={}", &tag[..p], name, first);
161+
let tail = &rest[end..];
162+
return format!("{}{}{}", head, new_value, tail);
163+
}
164+
}
165+
}
166+
}
167+
}
168+
tag.to_string()
169+
}
170+
171+
fn escape_attr(s: &str) -> String {
172+
s.replace('"', "&quot;")
173+
}
174+
175+
#[cfg(test)]
176+
#[path = "web_snapshot_tests.rs"]
177+
mod tests;
Lines changed: 44 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,44 @@
1+
//! web_snapshot 纯逻辑单测(v0.20.4 / REQ-305)。
2+
3+
use super::*;
4+
5+
#[test]
6+
fn resolve_url_forms() {
7+
assert_eq!(resolve_url("https://a.com/x/y.html", "/css/s.css").as_deref(), Some("https://a.com/css/s.css"));
8+
assert_eq!(resolve_url("https://a.com/x/y.html", "img/1.png").as_deref(), Some("https://a.com/x/img/1.png"));
9+
assert_eq!(resolve_url("https://a.com", "//cdn.e/x.js").as_deref(), Some("https://cdn.e/x.js"));
10+
assert_eq!(resolve_url("https://a.com/x", "https://b.com/z").as_deref(), Some("https://b.com/z"));
11+
assert!(resolve_url("https://a.com", "").is_none());
12+
assert!(resolve_url("not-a-url", "x").is_none());
13+
}
14+
15+
#[test]
16+
fn inline_styles_imgs_and_strip_external_scripts() {
17+
let html = r#"<html><head><link rel="stylesheet" href="/css/main.css">
18+
<script src="https://evil.example/x.js"></script></head>
19+
<body><img src="pic/logo.png" alt="logo"><script>var ok=1;</script><p>正文</p></body></html>"#;
20+
let mut resolver = |url: &str| match url {
21+
"https://a.com/css/main.css" => Some("aGFzaA==".to_string()), // base64('hash')
22+
"https://a.com/x/pic/logo.png" => Some("aWNvbg==".to_string()),
23+
_ => None,
24+
};
25+
let out = inline_html("https://a.com/x/y.html", html, &mut resolver);
26+
assert!(out.contains("data:text/css;base64,aGFzaA=="), "{}", out);
27+
assert!(out.contains("data:image/png;base64,aWNvbg=="), "{}", out);
28+
assert!(!out.contains("evil.example"), "外链脚本剔除");
29+
assert!(out.contains("var ok=1"), "行内脚本保持原文(快照只存不开)");
30+
assert!(out.contains("<p>正文</p>"));
31+
}
32+
33+
#[test]
34+
fn unresolvable_assets_keep_original_reference() {
35+
let html = r#"<img src="missing.png"><link rel="stylesheet" href="/gone.css">"#;
36+
let mut calls = 0;
37+
let out = inline_html("https://a.com/x/", html, &mut |_url: &str| {
38+
calls += 1;
39+
None
40+
});
41+
assert!(out.contains("missing.png"));
42+
assert!(out.contains("gone.css"));
43+
assert_eq!(calls, 2, "拉取失败保留原引用(降级链语义)");
44+
}

‎app/src/components/WebArticleView.tsx‎

Lines changed: 33 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@
88
*/
99
import { useEffect, useState } from "react";
1010
import { invoke } from "@tauri-apps/api/core";
11+
import { save } from "@tauri-apps/plugin-dialog";
1112

1213
interface WebPageView {
1314
session_id: number;
@@ -32,6 +33,29 @@ const btn: React.CSSProperties = { padding: "5px 10px", cursor: "pointer", fontS
3233
export default function WebArticleView({ sessionId, onToNote, onRemove }: Props) {
3334
const [page, setPage] = useState<WebPageView | null>(null);
3435
const [err, setErr] = useState("");
36+
const [snapMsg, setSnapMsg] = useState("");
37+
const [snapBusy, setSnapBusy] = useState(false);
38+
39+
/** REQ-305:整页快照(静态内联档——样式/图内联,外链脚本剔除防 XSS) */
40+
const doSnapshot = async () => {
41+
if (!page) return;
42+
setSnapBusy(true);
43+
setSnapMsg("");
44+
setErr("");
45+
try {
46+
const path = await save({
47+
defaultPath: `${(page.site ?? "page")}-snapshot.html`,
48+
filters: [{ name: "HTML", extensions: ["html"] }],
49+
});
50+
if (!path) return;
51+
const r = await invoke<{ chars: number; assets: number }>("web_snapshot_export", { sessionId, path });
52+
setSnapMsg(`✓ 快照已保存(${r.chars} 字符 · 内联 ${r.assets} 项资源)——离线可开(不执行原文脚本)`);
53+
} catch (e) {
54+
setErr(String(e));
55+
} finally {
56+
setSnapBusy(false);
57+
}
58+
};
3559

3660
useEffect(() => {
3761
void invoke<WebPageView | null>("web_page_get", { sessionId })
@@ -57,11 +81,20 @@ export default function WebArticleView({ sessionId, onToNote, onRemove }: Props)
5781
>
5882
📝 转为笔记
5983
</button>
84+
<button
85+
style={{ ...btn, borderRadius: 6 }}
86+
disabled={snapBusy}
87+
title="整页快照(静态内联 HTML——样式/图内联,外链脚本剔除)"
88+
onClick={() => void doSnapshot()}
89+
>
90+
{snapBusy ? "快照中…" : "📸 整页快照"}
91+
</button>
6092
<button style={btn} onClick={() => onRemove(sessionId)}>
6193
删除
6294
</button>
6395
</span>
6496
</div>
97+
{snapMsg && <div style={{ fontSize: 12, color: "#047857", marginBottom: 8 }}>{snapMsg}</div>}
6598
<div style={{ fontSize: 11, marginBottom: 8 }}>
6699
🔗 <a href={page.url} target="_blank" rel="noreferrer" style={{ color: "#2563eb", wordBreak: "break-all" }}>{page.url}</a>
67100
</div>

0 commit comments

Comments
 (0)