From 5c2441464db9f2110b04d41b7dd17789b7538cd3 Mon Sep 17 00:00:00 2001
From: yxx <2412119399@qq.com>
Date: Fri, 4 Sep 2026 09:02:33 +0800
Subject: [PATCH 1/8] =?UTF-8?q?feat(export):=20=E4=BA=A4=E4=BB=98=20Markdo?=
=?UTF-8?q?wn=20=E2=86=92=20HTML=20=E5=AF=BC=E5=87=BA=E6=9C=8D=E5=8A=A1?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
实现 Export Service 完整生命周期:mistune AST → Document AST → HtmlExporter 渲染完整 HTML5,异步任务注册表 + 取消 + 24h 产物过期。新增 5 个 /api/exports 端点与 15 项测试;pdf/docx 与函数图像静态渲染留待后续 PR。
---
.gitignore | 2 +
README.md | 4 +-
backend/app/config.py | 2 +
backend/app/contracts.py | 94 +++++-
backend/app/export/__init__.py | 8 +
backend/app/export/document.py | 46 +++
backend/app/export/exporters/__init__.py | 1 +
backend/app/export/exporters/html.py | 216 +++++++++++++
backend/app/export/markdown.py | 214 +++++++++++++
backend/app/export/service.py | 272 ++++++++++++++++
backend/app/routes.py | 86 ++++-
backend/pyproject.toml | 1 +
backend/tests/test_export.py | 303 ++++++++++++++++++
backend/uv.lock | 11 +
docs/README.md | 1 +
.../AI笔记软件技术栈说明-团队版-v2.3.md | 4 +-
docs/contracts/第二阶段接口契约-开发版.md | 25 +-
docs/development/Export开发说明.md | 85 +++++
18 files changed, 1358 insertions(+), 17 deletions(-)
create mode 100644 backend/app/export/__init__.py
create mode 100644 backend/app/export/document.py
create mode 100644 backend/app/export/exporters/__init__.py
create mode 100644 backend/app/export/exporters/html.py
create mode 100644 backend/app/export/markdown.py
create mode 100644 backend/app/export/service.py
create mode 100644 backend/tests/test_export.py
create mode 100644 docs/development/Export开发说明.md
diff --git a/.gitignore b/.gitignore
index 84abf25..abd3743 100644
--- a/.gitignore
+++ b/.gitignore
@@ -14,6 +14,8 @@ backend/.env
# 运行期生成的 SQLite 索引(vault 下的 Markdown 测试数据需提交)
backend/data/*.db*
backend/data/credentials/
+# 运行期导出的 HTML/PDF/DOCX 产物(不提交)
+backend/data/exports/
# 阶段验收笔记(验收用,不提交)
backend/data/vault/验收/
# 本机 MCP 配置、授权状态及服务器工作目录不得提交。
diff --git a/README.md b/README.md
index 60ca8e2..2e38266 100644
--- a/README.md
+++ b/README.md
@@ -2,7 +2,7 @@
> 本文件用于团队开发期间快速配置环境和启动项目,不是正式的项目 README。
-> 当前基线:2026-09-03。第一阶段 Web 联调前后端已经完成;第二阶段已完成 Workspace 去 Mock、Agent Trace 持久化与 SSE 恢复、stdio MCP Bridge、隔离 Plugin Host、Plugin Command/Settings,以及独立 MCP Server 配置中心 C.1(stdio、Streamable HTTP 与旧 SSE 兼容)。真实音频、Provider 协议增强、Benchmark、导出、主题包、Trace 可视化、Mermaid 与函数图像仍在后续开发;Tauri Host、Stronghold、原生多 Vault 文件系统和 Sync Server 尚未接入。
+> 当前基线:2026-09-03。第一阶段 Web 联调前后端已经完成;第二阶段已完成 Workspace 去 Mock、Agent Trace 持久化与 SSE 恢复、stdio MCP Bridge、隔离 Plugin Host、Plugin Command/Settings,以及独立 MCP Server 配置中心 C.1(stdio、Streamable HTTP 与旧 SSE 兼容)、RAG Benchmark 与 Markdown → HTML 导出。真实音频、Provider 协议增强、Agent Benchmark、PDF/DOCX 导出、主题包、Trace 可视化、Mermaid 与函数图像仍在后续开发;Tauri Host、Stronghold、原生多 Vault 文件系统和 Sync Server 尚未接入。
## 当前目录
@@ -118,7 +118,7 @@ cd frontend
pnpm test
```
-当前回归基线为后端 218 项测试、前端 32 项测试,且 TypeScript 类型检查和生产构建通过。测试数量会随功能增长,以本地实际输出和 CI 为准。
+当前回归基线为后端 467 项测试、前端 32 项测试,且 TypeScript 类型检查和生产构建通过。测试数量会随功能增长,以本地实际输出和 CI 为准。
构建产物位于 `frontend/dist`,该目录不提交到 Git。
diff --git a/backend/app/config.py b/backend/app/config.py
index 00334c9..3d3d753 100644
--- a/backend/app/config.py
+++ b/backend/app/config.py
@@ -25,6 +25,7 @@ class Settings:
vault_path: Path
attachments_path: Path
benchmark_datasets_path: Path
+ exports_path: Path
@lru_cache
@@ -45,4 +46,5 @@ def get_settings() -> Settings:
benchmark_datasets_path=Path(
os.getenv("APP_BENCHMARK_DATASETS_PATH", str(data_dir / "benchmarks"))
),
+ exports_path=Path(os.getenv("APP_EXPORTS_PATH", str(data_dir / "exports"))),
)
diff --git a/backend/app/contracts.py b/backend/app/contracts.py
index 22b47e8..5ac13bb 100644
--- a/backend/app/contracts.py
+++ b/backend/app/contracts.py
@@ -2,7 +2,14 @@ from datetime import datetime
from enum import Enum
from typing import Annotated, Any, Literal
-from pydantic import BaseModel, ConfigDict, Field, SecretStr, field_validator
+from pydantic import (
+ BaseModel,
+ ConfigDict,
+ Field,
+ SecretStr,
+ field_validator,
+ model_validator,
+)
class Contract(BaseModel):
@@ -1160,3 +1167,88 @@ class BenchmarkReport(Contract):
cases: list[RAGCaseResult] = Field(default_factory=list)
error: str | None = None
error_code: str | None = None
+
+
+# Export(多格式文档导出)
+class ExportStatus(str, Enum):
+ queued = "queued"
+ running = "running"
+ completed = "completed"
+ failed = "failed"
+ cancelled = "cancelled"
+
+
+class ExportFormat(str, Enum):
+ html = "html"
+ pdf = "pdf"
+ docx = "docx"
+
+
+class ExportSourceType(str, Enum):
+ note = "note"
+ markdown = "markdown"
+
+
+class ExportSource(Contract):
+ """导出源:note 引用已索引笔记,markdown 用于未保存预览(不持久化)。"""
+
+ type: ExportSourceType
+ note_id: str | None = None
+ markdown: str | None = None
+
+ @model_validator(mode="after")
+ def _validate_source(self) -> "ExportSource":
+ if self.type == ExportSourceType.note and not self.note_id:
+ raise ValueError("note source requires note_id")
+ if self.type == ExportSourceType.markdown and not self.markdown:
+ raise ValueError("markdown source requires markdown")
+ return self
+
+
+class ExportOptions(Contract):
+ theme_id: str = "light"
+ include_title: bool = True
+ include_metadata: bool = False
+ page_size: str = "A4"
+ code_theme: str = "github-light"
+
+
+class ExportRequest(Contract):
+ source: ExportSource
+ format: ExportFormat
+ options: ExportOptions = Field(default_factory=ExportOptions)
+
+
+class ExportProgress(Contract):
+ phase: str
+ current: int
+ total: int
+ percent: float | None = None
+ message: str | None = None
+
+
+class ExportFile(Contract):
+ file_name: str
+ mime_type: str
+ size: int
+ sha256: str
+ expires_at: datetime
+
+
+class ExportJob(Contract):
+ job_id: str
+ status: ExportStatus
+ format: ExportFormat
+ progress: ExportProgress | None = None
+ file: ExportFile | None = None
+ warnings: list[str] = Field(default_factory=list)
+ error: str | None = None
+ error_code: str | None = None
+ created_at: datetime
+ started_at: datetime | None = None
+ completed_at: datetime | None = None
+
+
+class ExportJobListResponse(Contract):
+ items: list[ExportJob] = Field(default_factory=list)
+ page: PageMeta = Field(default_factory=PageMeta)
diff --git a/backend/app/export/__init__.py b/backend/app/export/__init__.py
new file mode 100644
index 0000000..d5a6faa
--- /dev/null
+++ b/backend/app/export/__init__.py
@@ -0,0 +1,8 @@
+"""Export Service:多格式文档导出(首批 HTML)。
+
+模块划分:
+- document.py Document AST 内部协议 + DocumentExporter Protocol + ExportResult
+- markdown.py mistune → Document AST 解析
+- exporters/html.py HtmlExporter(Document AST → HTML5)
+- service.py 导出任务注册表、后台执行、取消与文件生命周期
+"""
diff --git a/backend/app/export/document.py b/backend/app/export/document.py
new file mode 100644
index 0000000..e09f5cb
--- /dev/null
+++ b/backend/app/export/document.py
@@ -0,0 +1,46 @@
+"""Document AST:导出器的内部中间表示(Internal Protocol,不放入 contracts.py)。
+
+契约 §10.3 规定节点用稳定判别字段 node_id / type / attributes / children / text,
+类型专有信息统一放 attributes(如 heading 的 level、link 的 href、image 的 src)。
+导出器据此递归渲染,对无法表示的节点记 warning,不静默丢弃。
+"""
+
+from __future__ import annotations
+
+from typing import Any, Protocol
+
+from pydantic import BaseModel, ConfigDict, Field
+
+from app.contracts import ExportOptions
+
+
+class DocumentNode(BaseModel):
+ """递归文档节点;type 取契约 §10.3 首批 node type 之一。"""
+
+ model_config = ConfigDict(extra="forbid")
+
+ type: str
+ node_id: str
+ attributes: dict[str, Any] = Field(default_factory=dict)
+ children: list["DocumentNode"] = Field(default_factory=list)
+ text: str = ""
+
+
+class Document(DocumentNode):
+ """根节点,type 固定为 document。"""
+
+ type: str = "document"
+
+
+class DocumentExporter(Protocol):
+ """导出器协议(契约 §10.3):把 Document AST 渲染为指定格式的产物。"""
+
+ async def export(self, document: Document, options: ExportOptions) -> "ExportResult": ...
+
+
+class ExportResult(BaseModel):
+ model_config = ConfigDict(extra="forbid")
+
+ content: bytes
+ mime_type: str
+ warnings: list[str] = Field(default_factory=list)
diff --git a/backend/app/export/exporters/__init__.py b/backend/app/export/exporters/__init__.py
new file mode 100644
index 0000000..018a8ab
--- /dev/null
+++ b/backend/app/export/exporters/__init__.py
@@ -0,0 +1 @@
+"""Export 渲染器:Document AST → 具体格式产物。"""
diff --git a/backend/app/export/exporters/html.py b/backend/app/export/exporters/html.py
new file mode 100644
index 0000000..0bcad16
--- /dev/null
+++ b/backend/app/export/exporters/html.py
@@ -0,0 +1,216 @@
+"""HtmlExporter:Document AST → 完整 HTML5 文档(内嵌基础 CSS)。
+
+对无法静态表达的节点(mermaid / function_plot)渲染为占位代码块并记 warning,不静默丢失;
+严重内容缺失由 service 层以 EXPORT_UNSUPPORTED_CONTENT 判定,本层只负责逐节点渲染。
+"""
+
+from __future__ import annotations
+
+import html
+from datetime import datetime
+
+from app.contracts import ExportOptions
+from app.export.document import Document, DocumentNode, ExportResult
+
+_MERMAID_WARNING = "mermaid 需前端渲染,已保留为占位代码块"
+_FUNCTION_PLOT_WARNING = "函数图像渲染将在后续版本提供,已保留为占位代码块"
+
+_BASE_CSS = """
+body { margin: 0; background: #f6f7f9; color: #1f2328; font: 15px/1.7 -apple-system, 'Segoe UI', 'Microsoft YaHei', sans-serif; }
+article { max-width: 860px; margin: 0 auto; padding: 40px 48px; background: #fff; }
+article.theme-dark { background: #0d1117; color: #c9d1d9; }
+h1, h2, h3, h4, h5, h6 { line-height: 1.3; margin: 1.4em 0 0.6em; }
+h1.title { margin-top: 0; }
+p { margin: 0.6em 0; }
+a { color: #0969da; }
+code { font-family: 'JetBrains Mono', Consolas, monospace; font-size: 0.9em; background: #f0f1f3; padding: 0.15em 0.35em; border-radius: 3px; }
+pre { background: #f6f8fa; padding: 14px 16px; border-radius: 6px; overflow-x: auto; }
+pre.code-theme-github-dark { background: #0d1117; color: #c9d1d9; }
+pre code { background: none; padding: 0; }
+pre.mermaid, pre.function-plot { border: 1px dashed #d0d7de; }
+blockquote { margin: 0.8em 0; padding: 0.2em 1em; border-left: 4px solid #d0d7de; color: #57606a; }
+img { max-width: 100%; }
+table { border-collapse: collapse; margin: 0.8em 0; }
+th, td { border: 1px solid #d0d7de; padding: 6px 12px; }
+th { background: #f6f8fa; }
+dl.metadata { font-size: 0.85em; color: #57606a; border-top: 1px solid #eaeef2; border-bottom: 1px solid #eaeef2; padding: 0.6em 0; }
+dl.metadata dt { display: inline; font-weight: 600; margin-right: 0.4em; }
+dl.metadata dd { display: inline; margin: 0 1.2em 0 0; }
+.math, .math-block { overflow-x: auto; padding: 0.4em 0; }
+.task-list-item { list-style: none; }
+.task-list-item input { margin-right: 0.4em; }
+hr { border: none; border-top: 1px solid #d0d7de; margin: 1.4em 0; }
+""".strip()
+
+
+class HtmlExporter:
+ """实现 DocumentExporter:递归渲染 Document AST 为完整 HTML5 文档。"""
+
+ async def export(self, document: Document, options: ExportOptions) -> ExportResult:
+ self._options = options
+ warnings: list[str] = []
+ body = self._render_children(document.children, warnings)
+ content = self._assemble(document, options, body, warnings)
+ return ExportResult(
+ content=content.encode("utf-8"), mime_type="text/html", warnings=warnings
+ )
+
+ def _assemble(
+ self, document: Document, options: ExportOptions, body: str, warnings: list[str]
+ ) -> str:
+ title = str(document.attributes.get("title") or "")
+ parts = [
+ "",
+ '',
+ "
",
+ ' ',
+ ' ',
+ ]
+ if title:
+ parts.append(f"{html.escape(title)} ")
+ parts.append(f"")
+ parts.append("")
+ parts.append("")
+ parts.append(f'')
+ if options.include_title and title:
+ parts.append(f'{html.escape(title)} ')
+ if options.include_metadata:
+ metadata = document.attributes.get("metadata")
+ if metadata:
+ parts.append(self._render_metadata(metadata))
+ parts.append(body)
+ parts.append(" ")
+ parts.append("")
+ parts.append("")
+ return "\n".join(parts) + "\n"
+
+ def _render_metadata(self, metadata: dict) -> str:
+ entries = ["']
+ for key, value in metadata.items():
+ entries.append(f"{html.escape(str(key))} ")
+ entries.append(f"{html.escape(self._fmt_meta_value(value))} ")
+ entries.append(" ")
+ return "".join(entries)
+
+ @staticmethod
+ def _fmt_meta_value(value: object) -> str:
+ if isinstance(value, datetime):
+ return value.isoformat()
+ if isinstance(value, list):
+ return ", ".join(str(item) for item in value)
+ return str(value)
+
+ def _render_children(self, children: list[DocumentNode], warnings: list[str]) -> str:
+ return "".join(self._render_node(child, warnings) for child in children)
+
+ def _render_node(self, node: DocumentNode, warnings: list[str]) -> str:
+ handler = getattr(self, f"_render_{node.type}", None)
+ if handler is not None:
+ return handler(node, warnings)
+ warnings.append(f"无法表示的节点类型已跳过:{node.type}")
+ return ""
+
+ # --- 块级 ---
+ def _render_heading(self, node: DocumentNode, warnings: list[str]) -> str:
+ level = max(1, min(6, int(node.attributes.get("level", 1))))
+ return f"{self._render_children(node.children, warnings)} "
+
+ def _render_paragraph(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"{self._render_children(node.children, warnings)}
"
+
+ def _render_blockquote(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"{self._render_children(node.children, warnings)} "
+
+ def _render_list(self, node: DocumentNode, warnings: list[str]) -> str:
+ tag = "ol" if node.attributes.get("ordered") else "ul"
+ return f"<{tag}>{self._render_children(node.children, warnings)}{tag}>"
+
+ def _render_list_item(self, node: DocumentNode, warnings: list[str]) -> str:
+ inner = self._render_children(node.children, warnings)
+ if node.attributes.get("task"):
+ checked = " checked" if node.attributes.get("checked") else ""
+ return (
+ ''
+ f' {inner} '
+ )
+ return f"{inner} "
+
+ def _render_table(self, node: DocumentNode, warnings: list[str]) -> str:
+ rows = node.children
+ head_rows = [r for r in rows if r.attributes.get("head")]
+ body_rows = [r for r in rows if not r.attributes.get("head")]
+ parts = [""]
+ if head_rows:
+ parts.append("")
+ parts.extend(self._render_node(r, warnings) for r in head_rows)
+ parts.append(" ")
+ if body_rows:
+ parts.append("")
+ parts.extend(self._render_node(r, warnings) for r in body_rows)
+ parts.append(" ")
+ parts.append("
")
+ return "".join(parts)
+
+ def _render_table_row(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"{self._render_children(node.children, warnings)} "
+
+ def _render_table_cell(self, node: DocumentNode, warnings: list[str]) -> str:
+ tag = "th" if node.attributes.get("head") else "td"
+ return f"<{tag}>{self._render_children(node.children, warnings)}{tag}>"
+
+ def _render_code_block(self, node: DocumentNode, warnings: list[str]) -> str:
+ lang = str(node.attributes.get("language") or "")
+ code = html.escape(node.text)
+ lang_cls = f' class="language-{html.escape(lang)}"' if lang else ""
+ theme = html.escape(self._options.code_theme)
+ return f'{code} '
+
+ def _render_thematic_break(self, node: DocumentNode, warnings: list[str]) -> str:
+ return " "
+
+ def _render_mermaid(self, node: DocumentNode, warnings: list[str]) -> str:
+ warnings.append(_MERMAID_WARNING)
+ return f'{html.escape(node.text)} '
+
+ def _render_function_plot(self, node: DocumentNode, warnings: list[str]) -> str:
+ warnings.append(_FUNCTION_PLOT_WARNING)
+ return f'{html.escape(node.text)} '
+
+ def _render_math_block(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f'$${html.escape(node.text)}$$
'
+
+ # --- 行内 ---
+ def _render_text(self, node: DocumentNode, warnings: list[str]) -> str:
+ return html.escape(node.text)
+
+ def _render_emphasis(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"{self._render_children(node.children, warnings)} "
+
+ def _render_strong(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"{self._render_children(node.children, warnings)} "
+
+ def _render_link(self, node: DocumentNode, warnings: list[str]) -> str:
+ href = html.escape(str(node.attributes.get("href") or ""))
+ title = str(node.attributes.get("title") or "")
+ attrs = [f'href="{href}"']
+ if title:
+ attrs.append(f'title="{html.escape(title)}"')
+ return f"{self._render_children(node.children, warnings)} "
+
+ def _render_codespan(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"{html.escape(node.text)}"
+
+ def _render_image(self, node: DocumentNode, warnings: list[str]) -> str:
+ src = html.escape(str(node.attributes.get("src") or ""))
+ alt = html.escape(str(node.attributes.get("alt") or ""))
+ title = str(node.attributes.get("title") or "")
+ attrs = [f'src="{src}"', f'alt="{alt}"']
+ if title:
+ attrs.append(f'title="{html.escape(title)}"')
+ return f" "
+
+ def _render_math_inline(self, node: DocumentNode, warnings: list[str]) -> str:
+ return f"\\({html.escape(node.text)}\\)"
+
+ def _render_linebreak(self, node: DocumentNode, warnings: list[str]) -> str:
+ return " "
diff --git a/backend/app/export/markdown.py b/backend/app/export/markdown.py
new file mode 100644
index 0000000..181d190
--- /dev/null
+++ b/backend/app/export/markdown.py
@@ -0,0 +1,214 @@
+"""Markdown → Document AST:用 mistune 的 ast renderer 产出通用 token,再映射为内部节点。
+
+选用 mistune 内置 'ast' renderer 而非自写 BaseRenderer,是因为 mistune 的行内渲染按
+字符串拼接、无法承载结构化子节点;ast renderer 直接给出带 children/attrs/raw 的 token
+树,映射层只做 token → DocumentNode 的搬运,不掺入任何 HTML。
+"""
+
+from __future__ import annotations
+
+import mistune
+
+from app.export.document import Document, DocumentNode
+
+_PLUGINS = ["table", "math", "url", "task_lists"]
+
+# fenced code 语言分流:命中则转为专用节点,其余按普通代码块
+_MERMAID_LANG = "mermaid"
+_FUNCTION_PLOT_LANGS = {"function_plot", "functionplot"}
+
+
+def parse_document(markdown: str) -> Document:
+ """把 Markdown 文本解析为 Document AST 根节点。"""
+ renderer = mistune.create_markdown(renderer="ast", plugins=_PLUGINS)
+ tokens = renderer(markdown)
+ mapper = _AstMapper()
+ return Document(node_id=mapper.next_id(), children=mapper.map_blocks(tokens))
+
+
+class _AstMapper:
+ """token 树 → DocumentNode 树的映射器;node_id 按遍历顺序递增,无需跨请求稳定。"""
+
+ def __init__(self) -> None:
+ self._seq = 0
+
+ def next_id(self) -> str:
+ self._seq += 1
+ return f"node_{self._seq:03d}"
+
+ def map_blocks(self, tokens: list[dict]) -> list[DocumentNode]:
+ nodes: list[DocumentNode] = []
+ for token in tokens:
+ node = self.map_block(token)
+ if node is not None:
+ nodes.append(node)
+ return nodes
+
+ def map_block(self, token: dict) -> DocumentNode | None:
+ kind = token["type"]
+ if kind == "heading":
+ return DocumentNode(
+ type="heading",
+ node_id=self.next_id(),
+ attributes={"level": token["attrs"]["level"]},
+ children=self.map_inline(token.get("children", [])),
+ )
+ if kind in ("paragraph", "block_text"):
+ # block_text 是列表项内的段落块,仍按 paragraph 表达,由 list_item 包裹
+ return DocumentNode(
+ type="paragraph",
+ node_id=self.next_id(),
+ children=self.map_inline(token.get("children", [])),
+ )
+ if kind == "list":
+ return DocumentNode(
+ type="list",
+ node_id=self.next_id(),
+ attributes={"ordered": bool(token.get("attrs", {}).get("ordered"))},
+ children=[self.map_list_item(child) for child in token.get("children", [])],
+ )
+ if kind == "block_code":
+ return self._map_code(token)
+ if kind == "block_quote":
+ return DocumentNode(
+ type="blockquote",
+ node_id=self.next_id(),
+ children=self.map_blocks(token.get("children", [])),
+ )
+ if kind == "table":
+ return self._map_table(token)
+ if kind == "block_math":
+ return DocumentNode(
+ type="math_block", node_id=self.next_id(), text=token.get("raw", "")
+ )
+ if kind == "thematic_break":
+ return DocumentNode(type="thematic_break", node_id=self.next_id())
+ if kind == "blank_line":
+ return None
+ # 未知块级 token(如 block_html)保守保留原文,避免静默丢失
+ raw = token.get("raw", "")
+ if raw:
+ return DocumentNode(type="paragraph", node_id=self.next_id(), text=raw)
+ return None
+
+ def map_list_item(self, token: dict) -> DocumentNode:
+ """列表项:block_text 展平为行内子节点,嵌套 list 保留为子节点。"""
+ attributes: dict = {}
+ if token["type"] == "task_list_item":
+ attributes = {"task": True, "checked": bool(token.get("attrs", {}).get("checked"))}
+ children: list[DocumentNode] = []
+ for child in token.get("children", []):
+ if child["type"] == "block_text":
+ children.extend(self.map_inline(child.get("children", [])))
+ elif child["type"] == "list":
+ children.append(self.map_block(child))
+ else:
+ node = self.map_block(child)
+ if node is not None:
+ children.append(node)
+ return DocumentNode(
+ type="list_item", node_id=self.next_id(), attributes=attributes, children=children
+ )
+
+ def map_inline(self, tokens: list[dict]) -> list[DocumentNode]:
+ nodes: list[DocumentNode] = []
+ for token in tokens:
+ node = self.map_inline_token(token)
+ if node is not None:
+ nodes.append(node)
+ return nodes
+
+ def map_inline_token(self, token: dict) -> DocumentNode | None:
+ kind = token["type"]
+ if kind == "text":
+ return DocumentNode(type="text", node_id=self.next_id(), text=token.get("raw", ""))
+ if kind == "strong":
+ return DocumentNode(
+ type="strong", node_id=self.next_id(),
+ children=self.map_inline(token.get("children", [])),
+ )
+ if kind == "emphasis":
+ return DocumentNode(
+ type="emphasis", node_id=self.next_id(),
+ children=self.map_inline(token.get("children", [])),
+ )
+ if kind == "link":
+ attrs = token.get("attrs", {})
+ attributes = {"href": attrs.get("url", "")}
+ if attrs.get("title"):
+ attributes["title"] = attrs["title"]
+ return DocumentNode(
+ type="link", node_id=self.next_id(), attributes=attributes,
+ children=self.map_inline(token.get("children", [])),
+ )
+ if kind == "codespan":
+ return DocumentNode(type="codespan", node_id=self.next_id(), text=token.get("raw", ""))
+ if kind == "image":
+ attrs = token.get("attrs", {})
+ attributes = {"src": attrs.get("src", "")}
+ if attrs.get("alt"):
+ attributes["alt"] = attrs["alt"]
+ if attrs.get("title"):
+ attributes["title"] = attrs["title"]
+ return DocumentNode(type="image", node_id=self.next_id(), attributes=attributes)
+ if kind == "inline_math":
+ return DocumentNode(
+ type="math_inline", node_id=self.next_id(), text=token.get("raw", "")
+ )
+ if kind == "softbreak":
+ # HTML 中换行会折叠为空白,软换行按空格表达
+ return DocumentNode(type="text", node_id=self.next_id(), text=" ")
+ if kind == "linebreak":
+ return DocumentNode(type="linebreak", node_id=self.next_id())
+ # 未知行内 token 保守保留原文
+ raw = token.get("raw", "")
+ if raw:
+ return DocumentNode(type="text", node_id=self.next_id(), text=raw)
+ return None
+
+ def _map_code(self, token: dict) -> DocumentNode:
+ info = (token.get("attrs", {}).get("info") or "").strip()
+ lang = info.split()[0].lower() if info else ""
+ code = token.get("raw", "").rstrip("\n")
+ if lang == _MERMAID_LANG:
+ return DocumentNode(type="mermaid", node_id=self.next_id(), text=code)
+ if lang in _FUNCTION_PLOT_LANGS:
+ return DocumentNode(type="function_plot", node_id=self.next_id(), text=code)
+ attributes = {"language": lang} if lang else {}
+ return DocumentNode(
+ type="code_block", node_id=self.next_id(), attributes=attributes, text=code
+ )
+
+ def _map_table(self, token: dict) -> DocumentNode:
+ rows: list[DocumentNode] = []
+ for child in token.get("children", []):
+ if child["type"] == "table_head":
+ rows.append(self._map_table_row(child, head=True))
+ elif child["type"] == "table_body":
+ for row in child.get("children", []):
+ if row["type"] == "table_row":
+ rows.append(self._map_table_row(row, head=False))
+ elif child["type"] == "table_row":
+ rows.append(self._map_table_row(child, head=False))
+ return DocumentNode(type="table", node_id=self.next_id(), children=rows)
+
+ def _map_table_row(self, token: dict, *, head: bool) -> DocumentNode:
+ cells: list[DocumentNode] = []
+ for cell in token.get("children", []):
+ if cell["type"] != "table_cell":
+ continue
+ attrs = cell.get("attrs", {})
+ cell_attributes = {"head": bool(attrs.get("head", head))}
+ if attrs.get("align"):
+ cell_attributes["align"] = attrs["align"]
+ cells.append(
+ DocumentNode(
+ type="table_cell",
+ node_id=self.next_id(),
+ attributes=cell_attributes,
+ children=self.map_inline(cell.get("children", [])),
+ )
+ )
+ return DocumentNode(
+ type="table_row", node_id=self.next_id(), attributes={"head": head}, children=cells
+ )
diff --git a/backend/app/export/service.py b/backend/app/export/service.py
new file mode 100644
index 0000000..73361ad
--- /dev/null
+++ b/backend/app/export/service.py
@@ -0,0 +1,272 @@
+"""Export 服务:任务注册表、后台渲染、取消与产物生命周期。
+
+与 Benchmark 一致采用「创建即返回 queued、后台 Task 异步执行」的内存模型:任务与产物
+暂存内存与 exports 目录,不持久化到 SQLite。导出是单阶段渲染,无 SSE 事件流,取消主要
+在渲染前/后让出执行权的边界生效;产物带 24h 过期时间,过期后不可下载。
+"""
+
+from __future__ import annotations
+
+import asyncio
+import hashlib
+import logging
+import re
+from datetime import datetime, timedelta, timezone
+from pathlib import Path
+from uuid import uuid4
+
+from app.config import get_settings
+from app.contracts import (
+ ExportFile,
+ ExportFormat,
+ ExportJob,
+ ExportOptions,
+ ExportProgress,
+ ExportRequest,
+ ExportSource,
+ ExportSourceType,
+ ExportStatus,
+)
+from app.errors import ApiError
+from app.export.document import Document
+from app.export.exporters.html import HtmlExporter
+from app.export.markdown import parse_document
+from app.services import note_service
+
+logger = logging.getLogger(__name__)
+
+_jobs: dict[str, ExportJob] = {}
+_tasks: dict[str, asyncio.Task] = {}
+_cancel_flags: dict[str, asyncio.Event] = {}
+MAX_JOBS = 100
+# markdown 源大小上限,防止未保存预览塞爆内存/产物
+MAX_MARKDOWN_CHARS = 200_000
+# 产物有效期
+FILE_TTL = timedelta(hours=24)
+
+_INVALID_FILE_CHARS = re.compile(r'[\\/:*?"<>|]')
+
+
+class ExportCancelled(Exception):
+ """导出在渲染前被取消时抛出,用于标记 cancelled。"""
+
+
+def _now() -> datetime:
+ return datetime.now(timezone.utc)
+
+
+def _safe_download_name(title: str) -> str:
+ """清洗标题得到安全的下载文件名;空标题回退到 export。"""
+ name = _INVALID_FILE_CHARS.sub("_", title).strip() or "export"
+ return name[:80]
+
+
+def _forget(job_id: str) -> None:
+ _jobs.pop(job_id, None)
+ _tasks.pop(job_id, None)
+ _cancel_flags.pop(job_id, None)
+
+
+def _evict_terminal() -> bool:
+ """超过容量时淘汰最旧的终态任务;全为活动任务无法淘汰时返回 False。"""
+ terminal = (ExportStatus.completed, ExportStatus.failed, ExportStatus.cancelled)
+ while len(_jobs) >= MAX_JOBS:
+ victim = next((jid for jid, job in _jobs.items() if job.status in terminal), None)
+ if victim is None:
+ return False
+ _forget(victim)
+ return True
+
+
+async def _resolve_source(source: ExportSource) -> tuple[str, str, dict | None]:
+ """把导出源解析为 (markdown, title, metadata);metadata 仅 note 源提供。"""
+ if source.type == ExportSourceType.note:
+ note = await note_service.get_note(source.note_id)
+ if note is None:
+ raise ApiError(
+ 404,
+ "EXPORT_SOURCE_NOT_FOUND",
+ "note not found",
+ {"note_id": source.note_id},
+ )
+ metadata = {
+ "file_path": note.file_path,
+ "tags": note.tags,
+ "created_at": note.created_at,
+ "updated_at": note.updated_at,
+ }
+ return note.markdown, note.title, metadata
+
+ markdown = source.markdown or ""
+ if not markdown.strip():
+ raise ApiError(400, "EXPORT_OPTIONS_INVALID", "markdown source must not be empty")
+ if len(markdown) > MAX_MARKDOWN_CHARS:
+ raise ApiError(
+ 400,
+ "EXPORT_OPTIONS_INVALID",
+ f"markdown source exceeds {MAX_MARKDOWN_CHARS} characters",
+ {"size": len(markdown), "limit": MAX_MARKDOWN_CHARS},
+ )
+ return markdown, "", None
+
+
+async def create_export(request: ExportRequest) -> ExportJob:
+ """创建导出任务,立即返回 queued 的 ExportJob,由后台 Task 渲染。"""
+ if request.format != ExportFormat.html:
+ raise ApiError(
+ 400,
+ "EXPORT_FORMAT_UNSUPPORTED",
+ "PDF/DOCX 暂未实现,当前仅支持 HTML",
+ {"format": request.format.value},
+ )
+ markdown, title, metadata = await _resolve_source(request.source)
+
+ if not _evict_terminal():
+ raise ApiError(
+ 429,
+ "EXPORT_CAPACITY_EXCEEDED",
+ "Export capacity exceeded; wait for active jobs to finish.",
+ {},
+ )
+
+ job_id = "export_" + uuid4().hex[:12]
+ job = ExportJob(
+ job_id=job_id,
+ status=ExportStatus.queued,
+ format=request.format,
+ created_at=_now(),
+ )
+ _jobs[job_id] = job
+ _cancel_flags[job_id] = asyncio.Event()
+ _tasks[job_id] = asyncio.create_task(
+ _execute(job_id, markdown, title, metadata, request.options)
+ )
+ return job
+
+
+async def _execute(
+ job_id: str,
+ markdown: str,
+ title: str,
+ metadata: dict | None,
+ options: ExportOptions,
+) -> None:
+ """后台渲染:解析 → 导出 → 写文件 → 挂载产物元信息。"""
+ cancel_event = _cancel_flags[job_id]
+ _jobs[job_id] = _jobs[job_id].model_copy(
+ update={
+ "status": ExportStatus.running,
+ "started_at": _now(),
+ "progress": ExportProgress(phase="rendering", current=0, total=1, percent=0.0),
+ }
+ )
+ try:
+ # 让出一次,使「创建后立即取消」的 queued 任务能及时进入 cancelled
+ await asyncio.sleep(0)
+ if cancel_event.is_set():
+ raise ExportCancelled()
+
+ document = parse_document(markdown)
+ document.attributes["title"] = title
+ if metadata:
+ document.attributes["metadata"] = metadata
+
+ exporter = HtmlExporter()
+ result = await exporter.export(document, options)
+ if cancel_event.is_set():
+ raise ExportCancelled()
+
+ out_dir = get_settings().exports_path
+ out_dir.mkdir(parents=True, exist_ok=True)
+ path = out_dir / f"{job_id}.html"
+ path.write_bytes(result.content)
+
+ completed_at = _now()
+ _jobs[job_id] = _jobs[job_id].model_copy(
+ update={
+ "status": ExportStatus.completed,
+ "progress": ExportProgress(
+ phase="completed", current=1, total=1, percent=1.0
+ ),
+ "file": ExportFile(
+ file_name=f"{_safe_download_name(title)}.html",
+ mime_type=result.mime_type,
+ size=len(result.content),
+ sha256=hashlib.sha256(result.content).hexdigest(),
+ expires_at=completed_at + FILE_TTL,
+ ),
+ "warnings": result.warnings,
+ "completed_at": completed_at,
+ }
+ )
+ except ExportCancelled:
+ _jobs[job_id] = _jobs[job_id].model_copy(
+ update={
+ "status": ExportStatus.cancelled,
+ "completed_at": _now(),
+ }
+ )
+ except Exception as exc: # 渲染失败不拖垮服务,只记日志与项目错误码
+ logger.exception("Export failed: job_id=%s", job_id)
+ _jobs[job_id] = _jobs[job_id].model_copy(
+ update={
+ "status": ExportStatus.failed,
+ "error": "Export render failed.",
+ "error_code": "EXPORT_RENDER_FAILED",
+ "completed_at": _now(),
+ }
+ )
+ finally:
+ _cancel_flags.pop(job_id, None)
+
+
+def list_exports(
+ status: ExportStatus | None = None,
+ format: ExportFormat | None = None,
+ limit: int = 50,
+ offset: int = 0,
+) -> tuple[list[ExportJob], int]:
+ jobs = list(_jobs.values())
+ if status is not None:
+ jobs = [j for j in jobs if j.status == status]
+ if format is not None:
+ jobs = [j for j in jobs if j.format == format]
+ jobs.sort(key=lambda j: j.created_at, reverse=True)
+ total = len(jobs)
+ return jobs[offset : offset + limit], total
+
+
+def get_export(job_id: str) -> ExportJob | None:
+ return _jobs.get(job_id)
+
+
+def cancel_export(job_id: str) -> ExportJob | None:
+ """取消导出:仅 queued/running 可取消,后台 Task 在让出边界标记 cancelled。"""
+ job = _jobs.get(job_id)
+ if job is None:
+ return None
+ if job.status in (ExportStatus.queued, ExportStatus.running):
+ _cancel_flags[job_id].set()
+ return job
+
+
+def get_export_file(job_id: str) -> Path:
+ """返回可下载产物的存储路径;未完成返回 404、过期返回 410。"""
+ job = _jobs.get(job_id)
+ if job is None:
+ raise ApiError(404, "EXPORT_JOB_NOT_FOUND", "export job not found", {"job_id": job_id})
+ if job.status != ExportStatus.completed or job.file is None:
+ raise ApiError(
+ 404, "EXPORT_JOB_NOT_FOUND", "export file not ready", {"job_id": job_id}
+ )
+ if job.file.expires_at <= _now():
+ raise ApiError(410, "EXPORT_FILE_EXPIRED", "export file has expired", {"job_id": job_id})
+ return get_settings().exports_path / f"{job_id}.html"
+
+
+async def wait_for_export(job_id: str) -> ExportJob | None:
+ """等待后台任务结束(测试/轮询用);无任务时直接返回当前状态。"""
+ task = _tasks.get(job_id)
+ if task is not None:
+ await task
+ return _jobs.get(job_id)
diff --git a/backend/app/routes.py b/backend/app/routes.py
index 38d319b..6bd8267 100644
--- a/backend/app/routes.py
+++ b/backend/app/routes.py
@@ -5,7 +5,7 @@ from datetime import datetime, timezone
from uuid import uuid4
from fastapi import APIRouter, Header, Query
-from fastapi.responses import StreamingResponse
+from fastapi.responses import FileResponse, StreamingResponse
from app.agent import AgentCapacityError, AgentRunNotFoundError
from app.container import container
@@ -48,6 +48,11 @@ from app.contracts import (
ModelRoutingResponse,
SpeakerMatchRequest,
SpeakerMatchResult,
+ ExportFormat,
+ ExportJob,
+ ExportJobListResponse,
+ ExportRequest,
+ ExportStatus,
Note,
NoteCreateRequest,
NoteListResponse,
@@ -96,8 +101,10 @@ from app.contracts import (
from app.agent import AgentCapacityError, AgentRunNotFoundError
from app.benchmarks import datasets as benchmark_datasets
from app.benchmarks import service as benchmark_service
+from app.config import get_settings
from app.container import container
from app.errors import ApiError
+from app.export import service as export_service
from app.extensions import ExtensionError
from app.extensions.mcp_registry import McpRegistryError
from app.providers.base import ProviderError
@@ -1310,3 +1317,80 @@ async def get_benchmark_report(run_id: str) -> BenchmarkReport:
404, "BENCHMARK_RUN_NOT_FOUND", "benchmark report not found", {"run_id": run_id}
)
return report
+
+
+@router.post(
+ "/exports",
+ response_model=ExportJob,
+ status_code=202,
+ tags=["Export"],
+)
+async def create_export(request: ExportRequest) -> ExportJob:
+ return await export_service.create_export(request)
+
+
+@router.get(
+ "/exports",
+ response_model=ExportJobListResponse,
+ tags=["Export"],
+)
+async def list_exports(
+ status: ExportStatus | None = Query(default=None),
+ format: ExportFormat | None = Query(default=None),
+ limit: int = Query(default=50, ge=1, le=200),
+ offset: int = Query(default=0, ge=0),
+) -> ExportJobListResponse:
+ items, total = export_service.list_exports(
+ status=status, format=format, limit=limit, offset=offset
+ )
+ return ExportJobListResponse(
+ items=items, page=PageMeta(total=total, limit=limit, offset=offset)
+ )
+
+
+@router.get(
+ "/exports/{job_id}",
+ response_model=ExportJob,
+ tags=["Export"],
+)
+async def get_export(job_id: str) -> ExportJob:
+ job = export_service.get_export(job_id)
+ if job is None:
+ raise ApiError(
+ 404, "EXPORT_JOB_NOT_FOUND", "export job not found", {"job_id": job_id}
+ )
+ return job
+
+
+@router.get(
+ "/exports/{job_id}/file",
+ tags=["Export"],
+)
+async def get_export_file(job_id: str) -> FileResponse:
+ path = export_service.get_export_file(job_id) # 未完成/过期分别抛 404/410
+ job = export_service.get_export(job_id)
+ if job is None or job.file is None:
+ raise ApiError(
+ 404, "EXPORT_JOB_NOT_FOUND", "export file not ready", {"job_id": job_id}
+ )
+ return FileResponse(
+ path=path,
+ media_type=job.file.mime_type,
+ filename=job.file.file_name,
+ )
+
+
+@router.post(
+ "/exports/{job_id}/cancel",
+ response_model=OperationResponse,
+ tags=["Export"],
+)
+async def cancel_export(job_id: str) -> OperationResponse:
+ job = export_service.cancel_export(job_id)
+ if job is None:
+ raise ApiError(
+ 404, "EXPORT_JOB_NOT_FOUND", "export job not found", {"job_id": job_id}
+ )
+ return OperationResponse(
+ status="accepted", resource_id=job_id, message="Export cancellation accepted."
+ )
diff --git a/backend/pyproject.toml b/backend/pyproject.toml
index 6c3d4c0..6862d00 100644
--- a/backend/pyproject.toml
+++ b/backend/pyproject.toml
@@ -9,6 +9,7 @@ dependencies = [
"fastapi>=0.116,<1.0",
"httpx>=0.28,<1.0",
"jsonschema>=4.25,<5.0",
+ "mistune>=3.0,<4.0",
"pyyaml>=6.0,<7.0",
"referencing>=0.36,<1.0",
"sqlite-vec>=0.1.9",
diff --git a/backend/tests/test_export.py b/backend/tests/test_export.py
new file mode 100644
index 0000000..9bbad58
--- /dev/null
+++ b/backend/tests/test_export.py
@@ -0,0 +1,303 @@
+"""Export Service 的单元与端到端测试。
+
+沿用 conftest 隔离机制:APP_DATA_DIR / DB / Vault / exports 目录都落在临时目录,
+不读写真实数据。导出采用「创建即 queued + 后台 Task 执行」的异步模型,测试在同一
+事件循环内创建并等待后台任务结束,得到终态 ExportJob 后再断言。
+"""
+
+from __future__ import annotations
+
+import asyncio
+from datetime import datetime, timedelta, timezone
+
+import pytest
+from pydantic import ValidationError
+
+from app.config import get_settings
+from app.contracts import (
+ ExportFormat,
+ ExportOptions,
+ ExportRequest,
+ ExportSource,
+ ExportSourceType,
+ ExportStatus,
+)
+from app.errors import ApiError
+from app.export import service as export_service
+from app.export.exporters.html import HtmlExporter
+from app.export.markdown import parse_document
+
+MD = """# 进程调度
+
+一些 **加粗** 和 *斜体*,[链接](https://a.b) 与 `code`。
+
+- 项目一
+- 项目二
+
+```python
+print(1)
+```
+
+```mermaid
+graph LR
+```
+
+```function_plot
+y = x
+```
+
+| a | b |
+|---|---|
+| 1 | 2 |
+
+行内 $x^2$ 与块级
+$$
+y = mx + b
+$$
+"""
+
+
+@pytest.fixture(autouse=True)
+def _reset_export_state():
+ """清空内存注册表,避免跨用例的任务/取消标志互相污染。"""
+ export_service._jobs.clear()
+ export_service._tasks.clear()
+ export_service._cancel_flags.clear()
+ yield
+ export_service._jobs.clear()
+ export_service._tasks.clear()
+ export_service._cancel_flags.clear()
+
+
+def _create_and_wait(request: ExportRequest) -> object:
+ """创建导出并在同一事件循环内等待后台任务结束,返回终态 ExportJob。"""
+
+ async def _execute():
+ job = await export_service.create_export(request)
+ return await export_service.wait_for_export(job.job_id)
+
+ return asyncio.run(_execute())
+
+
+# --------------------------------------------------------------------------- #
+# markdown → Document AST
+# --------------------------------------------------------------------------- #
+def _types(nodes) -> list[str]:
+ return [n.type for n in nodes]
+
+
+def test_parse_document_heading_and_inline() -> None:
+ doc = parse_document("# 标题\n\n一段 **加粗** 和 [链接](https://a.b)。")
+
+ assert doc.type == "document"
+ heading = doc.children[0]
+ assert heading.type == "heading"
+ assert heading.attributes["level"] == 1
+
+ para = doc.children[1]
+ assert para.type == "paragraph"
+ kinds = _types(para.children)
+ assert "text" in kinds
+ assert "strong" in kinds
+ assert "link" in kinds
+
+ link = next(c for c in para.children if c.type == "link")
+ assert link.attributes["href"] == "https://a.b"
+
+
+def test_parse_document_list_and_code_fencing() -> None:
+ doc = parse_document("- a\n- b\n\n```mermaid\ngraph LR\n```\n\n```function_plot\ny=x\n```\n\n```python\nx\n```")
+
+ kinds = [c.type for c in doc.children]
+ assert kinds[0] == "list"
+ assert kinds[1] == "mermaid"
+ assert kinds[2] == "function_plot"
+ assert kinds[3] == "code_block"
+
+ code = doc.children[3]
+ assert code.attributes["language"] == "python"
+ assert code.text == "x"
+
+
+def test_parse_document_table_and_math() -> None:
+ doc = parse_document("| a | b |\n|---|---|\n| 1 | 2 |\n\n$x^2$\n\n$$\ny=mx\n$$")
+
+ table = doc.children[0]
+ assert table.type == "table"
+ assert table.children[0].type == "table_row"
+ assert table.children[0].children[0].attributes["head"] is True
+
+ # 表格后是「行内数学所在段落」与「块级数学」
+ kinds = [c.type for c in doc.children[1:]]
+ assert "paragraph" in kinds
+ assert "math_block" in kinds
+
+
+# --------------------------------------------------------------------------- #
+# HtmlExporter
+# --------------------------------------------------------------------------- #
+async def _render(markdown: str, *, title: str = "") -> str:
+ doc = parse_document(markdown)
+ doc.attributes["title"] = title
+ result = await HtmlExporter().export(doc, ExportOptions())
+ return result.content.decode("utf-8")
+
+
+def test_html_exporter_renders_basic_nodes_and_escapes() -> None:
+ html = asyncio.run(_render("# 标题\n\n**加粗** [链接](https://a.b) 与 原始 。"))
+
+ assert "标题 " in html
+ assert "加粗 " in html
+ assert '链接 ' in html
+ # 原始 HTML 必须被转义,不能注入文档
+ assert "<b>原始</b>" in html
+ assert "原始 " not in html
+
+
+def test_html_exporter_marks_mermaid_and_function_plot() -> None:
+ result = asyncio.run(HtmlExporter().export(parse_document("```mermaid\ngraph LR\n```"), ExportOptions()))
+
+ html = result.content.decode("utf-8")
+ assert 'graph LR ' in html
+ assert any("mermaid" in w for w in result.warnings)
+
+
+def test_html_exporter_include_title_and_metadata() -> None:
+ doc = parse_document("正文")
+ doc.attributes["title"] = "操作系统复习"
+ doc.attributes["metadata"] = {"tags": ["os", "复习"]}
+
+ opts = ExportOptions(include_title=True, include_metadata=True)
+ result = asyncio.run(HtmlExporter().export(doc, opts))
+ html = result.content.decode("utf-8")
+
+ assert '操作系统复习 ' in html
+ assert "os, 复习" in html
+
+
+# --------------------------------------------------------------------------- #
+# ExportService
+# --------------------------------------------------------------------------- #
+def _markdown_request(markdown: str, *, format: ExportFormat = ExportFormat.html) -> ExportRequest:
+ return ExportRequest(
+ source=ExportSource(type=ExportSourceType.markdown, markdown=markdown),
+ format=format,
+ )
+
+
+def test_export_markdown_source_completes_and_writes_file() -> None:
+ finished = _create_and_wait(_markdown_request(MD))
+
+ assert finished.status == ExportStatus.completed
+ assert finished.file is not None
+ assert finished.file.mime_type == "text/html"
+ assert finished.file.size > 0
+ assert len(finished.file.sha256) == 64
+
+ path = get_settings().exports_path / f"{finished.job_id}.html"
+ assert path.exists()
+ content = path.read_text(encoding="utf-8")
+ assert "进程调度" in content
+
+
+def test_export_note_source_resolves_title_and_metadata() -> None:
+ from app.services import note_service
+
+ async def _go():
+ note = await note_service.create_note(
+ title="操作系统复习", markdown="# 进程调度\n\n内容。", folder="导出", tags=["os"]
+ )
+ request = ExportRequest(
+ source=ExportSource(type=ExportSourceType.note, note_id=note.note_id),
+ format=ExportFormat.html,
+ options=ExportOptions(include_metadata=True),
+ )
+ job = await export_service.create_export(request)
+ return await export_service.wait_for_export(job.job_id)
+
+ finished = asyncio.run(_go())
+ assert finished.status == ExportStatus.completed
+ assert finished.file is not None
+ assert finished.file.file_name == "操作系统复习.html"
+ content = (get_settings().exports_path / f"{finished.job_id}.html").read_text(encoding="utf-8")
+ assert "操作系统复习" in content
+ assert "进程调度" in content
+
+
+def test_export_pdf_unsupported() -> None:
+ with pytest.raises(ApiError) as exc:
+ asyncio.run(
+ export_service.create_export(_markdown_request("# x", format=ExportFormat.pdf))
+ )
+ assert exc.value.status_code == 400
+ assert exc.value.code == "EXPORT_FORMAT_UNSUPPORTED"
+
+
+def test_export_unknown_note_404() -> None:
+ request = ExportRequest(
+ source=ExportSource(type=ExportSourceType.note, note_id="note_missing"),
+ format=ExportFormat.html,
+ )
+ with pytest.raises(ApiError) as exc:
+ asyncio.run(export_service.create_export(request))
+ assert exc.value.status_code == 404
+ assert exc.value.code == "EXPORT_SOURCE_NOT_FOUND"
+
+
+def test_export_empty_markdown_invalid() -> None:
+ with pytest.raises(ApiError) as exc:
+ asyncio.run(export_service.create_export(_markdown_request(" ")))
+ assert exc.value.status_code == 400
+ assert exc.value.code == "EXPORT_OPTIONS_INVALID"
+
+
+def test_export_cancel_queued_job() -> None:
+ async def _go():
+ job = await export_service.create_export(_markdown_request("# x"))
+ cancelled = export_service.cancel_export(job.job_id)
+ assert cancelled is not None
+ return await export_service.wait_for_export(job.job_id)
+
+ finished = asyncio.run(_go())
+ assert finished.status == ExportStatus.cancelled
+ assert finished.file is None
+
+
+def test_export_file_expired_410() -> None:
+ async def _go():
+ job = await export_service.create_export(_markdown_request("# x"))
+ finished = await export_service.wait_for_export(job.job_id)
+ past = datetime.now(timezone.utc) - timedelta(hours=1)
+ export_service._jobs[job.job_id] = finished.model_copy(
+ update={"file": finished.file.model_copy(update={"expires_at": past})}
+ )
+ return job.job_id
+
+ job_id = asyncio.run(_go())
+ with pytest.raises(ApiError) as exc:
+ export_service.get_export_file(job_id)
+ assert exc.value.status_code == 410
+ assert exc.value.code == "EXPORT_FILE_EXPIRED"
+
+
+def test_export_list_and_get() -> None:
+ finished = _create_and_wait(_markdown_request("# 列表测试"))
+
+ items, total = export_service.list_exports(limit=50, offset=0)
+ assert total == 1
+ assert items[0].job_id == finished.job_id
+
+ got = export_service.get_export(finished.job_id)
+ assert got is not None and got.status == ExportStatus.completed
+
+ assert export_service.get_export("export_missing") is None
+
+
+# --------------------------------------------------------------------------- #
+# 契约校验
+# --------------------------------------------------------------------------- #
+def test_export_source_requires_matching_field() -> None:
+ with pytest.raises(ValidationError):
+ ExportSource(type=ExportSourceType.note, note_id=None)
+ with pytest.raises(ValidationError):
+ ExportSource(type=ExportSourceType.markdown, markdown=None)
diff --git a/backend/uv.lock b/backend/uv.lock
index 16eefd2..281c3eb 100644
--- a/backend/uv.lock
+++ b/backend/uv.lock
@@ -364,6 +364,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/41/45/1a4ed80516f02155c51f51e8cedb3c1902296743db0bbc66608a0db2814f/jsonschema_specifications-2025.9.1-py3-none-any.whl", hash = "sha256:98802fee3a11ee76ecaca44429fda8a41bff98b00a0f2838151b113f210cc6fe", size = 18437, upload-time = "2025-09-08T01:34:57.871Z" },
]
+[[package]]
+name = "mistune"
+version = "3.3.4"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/7b/92/328a294a6de83bacb95bed01f04e0eaff4e3616ee359fc821a5dfc539b02/mistune-3.3.4.tar.gz", hash = "sha256:58b5c96d6fcb61190dfe5fae498d2b2065f99cf61e9649418fd54cf1ada86dfe", size = 121426, upload-time = "2026-07-22T05:22:30.89Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/77/e4/288365afae98953bc01de09f686f40d8ee84578135aa7767d5d4e60b5278/mistune-3.3.4-py3-none-any.whl", hash = "sha256:ee015381e955e370962968befe1d729ab60fafb6a715ac6751763fbce38c8d4a", size = 66862, upload-time = "2026-07-22T05:22:29.419Z" },
+]
+
[[package]]
name = "notes-agent-backend"
version = "0.1.0"
@@ -373,6 +382,7 @@ dependencies = [
{ name = "fastapi" },
{ name = "httpx" },
{ name = "jsonschema" },
+ { name = "mistune" },
{ name = "pyyaml" },
{ name = "referencing" },
{ name = "sqlite-vec" },
@@ -390,6 +400,7 @@ requires-dist = [
{ name = "fastapi", specifier = ">=0.116,<1.0" },
{ name = "httpx", specifier = ">=0.28,<1.0" },
{ name = "jsonschema", specifier = ">=4.25,<5.0" },
+ { name = "mistune", specifier = ">=3.0,<4.0" },
{ name = "pyyaml", specifier = ">=6.0,<7.0" },
{ name = "referencing", specifier = ">=0.36,<1.0" },
{ name = "sqlite-vec", specifier = ">=0.1.9" },
diff --git a/docs/README.md b/docs/README.md
index a6dda3b..f1e96a1 100644
--- a/docs/README.md
+++ b/docs/README.md
@@ -31,6 +31,7 @@
- [AI Core 与 Agent Core 开发说明](development/AI-Core与Agent-Core开发说明.md)
- [Knowledge 与 Retrieval Core 开发说明](development/Knowledge与Retrieval-Core开发说明.md)
- [Benchmark 开发说明](development/Benchmark开发说明.md)
+- [Export 开发说明](development/Export开发说明.md)
- [模型提供商与模型发现开发说明](development/模型提供商与模型发现开发说明.md)
- [MCP Bridge 与 Plugin Host 开发说明](development/MCP-Bridge与Plugin-Host开发说明.md)
- [独立 MCP Server 配置中心开发说明](development/独立MCP-Server配置中心开发说明.md)
diff --git a/docs/architecture/AI笔记软件技术栈说明-团队版-v2.3.md b/docs/architecture/AI笔记软件技术栈说明-团队版-v2.3.md
index ce7ce33..dc0cb74 100644
--- a/docs/architecture/AI笔记软件技术栈说明-团队版-v2.3.md
+++ b/docs/architecture/AI笔记软件技术栈说明-团队版-v2.3.md
@@ -2354,7 +2354,7 @@ Quality
└── Retrieval 参数调优
Content Output
-├── Markdown → HTML / PDF / DOCX
+├── Markdown → HTML(已实现)/ PDF / DOCX(暂缓)
├── Mermaid 编辑、预览与静态导出
└── Function Plot 解析、预览与静态导出
@@ -2365,7 +2365,7 @@ Frontend Extension
└── Plugin Settings UI
```
-上述列表描述第二阶段技术范围,其中 stdio MCP Bridge、Plugin Command Contribution 和 Plugin Settings Contribution 后端 Contract 已实现,其余能力以各自开发说明的状态为准。每项功能必须继续经过现有 Service、Contract、Permission 和 Adapter 边界,不因 Demo 需要在 Vue 组件、Router 或 Agent Runtime 中直接绑定第三方协议。
+上述列表描述第二阶段技术范围,其中 stdio MCP Bridge、Plugin Command Contribution、Plugin Settings Contribution 后端 Contract 与 Markdown → HTML 导出已实现,其余能力以各自开发说明的状态为准。每项功能必须继续经过现有 Service、Contract、Permission 和 Adapter 边界,不因 Demo 需要在 Vue 组件、Router 或 Agent Runtime 中直接绑定第三方协议。
第三阶段处理:
diff --git a/docs/contracts/第二阶段接口契约-开发版.md b/docs/contracts/第二阶段接口契约-开发版.md
index dd0da1f..ab4fbb7 100644
--- a/docs/contracts/第二阶段接口契约-开发版.md
+++ b/docs/contracts/第二阶段接口契约-开发版.md
@@ -66,11 +66,11 @@
| Benchmark | POST | `/api/benchmarks/agent/runs` | 暂缓 | 创建 Agent Benchmark(依赖 Agent Runtime 完成后交付) |
| Benchmark | GET | `/api/benchmarks/runs` | 已实现 | 分页获取 Benchmark Run |
| Benchmark | GET/POST | `/api/benchmarks/runs/{run_id}/*` | 计划新增 | 查询、订阅、取消和读取报告 |
-| Export | POST | `/api/exports` | 计划新增 | 创建 HTML/PDF/DOCX 导出任务 |
-| Export | GET | `/api/exports` | 计划新增 | 分页获取导出任务 |
-| Export | GET | `/api/exports/{job_id}` | 计划新增 | 查询导出任务 |
-| Export | GET | `/api/exports/{job_id}/file` | 计划新增 | 下载已完成产物 |
-| Export | POST | `/api/exports/{job_id}/cancel` | 计划新增 | 取消导出任务 |
+| Export | POST | `/api/exports` | 已实现(HTML) | 创建导出任务;`pdf`/`docx` 暂缓,返回 `EXPORT_FORMAT_UNSUPPORTED` |
+| Export | GET | `/api/exports` | 已实现(HTML) | 分页获取导出任务 |
+| Export | GET | `/api/exports/{job_id}` | 已实现(HTML) | 查询导出任务 |
+| Export | GET | `/api/exports/{job_id}/file` | 已实现(HTML) | 下载已完成产物 |
+| Export | POST | `/api/exports/{job_id}/cancel` | 已实现(HTML) | 取消导出任务 |
| Theme | Host Contract | `ThemePackageService` | 计划新增 | 导入、预览、启停和卸载主题包 |
| Renderer | 内部 Contract | `StaticRenderer` | 计划新增 | Mermaid/Function Plot 预览和导出复用 |
@@ -1078,6 +1078,8 @@ VECTOR_INDEX_REBUILD_REQUIRED
## 10. Export Service
+> 实现状态:HTML 导出已实现(`backend/app/export/`),`pdf`/`docx` 暂缓——请求这两个格式返回 `EXPORT_FORMAT_UNSUPPORTED`。函数图像与 Mermaid 在 HTML 中以占位代码块保留并记 warning,静态渲染由 §10.4 的 Render Contract 在后续 PR 补齐。
+
### 10.1 创建导出任务
`POST /api/exports`,返回 `202 ExportJob`。
@@ -1088,7 +1090,7 @@ VECTOR_INDEX_REBUILD_REQUIRED
"type": "note",
"note_id": "note_123"
},
- "format": "pdf",
+ "format": "html",
"options": {
"theme_id": "light",
"include_title": true,
@@ -1099,7 +1101,7 @@ VECTOR_INDEX_REBUILD_REQUIRED
}
```
-`source.type` 首批支持 `note` 和 `markdown`。`markdown` 来源用于尚未保存的预览,字段大小受限且不持久化到 Trace。`format` 固定为 `html`、`pdf`、`docx`。
+`source.type` 首批支持 `note` 和 `markdown`。`note` 来源通过 `source.note_id` 引用已建索引笔记;`markdown` 来源用于尚未保存的预览,内容放在 `source.markdown` 字段,大小限制为 200 000 字符、不持久化到 Trace。`format` 可取 `html`、`pdf`、`docx`,但当前仅 `html` 已实现,`pdf`/`docx` 返回 `EXPORT_FORMAT_UNSUPPORTED`。
响应:
@@ -1107,11 +1109,12 @@ VECTOR_INDEX_REBUILD_REQUIRED
{
"job_id": "export_123",
"status": "queued",
- "format": "pdf",
+ "format": "html",
"progress": null,
"file": null,
"warnings": [],
"error": null,
+ "error_code": null,
"created_at": "2026-08-31T10:30:00Z",
"started_at": null,
"completed_at": null
@@ -1127,14 +1130,14 @@ VECTOR_INDEX_REBUILD_REQUIRED
| POST | `/api/exports/{job_id}/cancel` | `OperationResponse` |
| GET | `/api/exports/{job_id}/file` | 文件流 |
-下载响应设置正确 `Content-Type`、经过清理的 `Content-Disposition` 文件名和 `Content-Length`。未完成、失败或过期 Job 不返回空文件。
+下载响应设置正确 `Content-Type`、经过清理的 `Content-Disposition` 文件名和 `Content-Length`。未完成、失败或过期的 Job 不返回空文件:未完成/失败返回 `EXPORT_JOB_NOT_FOUND`(404),产物过期(超过 `expires_at`)返回 `EXPORT_FILE_EXPIRED`(410)。
完成 Job 的 file:
```json
{
- "file_name": "操作系统复习.pdf",
- "mime_type": "application/pdf",
+ "file_name": "操作系统复习.html",
+ "mime_type": "text/html",
"size": 1048576,
"sha256": "...",
"expires_at": "2026-09-01T10:30:00Z"
diff --git a/docs/development/Export开发说明.md b/docs/development/Export开发说明.md
new file mode 100644
index 0000000..dac42f7
--- /dev/null
+++ b/docs/development/Export开发说明.md
@@ -0,0 +1,85 @@
+# Export 开发说明
+
+> 所属模块:Export Service(后端,负责人 yxx)。本次交付「多格式文档导出」第一步:Markdown → HTML 的完整生命周期;PDF/DOCX 与函数图像静态渲染在后续 PR 补齐。契约对应 [第二阶段接口契约 §10](../contracts/第二阶段接口契约-开发版.md)。
+
+## 定位
+
+Export Service 把笔记或未保存的 Markdown 文本渲染为可下载的 HTML 文件。采用与 Benchmark 一致的「创建即返回 queued、后台 asyncio.Task 执行」的内存模型,产物带 24h 过期时间,过期后不可下载。导出是轮询式(无 SSE 事件流),客户端通过 `GET /api/exports/{job_id}` 轮询状态,完成后走 `GET /api/exports/{job_id}/file` 下载。
+
+## 模块布局
+
+```text
+backend/app/export/
+├── __init__.py 包说明
+├── document.py Document AST 内部协议 + DocumentExporter Protocol + ExportResult
+├── markdown.py mistune 'ast' renderer → Document AST
+├── exporters/
+│ ├── __init__.py
+│ └── html.py HtmlExporter(Document AST → 完整 HTML5)
+└── service.py ExportService(注册表 + 后台渲染 + 取消 + 产物生命周期)
+```
+
+HTTP DTO(`ExportStatus` / `ExportFormat` / `ExportSource` / `ExportOptions` / `ExportJob` 等)放在 [app/contracts.py](../../backend/app/contracts.py),与 Benchmark DTO 同层;`DocumentNode` / `ExportResult` 属导出器内部协议,放在 `export/document.py`,不进入 HTTP 契约。
+
+## 接口
+
+| 方法 | 路径 | 用途 |
+| --- | --- | --- |
+| POST | `/api/exports` | 创建导出任务(202) |
+| GET | `/api/exports?status=&format=&limit=&offset=` | 分页获取任务 |
+| GET | `/api/exports/{job_id}` | 查询任务状态 |
+| GET | `/api/exports/{job_id}/file` | 下载已完成产物 |
+| POST | `/api/exports/{job_id}/cancel` | 取消任务 |
+
+`source.type` 支持 `note`(引用已建索引笔记)与 `markdown`(未保存预览,字段为 `source.markdown`,上限 200 000 字符)。当前仅 `format=html` 实现,`pdf`/`docx` 返回 `EXPORT_FORMAT_UNSUPPORTED`。
+
+## Markdown → Document AST
+
+解析用 [mistune](https://github.com/lepture/mistune) 的内置 `renderer="ast"`(非自写 `BaseRenderer`),因为 mistune 的行内渲染按字符串拼接、无法承载结构化子节点;ast renderer 直接给出带 `children`/`attrs`/`raw` 的 token 树,`_AstMapper` 只做 token → `DocumentNode` 的搬运,不掺入任何 HTML。插件启用 `table`、`math`、`url`、`task_lists`。
+
+fenced code 按语言分流:`mermaid` → `mermaid` 节点、`function_plot`/`functionplot` → `function_plot` 节点,其余 → `code_block`(`attributes.language`)。`node_id` 按遍历顺序 `node_{seq:03d}` 生成,仅渲染内部使用,无需跨请求稳定。
+
+## HtmlExporter
+
+递归渲染 Document AST 为完整 HTML5 文档(`` + `` 内嵌基础 CSS + ``),标题/正文/元信息文本一律 `html.escape`。`mermaid` 与 `function_plot` 无法静态表达,渲染为占位 ``/`` 并记 warning,不静默丢失;`code_theme` 仅作为代码容器 class,不引入 JS 高亮库。无法表示的节点统一 `warnings.append(...)` 跳过。
+
+## 运行生命周期
+
+`queued → running → completed | failed | cancelled`。
+
+- 创建时校验:`format` 非 html → `EXPORT_FORMAT_UNSUPPORTED`;`note` 源不存在 → `EXPORT_SOURCE_NOT_FOUND`(404);`markdown` 源为空或超上限 → `EXPORT_OPTIONS_INVALID`。
+- 内存注册表上限 `MAX_JOBS=100`,超限只淘汰终态任务;满容量且全为活动任务时返回 `EXPORT_CAPACITY_EXCEEDED`(429)。
+- 后台渲染在解析前后各让出一次执行权,使「创建后立即取消」的 queued 任务能及时进入 cancelled。
+- 失败只向公开响应暴露项目错误码与安全消息,详细异常进入日志。
+
+## 产物生命周期
+
+产物写入 `settings.exports_path`(默认 `backend/data/exports/`,可通过 `APP_EXPORTS_PATH` 覆盖,已加入 `.gitignore`),文件名为 `{job_id}.html`,下载 `Content-Disposition` 用 `_safe_download_name` 清洗标题得到。`ExportFile` 记录 `sha256`、`size` 与 `expires_at`(`completed_at + 24h`),过期返回 `EXPORT_FILE_EXPIRED`(410)。
+
+## 错误码
+
+```text
+EXPORT_SOURCE_NOT_FOUND 404
+EXPORT_FORMAT_UNSUPPORTED 400
+EXPORT_OPTIONS_INVALID 400
+EXPORT_RENDER_FAILED 500
+EXPORT_UNSUPPORTED_CONTENT 422
+EXPORT_JOB_NOT_FOUND 404
+EXPORT_FILE_EXPIRED 410
+EXPORT_CAPACITY_EXCEEDED 429
+```
+
+## 测试
+
+```powershell
+cd backend
+uv run pytest -q
+```
+
+`tests/test_export.py` 覆盖 Markdown 解析(标题/行内/列表/代码分流/表格/数学)、HTML 渲染(标签 + 转义 + warning)、Service 端到端(note 源与 markdown 源、pdf 拒绝、未知 note、取消、list/get、过期 410)与 `ExportSource` 契约校验。
+
+## 范围外(后续 PR)
+
+- PDF / DOCX 导出(`python-docx` 等底层库在 PoC 后冻结,封装在 Exporter Adapter 内)。
+- 函数图像绘制(FunctionPlot 结构化模型 + 白名单表达式解析器 + SVG 静态渲染,契约 §10.4/§12)。
+- 代码语法高亮(当前仅 CSS class 占位)。
--
2.43.0
From 64af1f516550e0f7d4b6cb2b8651b400fb936f2b Mon Sep 17 00:00:00 2001
From: yxx <2412119399@qq.com>
Date: Fri, 4 Sep 2026 22:24:16 +0800
Subject: [PATCH 2/8] =?UTF-8?q?fix(export):=20=E4=BF=AE=E5=A4=8D=20PR=20#1?=
=?UTF-8?q?7=20=E5=AE=A1=E9=98=85=E9=97=AE=E9=A2=98=EF=BC=881=20P1=20+=205?=
=?UTF-8?q?=20P2=EF=BC=89?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
- P1 链接/图片 URL 协议白名单校验,危险协议降级为纯文本 + warning
- P2 图片 AST 字段映射(src=attrs.url,alt 取 children 文本)
- P2 原始 HTML 块转义保留,正文不丢失 + warning
- P2 过期/淘汰/重启清理导出产物文件
- P2 解析与渲染移入 asyncio.to_thread,运行中取消生效
- P2 function-plot 围栏别名补全
- 回归测试覆盖全部修复
Co-Authored-By: Claude Code
---
backend/app/export/exporters/html.py | 51 ++++++++++--
backend/app/export/markdown.py | 27 +++++--
backend/app/export/service.py | 48 ++++++++++--
backend/app/main.py | 3 +
backend/tests/test_export.py | 111 +++++++++++++++++++++++++++
5 files changed, 221 insertions(+), 19 deletions(-)
diff --git a/backend/app/export/exporters/html.py b/backend/app/export/exporters/html.py
index 0bcad16..ca8436f 100644
--- a/backend/app/export/exporters/html.py
+++ b/backend/app/export/exporters/html.py
@@ -8,12 +8,28 @@ from __future__ import annotations
import html
from datetime import datetime
+from urllib.parse import urlparse
from app.contracts import ExportOptions
from app.export.document import Document, DocumentNode, ExportResult
_MERMAID_WARNING = "mermaid 需前端渲染,已保留为占位代码块"
_FUNCTION_PLOT_WARNING = "函数图像渲染将在后续版本提供,已保留为占位代码块"
+_RAW_HTML_WARNING = "原始 HTML 已按纯文本转义保留"
+
+# 链接/图片地址允许的协议;无 scheme 的相对地址视为安全,其余协议一律降级
+_ALLOWED_URL_SCHEMES = frozenset({"http", "https", "mailto"})
+
+
+def _safe_url(url: str) -> str | None:
+ """校验 URL 协议;安全返回原串,不安全返回 None。"""
+ url = url.strip()
+ if not url:
+ return None
+ scheme = urlparse(url).scheme.lower()
+ if scheme and scheme not in _ALLOWED_URL_SCHEMES:
+ return None
+ return url
_BASE_CSS = """
body { margin: 0; background: #f6f7f9; color: #1f2328; font: 15px/1.7 -apple-system, 'Segoe UI', 'Microsoft YaHei', sans-serif; }
@@ -46,7 +62,8 @@ hr { border: none; border-top: 1px solid #d0d7de; margin: 1.4em 0; }
class HtmlExporter:
"""实现 DocumentExporter:递归渲染 Document AST 为完整 HTML5 文档。"""
- async def export(self, document: Document, options: ExportOptions) -> ExportResult:
+ def render(self, document: Document, options: ExportOptions) -> ExportResult:
+ """同步渲染;CPU 密集,调用方应放入线程执行,避免阻塞事件循环。"""
self._options = options
warnings: list[str] = []
body = self._render_children(document.children, warnings)
@@ -55,6 +72,10 @@ class HtmlExporter:
content=content.encode("utf-8"), mime_type="text/html", warnings=warnings
)
+ async def export(self, document: Document, options: ExportOptions) -> ExportResult:
+ """契约要求的 async 接口;渲染本身同步,直接转发到 render。"""
+ return self.render(document, options)
+
def _assemble(
self, document: Document, options: ExportOptions, body: str, warnings: list[str]
) -> str:
@@ -179,6 +200,11 @@ class HtmlExporter:
def _render_math_block(self, node: DocumentNode, warnings: list[str]) -> str:
return f'$${html.escape(node.text)}$$
'
+ def _render_html_block(self, node: DocumentNode, warnings: list[str]) -> str:
+ # 原始 HTML 不可信,转义为纯文本展示,保证正文不丢且无注入风险
+ warnings.append(_RAW_HTML_WARNING)
+ return f'{html.escape(node.text)}
'
+
# --- 行内 ---
def _render_text(self, node: DocumentNode, warnings: list[str]) -> str:
return html.escape(node.text)
@@ -190,21 +216,32 @@ class HtmlExporter:
return f"{self._render_children(node.children, warnings)} "
def _render_link(self, node: DocumentNode, warnings: list[str]) -> str:
- href = html.escape(str(node.attributes.get("href") or ""))
+ inner = self._render_children(node.children, warnings)
+ href = str(node.attributes.get("href") or "")
+ safe_href = _safe_url(href)
+ if safe_href is None:
+ # 危险协议(如 javascript:)降级为纯文本,不输出可点击链接
+ warnings.append(f"链接协议不安全,已降级为纯文本:{href!r}")
+ return inner
title = str(node.attributes.get("title") or "")
- attrs = [f'href="{href}"']
+ attrs = [f'href="{html.escape(safe_href)}"']
if title:
attrs.append(f'title="{html.escape(title)}"')
- return f"{self._render_children(node.children, warnings)} "
+ return f"{inner} "
def _render_codespan(self, node: DocumentNode, warnings: list[str]) -> str:
return f"{html.escape(node.text)}"
def _render_image(self, node: DocumentNode, warnings: list[str]) -> str:
- src = html.escape(str(node.attributes.get("src") or ""))
- alt = html.escape(str(node.attributes.get("alt") or ""))
+ src = str(node.attributes.get("src") or "")
+ alt = str(node.attributes.get("alt") or "")
+ safe_src = _safe_url(src)
+ if safe_src is None:
+ # 危险协议(如 data:/javascript:)跳过图片,仅输出 alt 文本
+ warnings.append(f"图片地址不安全,已跳过:{src!r}")
+ return html.escape(alt) if alt else ""
title = str(node.attributes.get("title") or "")
- attrs = [f'src="{src}"', f'alt="{alt}"']
+ attrs = [f'src="{html.escape(safe_src)}"', f'alt="{html.escape(alt)}"']
if title:
attrs.append(f'title="{html.escape(title)}"')
return f" "
diff --git a/backend/app/export/markdown.py b/backend/app/export/markdown.py
index 181d190..bf9934d 100644
--- a/backend/app/export/markdown.py
+++ b/backend/app/export/markdown.py
@@ -15,7 +15,7 @@ _PLUGINS = ["table", "math", "url", "task_lists"]
# fenced code 语言分流:命中则转为专用节点,其余按普通代码块
_MERMAID_LANG = "mermaid"
-_FUNCTION_PLOT_LANGS = {"function_plot", "functionplot"}
+_FUNCTION_PLOT_LANGS = {"function-plot", "function_plot", "functionplot"}
def parse_document(markdown: str) -> Document:
@@ -85,10 +85,19 @@ class _AstMapper:
return DocumentNode(type="thematic_break", node_id=self.next_id())
if kind == "blank_line":
return None
- # 未知块级 token(如 block_html)保守保留原文,避免静默丢失
+ if kind == "block_html":
+ # 原始 HTML 块降级为纯文本节点,由 HtmlExporter 转义并记 warning,避免静默丢失正文
+ return DocumentNode(
+ type="html_block", node_id=self.next_id(), text=token.get("raw", "")
+ )
+ # 未知块级 token 保守保留原文;映射为带 text 子节点的 paragraph,避免被渲染层丢弃
raw = token.get("raw", "")
if raw:
- return DocumentNode(type="paragraph", node_id=self.next_id(), text=raw)
+ return DocumentNode(
+ type="paragraph",
+ node_id=self.next_id(),
+ children=[DocumentNode(type="text", node_id=self.next_id(), text=raw)],
+ )
return None
def map_list_item(self, token: dict) -> DocumentNode:
@@ -144,10 +153,16 @@ class _AstMapper:
if kind == "codespan":
return DocumentNode(type="codespan", node_id=self.next_id(), text=token.get("raw", ""))
if kind == "image":
+ # mistune 图片 token:src 在 attrs.url,alt 来自 children 的文本,title 在 attrs.title
attrs = token.get("attrs", {})
- attributes = {"src": attrs.get("src", "")}
- if attrs.get("alt"):
- attributes["alt"] = attrs["alt"]
+ alt = "".join(
+ child.get("raw", "")
+ for child in token.get("children", [])
+ if child.get("type") == "text"
+ )
+ attributes = {"src": attrs.get("url", "")}
+ if alt:
+ attributes["alt"] = alt
if attrs.get("title"):
attributes["title"] = attrs["title"]
return DocumentNode(type="image", node_id=self.next_id(), attributes=attributes)
diff --git a/backend/app/export/service.py b/backend/app/export/service.py
index 73361ad..6730246 100644
--- a/backend/app/export/service.py
+++ b/backend/app/export/service.py
@@ -28,7 +28,7 @@ from app.contracts import (
ExportStatus,
)
from app.errors import ApiError
-from app.export.document import Document
+from app.export.document import Document, ExportResult
from app.export.exporters.html import HtmlExporter
from app.export.markdown import parse_document
from app.services import note_service
@@ -61,10 +61,44 @@ def _safe_download_name(title: str) -> str:
return name[:80]
+def _export_path(job_id: str) -> Path:
+ return get_settings().exports_path / f"{job_id}.html"
+
+
+def _delete_file(job_id: str) -> None:
+ """删除导出产物文件;文件不存在时忽略。"""
+ try:
+ _export_path(job_id).unlink(missing_ok=True)
+ except OSError:
+ logger.warning("Failed to delete export file: %s", job_id)
+
+
+def cleanup_orphan_files() -> int:
+ """清理 exports 目录下无对应内存任务的孤立产物(服务重启后调用)。"""
+ exports_dir = get_settings().exports_path
+ if not exports_dir.is_dir():
+ return 0
+ removed = 0
+ for path in exports_dir.glob("*.html"):
+ if path.stem not in _jobs:
+ try:
+ path.unlink()
+ removed += 1
+ except OSError:
+ logger.warning("Failed to delete orphan export file: %s", path)
+ return removed
+
+
+def _render_document(document: Document, options: ExportOptions) -> ExportResult:
+ """同步渲染辅助,供 asyncio.to_thread 调用;每次新建实例避免跨线程复用。"""
+ return HtmlExporter().render(document, options)
+
+
def _forget(job_id: str) -> None:
_jobs.pop(job_id, None)
_tasks.pop(job_id, None)
_cancel_flags.pop(job_id, None)
+ _delete_file(job_id)
def _evict_terminal() -> bool:
@@ -166,19 +200,20 @@ async def _execute(
if cancel_event.is_set():
raise ExportCancelled()
- document = parse_document(markdown)
+ # 解析与渲染都是 CPU 密集的同步工作,放入线程执行避免阻塞事件循环,
+ # 使运行中的取消能在渲染边界生效;写文件前再次检查取消。
+ document = await asyncio.to_thread(parse_document, markdown)
document.attributes["title"] = title
if metadata:
document.attributes["metadata"] = metadata
- exporter = HtmlExporter()
- result = await exporter.export(document, options)
+ result = await asyncio.to_thread(_render_document, document, options)
if cancel_event.is_set():
raise ExportCancelled()
out_dir = get_settings().exports_path
out_dir.mkdir(parents=True, exist_ok=True)
- path = out_dir / f"{job_id}.html"
+ path = _export_path(job_id)
path.write_bytes(result.content)
completed_at = _now()
@@ -260,8 +295,9 @@ def get_export_file(job_id: str) -> Path:
404, "EXPORT_JOB_NOT_FOUND", "export file not ready", {"job_id": job_id}
)
if job.file.expires_at <= _now():
+ _forget(job_id) # 过期即清理内存记录与产物文件
raise ApiError(410, "EXPORT_FILE_EXPIRED", "export file has expired", {"job_id": job_id})
- return get_settings().exports_path / f"{job_id}.html"
+ return _export_path(job_id)
async def wait_for_export(job_id: str) -> ExportJob | None:
diff --git a/backend/app/main.py b/backend/app/main.py
index bfaaf91..0ed0bba 100644
--- a/backend/app/main.py
+++ b/backend/app/main.py
@@ -8,6 +8,7 @@ from starlette.exceptions import HTTPException as StarletteHttpException
from app.config import get_settings
from app.container import container
from app.errors import ApiError, api_error_handler, http_error_handler, validation_error_handler
+from app.export import service as export_service
from app.routes import router as api_router
from app.schemas import HealthResponse, ServiceStatusResponse
@@ -16,6 +17,8 @@ settings = get_settings()
@asynccontextmanager
async def lifespan(_: FastAPI):
+ # 重启后内存注册表为空,清理上一次运行遗留的导出产物,避免磁盘垃圾堆积。
+ export_service.cleanup_orphan_files()
yield
# 第三方 MCP Server 必须跟随 AI Core 退出,不能遗留孤儿进程。
container.plugins.shutdown()
diff --git a/backend/tests/test_export.py b/backend/tests/test_export.py
index 9bbad58..38fda15 100644
--- a/backend/tests/test_export.py
+++ b/backend/tests/test_export.py
@@ -16,6 +16,7 @@ from pydantic import ValidationError
from app.config import get_settings
from app.contracts import (
ExportFormat,
+ ExportJob,
ExportOptions,
ExportRequest,
ExportSource,
@@ -133,6 +134,21 @@ def test_parse_document_table_and_math() -> None:
assert "math_block" in kinds
+def test_parse_document_image_maps_src_alt_title() -> None:
+ doc = parse_document('')
+ img = doc.children[0].children[0]
+ assert img.type == "image"
+ assert img.attributes["src"] == "https://a.b/img.png"
+ assert img.attributes["alt"] == "替代文本"
+ assert img.attributes["title"] == "标题"
+
+
+def test_parse_document_function_plot_dash_alias() -> None:
+ doc = parse_document("```function-plot\ny = x^2\n```")
+ assert doc.children[0].type == "function_plot"
+ assert doc.children[0].text == "y = x^2"
+
+
# --------------------------------------------------------------------------- #
# HtmlExporter
# --------------------------------------------------------------------------- #
@@ -162,6 +178,38 @@ def test_html_exporter_marks_mermaid_and_function_plot() -> None:
assert any("mermaid" in w for w in result.warnings)
+def test_html_exporter_rejects_unsafe_link_protocol() -> None:
+ result = asyncio.run(
+ HtmlExporter().export(parse_document("[点我](javascript:alert(1))"), ExportOptions())
+ )
+ html = result.content.decode("utf-8")
+ assert "javascript:" not in html
+ assert "点我" in html
+ assert any("不安全" in w for w in result.warnings)
+
+
+def test_html_exporter_rejects_unsafe_image_protocol() -> None:
+ result = asyncio.run(
+ HtmlExporter().export(parse_document("![alt](data:text/html,