feat(export): 新增 PDF/DOCX 导出与 StaticRenderer 内部契约

- 新增 PdfExporter(reportlab)与 DocxExporter(python-docx),实现与
  HtmlExporter 一致的同步 render + 异步 export,v1 文本优先(标题/段落/
  行内强调与链接/列表/引用/表格/代码块/数学文本),function_plot 与 mermaid
  保留源码占位并记 warning。
- service 层加 _EXPORTERS 注册表按格式分发,删除 format!=html 硬限制,
  扩展名/MIME/产物清理泛化到 html/pdf/docx 三种格式。
- 新增 app/plot/renderer.py:StaticRenderRequest + StaticRenderer Protocol +
  FunctionPlotStaticRenderer + MermaidStaticRenderer;HtmlExporter 改经
  FunctionPlotStaticRenderer 消费,去除对 render_svg 的直接依赖。
- 补齐 PDF/DOCX 魔法字节、CJK 字体、占位 warning 与 StaticRenderer 契约测试。
- 更新 Export开发说明.md。

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
yxx
2026-09-06 17:04:58 +08:00
co-authored by Claude Code
parent 966a94cad8
commit 406dd42571
11 changed files with 1334 additions and 51 deletions
+37
View File
@@ -0,0 +1,37 @@
"""导出器共享工具:URL 协议校验与占位 warning 文案。
html / pdf / docx 三个导出器共用同一套安全规则,避免各写一份导致行为漂移。
"""
from __future__ import annotations
from datetime import datetime
from urllib.parse import urlparse
# 链接/图片地址允许的协议;无 scheme 的相对地址视为安全,其余协议一律降级
ALLOWED_URL_SCHEMES = frozenset({"http", "https", "mailto"})
MERMAID_WARNING = "mermaid 需前端渲染,已保留为占位代码块"
RAW_HTML_WARNING = "原始 HTML 已按纯文本转义保留"
# PDF/DOCX 暂不支持静态渲染函数图像,统一回退源码占位
PLOT_PLACEHOLDER_WARNING = "函数图像:该格式暂不支持静态渲染,已保留为源码占位"
def safe_url(url: str) -> str | None:
"""校验 URL 协议;安全返回原串,不安全返回 None。"""
url = url.strip()
if not url:
return None
scheme = urlparse(url).scheme.lower()
if scheme and scheme not in ALLOWED_URL_SCHEMES:
return None
return url
def format_meta_value(value: object) -> str:
"""把元数据值转成可读文本:datetime 转 ISO、列表用逗号连接。"""
if isinstance(value, datetime):
return value.isoformat()
if isinstance(value, list):
return ", ".join(str(item) for item in value)
return str(value)
+330
View File
@@ -0,0 +1,330 @@
"""DocxExporterDocument AST → DOCXpython-docx)。
v1 为文本优先:标题/段落/行内强调与链接/列表/引用/表格/代码块/数学文本均可导出;
function_plot 与 mermaid 保留源码占位并记 warning。中文字体通过 Normal 样式挂载
w:eastAsia=宋体,保证 Word 打开时中文正常显示;bold/italic 由 Word 原生渲染。
"""
from __future__ import annotations
from io import BytesIO
from docx import Document as DocxDocument
from docx.enum.text import WD_ALIGN_PARAGRAPH
from docx.opc.constants import RELATIONSHIP_TYPE
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import Inches, Mm, Pt, RGBColor
from app.contracts import ExportOptions
from app.export.document import Document, DocumentNode, ExportResult
from app.export.exporters._common import (
MERMAID_WARNING,
PLOT_PLACEHOLDER_WARNING,
RAW_HTML_WARNING,
format_meta_value,
safe_url,
)
_MIME = "application/vnd.openxmlformats-officedocument.wordprocessingml.document"
_HEADING_SIZES = {1: 20, 2: 16, 3: 14, 4: 12, 5: 11, 6: 10.5}
def _plain_text(children: list[DocumentNode]) -> str:
"""递归拼接行内节点的纯文本,供标题/链接文字等需要纯文本处使用。"""
parts: list[str] = []
for child in children:
if child.type == "text":
parts.append(child.text)
elif child.children:
parts.append(_plain_text(child.children))
elif child.text:
parts.append(child.text)
return "".join(parts)
class DocxExporter:
"""实现 DocumentExporter:递归渲染 Document AST 为 DOCX 字节流。"""
def render(self, document: Document, options: ExportOptions) -> ExportResult:
"""同步渲染;CPU 密集,调用方应放入线程执行,避免阻塞事件循环。"""
self._doc = DocxDocument()
self._configure_normal_style()
self._configure_page(options)
warnings: list[str] = []
self._render_header(document, options, warnings)
self._render_children(document.children, warnings)
buf = BytesIO()
self._doc.save(buf)
return ExportResult(content=buf.getvalue(), mime_type=_MIME, warnings=warnings)
async def export(self, document: Document, options: ExportOptions) -> ExportResult:
"""契约要求的 async 接口;渲染本身同步,直接转发到 render。"""
return self.render(document, options)
def _configure_normal_style(self) -> None:
"""Normal 样式挂载 CJK 字体;拉丁用 Calibri,中文用宋体。"""
style = self._doc.styles["Normal"]
style.font.name = "Calibri"
style.font.size = Pt(11)
rfonts = style.element.get_or_add_rPr().get_or_add_rFonts()
rfonts.set(qn("w:eastAsia"), "宋体")
def _configure_page(self, options: ExportOptions) -> None:
section = self._doc.sections[0]
size = (options.page_size or "A4").lower()
if size == "a4":
section.page_width = Mm(210)
section.page_height = Mm(297)
elif size == "letter":
section.page_width = Inches(8.5)
section.page_height = Inches(11)
# --- 文档头部 ---
def _render_header(self, document: Document, options: ExportOptions, warnings: list[str]) -> None:
title = str(document.attributes.get("title") or "")
if options.include_title and title:
p = self._doc.add_paragraph()
run = p.add_run(title)
run.bold = True
run.font.size = Pt(22)
p.paragraph_format.space_after = Pt(12)
if options.include_metadata:
metadata = document.attributes.get("metadata")
if metadata:
for key, value in metadata.items():
p = self._doc.add_paragraph()
run = p.add_run(f"{key}: {format_meta_value(value)}")
run.font.size = Pt(9)
run.font.color.rgb = RGBColor(0x57, 0x60, 0x6A)
# --- 块级 ---
def _render_children(self, children: list[DocumentNode], warnings: list[str]) -> None:
for child in children:
self._render_block(child, warnings)
def _render_block(self, node: DocumentNode, warnings: list[str]) -> None:
handler = getattr(self, f"_block_{node.type}", None)
if handler is not None:
handler(node, warnings)
else:
warnings.append(f"无法表示的节点类型已跳过:{node.type}")
def _block_heading(self, node: DocumentNode, warnings: list[str]) -> None:
level = max(1, min(6, int(node.attributes.get("level", 1))))
p = self._doc.add_paragraph()
run = p.add_run(_plain_text(node.children))
run.bold = True
run.font.size = Pt(_HEADING_SIZES[level])
p.paragraph_format.space_before = Pt(14 if level <= 2 else 10)
p.paragraph_format.space_after = Pt(6)
def _block_paragraph(self, node: DocumentNode, warnings: list[str]) -> None:
p = self._doc.add_paragraph()
self._render_inline(p, node.children, warnings)
def _block_blockquote(self, node: DocumentNode, warnings: list[str]) -> None:
p = self._doc.add_paragraph()
self._render_inline(p, node.children, warnings)
p.paragraph_format.left_indent = Pt(16)
for run in p.runs:
run.font.color.rgb = RGBColor(0x57, 0x60, 0x6A)
def _block_list(self, node: DocumentNode, warnings: list[str], level: int = 0) -> None:
ordered = bool(node.attributes.get("ordered"))
for index, item in enumerate(node.children, start=1):
self._block_list_item(item, warnings, ordered, index, level)
def _block_list_item(
self,
item: DocumentNode,
warnings: list[str],
ordered: bool,
index: int,
level: int,
) -> None:
if item.attributes.get("task"):
marker = "" if item.attributes.get("checked") else ""
else:
marker = f"{index}. " if ordered else ""
indent = Pt(18 + 18 * level)
first = True
for child in item.children:
if child.type == "list":
self._block_list(child, warnings, level + 1)
continue
if child.type == "paragraph":
p = self._doc.add_paragraph()
p.paragraph_format.left_indent = indent
if first:
self._add_run(p, marker)
first = False
self._render_inline(p, child.children, warnings)
elif child.children:
# 直接行内子节点:拼进一个段落
p = self._doc.add_paragraph()
p.paragraph_format.left_indent = indent
if first:
self._add_run(p, marker)
first = False
self._render_inline(p, child.children, warnings)
else:
self._render_block(child, warnings)
first = False
def _block_table(self, node: DocumentNode, warnings: list[str]) -> None:
rows = node.children
ncols = max((len(r.children) for r in rows), default=0)
if not rows or ncols == 0:
return
table = self._doc.add_table(rows=len(rows), cols=ncols)
table.style = "Table Grid"
for ri, row in enumerate(rows):
head = bool(row.attributes.get("head"))
for ci in range(ncols):
cell = table.cell(ri, ci)
p = cell.paragraphs[0]
if ci < len(row.children):
self._render_inline(p, row.children[ci].children, warnings, bold=head)
def _block_code_block(self, node: DocumentNode, warnings: list[str]) -> None:
lines = node.text.split("\n")
p = self._doc.add_paragraph()
self._shade_paragraph(p)
p.paragraph_format.left_indent = Pt(8)
p.paragraph_format.right_indent = Pt(8)
p.paragraph_format.space_before = Pt(6)
p.paragraph_format.space_after = Pt(8)
for i, line in enumerate(lines):
run = p.add_run(line)
run.font.name = "Consolas"
run.font.size = Pt(10)
if i < len(lines) - 1:
run.add_break()
def _block_thematic_break(self, node: DocumentNode, warnings: list[str]) -> None:
p = self._doc.add_paragraph()
pPr = p._p.get_or_add_pPr()
pBdr = OxmlElement("w:pBdr")
bottom = OxmlElement("w:bottom")
bottom.set(qn("w:val"), "single")
bottom.set(qn("w:sz"), "6")
bottom.set(qn("w:space"), "1")
bottom.set(qn("w:color"), "D0D7DE")
pBdr.append(bottom)
pPr.append(pBdr)
def _block_mermaid(self, node: DocumentNode, warnings: list[str]) -> None:
warnings.append(MERMAID_WARNING)
self._block_code_block(node, warnings)
def _block_function_plot(self, node: DocumentNode, warnings: list[str]) -> None:
warnings.append(PLOT_PLACEHOLDER_WARNING)
self._block_code_block(node, warnings)
def _block_math_block(self, node: DocumentNode, warnings: list[str]) -> None:
p = self._doc.add_paragraph()
p.alignment = WD_ALIGN_PARAGRAPH.CENTER
p.add_run(f"$${node.text}$$")
def _block_html_block(self, node: DocumentNode, warnings: list[str]) -> None:
# 原始 HTML 不可信,按纯文本保留正文
warnings.append(RAW_HTML_WARNING)
self._doc.add_paragraph(node.text)
# --- 行内(写入 run ---
def _render_inline(
self,
paragraph,
children: list[DocumentNode],
warnings: list[str],
bold: bool = False,
italic: bool = False,
) -> None:
for child in children:
self._render_inline_node(paragraph, child, warnings, bold, italic)
def _render_inline_node(
self, paragraph, node: DocumentNode, warnings: list[str], bold: bool, italic: bool
) -> None:
t = node.type
if t == "text":
self._add_run(paragraph, node.text, bold=bold, italic=italic)
elif t == "strong":
self._render_inline(paragraph, node.children, warnings, bold=True, italic=italic)
elif t == "emphasis":
self._render_inline(paragraph, node.children, warnings, bold=bold, italic=True)
elif t == "codespan":
self._add_run(paragraph, node.text, code=True)
elif t == "link":
inner = _plain_text(node.children)
href = str(node.attributes.get("href") or "")
safe_href = safe_url(href)
if safe_href is None:
warnings.append(f"链接协议不安全,已降级为纯文本:{href!r}")
self._render_inline(paragraph, node.children, warnings, bold, italic)
else:
self._add_hyperlink(paragraph, safe_href, inner)
elif t == "image":
src = str(node.attributes.get("src") or "")
alt = str(node.attributes.get("alt") or "")
if safe_url(src) is None:
warnings.append(f"图片地址不安全,已跳过:{src!r}")
else:
warnings.append("图片未内嵌到 DOCX,已用替代文本表示")
if alt:
self._add_run(paragraph, alt)
elif t == "math_inline":
self._add_run(paragraph, f"\\({node.text}\\)")
elif t == "linebreak":
self._add_run(paragraph, "").add_break()
else:
warnings.append(f"无法表示的行内节点已跳过:{t}")
def _add_run(self, paragraph, text: str, bold: bool = False, italic: bool = False, code: bool = False):
run = paragraph.add_run(text)
run.bold = bold
run.italic = italic
if code:
run.font.name = "Consolas"
run.font.size = Pt(10)
return run
def _add_hyperlink(self, paragraph, url: str, text: str) -> None:
"""写入可点击的超链接 runpython-docx 无公开 API,需手写 w:hyperlink)。"""
part = paragraph.part
r_id = part.relate_to(url, RELATIONSHIP_TYPE.HYPERLINK, is_external=True)
hyperlink = OxmlElement("w:hyperlink")
hyperlink.set(qn("r:id"), r_id)
run = OxmlElement("w:r")
rPr = OxmlElement("w:rPr")
rFonts = OxmlElement("w:rFonts")
rFonts.set(qn("w:ascii"), "Calibri")
rFonts.set(qn("w:hAnsi"), "Calibri")
rFonts.set(qn("w:eastAsia"), "宋体")
rPr.append(rFonts)
color = OxmlElement("w:color")
color.set(qn("w:val"), "0969DA")
rPr.append(color)
u = OxmlElement("w:u")
u.set(qn("w:val"), "single")
rPr.append(u)
run.append(rPr)
t = OxmlElement("w:t")
t.text = text
t.set(qn("xml:space"), "preserve")
run.append(t)
hyperlink.append(run)
paragraph._p.append(hyperlink)
def _shade_paragraph(self, paragraph, fill: str = "F2F2F2") -> None:
"""给段落加浅灰底纹,用于代码块占位。"""
pPr = paragraph._p.get_or_add_pPr()
shd = OxmlElement("w:shd")
shd.set(qn("w:val"), "clear")
shd.set(qn("w:color"), "auto")
shd.set(qn("w:fill"), fill)
pPr.append(shd)
+7 -4
View File
@@ -13,8 +13,7 @@ from urllib.parse import urlparse
from app.contracts import ExportOptions
from app.export.document import Document, DocumentNode, ExportResult
from app.plot.parser import parse_source
from app.plot.render import render_svg
from app.plot.renderer import FunctionPlotStaticRenderer, StaticRenderRequest
_MERMAID_WARNING = "mermaid 需前端渲染,已保留为占位代码块"
_RAW_HTML_WARNING = "原始 HTML 已按纯文本转义保留"
@@ -77,6 +76,7 @@ class HtmlExporter:
self._options = options
self._plot_count = 0
self._plot_nodes = 0
self._plot_renderer = FunctionPlotStaticRenderer()
warnings: list[str] = []
body = self._render_children(document.children, warnings)
content = self._assemble(document, options, body, warnings)
@@ -221,7 +221,10 @@ class HtmlExporter:
# 解析与渲染共同纳入局部异常回退:单个图像失败只回退占位 + warning,
# 绝不阻断整篇导出(含复杂表达式触发的 RecursionError 等异常)。
try:
parsed = parse_source(node.text)
request = StaticRenderRequest(
kind="function_plot", source=node.text, theme=self._options.theme_id
)
parsed = self._plot_renderer.parse(request)
for diag in parsed.diagnostics:
warnings.append(self._format_plot_diagnostic(diag))
if parsed.plot is None:
@@ -233,7 +236,7 @@ class HtmlExporter:
)
return f'<pre class="function-plot">{html.escape(node.text)}</pre>'
self._plot_nodes += parsed.plot.node_count
rendered = render_svg(parsed.plot)
rendered = self._plot_renderer.render_plot(parsed.plot)
except Exception as exc:
warnings.append(f"函数图像:解析或渲染失败,已回退占位({exc}")
return f'<pre class="function-plot">{html.escape(node.text)}</pre>'
+303
View File
@@ -0,0 +1,303 @@
"""PdfExporterDocument AST → PDFreportlab platypus)。
v1 为文本优先:标题/段落/行内强调与链接/列表/引用/表格/代码块/数学文本均可导出;
function_plot 与 mermaid 保留源码占位并记 warning。中文字体用 reportlab 内置
STSong-Light CID 字体,避免外部字体依赖。CID 字体无独立 bold/italic 字重,
故行内强调退化为普通文本(内容不丢、样式简化),标题靠字号区分层级。
"""
from __future__ import annotations
import html as _html
from io import BytesIO
from reportlab.lib.enums import TA_CENTER
from reportlab.lib.pagesizes import A4, letter
from reportlab.lib.styles import ParagraphStyle
from reportlab.lib.units import mm
from reportlab.pdfbase import pdfmetrics
from reportlab.pdfbase.cidfonts import UnicodeCIDFont
from reportlab.platypus import (
Paragraph,
Preformatted,
SimpleDocTemplate,
Spacer,
Table,
TableStyle,
)
from reportlab.platypus.flowables import HRFlowable
from app.contracts import ExportOptions
from app.export.document import Document, DocumentNode, ExportResult
from app.export.exporters._common import (
MERMAID_WARNING,
PLOT_PLACEHOLDER_WARNING,
RAW_HTML_WARNING,
format_meta_value,
safe_url,
)
_FONT = "STSong-Light"
pdfmetrics.registerFont(UnicodeCIDFont(_FONT))
_MIME = "application/pdf"
_PAGE_SIZES = {"a4": A4, "letter": letter}
# 标题字号随层级递减;标题不依赖粗体(CID 无粗体字重),靠字号拉开层级
_HEADING_SIZES = {1: 20, 2: 16, 3: 14, 4: 12, 5: 11, 6: 10.5}
def _make_styles() -> dict[str, ParagraphStyle]:
body = ParagraphStyle(
"pdf-body",
fontName=_FONT,
fontSize=10.5,
leading=16,
spaceAfter=6,
)
title = ParagraphStyle("pdf-title", parent=body, fontSize=22, leading=28, spaceAfter=12)
quote = ParagraphStyle(
"pdf-quote",
parent=body,
leftIndent=14,
textColor="#57606a",
spaceBefore=4,
spaceAfter=6,
)
code = ParagraphStyle(
"pdf-code",
parent=body,
fontSize=9,
leading=12,
leftIndent=6,
rightIndent=6,
backColor="#f6f8fa",
borderColor="#d0d7de",
borderWidth=0.5,
borderPadding=6,
spaceBefore=4,
spaceAfter=8,
)
math = ParagraphStyle("pdf-math", parent=body, alignment=TA_CENTER, spaceBefore=6)
cell = ParagraphStyle("pdf-cell", parent=body, fontSize=10, leading=14, spaceAfter=0)
cell_head = ParagraphStyle(
"pdf-cell-head", parent=cell, textColor="#1f2328", fontSize=10
)
meta = ParagraphStyle("pdf-meta", parent=body, fontSize=8.5, leading=13, textColor="#57606a")
styles: dict[str, ParagraphStyle] = {
"body": body,
"title": title,
"quote": quote,
"code": code,
"math": math,
"cell": cell,
"cell_head": cell_head,
"meta": meta,
}
for level, size in _HEADING_SIZES.items():
styles[f"h{level}"] = ParagraphStyle(
f"pdf-h{level}",
parent=body,
fontSize=size,
leading=size * 1.4,
spaceBefore=14 if level <= 2 else 10,
spaceAfter=6,
)
return styles
class PdfExporter:
"""实现 DocumentExporter:递归渲染 Document AST 为 PDF 字节流。"""
def render(self, document: Document, options: ExportOptions) -> ExportResult:
"""同步渲染;CPU 密集,调用方应放入线程执行,避免阻塞事件循环。"""
self._styles = _make_styles()
warnings: list[str] = []
page = _PAGE_SIZES.get((options.page_size or "A4").lower(), A4)
buf = BytesIO()
doc = SimpleDocTemplate(
buf,
pagesize=page,
leftMargin=20 * mm,
rightMargin=20 * mm,
topMargin=18 * mm,
bottomMargin=18 * mm,
title=str(document.attributes.get("title") or "") or None,
)
story: list = []
self._render_header(document, options, story)
self._render_children(document.children, story, warnings)
doc.build(story)
return ExportResult(content=buf.getvalue(), mime_type=_MIME, warnings=warnings)
async def export(self, document: Document, options: ExportOptions) -> ExportResult:
"""契约要求的 async 接口;渲染本身同步,直接转发到 render。"""
return self.render(document, options)
# --- 文档头部 ---
def _render_header(self, document: Document, options: ExportOptions, story: list) -> None:
title = str(document.attributes.get("title") or "")
if options.include_title and title:
story.append(Paragraph(_html.escape(title), self._styles["title"]))
if options.include_metadata:
metadata = document.attributes.get("metadata")
if metadata:
for key, value in metadata.items():
text = f"{_html.escape(str(key))}: {_html.escape(format_meta_value(value))}"
story.append(Paragraph(text, self._styles["meta"]))
# --- 块级 ---
def _render_children(self, children: list[DocumentNode], story: list, warnings: list[str]) -> None:
for child in children:
self._render_block(child, story, warnings)
def _render_block(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
handler = getattr(self, f"_block_{node.type}", None)
if handler is not None:
handler(node, story, warnings)
else:
warnings.append(f"无法表示的节点类型已跳过:{node.type}")
def _block_heading(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
level = max(1, min(6, int(node.attributes.get("level", 1))))
inline = self._render_inline(node.children, warnings)
story.append(Paragraph(inline, self._styles[f"h{level}"]))
def _block_paragraph(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
story.append(Paragraph(self._render_inline(node.children, warnings), self._styles["body"]))
def _block_blockquote(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
story.append(Paragraph(self._render_inline(node.children, warnings), self._styles["quote"]))
def _block_list(self, node: DocumentNode, story: list, warnings: list[str], indent: int = 14) -> None:
ordered = bool(node.attributes.get("ordered"))
for index, item in enumerate(node.children, start=1):
self._block_list_item(item, story, warnings, ordered, index, indent)
def _block_list_item(
self,
item: DocumentNode,
story: list,
warnings: list[str],
ordered: bool,
index: int,
indent: int,
) -> None:
if item.attributes.get("task"):
marker = "" if item.attributes.get("checked") else ""
else:
marker = f"{index}. " if ordered else ""
style = ParagraphStyle(
f"pdf-li-{indent}",
parent=self._styles["body"],
leftIndent=indent,
firstLineIndent=-7,
spaceAfter=2,
)
# 列表项内容通常是单个段落或直接行内节点,嵌套列表单独递归加深缩进
parts: list[str] = []
for child in item.children:
if child.type == "list":
self._block_list(child, story, warnings, indent + 14)
elif child.type == "paragraph":
parts.append(self._render_inline(child.children, warnings))
elif child.children:
parts.append(self._render_inline(child.children, warnings))
else:
parts.append(_html.escape(child.text))
story.append(Paragraph(marker + "<br/>".join(parts), style))
def _block_table(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
rows = node.children
if not rows:
return
data: list[list[Paragraph]] = []
head_row_count = 0
for row in rows:
head = bool(row.attributes.get("head"))
if head:
head_row_count += 1
cells = [
Paragraph(
self._render_inline(cell.children, warnings),
self._styles["cell_head" if cell.attributes.get("head") else "cell"],
)
for cell in row.children
]
data.append(cells)
table = Table(data, repeatRows=head_row_count)
commands = [
("GRID", (0, 0), (-1, -1), 0.5, "#d0d7de"),
("VALIGN", (0, 0), (-1, -1), "TOP"),
("LEFTPADDING", (0, 0), (-1, -1), 6),
("RIGHTPADDING", (0, 0), (-1, -1), 6),
("TOPPADDING", (0, 0), (-1, -1), 4),
("BOTTOMPADDING", (0, 0), (-1, -1), 4),
]
if head_row_count:
commands.append(("BACKGROUND", (0, 0), (-1, head_row_count - 1), "#f6f8fa"))
table.setStyle(TableStyle(commands))
story.append(table)
def _block_code_block(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
story.append(Preformatted(node.text, self._styles["code"]))
def _block_thematic_break(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
story.append(Spacer(1, 4))
story.append(HRFlowable(width="100%", color="#d0d7de", thickness=0.5))
story.append(Spacer(1, 6))
def _block_mermaid(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
warnings.append(MERMAID_WARNING)
story.append(Preformatted(node.text, self._styles["code"]))
def _block_function_plot(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
warnings.append(PLOT_PLACEHOLDER_WARNING)
story.append(Preformatted(node.text, self._styles["code"]))
def _block_math_block(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
story.append(Paragraph(f"$${_html.escape(node.text)}$$", self._styles["math"]))
def _block_html_block(self, node: DocumentNode, story: list, warnings: list[str]) -> None:
# 原始 HTML 不可信,按纯文本保留正文
warnings.append(RAW_HTML_WARNING)
story.append(Paragraph(_html.escape(node.text), self._styles["body"]))
# --- 行内(产出 reportlab Paragraph 标记文本) ---
def _render_inline(self, children: list[DocumentNode], warnings: list[str]) -> str:
return "".join(self._render_inline_node(child, warnings) for child in children)
def _render_inline_node(self, node: DocumentNode, warnings: list[str]) -> str:
t = node.type
if t == "text":
return _html.escape(node.text)
if t in ("strong", "emphasis"):
return self._render_inline(node.children, warnings)
if t == "codespan":
return f'<font size="9">{_html.escape(node.text)}</font>'
if t == "link":
inner = self._render_inline(node.children, warnings)
href = str(node.attributes.get("href") or "")
safe_href = safe_url(href)
if safe_href is None:
warnings.append(f"链接协议不安全,已降级为纯文本:{href!r}")
return inner
return f'<a href="{_html.escape(safe_href)}">{inner}</a>'
if t == "image":
src = str(node.attributes.get("src") or "")
alt = str(node.attributes.get("alt") or "")
if safe_url(src) is None:
warnings.append(f"图片地址不安全,已跳过:{src!r}")
else:
warnings.append("图片未内嵌到 PDF,已用替代文本表示")
return _html.escape(alt) if alt else ""
if t == "math_inline":
return f"\\({_html.escape(node.text)}\\)"
if t == "linebreak":
return "<br/>"
warnings.append(f"无法表示的行内节点已跳过:{t}")
return ""
+46 -27
View File
@@ -29,7 +29,9 @@ from app.contracts import (
)
from app.errors import ApiError
from app.export.document import Document, ExportResult
from app.export.exporters.docx import DocxExporter
from app.export.exporters.html import HtmlExporter
from app.export.exporters.pdf import PdfExporter
from app.export.markdown import parse_document
from app.services import note_service
@@ -52,6 +54,24 @@ FILE_TTL = timedelta(hours=24)
_INVALID_FILE_CHARS = re.compile(r'[\\/:*?"<>|]')
# 格式 → 导出器;新增格式只需在此登记,路由与任务模型无需改动
_EXPORTERS: dict[ExportFormat, type] = {
ExportFormat.html: HtmlExporter,
ExportFormat.pdf: PdfExporter,
ExportFormat.docx: DocxExporter,
}
# 格式 → 文件扩展名(用于落盘文件名与产物清理)
_EXTENSIONS: dict[ExportFormat, str] = {
ExportFormat.html: ".html",
ExportFormat.pdf: ".pdf",
ExportFormat.docx: ".docx",
}
def _extension_for(format: ExportFormat) -> str:
return _EXTENSIONS[format]
class ExportCancelled(Exception):
"""导出在渲染前被取消时抛出,用于标记 cancelled。"""
@@ -71,14 +91,14 @@ def _safe_download_name(title: str) -> str:
return name[:80]
def _export_path(job_id: str) -> Path:
return get_settings().exports_path / f"{job_id}.html"
def _export_path(job_id: str, ext: str) -> Path:
return get_settings().exports_path / f"{job_id}{ext}"
def _delete_file(job_id: str) -> None:
def _delete_file(job_id: str, ext: str) -> None:
"""删除导出产物文件;文件不存在时忽略。"""
try:
_export_path(job_id).unlink(missing_ok=True)
_export_path(job_id, ext).unlink(missing_ok=True)
except OSError:
logger.warning("Failed to delete export file: %s", job_id)
@@ -89,26 +109,30 @@ def cleanup_orphan_files() -> int:
if not exports_dir.is_dir():
return 0
removed = 0
for path in exports_dir.glob("*.html"):
if path.stem not in _jobs:
try:
path.unlink()
removed += 1
except OSError:
logger.warning("Failed to delete orphan export file: %s", path)
for ext in _EXTENSIONS.values():
for path in exports_dir.glob(f"*{ext}"):
if path.stem not in _jobs:
try:
path.unlink()
removed += 1
except OSError:
logger.warning("Failed to delete orphan export file: %s", path)
return removed
def _render_document(document: Document, options: ExportOptions) -> ExportResult:
"""同步渲染辅助,供 asyncio.to_thread 调用;每次新建实例避免跨线程复用。"""
return HtmlExporter().render(document, options)
def _render_document(document: Document, options: ExportOptions, format: ExportFormat) -> ExportResult:
"""按 format 分发到对应导出器;每次新建实例避免跨线程复用。"""
exporter_cls = _EXPORTERS[format]
return exporter_cls().render(document, options)
def _forget(job_id: str) -> None:
job = _jobs.get(job_id)
ext = _extension_for(job.format) if job is not None else ".html"
_jobs.pop(job_id, None)
_tasks.pop(job_id, None)
_cancel_flags.pop(job_id, None)
_delete_file(job_id)
_delete_file(job_id, ext)
def _evict_terminal() -> bool:
@@ -163,13 +187,6 @@ async def _resolve_source(source: ExportSource) -> tuple[str, str, dict | None]:
async def create_export(request: ExportRequest) -> ExportJob:
"""创建导出任务,立即返回 queued 的 ExportJob,由后台 Task 渲染。"""
if request.format != ExportFormat.html:
raise ApiError(
400,
"EXPORT_FORMAT_UNSUPPORTED",
"PDF/DOCX 暂未实现,当前仅支持 HTML",
{"format": request.format.value},
)
markdown, title, metadata = await _resolve_source(request.source)
if not _evict_terminal():
@@ -190,13 +207,14 @@ async def create_export(request: ExportRequest) -> ExportJob:
_jobs[job_id] = job
_cancel_flags[job_id] = asyncio.Event()
_tasks[job_id] = asyncio.create_task(
_execute(job_id, markdown, title, metadata, request.options)
_execute(job_id, request.format, markdown, title, metadata, request.options)
)
return job
async def _execute(
job_id: str,
format: ExportFormat,
markdown: str,
title: str,
metadata: dict | None,
@@ -227,15 +245,16 @@ async def _execute(
if metadata:
document.attributes["metadata"] = metadata
result = await asyncio.to_thread(_render_document, document, options)
result = await asyncio.to_thread(_render_document, document, options, format)
if cancel_event.is_set():
raise ExportCancelled()
if len(result.content) > MAX_EXPORT_BYTES:
raise ExportTooLarge()
ext = _extension_for(format)
out_dir = get_settings().exports_path
out_dir.mkdir(parents=True, exist_ok=True)
path = _export_path(job_id)
path = _export_path(job_id, ext)
path.write_bytes(result.content)
completed_at = _now()
@@ -246,7 +265,7 @@ async def _execute(
phase="completed", current=1, total=1, percent=1.0
),
"file": ExportFile(
file_name=f"{_safe_download_name(title)}.html",
file_name=f"{_safe_download_name(title)}{ext}",
mime_type=result.mime_type,
size=len(result.content),
sha256=hashlib.sha256(result.content).hexdigest(),
@@ -328,7 +347,7 @@ def get_export_file(job_id: str) -> Path:
if job.file.expires_at <= _now():
_forget(job_id) # 过期即清理内存记录与产物文件
raise ApiError(410, "EXPORT_FILE_EXPIRED", "export file has expired", {"job_id": job_id})
return _export_path(job_id)
return _export_path(job_id, _extension_for(job.format))
async def wait_for_export(job_id: str) -> ExportJob | None:
+66
View File
@@ -0,0 +1,66 @@
"""StaticRenderer 内部契约(契约 §10.4)。
把「静态可视化」抽象为统一请求/协议:导出器只面向 StaticRenderer,不再直接调用
``render_svg`` 等具体实现。后端当前仅能静态渲染函数图像;Mermaid 后端无渲染能力,
返回占位结果交前端渲染。
"""
from __future__ import annotations
from typing import Literal, Protocol
from pydantic import BaseModel, Field
from app.plot.model import FunctionPlot, FunctionPlotParseResult, StaticRenderResult
from app.plot.parser import parse_source
from app.plot.render import render_svg
class StaticRenderRequest(BaseModel):
"""一次静态渲染请求;source_hash 供缓存/去重,theme 供主题化渲染。"""
kind: Literal["function_plot", "mermaid"]
source: str
source_hash: str = ""
theme: str | None = None
width: int | None = None
height: int | None = None
class StaticRenderer(Protocol):
"""静态渲染器协议:请求 → 渲染结果(content 为可直接内嵌的标记)。"""
def render(self, request: StaticRenderRequest) -> StaticRenderResult: ...
class FunctionPlotStaticRenderer:
"""函数图像渲染器:parse_source 解析 → render_svg 输出内嵌 SVG。
``parse`` 与 ``render_plot`` 拆开,供导出器在渲染前先拿 node_count 做文档级
累计复杂度预算、并消费解析诊断。
"""
def parse(self, request: StaticRenderRequest) -> FunctionPlotParseResult:
return parse_source(request.source)
def render(self, request: StaticRenderRequest) -> StaticRenderResult:
parsed = self.parse(request)
if parsed.plot is None:
raise ValueError("function-plot source has no valid plot")
return self.render_plot(parsed.plot)
def render_plot(self, plot: FunctionPlot) -> StaticRenderResult:
return render_svg(plot)
class MermaidStaticRenderer:
"""Mermaid 后端无渲染能力:返回空占位结果,交前端渲染。"""
def render(self, request: StaticRenderRequest) -> StaticRenderResult:
return StaticRenderResult(
content="",
mime_type="text/plain",
width=0,
height=0,
warnings=["mermaid 需前端渲染,已保留为占位代码块"],
)