From f3847f3e4b4406ab590f01b3105daf6b4497df7d Mon Sep 17 00:00:00 2001 From: SkyJourney Date: Fri, 31 Jul 2026 15:19:34 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=AE=8C=E6=88=90=20DOCX=20=E9=80=9A?= =?UTF-8?q?=E7=94=A8=E7=BB=93=E6=9E=84=E6=98=A0=E5=B0=84=E4=B8=8E=E4=B8=A5?= =?UTF-8?q?=E6=A0=BC=E6=94=B6=E5=8F=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- apps/server/scripts/verify-docx-http.mjs | 67 +- apps/server/src/app.ts | 6 + docs/PROGRESS.md | 35 +- docs/V0.6.0_DOCX_THEME_AUDIT.md | 87 ++ package.json | 2 +- packages/application/README.md | 8 +- .../application/src/docx-export-service.ts | 1 + .../tests/docx-export-service.test.ts | 6 + packages/docx-engine/README.md | 16 +- .../docx-engine/assets/docx-media-filter.lua | 98 +- .../scripts/export-docx-with-word.ps1 | 4 +- .../scripts/verify-docx-acceptance.mjs | 27 + .../scripts/verify-pandoc-conversion.mjs | 24 + .../scripts/verify-theme-docx-matrix.mjs | 186 +++- .../docx-engine/src/acceptance-validator.ts | 195 ++++ .../src/document-structure-transform.ts | 864 ++++++++++++++++++ packages/docx-engine/src/index.ts | 2 + packages/docx-engine/src/ooxml.ts | 391 +++++++- packages/docx-engine/src/pandoc-converter.ts | 41 +- packages/docx-engine/src/pandoc-structure.ts | 240 +++++ packages/docx-engine/src/reference-package.ts | 33 +- packages/docx-engine/src/styles-transform.ts | 23 +- packages/docx-engine/src/validator.ts | 59 +- .../tests/acceptance-validator.test.ts | 53 +- .../document-structure-transform.test.ts | 251 +++++ packages/docx-engine/tests/ooxml.test.ts | 147 +++ .../tests/pandoc-converter.test.ts | 22 + .../tests/pandoc-structure.test.ts | 239 +++++ .../tests/reference-builder.test.ts | 49 +- packages/docx-theme-engine/src/normalizer.ts | 38 +- .../tests/normalizer.test.ts | 25 + 31 files changed, 3175 insertions(+), 64 deletions(-) create mode 100644 packages/docx-engine/src/document-structure-transform.ts create mode 100644 packages/docx-engine/src/pandoc-structure.ts create mode 100644 packages/docx-engine/tests/document-structure-transform.test.ts create mode 100644 packages/docx-engine/tests/ooxml.test.ts create mode 100644 packages/docx-engine/tests/pandoc-structure.test.ts diff --git a/apps/server/scripts/verify-docx-http.mjs b/apps/server/scripts/verify-docx-http.mjs index d65e3f4..da83f53 100644 --- a/apps/server/scripts/verify-docx-http.mjs +++ b/apps/server/scripts/verify-docx-http.mjs @@ -7,6 +7,7 @@ import { defaultExportConfig } from "@md-to-pdf/core"; import { inspectDocxAcceptance } from "@md-to-pdf/docx-engine"; +import { DOCX_STYLE_SLOTS } from "@md-to-pdf/docx-theme-engine"; import { buildApp } from "../dist/app.js"; const directory = path.dirname(fileURLToPath(import.meta.url)); @@ -75,7 +76,66 @@ const exportConfig = { } } }; +const computedStyle = { + fontFamily: '"Microsoft YaHei", sans-serif', + fontSize: "16px", + fontWeight: "400", + fontStyle: "normal", + color: "rgb(34, 34, 34)", + backgroundColor: "rgba(0, 0, 0, 0)", + lineHeight: "24px", + letterSpacing: "normal", + textAlign: "start", + direction: "ltr", + textIndent: "0px", + textDecorationLine: "none", + marginTop: "0px", + marginRight: "0px", + marginBottom: "8px", + marginLeft: "0px", + paddingTop: "0px", + paddingRight: "0px", + paddingBottom: "0px", + paddingLeft: "0px", + borderTop: "0px none rgb(34, 34, 34)", + borderRight: "0px none rgb(34, 34, 34)", + borderBottom: "0px none rgb(34, 34, 34)", + borderLeft: "0px none rgb(34, 34, 34)", + width: "640px", + maxWidth: "none", + minHeight: "0px", + height: "24px", + breakBefore: "auto", + breakAfter: "auto", + breakInside: "auto", + display: "block", + flexDirection: "row", + alignItems: "normal", + justifyContent: "normal", + outline: "0px none rgb(34, 34, 34)", + outlineOffset: "0px" +}; const mediaAdapter = { + themeStyleBaseUrl: + "http://localhost:5173/preview-frame.html", + async captureThemeStyle(request) { + return { + schemaVersion: 1, + themeId: request.themeId, + themeFingerprint: request.themeFingerprint, + viewport: { + widthPx: 794, + heightPx: 1123, + deviceScaleFactor: 1 + }, + rootFontSizePx: 16, + slots: DOCX_STYLE_SLOTS.map(({ name }) => ({ + slot: name, + matched: true, + computed: computedStyle + })) + }; + }, async capture() { return { plan: { @@ -96,7 +156,7 @@ const pdfGenerator = { }; const beforeTemporaryDirectories = temporaryDirectories(); const app = buildApp({ - logger: false, + logger: { level: "error" }, pdfGenerator, docxMediaAdapter: mediaAdapter }); @@ -128,7 +188,10 @@ try { await app.close(); } -assert(response.statusCode === 200, `DOCX HTTP 返回 ${response.statusCode}`); +assert( + response.statusCode === 200, + `DOCX HTTP 返回 ${response.statusCode}:${response.body}` +); assert( response.headers["content-type"]?.includes(DOCX_MIME_TYPE), `DOCX MIME 无效:${response.headers["content-type"]}` diff --git a/apps/server/src/app.ts b/apps/server/src/app.ts index 5e970a2..83d7efd 100644 --- a/apps/server/src/app.ts +++ b/apps/server/src/app.ts @@ -516,6 +516,12 @@ export function buildApp(options: BuildAppOptions = {}) { }); } if (error instanceof DocxExportServiceError) { + if (error.statusCode >= 500) { + request.log.error( + { error, cause: error.cause }, + "DOCX generation service failed" + ); + } if (error.retryable) { reply.header("retry-after", "5"); } diff --git a/docs/PROGRESS.md b/docs/PROGRESS.md index c35df95..7e7011e 100644 --- a/docs/PROGRESS.md +++ b/docs/PROGRESS.md @@ -1,6 +1,6 @@ # Markdown PDF 导出器进度 -最后更新:2026-07-30 +最后更新:2026-07-31 ## 1. 当前概况 @@ -207,6 +207,31 @@ Server Playwright 与 Desktop Electron 适配器,按主题 CSS SHA-256 当前结构样式尚未绑定到 Pandoc AST,封面、版头、分节与标题去重留给 阶段 12C。 +`v0.6.0` 阶段 12C 已完成通用 Pandoc 结构映射和最终 OOXML 收口。统一 +语义文档模型先生成主题无关的结构计划,静态 Lua Filter 将公文版头、 +文号、签发人、落款、版记、简报元数据及项目/标书封面注入可编辑 Word +段落,并绑定标准 Markdown 样式和 31 个 `Md*` 结构样式;转换过程不按 +主题 ID 分支。最终 DOCX 后处理生成真实分节,支持封面无页眉页脚、正文 +页码从 1 重启和 `SECTIONPAGES`,同时收口容器边框、背景、分页控制、 +表格内容区全宽、固定网格和行不拆分。标题策略会抑制 YAML 与首个 H1 +重复,所有内部结构标记在输出前强制清除。根级 +`npm run verify:docx-themes` 现在强制先用 Playwright Chromium 151 +重新采集 14 套主题各 56 个真实槽位,再由 Pandoc 3.9.0.2 生成最终矩阵; +结果为 11/11 表格全宽、8/8 结构化主题字段完整、4/4 独立封面分节合格、 +重复标题失败 0、内部标记残留 0。下一步阶段 12D 继续使用真实 Word/WPS +做视觉、编辑和互存回归,自动结构通过不替代客户端验收。 + +阶段 12C-R1 已完成 WordprocessingML 严格性修复。OOXML 写入层现在按 +模式顺序生成和规范化段落、文本、表格、单元格、分节及条件表格样式属性; +样式生成将 CSS `justify` 映射为合法的 `w:jc="both"`,表格宽度仅写入 +正文表格而不写入表格样式,最终转换会移除 Pandoc 追加的重复样式 ID。 +内置校验器会拒绝乱序属性、重复样式 ID、非法两端对齐枚举和表格样式 +`w:tblW`。Microsoft Open XML SDK 2.20 审计确认 14 份最终 DOCX 的上述 +错误全部归零;Microsoft Word 和 WPS 均以禁止修复的只读方式正常打开 +14/14,Word 成功导出 31 页 PDF。逐页检查未发现内容丢失、裁切、重叠或 +空白正文。下一步按 R2 字体嵌入、R3 PDF 与 DOCX 视觉差异引擎、R4 全主题 +深度回归的顺序推进。 + ## 2. 已完成 ### 2.1 项目骨架 @@ -1140,9 +1165,11 @@ ECharts 第二阶段浏览器与 PDF 验证结果: 共用同一语义树,现有 HTML DOM 保持兼容; - 阶段 12B:已将跨端主题令牌接入动态 `reference.docx`,14 主题双引擎 令牌和真实 Pandoc 模板矩阵通过; -- 阶段 12C:完成 Pandoc 结构映射、封面分节、标题去重、表格宽度和 - 分页控制; -- 阶段 12D:完成 14 套主题的真实 Word/WPS 视觉与结构回归; +- 阶段 12C:已完成通用 Pandoc 结构映射、封面真实分节、标题去重、 + 表格内容区全宽、分页控制、OOXML 严格性和 14 主题真实 Word/WPS + 无修复打开门禁; +- 阶段 12D:依次完成可分发字体嵌入、PDF 与 DOCX 自动视觉差异、14 套 + 主题真实 Word/WPS 编辑互存和深度视觉回归; - 阶段 13:完成 Word/WPS 双向互存、外部主题兼容、体积和正式发布验收。 每个阶段验收通过后创建一个独立提交,再进入下一阶段。当前阶段不得混入 diff --git a/docs/V0.6.0_DOCX_THEME_AUDIT.md b/docs/V0.6.0_DOCX_THEME_AUDIT.md index 81cfc20..08b131e 100644 --- a/docs/V0.6.0_DOCX_THEME_AUDIT.md +++ b/docs/V0.6.0_DOCX_THEME_AUDIT.md @@ -2,6 +2,10 @@ 审计日期:2026-07-30 +> 本文第 3~8 节记录阶段 10 的首轮审计基线。阶段 12C 已于 +> 2026-07-31 完成对应结构修复和自动化门禁,最新状态见第 9 节;真实 +> Word/WPS 视觉、编辑和互存结论仍需阶段 12D 复验。 + ## 1. 审计目标 本轮按 v0.6.0 已冻结的两条 DOCX 发布门禁检查全部内置主题: @@ -258,3 +262,86 @@ Word 实际导出的 PDF 中可见字体符合当前样式意图: 在完成 P0 项目前,不建议将阶段 10 标记为通过,也不建议创建 v0.6.0 正式发布提交或版本标签。 + +## 9. 阶段 12C 自动修复复验 + +阶段 12C 保留 Pandoc 作为可编辑 OOXML 底座,并增加项目自有的通用转换 +层:统一语义文档模型生成结构计划,Lua Filter 注入可编辑段落与样式, +最终 OOXML 收口层写入真实分节、页码重启、表格宽度和分页属性。映射只 +依赖语义角色与主题令牌,不包含主题 ID 分支,因此同一机制也可服务未来 +外部主题。 + +2026-07-31 使用 Playwright Chromium 151 重新采集 14 套主题各 56 个 +计算样式槽位,并由 Pandoc 3.9.0.2 重新生成全部 DOCX,自动结构结果如下: + +| 检查项 | 阶段 12C 结果 | 结论 | +| --- | ---: | --- | +| 主题使用本轮真实 Chromium 令牌 | 14/14 | 通过 | +| 结构化主题字段完整映射 | 8/8 | 通过 | +| 项目报告与标书独立封面真实分节 | 4/4 | 通过 | +| 封面隐藏页眉页脚、正文页码从 1 重启 | 4/4 | 通过 | +| 封面文档使用 `SECTIONPAGES` | 4/4 | 通过 | +| 含表格主题的表格铺满内容区 | 11/11 | 通过 | +| YAML 标题与正文首个 H1 去重 | 14/14 | 通过 | +| 内部结构标记清理 | 14/14 | 通过 | +| 禁止 `altChunk`、保留原生可编辑结构 | 14/14 | 通过 | + +统一复现命令仍为: + +```powershell +npm run verify:docx-themes +``` + +该命令不会回退到合成令牌或陈旧快照。它先生成本轮 Chromium 令牌,再 +执行 14 主题 Pandoc 矩阵和硬断言。阶段 10 的 P0/P1 结构阻塞因此已由 +自动门禁关闭;“完整主题样式映射”的最终判定仍保留到阶段 12D,必须在 +Microsoft Word 与 WPS 中完成视觉检查、编辑、保存、重开和双向互存后 +才能通过。 + +## 10. 阶段 12C-R1 OOXML 严格性复验 + +2026-07-31 对阶段 12C 生成的 14 份 DOCX 做 Microsoft Word 原生打开 +时,最初仅 7 份可直接打开,另外 7 份触发“发现无法读取的内容”且无法 +恢复。WPS 可以容错打开,但会报告缺失字体,因此不能据此判定 OOXML +合格。 + +深度审计确认根因集中在生成样式与 OOXML 模式约束: + +- Pandoc 消费 `custom-style` 后会追加同名降级样式,造成重复 + `w:styleId`; +- CSS `text-align: justify` 被直接写成非法的 `w:jc="justify"`,正确 + WordprocessingML 枚举应为 `both`; +- 表格条件样式中的 `w:rPr` 与 `w:tcPr` 顺序不符合模式; +- `w:tblW` 被写入表格样式层,和正文表格的直接宽度属性混在一起; +- 多类段落、文本、表格、单元格和分节属性缺少统一的有序写入门禁。 + +通用引擎已经增加有序插入、序列化规范化和严格校验,并在最终 DOCX +收口层保留前置主题样式、移除 Pandoc 同名降级样式。修复不按主题 ID +分支,也不改变正文表格的内容区全宽属性。 + +复验结果: + +| 检查项 | 结果 | 结论 | +| --- | ---: | --- | +| 14 主题结构矩阵 | 14/14 | 通过 | +| Microsoft Open XML SDK 非兼容性错误 | 0 | 通过 | +| Microsoft Word 禁止修复打开 | 14/14 | 通过 | +| WPS 原生只读打开 | 14/14 | 通过 | +| Word 原生 PDF 导出 | 14/14,共 31 页 | 通过 | +| 31 页逐页结构检查 | 31/31 | 通过 | + +Open XML SDK 2.20 仍会按 Office 2007 模式报告 Pandoc 表格上的 Office +2010 `tblLook` 六个兼容属性;这些属性在本轮 Word/WPS 实测中正常,不计 +为生成错误。 + +逐页检查没有发现内容丢失、裁切、重叠或空白正文,但保留以下阶段 12D +视觉问题: + +- 深色表头文字对比度不足; +- 部分短文档尾页留白过大; +- `gov-red-standard` 在 Word 与 WPS 中分别为 2 页和 3 页,说明字体与 + 客户端排版差异尚未收口; +- 当前主题矩阵的 Mermaid 仍使用黑色占位 PNG,不代表真实媒体效果。 + +因此 R1 只关闭 OOXML 合法性和 Word/WPS 无修复打开门禁。字体可移植性、 +PDF 与 DOCX 自动视觉差异、真实媒体和双向编辑互存继续由 R2~R4 验收。 diff --git a/package.json b/package.json index 773eeeb..493227f 100644 --- a/package.json +++ b/package.json @@ -29,7 +29,7 @@ "verify:docx-reference": "npm run build -w @md-to-pdf/core && npm run build -w @md-to-pdf/docx-engine && npm run verify:pandoc -w @md-to-pdf/docx-engine", "verify:docx-conversion": "npm run build -w @md-to-pdf/core && npm run build -w @md-to-pdf/docx-engine && npm run verify:conversion -w @md-to-pdf/docx-engine", "verify:docx-matrix": "npm run build -w @md-to-pdf/core && npm run build -w @md-to-pdf/docx-engine && npm run verify:acceptance -w @md-to-pdf/docx-engine", - "verify:docx-themes": "npm run build:web-runtime && npm run verify:themes -w @md-to-pdf/docx-engine", + "verify:docx-themes": "npm run build:web-runtime && npm run build -w @md-to-pdf/server && npm run verify:docx-theme-styles -w @md-to-pdf/server && npm run verify:themes -w @md-to-pdf/docx-engine", "verify:docx-theme-styles": "npm run build -w @md-to-pdf/core && npm run build -w @md-to-pdf/semantic-document && npm run build -w @md-to-pdf/docx-theme-engine && npm run build -w @md-to-pdf/docx-engine && npm run build -w @md-to-pdf/renderer && npm run build -w @md-to-pdf/application && npm run build -w @md-to-pdf/server && npm run verify:docx-theme-styles -w @md-to-pdf/server && npm run verify:docx-theme-styles -w @md-to-pdf/desktop", "verify:docx-acceptance": "npm run build:web-runtime && npm run build -w @md-to-pdf/server && npm run build -w @md-to-pdf/desktop && npm run verify:pandoc -w @md-to-pdf/docx-engine && npm run verify:conversion -w @md-to-pdf/docx-engine && npm run verify:acceptance -w @md-to-pdf/docx-engine && npm run verify:docx-http -w @md-to-pdf/server && npm run test -w @md-to-pdf/desktop -- desktop-docx-save.test.ts", "verify:docx-http": "npm run build:web-runtime && npm run build -w @md-to-pdf/server && npm run verify:docx-http -w @md-to-pdf/server", diff --git a/packages/application/README.md b/packages/application/README.md index 8f5652b..4c8fa40 100644 --- a/packages/application/README.md +++ b/packages/application/README.md @@ -50,8 +50,9 @@ Web 端通过 `apps/server` 的 HTTP API 调用;桌面端在主进程中创建 Desktop 才会以 Markdown 所在目录为边界解析相对资源。 `prepareDocxExport()` 统一校验 DOCX 请求、解析图片资源、查找主题并返回 -源请求、安全渲染文档、主题清单和主题 CSS。它不调用 Pandoc;后续 DOCX -引擎只消费该准备结果,避免 Server 与 Desktop 重复实现文档准备逻辑。 +源请求、安全渲染文档、统一语义文档、主题清单和主题 CSS。它不调用 +Pandoc;后续 DOCX 引擎直接消费同一语义树生成结构计划,避免 Server、 +Desktop、HTML/PDF 与 DOCX 重复解释 Front Matter。 `prepareDocxMedia()` 根据纸张方向、尺寸、主题默认页边距和用户配置计算 内容区,调用平台捕获适配器,并校验捕获计划、PNG 签名、像素尺寸、单图 @@ -63,7 +64,8 @@ Desktop 才会以 Markdown 所在目录为边界解析相对资源。 `explicit` 模式不启动浏览器,直接使用清单覆盖和预设降级。 `DocxExportService` 串联 capability 探测、请求准备、平台媒体捕获、主题 -令牌准备、Pandoc 转换和最终 OOXML 校验。媒体与主题令牌并行准备。默认并发为 1、队列为 4、总超时为 90 秒; +令牌准备、Pandoc 结构转换、最终 OOXML 分节收口和验收校验。媒体与主题 +令牌并行准备。默认并发为 1、队列为 4、总超时为 90 秒; 可以通过 `DOCX_CONCURRENCY`、`DOCX_MAX_QUEUE` 和 `DOCX_TIMEOUT_MS` 配置。排队、探测、准备、主题样式、媒体、模板、Pandoc、校验和总耗时使用共享协议 返回;关闭服务时会取消活动任务、拒绝排队任务并等待清理完成。 diff --git a/packages/application/src/docx-export-service.ts b/packages/application/src/docx-export-service.ts index efc80d7..1bd4b80 100644 --- a/packages/application/src/docx-export-service.ts +++ b/packages/application/src/docx-export-service.ts @@ -490,6 +490,7 @@ export class DocxExportService { theme: prepared.theme.manifest, themeTokens: themeTokens.tokens, metadata: prepared.document.metadata, + semanticDocument: prepared.document.semanticDocument, media }; } diff --git a/packages/application/tests/docx-export-service.test.ts b/packages/application/tests/docx-export-service.test.ts index 996f1ea..a54f5a6 100644 --- a/packages/application/tests/docx-export-service.test.ts +++ b/packages/application/tests/docx-export-service.test.ts @@ -185,6 +185,12 @@ describe("DOCX 共享导出服务", () => { expect(result.docx).toEqual(new Uint8Array([1, 2, 3])); expect(result.fileName).toBe("报告_.docx"); + expect(convert).toHaveBeenCalledWith( + expect.objectContaining({ + semanticDocument: prepared.document.semanticDocument + }), + expect.any(AbortSignal) + ); expect(result.diagnostics).toEqual({ warnings: [], echartsErrors: [], diff --git a/packages/docx-engine/README.md b/packages/docx-engine/README.md index c746a53..a538a25 100644 --- a/packages/docx-engine/README.md +++ b/packages/docx-engine/README.md @@ -5,7 +5,10 @@ 兼容样式预设映射为 Word 样式,并生成纸张、页边距、字体、段落、代码、表格、 页眉、页脚和页码均已映射的动态模板。Pandoc 转换器通过静态 Lua Filter 将普通图片、Mermaid 和 ECharts 节点替换为受控 PNG,同时保留 Markdown -正文、标题、列表、表格、代码和公式的可编辑文档结构。 +正文、标题、列表、表格、代码和公式的可编辑文档结构。统一语义文档模型 +会先转换为主题无关的 Pandoc 结构计划,Lua Filter 再绑定标准 Markdown +样式与 `Md*` 结构样式;生成后的 OOXML 收口层负责真实分节、封面页眉页脚 +隔离、正文页码重启、表格内容区全宽和分页控制。 ## 设计边界 @@ -15,7 +18,11 @@ - 不解析任意主题 CSS,只消费 `@md-to-pdf/docx-theme-engine` 的受限 语义令牌;`docxStyle` 预设在槽位缺失时提供兼容降级; - 槽位只映射到稳定的标准或 `Md*` Word 样式,不包含主题 ID 分支; +- Front Matter 结构、标题策略和分节意图只消费统一语义文档模型,不按 + 主题 ID 编写转换分支; - 生成结果会复验全部 XML、包内关系、内容类型、最终节和关键样式; +- 最终 DOCX 必须清除全部内部结构标记,并校验实际样式使用、分节、 + `SECTIONPAGES`、全宽表格和标题去重; - 每次转换使用独立临时目录、隔离 Pandoc data 目录并在 `finally` 清理; - 媒体映射检查顺序、PNG、像素、单图/总大小和物理显示尺寸; - 并发排队、跨端用例编排和错误协议映射由 @@ -40,6 +47,7 @@ npm run verify:docx-reference npm run verify:docx-conversion npm run verify:docx-acceptance npm run verify:docx-theme-styles +npm run verify:docx-themes ``` `verify:docx-reference` 要求本机 `PATH` 中存在 Pandoc 3.9.0.2,也可以通过 @@ -59,3 +67,9 @@ npm run verify:docx-theme-styles 内置主题的 56 个槽位,检查跨引擎令牌一致性,并使用固定 Pandoc 默认 模板生成 14 份动态 `reference.docx`。模板矩阵检查标准 Markdown 样式、 结构化 `Md*` 样式、正文字体、字号、字体表和缓存指纹。 + +根级 `verify:docx-themes` 会先重新构建运行时并使用 Playwright Chromium +生成本轮真实主题令牌,再以固定 Pandoc 生成 14 份最终 DOCX。矩阵硬门禁 +覆盖标准样式、原生页眉页脚、结构字段完整率、封面真实分节、正文页码 +重启、`SECTIONPAGES`、内容区全宽表格、标题去重、内部标记清理和 +`altChunk` 禁用;不会使用陈旧快照或合成令牌降级。 diff --git a/packages/docx-engine/assets/docx-media-filter.lua b/packages/docx-engine/assets/docx-media-filter.lua index bcf9b03..064b074 100644 --- a/packages/docx-engine/assets/docx-media-filter.lua +++ b/packages/docx-engine/assets/docx-media-filter.lua @@ -1,18 +1,29 @@ local map_path = os.getenv("MD_TO_PDF_DOCX_MEDIA_MAP") +local structure_plan_path = + os.getenv("MD_TO_PDF_DOCX_STRUCTURE_PLAN") if map_path == nil or map_path == "" then error("DOCX media map path is missing") end - -local map_file, open_error = io.open(map_path, "rb") -if map_file == nil then - error("DOCX media map cannot be opened: " .. tostring(open_error)) +if structure_plan_path == nil or structure_plan_path == "" then + error("DOCX structure plan path is missing") end -local map_content = map_file:read("*a") -map_file:close() +local function read_json(path, description) + local file, open_error = io.open(path, "rb") + if file == nil then + error( + description .. " cannot be opened: " .. tostring(open_error) + ) + end + local content = file:read("*a") + file:close() + return pandoc.json.decode(content, false) +end -local media_map = pandoc.json.decode(map_content, false) +local media_map = read_json(map_path, "DOCX media map") +local structure_plan = + read_json(structure_plan_path, "DOCX structure plan") local counters = { image = 0, mermaid = 0, @@ -86,6 +97,69 @@ local function replace_code_block(block) return pandoc.Para { image } end +local function append_inlines(target, value) + target:extend(pandoc.Inlines(value)) +end + +local function structure_paragraph(block) + local inlines = pandoc.Inlines {} + for index, segment in ipairs(block.segments) do + if index > 1 then + if block.separator == "tab" then + inlines:insert( + pandoc.RawInline( + "openxml", + "" + ) + ) + else + inlines:insert(pandoc.Space()) + end + end + append_inlines(inlines, segment) + end + local paragraph = pandoc.Para(inlines) + if block.styleId == nil or block.styleId == "" then + return paragraph + end + return pandoc.Div( + { paragraph }, + pandoc.Attr("", {}, { ["custom-style"] = block.styleId }) + ) +end + +local function marker_paragraph(marker) + return pandoc.Para { pandoc.Str(marker) } +end + +local function structure_blocks(blocks) + local result = pandoc.Blocks {} + for _, block in ipairs(blocks) do + if block.kind == "paragraph" then + result:insert(structure_paragraph(block)) + elseif block.kind == "container" then + local children = structure_blocks(block.blocks) + result:insert(marker_paragraph(block.startMarker)) + result:extend(children) + result:insert(marker_paragraph(block.endMarker)) + elseif block.kind == "section-break" then + result:insert(marker_paragraph(block.marker)) + else + error("Unsupported DOCX structure block: " .. tostring(block.kind)) + end + end + return result +end + +local function suppress_first_body_heading(blocks) + for index, block in ipairs(blocks) do + if block.t == "Header" and block.level == 1 then + blocks:remove(index) + return + end + end +end + local function validate_counts() for _, kind in ipairs({ "image", "mermaid", "echarts" }) do if counters[kind] ~= #media_map[kind] then @@ -104,6 +178,16 @@ return { Image = replace_image, CodeBlock = replace_code_block } + if structure_plan.titlePolicy.metadataTitle == "suppress" then + transformed.meta.title = nil + end + if structure_plan.titlePolicy.firstBodyHeading == "suppress" then + suppress_first_body_heading(transformed.blocks) + end + local blocks = structure_blocks(structure_plan.prefix) + blocks:extend(transformed.blocks) + blocks:extend(structure_blocks(structure_plan.suffix)) + transformed.blocks = blocks validate_counts() return transformed end diff --git a/packages/docx-engine/scripts/export-docx-with-word.ps1 b/packages/docx-engine/scripts/export-docx-with-word.ps1 index 7cd1a9a..7684e2e 100644 --- a/packages/docx-engine/scripts/export-docx-with-word.ps1 +++ b/packages/docx-engine/scripts/export-docx-with-word.ps1 @@ -25,7 +25,7 @@ try { $word.DisplayAlerts = 0 $word.AutomationSecurity = 3 $word.Options.SaveNormalPrompt = $false - $document = $word.Documents.Open( + $document = $word.Documents.OpenNoRepairDialog( $resolvedInput, $false, $true, @@ -38,7 +38,7 @@ try { 0, 0, $false, - $true, + $false, 0, $true ) diff --git a/packages/docx-engine/scripts/verify-docx-acceptance.mjs b/packages/docx-engine/scripts/verify-docx-acceptance.mjs index 0ffdfe6..d495fea 100644 --- a/packages/docx-engine/scripts/verify-docx-acceptance.mjs +++ b/packages/docx-engine/scripts/verify-docx-acceptance.mjs @@ -11,6 +11,7 @@ import { PandocRuntime, inspectDocxAcceptance } from "@md-to-pdf/docx-engine"; +import { DOCX_STYLE_SLOT_NAMES } from "@md-to-pdf/docx-theme-engine"; const directory = path.dirname(fileURLToPath(import.meta.url)); const packageDirectory = path.resolve(directory, ".."); @@ -95,6 +96,23 @@ function createMedia() { }; } +function createThemeTokens(theme) { + return { + schemaVersion: 1, + themeId: theme.id, + themeFingerprint: "c".repeat(64), + mode: "auto-with-overrides", + basePreset: theme.docxStyle?.preset ?? "general", + slots: DOCX_STYLE_SLOT_NAMES.map((slot) => ({ + slot, + source: "preset-fallback", + confidence: "fallback", + style: { fontCandidates: [] } + })), + diagnostics: [] + }; +} + const technicalTheme = loadTheme("typora-like"); const officialTheme = loadTheme("gov-red-standard"); const commonExpectation = { @@ -287,6 +305,7 @@ for (const variant of variants) { language: "zh-CN", exportConfig: variant.exportConfig, theme: variant.theme, + themeTokens: createThemeTokens(variant.theme), metadata: { title: "v0.6.0 DOCX 自动验收", author: "Markdown PDF 导出器研发组", @@ -294,6 +313,14 @@ for (const variant of variants) { keywords: ["DOCX", "Pandoc", "OOXML"], language: "zh-CN" }, + semanticDocument: { + schemaVersion: 1, + titlePolicy: { + metadataTitle: "emit", + firstBodyHeading: "keep" + }, + regions: [] + }, media: createMedia() }); const report = inspectDocxAcceptance( diff --git a/packages/docx-engine/scripts/verify-pandoc-conversion.mjs b/packages/docx-engine/scripts/verify-pandoc-conversion.mjs index 55d57be..e10f5ee 100644 --- a/packages/docx-engine/scripts/verify-pandoc-conversion.mjs +++ b/packages/docx-engine/scripts/verify-pandoc-conversion.mjs @@ -9,6 +9,7 @@ import { PandocDocxConverter, PandocRuntime } from "@md-to-pdf/docx-engine"; +import { DOCX_STYLE_SLOT_NAMES } from "@md-to-pdf/docx-theme-engine"; import { unzipSync } from "fflate"; const directory = path.dirname(fileURLToPath(import.meta.url)); @@ -76,6 +77,20 @@ const runtime = new PandocRuntime( : {} ); const converter = new PandocDocxConverter({ runtime }); +const themeTokens = { + schemaVersion: 1, + themeId: theme.id, + themeFingerprint: "c".repeat(64), + mode: "auto-with-overrides", + basePreset: theme.docxStyle?.preset ?? "general", + slots: DOCX_STYLE_SLOT_NAMES.map((slot) => ({ + slot, + source: "preset-fallback", + confidence: "fallback", + style: { fontCandidates: [] } + })), + diagnostics: [] +}; const conversionInput = { markdown, fileName: "Pandoc 媒体转换验收.md", @@ -85,6 +100,7 @@ const conversionInput = { themeId: theme.id }, theme, + themeTokens, metadata: { title: "Pandoc 媒体转换验收", author: "Markdown PDF 导出器研发组", @@ -92,6 +108,14 @@ const conversionInput = { keywords: [], language: "zh-CN" }, + semanticDocument: { + schemaVersion: 1, + titlePolicy: { + metadataTitle: "emit", + firstBodyHeading: "keep" + }, + regions: [] + }, media: { resources: [ resource("image", 1, { width: 384, height: 192 }), diff --git a/packages/docx-engine/scripts/verify-theme-docx-matrix.mjs b/packages/docx-engine/scripts/verify-theme-docx-matrix.mjs index dd7547e..003fdf8 100644 --- a/packages/docx-engine/scripts/verify-theme-docx-matrix.mjs +++ b/packages/docx-engine/scripts/verify-theme-docx-matrix.mjs @@ -24,6 +24,12 @@ const outputDirectory = path.join( "output", "docx-theme-matrix" ); +const themeTokenReportPath = path.join( + repositoryDirectory, + "output", + "docx-theme-styles", + "snapshots.json" +); const decoder = new TextDecoder(); const placeholderPng = Buffer.from( "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=", @@ -49,6 +55,26 @@ function countMatches(value, pattern) { return value.match(pattern)?.length ?? 0; } +function assert(condition, message) { + if (!condition) { + throw new Error(message); + } +} + +function countOccurrences(value, search) { + let count = 0; + let offset = 0; + while (offset <= value.length - search.length) { + const index = value.indexOf(search, offset); + if (index < 0) { + break; + } + count += 1; + offset = index + search.length; + } + return count; +} + function readThemeManifests() { return fs .readdirSync(themesDirectory, { withFileTypes: true }) @@ -90,7 +116,13 @@ function structuralValues(document) { .map(String); } -function inspectDocument(content, theme, metadata, exportConfig) { +function inspectDocument( + content, + theme, + metadata, + semanticDocument, + exportConfig +) { const entries = unzipSync(content); const documentXml = decoder.decode(entries["word/document.xml"]); const stylesXml = decoder.decode(entries["word/styles.xml"]); @@ -120,6 +152,37 @@ function inspectDocument(content, theme, metadata, exportConfig) { documentXml, /]*w:type="page"/gu ); + const sections = [ + ...documentXml.matchAll( + /]*)?>[\s\S]*?<\/w:sectPr>/gu + ) + ].map((match) => match[0]); + const fullWidthTableCount = countMatches( + documentXml, + //gu + ); + const internalMarkerCount = countMatches( + documentXml, + /MD_TO_PDF_(?:CONTAINER|SECTION)_/gu + ); + const pandocTitleParagraphCount = countMatches( + documentXml, + //gu + ); + const visibleMetadataTitleCount = countOccurrences( + bodyText, + metadata.title + ); + const titlePolicyPassed = + semanticDocument.titlePolicy.metadataTitle === "suppress" + ? pandocTitleParagraphCount === 0 + : pandocTitleParagraphCount === 1; + const titleDeduplicationPassed = + semanticDocument.titlePolicy.metadataTitle === "suppress" && + semanticDocument.titlePolicy.firstBodyHeading === "keep" && + !metadata.document?.profile + ? visibleMetadataTitleCount === 1 + : titlePolicyPassed; return { bytes: content.byteLength, @@ -128,6 +191,14 @@ function inspectDocument(content, theme, metadata, exportConfig) { tableCount: countMatches(documentXml, /)/gu), drawingCount: countMatches(documentXml, /)/gu), explicitPageBreaks, + sectionCount: sections.length, + fullWidthTableCount, + internalMarkerCount, + pandocTitleParagraphCount, + visibleMetadataTitleCount, + titlePolicy: semanticDocument.titlePolicy, + titlePolicyPassed, + titleDeduplicationPassed, headerPartCount: Object.keys(entries).filter((name) => /^word\/header\d+\.xml$/u.test(name) ).length, @@ -138,6 +209,7 @@ function inspectDocument(content, theme, metadata, exportConfig) { !headerXml.includes(")/gu), styles: { normal: stylesXml.includes('w:styleId="Normal"'), @@ -157,7 +229,11 @@ function inspectDocument(content, theme, metadata, exportConfig) { expectsStandaloneCover, standaloneCoverDetected: expectsStandaloneCover && - explicitPageBreaks > 0 && + sections.length >= 2 && + !sections[0]?.includes("headerReference") && + !sections[0]?.includes("footerReference") && + sections.at(-1)?.includes('w:pgNumType w:start="1"') === + true && missingStructureValues.length === 0 }; } @@ -220,6 +296,49 @@ function createMedia(markdown) { totalBytes: placeholderPng.byteLength * resources.length }; } + +function readThemeTokenReport(themes) { + if (!fs.existsSync(themeTokenReportPath)) { + throw new Error( + `缺少本轮 Playwright 主题令牌报告:${path.relative(repositoryDirectory, themeTokenReportPath)}` + ); + } + const report = JSON.parse( + fs.readFileSync(themeTokenReportPath, "utf8") + ); + assert( + Array.isArray(report.tokenSets), + "Playwright 主题令牌报告缺少 tokenSets" + ); + assert( + report.tokenSets.length === themes.length, + `真实主题令牌数量应为 ${themes.length},实际为 ${report.tokenSets.length}` + ); + const tokensByTheme = new Map( + report.tokenSets.map((tokens) => { + assert( + tokens.slots?.length === 56, + `主题 ${tokens.themeId} 的真实令牌槽位不是 56` + ); + assert( + /^[a-f0-9]{64}$/u.test(tokens.themeFingerprint ?? ""), + `主题 ${tokens.themeId} 的 CSS 指纹无效` + ); + return [tokens.themeId, tokens]; + }) + ); + for (const theme of themes) { + assert( + tokensByTheme.has(theme.id), + `真实主题令牌缺少 ${theme.id}` + ); + } + return { + generatedAt: report.generatedAt, + chromiumVersion: report.chromiumVersion, + tokensByTheme + }; +} const runtime = new PandocRuntime( process.env.DOCX_PANDOC_PATH?.trim() ? { configuredPath: process.env.DOCX_PANDOC_PATH.trim() } @@ -238,6 +357,8 @@ if (capability.capability.detectedVersion !== DOCX_PANDOC_VERSION) { fs.mkdirSync(outputDirectory, { recursive: true }); const converter = new PandocDocxConverter({ runtime }); const themes = readThemeManifests(); +assert(themes.length === 14, `内置主题数量应为 14,实际为 ${themes.length}`); +const themeTokenReport = readThemeTokenReport(themes); const results = []; for (const theme of themes) { @@ -248,6 +369,7 @@ for (const theme of themes) { } const markdown = fs.readFileSync(samplePath, "utf8"); const rendered = renderMarkdown(markdown, { language: "zh-CN" }); + const themeTokens = themeTokenReport.tokensByTheme.get(theme.id); const exportConfig = { ...defaultExportConfig, name: `${theme.name} DOCX 主题验收`, @@ -268,7 +390,9 @@ for (const theme of themes) { language: rendered.metadata.language, exportConfig, theme, + themeTokens, metadata: rendered.metadata, + semanticDocument: rendered.semanticDocument, media: createMedia(markdown) }); } catch (error) { @@ -297,14 +421,72 @@ for (const theme of themes) { conversion.docx, theme, rendered.metadata, + rendered.semanticDocument, exportConfig ) }); } +for (const result of results) { + const { inspection } = result; + assert( + Object.values(inspection.styles).every(Boolean), + `主题 ${result.id} 缺少标准 Word 样式` + ); + assert( + inspection.nativeHeaderFooter, + `主题 ${result.id} 的页眉页脚包含表格模拟` + ); + assert( + inspection.altChunkCount === 0, + `主题 ${result.id} 包含 altChunk` + ); + assert( + inspection.internalMarkerCount === 0, + `主题 ${result.id} 残留内部结构标记` + ); + assert( + inspection.titlePolicyPassed, + `主题 ${result.id} 的标题策略未生效` + ); + assert( + inspection.titleDeduplicationPassed, + `主题 ${result.id} 存在重复标题` + ); + assert( + inspection.fullWidthTableCount === inspection.tableCount, + `主题 ${result.id} 存在非内容区全宽表格` + ); + if (inspection.profile) { + assert( + inspection.structureCoverage === 1, + `主题 ${result.id} 的结构字段覆盖率不是 100%` + ); + } + if (inspection.expectsStandaloneCover) { + assert( + inspection.standaloneCoverDetected, + `主题 ${result.id} 未生成合格的独立封面分节` + ); + assert( + inspection.sectionPagesField, + `主题 ${result.id} 未使用 SECTIONPAGES` + ); + } +} + const report = { generatedAt: new Date().toISOString(), pandocVersion: capability.capability.detectedVersion, + themeTokens: { + source: path.relative( + repositoryDirectory, + themeTokenReportPath + ), + generatedAt: themeTokenReport.generatedAt, + chromiumVersion: themeTokenReport.chromiumVersion, + tokenSetCount: themeTokenReport.tokensByTheme.size + }, themeCount: themes.length, outputDirectory: path.relative( repositoryDirectory, diff --git a/packages/docx-engine/src/acceptance-validator.ts b/packages/docx-engine/src/acceptance-validator.ts index 9017b9a..77b66fe 100644 --- a/packages/docx-engine/src/acceptance-validator.ts +++ b/packages/docx-engine/src/acceptance-validator.ts @@ -4,6 +4,7 @@ import { OFFICE_RELATIONSHIP_NAMESPACE, PACKAGE_RELATIONSHIP_NAMESPACE, WORD_NAMESPACE, + firstDirectChild, parseXmlPart, type XmlElement } from "./ooxml.js"; @@ -64,7 +65,15 @@ export interface DocxAcceptanceExpectation { minimumPngImages?: number; requiredImageAltText?: readonly string[]; requiredStyleIds?: readonly string[]; + requiredParagraphStyleIds?: readonly string[]; requirePageField?: boolean; + minimumSections?: number; + requireFirstSectionWithoutHeaderFooter?: boolean; + finalPageNumberStart?: number; + requireSectionPagesField?: boolean; + minimumFullWidthTables?: number; + maximumTextOccurrences?: Readonly>; + forbidInternalMarkers?: boolean; } export interface DocxAcceptanceReport { @@ -93,10 +102,15 @@ export interface DocxAcceptanceReport { drawingCount: number; pngImageCount: number; pageFieldCount: number; + sectionPageFieldCount: number; + sectionCount: number; + fullWidthTableCount: number; + internalMarkerCount: number; altChunkCount: number; }; imageAltText: string[]; styleIds: string[]; + appliedParagraphStyleIds: string[]; checks: Record; } @@ -213,6 +227,48 @@ function collectPageFieldCount( return count; } +function collectFieldCount( + entries: ReadonlyMap, + field: "SECTIONPAGES" +) { + let count = 0; + for (const [partName, content] of entries) { + if (!/^word\/(?:document|header\d+|footer\d+)\.xml$/u.test(partName)) { + continue; + } + const document = parseXmlPart(content, partName); + for (const element of Array.from( + document.getElementsByTagNameNS(WORD_NAMESPACE, "instrText") + )) { + if ( + new RegExp(`\\b${field}\\b`, "u").test( + element.textContent ?? "" + ) + ) { + count += 1; + } + } + } + return count; +} + +function countTextOccurrences(value: string, search: string) { + if (!search) { + return 0; + } + let count = 0; + let offset = 0; + while (offset <= value.length - search.length) { + const index = value.indexOf(search, offset); + if (index < 0) { + break; + } + count += 1; + offset = index + search.length; + } + return count; +} + export function inspectDocxAcceptance( content: Uint8Array, expectation: DocxAcceptanceExpectation @@ -313,6 +369,62 @@ export function inspectDocxAcceptance( expectation.page.marginsTwips[side] ); } + assertMinimum( + expectation.id, + "sections", + sections.length, + expectation.minimumSections + ); + if (expectation.requireFirstSectionWithoutHeaderFooter) { + const firstSection = sections[0]!; + const headerReferences = + firstSection.getElementsByTagNameNS( + WORD_NAMESPACE, + "headerReference" + ).length; + const footerReferences = + firstSection.getElementsByTagNameNS( + WORD_NAMESPACE, + "footerReference" + ).length; + assertEqual( + expectation.id, + "first-section-header-references", + headerReferences, + 0 + ); + assertEqual( + expectation.id, + "first-section-footer-references", + footerReferences, + 0 + ); + } + if (expectation.finalPageNumberStart !== undefined) { + const pageNumber = firstDirectChild( + section, + WORD_NAMESPACE, + "pgNumType" + ); + if (!pageNumber) { + fail( + expectation.id, + "final-page-number-start", + "最终节缺少 w:pgNumType" + ); + } + assertEqual( + expectation.id, + "final-page-number-start", + numericAttribute( + pageNumber, + "start", + expectation.id, + "final-page-number-start" + ), + expectation.finalPageNumberStart + ); + } const paragraphCount = document.getElementsByTagNameNS( WORD_NAMESPACE, @@ -326,6 +438,22 @@ export function inspectDocxAcceptance( WORD_NAMESPACE, "tbl" ).length; + const fullWidthTableCount = Array.from( + document.getElementsByTagNameNS(WORD_NAMESPACE, "tbl") + ).filter((table) => { + const properties = firstDirectChild( + table, + WORD_NAMESPACE, + "tblPr" + ); + const width = properties + ? firstDirectChild(properties, WORD_NAMESPACE, "tblW") + : undefined; + return ( + width?.getAttributeNS(WORD_NAMESPACE, "type") === "pct" && + width.getAttributeNS(WORD_NAMESPACE, "w") === "5000" + ); + }).length; const numberedParagraphCount = document.getElementsByTagNameNS( WORD_NAMESPACE, "numPr" @@ -368,6 +496,12 @@ export function inspectDocxAcceptance( tableCount, expectation.minimumTables ); + assertMinimum( + expectation.id, + "full-width-tables", + fullWidthTableCount, + expectation.minimumFullWidthTables + ); assertMinimum( expectation.id, "numbering", @@ -410,6 +544,30 @@ export function inspectDocxAcceptance( ); } } + for (const [text, maximum] of Object.entries( + expectation.maximumTextOccurrences ?? {} + )) { + const actual = countTextOccurrences(bodyText, text); + if (actual > maximum) { + fail( + expectation.id, + "text-occurrences", + `${JSON.stringify(text)} 最多允许 ${maximum} 次,实际 ${actual} 次` + ); + } + } + const internalMarkerCount = countTextOccurrences( + bodyText, + "MD_TO_PDF_" + ); + if (expectation.forbidInternalMarkers) { + assertEqual( + expectation.id, + "internal-markers", + internalMarkerCount, + 0 + ); + } if ((expectation.requiredFootnoteText?.length ?? 0) > 0) { const footnotesPart = entries.get("word/footnotes.xml"); if (!footnotesPart) { @@ -517,11 +675,40 @@ export function inspectDocxAcceptance( ); } } + const appliedParagraphStyleIds = Array.from( + document.getElementsByTagNameNS(WORD_NAMESPACE, "pStyle") + ) + .map((element) => + element.getAttributeNS(WORD_NAMESPACE, "val") + ) + .filter((value): value is string => Boolean(value)); + for (const requiredStyleId of + expectation.requiredParagraphStyleIds ?? []) { + if (!appliedParagraphStyleIds.includes(requiredStyleId)) { + fail( + expectation.id, + "applied-paragraph-styles", + `正文未使用样式 ${requiredStyleId}` + ); + } + } const pageFieldCount = collectPageFieldCount(entries); + const sectionPageFieldCount = collectFieldCount( + entries, + "SECTIONPAGES" + ); if (expectation.requirePageField) { assertMinimum(expectation.id, "page-field", pageFieldCount, 1); } + if (expectation.requireSectionPagesField) { + assertMinimum( + expectation.id, + "section-pages-field", + sectionPageFieldCount, + 1 + ); + } return { id: expectation.id, package: packageValidation, @@ -538,10 +725,15 @@ export function inspectDocxAcceptance( drawingCount, pngImageCount: pngParts.size, pageFieldCount, + sectionPageFieldCount, + sectionCount: sections.length, + fullWidthTableCount, + internalMarkerCount, altChunkCount }, imageAltText, styleIds, + appliedParagraphStyleIds, checks: { package: true, page: true, @@ -549,6 +741,9 @@ export function inspectDocxAcceptance( relationships: true, pngMedia: true, styles: true, + sections: true, + tables: true, + textOccurrences: true, noAltChunk: true } }; diff --git a/packages/docx-engine/src/document-structure-transform.ts b/packages/docx-engine/src/document-structure-transform.ts new file mode 100644 index 0000000..14e9a76 --- /dev/null +++ b/packages/docx-engine/src/document-structure-transform.ts @@ -0,0 +1,864 @@ +import type { + DocxSlotStyleToken, + DocxThemeTokenSet +} from "@md-to-pdf/docx-theme-engine"; +import { + WORD_NAMESPACE, + appendElement, + colorValue, + directChildren, + firstDirectChild, + parseXmlPart, + removeDirectChildren, + serializeXmlPart, + type XmlElement +} from "./ooxml.js"; +import type { + PandocStructureBlock, + PandocStructureContainer, + PandocStructurePlan, + PandocStructureSectionBreak +} from "./pandoc-structure.js"; +import { + readGeneratedDocxPackage, + writeGeneratedDocxPackage +} from "./reference-package.js"; +import { + createDocxTokenSlotMap, + DOCX_SLOT_WORD_STYLE_BINDINGS +} from "./token-style-map.js"; + +type DocxBorderToken = NonNullable< + NonNullable["top"] +>; + +export interface GeneratedDocxStructureReport { + containerCount: number; + sectionCount: number; + tableCount: number; + pageBreakAfterCount: number; + sectionPageFieldCount: number; +} + +export interface FinalizedGeneratedDocx { + content: Uint8Array; + report: GeneratedDocxStructureReport; +} + +interface ContainerDescriptor { + container: PandocStructureContainer; + followedBySection: boolean; +} + +function setWordAttribute( + element: XmlElement, + localName: string, + value: string +) { + element.setAttributeNS( + WORD_NAMESPACE, + `w:${localName}`, + value + ); +} + +function wordAttribute( + element: XmlElement, + localName: string +) { + return element.getAttributeNS(WORD_NAMESPACE, localName); +} + +function ensureFirstElement( + parent: XmlElement, + localName: string, + qualifiedName: string +) { + const existing = firstDirectChild( + parent, + WORD_NAMESPACE, + localName + ); + if (existing) { + return existing; + } + const element = parent.ownerDocument!.createElementNS( + WORD_NAMESPACE, + qualifiedName + ); + parent.insertBefore(element, parent.firstChild); + return element; +} + +function paragraphProperties(paragraph: XmlElement) { + return ensureFirstElement(paragraph, "pPr", "w:pPr"); +} + +function paragraphText(paragraph: XmlElement) { + return Array.from( + paragraph.getElementsByTagNameNS(WORD_NAMESPACE, "t") + ) + .map((element) => element.textContent ?? "") + .join(""); +} + +function bodyElements(body: XmlElement) { + return Array.from(body.childNodes).filter( + (node): node is XmlElement => node.nodeType === 1 + ); +} + +function markerParagraph( + body: XmlElement, + marker: string +): XmlElement { + const matches = bodyElements(body).filter( + (element) => + element.namespaceURI === WORD_NAMESPACE && + element.localName === "p" && + paragraphText(element) === marker + ); + if (matches.length !== 1) { + throw new Error( + `DOCX 结构标记数量无效:${marker}(${matches.length})` + ); + } + return matches[0]!; +} + +function flattenContainers( + blocks: readonly PandocStructureBlock[] +): PandocStructureContainer[] { + return blocks.flatMap((block) => + block.kind === "container" + ? [block, ...flattenContainers(block.blocks)] + : [] + ); +} + +function containerDescriptors( + plan: PandocStructurePlan +): ContainerDescriptor[] { + const descriptors: ContainerDescriptor[] = []; + for (const blocks of [plan.prefix, plan.suffix]) { + for (const [index, block] of blocks.entries()) { + if (block.kind !== "container") { + continue; + } + descriptors.push({ + container: block, + followedBySection: + blocks[index + 1]?.kind === "section-break" + }); + for (const nested of flattenContainers(block.blocks)) { + descriptors.push({ + container: nested, + followedBySection: false + }); + } + } + } + return descriptors; +} + +function sectionBreaks( + blocks: readonly PandocStructureBlock[] +): PandocStructureSectionBreak[] { + return blocks.flatMap((block) => { + if (block.kind === "section-break") { + return [block]; + } + return block.kind === "container" + ? sectionBreaks(block.blocks) + : []; + }); +} + +function toggleParagraphProperty( + paragraph: XmlElement, + localName: "keepLines" | "keepNext" | "pageBreakBefore", + enabled: boolean +) { + const properties = paragraphProperties(paragraph); + removeDirectChildren(properties, WORD_NAMESPACE, localName); + if (enabled) { + appendElement( + properties, + WORD_NAMESPACE, + `w:${localName}` + ); + } +} + +function appendPageBreak(paragraph: XmlElement): boolean { + const existing = Array.from( + paragraph.getElementsByTagNameNS(WORD_NAMESPACE, "br") + ).some( + (element) => wordAttribute(element, "type") === "page" + ); + if (existing) { + return false; + } + const run = appendElement( + paragraph, + WORD_NAMESPACE, + "w:r" + ); + appendElement(run, WORD_NAMESPACE, "w:br", { + "w:type": "page" + }); + return true; +} + +function borderAttributes(border: DocxBorderToken) { + return { + "w:val": border.style, + "w:sz": String( + Math.max(2, Math.min(96, Math.round(border.widthPt * 8))) + ), + "w:space": "0", + "w:color": colorValue(border.color) + }; +} + +function applyContainerToken( + paragraphs: readonly XmlElement[], + token: DocxSlotStyleToken | undefined, + followedBySection: boolean +) { + if (!token || paragraphs.length === 0) { + return 0; + } + for (const [index, paragraph] of paragraphs.entries()) { + const properties = paragraphProperties(paragraph); + if (token.backgroundColor) { + removeDirectChildren(properties, WORD_NAMESPACE, "shd"); + appendElement(properties, WORD_NAMESPACE, "w:shd", { + "w:val": "clear", + "w:color": "auto", + "w:fill": colorValue(token.backgroundColor) + }); + } + if (token.borders) { + const borders = + firstDirectChild(properties, WORD_NAMESPACE, "pBdr") ?? + appendElement(properties, WORD_NAMESPACE, "w:pBdr"); + for (const side of ["left", "right"] as const) { + removeDirectChildren(borders, WORD_NAMESPACE, side); + const border = token.borders[side]; + if (border) { + appendElement( + borders, + WORD_NAMESPACE, + `w:${side}`, + borderAttributes(border) + ); + } + } + if (index === 0) { + removeDirectChildren(borders, WORD_NAMESPACE, "top"); + if (token.borders.top) { + appendElement( + borders, + WORD_NAMESPACE, + "w:top", + borderAttributes(token.borders.top) + ); + } + } + if (index === paragraphs.length - 1) { + removeDirectChildren(borders, WORD_NAMESPACE, "bottom"); + if (token.borders.bottom) { + appendElement( + borders, + WORD_NAMESPACE, + "w:bottom", + borderAttributes(token.borders.bottom) + ); + } + } + } + if (token.keepLines) { + toggleParagraphProperty(paragraph, "keepLines", true); + if (index < paragraphs.length - 1) { + toggleParagraphProperty(paragraph, "keepNext", true); + } + } + if (token.keepWithNext) { + toggleParagraphProperty(paragraph, "keepNext", true); + } + if (index === 0 && token.pageBreakBefore) { + toggleParagraphProperty( + paragraph, + "pageBreakBefore", + true + ); + } + } + return token.pageBreakAfter && !followedBySection + ? Number(appendPageBreak(paragraphs.at(-1)!)) + : 0; +} + +function applyContainerRanges( + body: XmlElement, + plan: PandocStructurePlan, + tokens: DocxThemeTokenSet +) { + const slots = createDocxTokenSlotMap(tokens); + let pageBreakAfterCount = 0; + let containerCount = 0; + for (const descriptor of containerDescriptors(plan)) { + const { container } = descriptor; + const start = markerParagraph(body, container.startMarker); + const end = markerParagraph(body, container.endMarker); + const elements = bodyElements(body); + const startIndex = elements.indexOf(start); + const endIndex = elements.indexOf(end); + if (startIndex < 0 || endIndex <= startIndex) { + throw new Error( + `DOCX 容器标记顺序无效:${container.startMarker}` + ); + } + const paragraphs = elements + .slice(startIndex + 1, endIndex) + .filter( + (element) => + element.namespaceURI === WORD_NAMESPACE && + element.localName === "p" + ); + const token = container.slot + ? slots.get(container.slot)?.style + : undefined; + pageBreakAfterCount += applyContainerToken( + paragraphs, + token, + descriptor.followedBySection + ); + body.removeChild(start); + body.removeChild(end); + containerCount += 1; + } + return { containerCount, pageBreakAfterCount }; +} + +function finalSection(body: XmlElement) { + const section = firstDirectChild( + body, + WORD_NAMESPACE, + "sectPr" + ); + if (!section) { + throw new Error("DOCX 正文缺少最终节"); + } + return section; +} + +function setSectionPageNumber( + section: XmlElement, + start: number +) { + removeDirectChildren(section, WORD_NAMESPACE, "pgNumType"); + const pageNumber = section.ownerDocument!.createElementNS( + WORD_NAMESPACE, + "w:pgNumType" + ); + setWordAttribute(pageNumber, "start", String(start)); + const insertionPoint = + firstDirectChild(section, WORD_NAMESPACE, "cols") ?? + firstDirectChild(section, WORD_NAMESPACE, "formProt") ?? + firstDirectChild(section, WORD_NAMESPACE, "vAlign") ?? + firstDirectChild(section, WORD_NAMESPACE, "titlePg"); + section.insertBefore(pageNumber, insertionPoint ?? null); +} + +function applySections( + body: XmlElement, + plan: PandocStructurePlan +) { + const breaks = [ + ...sectionBreaks(plan.prefix), + ...sectionBreaks(plan.suffix) + ]; + if (breaks.length === 0) { + return 0; + } + const followingSection = finalSection(body); + for (const sectionBreak of breaks) { + const marker = markerParagraph(body, sectionBreak.marker); + const elements = bodyElements(body); + const markerIndex = elements.indexOf(marker); + const previous = [...elements.slice(0, markerIndex)] + .reverse() + .find( + (element) => + element.namespaceURI === WORD_NAMESPACE && + element.localName === "p" + ); + if (!previous) { + throw new Error( + `DOCX 分节标记之前缺少段落:${sectionBreak.marker}` + ); + } + const coverSection = followingSection.cloneNode( + true + ) as XmlElement; + removeDirectChildren( + coverSection, + WORD_NAMESPACE, + "headerReference", + "footerReference", + "pgNumType", + "titlePg", + "type", + "vAlign" + ); + const type = coverSection.ownerDocument!.createElementNS( + WORD_NAMESPACE, + "w:type" + ); + setWordAttribute(type, "val", "nextPage"); + coverSection.insertBefore( + type, + firstDirectChild(coverSection, WORD_NAMESPACE, "pgSz") ?? + coverSection.firstChild + ); + if (sectionBreak.verticalAlignment) { + appendElement(coverSection, WORD_NAMESPACE, "w:vAlign", { + "w:val": sectionBreak.verticalAlignment + }); + } + const properties = paragraphProperties(previous); + removeDirectChildren(properties, WORD_NAMESPACE, "sectPr"); + properties.appendChild(coverSection); + setSectionPageNumber( + followingSection, + sectionBreak.followingPageNumberStart + ); + body.removeChild(marker); + } + return breaks.length; +} + +function replaceSectionPageFields( + entries: Map, + enabled: boolean +) { + if (!enabled) { + return 0; + } + let count = 0; + for (const [partName, content] of entries) { + if (!/^word\/footer\d+\.xml$/u.test(partName)) { + continue; + } + const document = parseXmlPart(content, partName); + for (const instruction of Array.from( + document.getElementsByTagNameNS( + WORD_NAMESPACE, + "instrText" + ) + )) { + const value = instruction.textContent ?? ""; + const replaced = value.replace( + /\bNUMPAGES\b/gu, + "SECTIONPAGES" + ); + if (replaced === value) { + continue; + } + while (instruction.firstChild) { + instruction.removeChild(instruction.firstChild); + } + instruction.appendChild( + instruction.ownerDocument!.createTextNode(replaced) + ); + count += 1; + } + entries.set(partName, serializeXmlPart(document)); + } + return count; +} + +function normalizeGeneratedStyles( + entries: Map +) { + const partName = "word/styles.xml"; + const document = parseXmlPart(entries.get(partName)!, partName); + const styles = document.documentElement; + if ( + !styles || + styles.namespaceURI !== WORD_NAMESPACE || + styles.localName !== "styles" + ) { + throw new Error("word/styles.xml 的根元素无效"); + } + const seenStyleIds = new Set(); + for (const style of directChildren( + styles, + WORD_NAMESPACE, + "style" + )) { + const styleId = wordAttribute(style, "styleId"); + if (!styleId || !seenStyleIds.has(styleId)) { + if (styleId) { + seenStyleIds.add(styleId); + } + continue; + } + styles.removeChild(style); + } + for (const alignment of Array.from( + styles.getElementsByTagNameNS(WORD_NAMESPACE, "jc") + )) { + if (wordAttribute(alignment, "val") === "justify") { + setWordAttribute(alignment, "val", "both"); + } + } + for (const style of directChildren( + styles, + WORD_NAMESPACE, + "style" + )) { + if (wordAttribute(style, "type") !== "table") { + continue; + } + const properties = firstDirectChild( + style, + WORD_NAMESPACE, + "tblPr" + ); + if (properties) { + removeDirectChildren(properties, WORD_NAMESPACE, "tblW"); + } + } + entries.set(partName, serializeXmlPart(document)); +} + +function sectionContentWidth(section: XmlElement) { + const pageSize = firstDirectChild( + section, + WORD_NAMESPACE, + "pgSz" + ); + const margins = firstDirectChild( + section, + WORD_NAMESPACE, + "pgMar" + ); + const width = Number(pageSize && wordAttribute(pageSize, "w")); + const left = Number(margins && wordAttribute(margins, "left")); + const right = Number(margins && wordAttribute(margins, "right")); + const contentWidth = width - left - right; + if ( + !Number.isFinite(contentWidth) || + contentWidth <= 0 + ) { + throw new Error("DOCX 内容区宽度无效"); + } + return Math.round(contentWidth); +} + +function gridSpan(cell: XmlElement) { + const properties = firstDirectChild( + cell, + WORD_NAMESPACE, + "tcPr" + ); + const span = properties + ? firstDirectChild(properties, WORD_NAMESPACE, "gridSpan") + : undefined; + const value = Number(span && wordAttribute(span, "val")); + return Number.isInteger(value) && value > 0 ? value : 1; +} + +function tableColumnCount(table: XmlElement) { + return Math.max( + 1, + ...directChildren(table, WORD_NAMESPACE, "tr").map((row) => + directChildren(row, WORD_NAMESPACE, "tc").reduce( + (total, cell) => total + gridSpan(cell), + 0 + ) + ) + ); +} + +function scaledColumnWidths( + table: XmlElement, + count: number, + targetWidth: number +) { + const grid = firstDirectChild( + table, + WORD_NAMESPACE, + "tblGrid" + ); + const declared = grid + ? directChildren(grid, WORD_NAMESPACE, "gridCol").map( + (column) => Number(wordAttribute(column, "w")) + ) + : []; + const weights = + declared.length === count && + declared.every((value) => Number.isFinite(value) && value > 0) + ? declared + : Array.from({ length: count }, () => 1); + const total = weights.reduce((sum, value) => sum + value, 0); + const widths = weights.map((value) => + Math.max(1, Math.round((targetWidth * value) / total)) + ); + const lastIndex = widths.length - 1; + widths[lastIndex] = + widths[lastIndex]! + + targetWidth - widths.reduce((sum, value) => sum + value, 0); + return widths; +} + +function ensureTableGrid(table: XmlElement) { + const existing = firstDirectChild( + table, + WORD_NAMESPACE, + "tblGrid" + ); + if (existing) { + return existing; + } + const grid = table.ownerDocument!.createElementNS( + WORD_NAMESPACE, + "w:tblGrid" + ); + const firstRow = firstDirectChild( + table, + WORD_NAMESPACE, + "tr" + ); + table.insertBefore(grid, firstRow ?? null); + return grid; +} + +function applyTables( + document: ReturnType, + section: XmlElement, + tokens: DocxThemeTokenSet +) { + const tableToken = + createDocxTokenSlotMap(tokens).get("table")?.style; + const widthPercent = tableToken?.widthPercent ?? 100; + const targetWidth = Math.max( + 1, + Math.round( + (sectionContentWidth(section) * widthPercent) / 100 + ) + ); + const tables = Array.from( + document.getElementsByTagNameNS(WORD_NAMESPACE, "tbl") + ); + for (const table of tables) { + const properties = ensureFirstElement( + table, + "tblPr", + "w:tblPr" + ); + removeDirectChildren( + properties, + WORD_NAMESPACE, + "tblW", + "tblLayout" + ); + appendElement(properties, WORD_NAMESPACE, "w:tblW", { + "w:w": String(Math.round(widthPercent * 50)), + "w:type": "pct" + }); + appendElement(properties, WORD_NAMESPACE, "w:tblLayout", { + "w:type": "fixed" + }); + + const columnCount = tableColumnCount(table); + const widths = scaledColumnWidths( + table, + columnCount, + targetWidth + ); + const grid = ensureTableGrid(table); + for (const child of Array.from(grid.childNodes)) { + grid.removeChild(child); + } + for (const width of widths) { + appendElement(grid, WORD_NAMESPACE, "w:gridCol", { + "w:w": String(width) + }); + } + + for (const row of directChildren( + table, + WORD_NAMESPACE, + "tr" + )) { + if (tableToken?.keepLines) { + const rowProperties = ensureFirstElement( + row, + "trPr", + "w:trPr" + ); + removeDirectChildren( + rowProperties, + WORD_NAMESPACE, + "cantSplit" + ); + appendElement( + rowProperties, + WORD_NAMESPACE, + "w:cantSplit" + ); + } + let columnIndex = 0; + for (const cell of directChildren( + row, + WORD_NAMESPACE, + "tc" + )) { + const span = gridSpan(cell); + const cellWidth = widths + .slice(columnIndex, columnIndex + span) + .reduce((sum, value) => sum + value, 0); + columnIndex += span; + const cellProperties = ensureFirstElement( + cell, + "tcPr", + "w:tcPr" + ); + removeDirectChildren( + cellProperties, + WORD_NAMESPACE, + "tcW" + ); + appendElement( + cellProperties, + WORD_NAMESPACE, + "w:tcW", + { + "w:w": String(cellWidth), + "w:type": "dxa" + } + ); + } + } + } + return tables.length; +} + +function pageBreakAfterStyleIds(tokens: DocxThemeTokenSet) { + const ids = new Set(); + for (const entry of tokens.slots) { + if (!entry.style.pageBreakAfter) { + continue; + } + for (const binding of + DOCX_SLOT_WORD_STYLE_BINDINGS[entry.slot] ?? []) { + if (binding.type === "paragraph") { + ids.add(binding.styleId); + } + } + } + return ids; +} + +function applyParagraphPageBreaks( + document: ReturnType, + tokens: DocxThemeTokenSet +) { + const styleIds = pageBreakAfterStyleIds(tokens); + let count = 0; + for (const paragraph of Array.from( + document.getElementsByTagNameNS(WORD_NAMESPACE, "p") + )) { + const properties = firstDirectChild( + paragraph, + WORD_NAMESPACE, + "pPr" + ); + const style = properties + ? firstDirectChild(properties, WORD_NAMESPACE, "pStyle") + : undefined; + const styleId = style && wordAttribute(style, "val"); + if (styleId && styleIds.has(styleId)) { + count += Number(appendPageBreak(paragraph)); + } + } + return count; +} + +function ensureNoMarkers( + body: XmlElement, + plan: PandocStructurePlan +) { + const markers = new Set(); + for (const descriptor of containerDescriptors(plan)) { + markers.add(descriptor.container.startMarker); + markers.add(descriptor.container.endMarker); + } + for (const section of [ + ...sectionBreaks(plan.prefix), + ...sectionBreaks(plan.suffix) + ]) { + markers.add(section.marker); + } + const residual = bodyElements(body).find( + (element) => + element.localName === "p" && + markers.has(paragraphText(element)) + ); + if (residual) { + throw new Error( + `DOCX 结构标记未被消费:${paragraphText(residual)}` + ); + } +} + +export function finalizeGeneratedDocxStructure( + content: Uint8Array, + plan: PandocStructurePlan, + tokens: DocxThemeTokenSet +): FinalizedGeneratedDocx { + const source = readGeneratedDocxPackage(content); + const entries = new Map(source.entries); + const document = parseXmlPart( + entries.get("word/document.xml")!, + "word/document.xml" + ); + const body = document.getElementsByTagNameNS( + WORD_NAMESPACE, + "body" + )[0]; + if (!body) { + throw new Error("DOCX 正文结构无效"); + } + const containers = applyContainerRanges(body, plan, tokens); + const sectionCount = applySections(body, plan); + const final = finalSection(body); + const tableCount = applyTables(document, final, tokens); + const pageBreakAfterCount = + containers.pageBreakAfterCount + + applyParagraphPageBreaks(document, tokens); + ensureNoMarkers(body, plan); + entries.set( + "word/document.xml", + serializeXmlPart(document) + ); + const sectionPageFieldCount = replaceSectionPageFields( + entries, + sectionCount > 0 + ); + normalizeGeneratedStyles(entries); + return { + content: writeGeneratedDocxPackage(entries), + report: { + containerCount: containers.containerCount, + sectionCount, + tableCount, + pageBreakAfterCount, + sectionPageFieldCount + } + }; +} diff --git a/packages/docx-engine/src/index.ts b/packages/docx-engine/src/index.ts index 02f0424..987308b 100644 --- a/packages/docx-engine/src/index.ts +++ b/packages/docx-engine/src/index.ts @@ -1,8 +1,10 @@ export * from "./acceptance-validator.js"; +export * from "./document-structure-transform.js"; export * from "./header-footer-transform.js"; export * from "./ooxml.js"; export * from "./pandoc-process.js"; export * from "./pandoc-media.js"; +export * from "./pandoc-structure.js"; export * from "./pandoc-converter.js"; export * from "./pandoc-runtime-manifest.js"; export * from "./pandoc-runtime.js"; diff --git a/packages/docx-engine/src/ooxml.ts b/packages/docx-engine/src/ooxml.ts index 34398df..ac8531a 100644 --- a/packages/docx-engine/src/ooxml.ts +++ b/packages/docx-engine/src/ooxml.ts @@ -1,10 +1,11 @@ import { DOMParser, XMLSerializer } from "@xmldom/xmldom"; import type { Document as XmlDocument, - Element as XmlElement + Element as XmlElement, + Node as XmlNode } from "@xmldom/xmldom"; -export type { XmlDocument, XmlElement }; +export type { XmlDocument, XmlElement, XmlNode }; export const WORD_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"; @@ -20,6 +21,385 @@ export const DRAWING_NAMESPACE = const decoder = new TextDecoder("utf-8", { fatal: true }); const encoder = new TextEncoder(); +const wordprocessingChildOrders = new Map< + string, + readonly string[] +>([ + [ + "style", + [ + "name", + "aliases", + "basedOn", + "next", + "link", + "autoRedefine", + "hidden", + "uiPriority", + "semiHidden", + "unhideWhenUsed", + "qFormat", + "locked", + "personal", + "personalCompose", + "personalReply", + "rsid", + "pPr", + "rPr", + "tblPr", + "trPr", + "tcPr" + ] + ], + [ + "pPr", + [ + "pStyle", + "keepNext", + "keepLines", + "pageBreakBefore", + "framePr", + "widowControl", + "numPr", + "suppressLineNumbers", + "pBdr", + "shd", + "tabs", + "suppressAutoHyphens", + "kinsoku", + "wordWrap", + "overflowPunct", + "topLinePunct", + "autoSpaceDE", + "autoSpaceDN", + "bidi", + "adjustRightInd", + "snapToGrid", + "spacing", + "ind", + "contextualSpacing", + "mirrorIndents", + "suppressOverlap", + "jc", + "textDirection", + "textAlignment", + "textboxTightWrap", + "outlineLvl", + "divId", + "cnfStyle", + "rPr", + "sectPr", + "pPrChange" + ] + ], + [ + "rPr", + [ + "rStyle", + "rFonts", + "b", + "bCs", + "i", + "iCs", + "caps", + "smallCaps", + "strike", + "dstrike", + "outline", + "shadow", + "emboss", + "imprint", + "noProof", + "snapToGrid", + "vanish", + "webHidden", + "color", + "spacing", + "w", + "kern", + "position", + "sz", + "szCs", + "highlight", + "u", + "effect", + "bdr", + "shd", + "fitText", + "vertAlign", + "rtl", + "cs", + "em", + "lang", + "eastAsianLayout", + "specVanish", + "oMath", + "rPrChange" + ] + ], + [ + "numPr", + ["ilvl", "numId", "numberingChange", "ins"] + ], + [ + "pBdr", + ["top", "left", "bottom", "right", "between", "bar"] + ], + [ + "tblPr", + [ + "tblStyle", + "tblpPr", + "tblOverlap", + "bidiVisual", + "tblStyleRowBandSize", + "tblStyleColBandSize", + "tblW", + "jc", + "tblCellSpacing", + "tblInd", + "tblBorders", + "shd", + "tblLayout", + "tblCellMar", + "tblLook", + "tblCaption", + "tblDescription", + "tblPrChange" + ] + ], + [ + "trPr", + [ + "cnfStyle", + "divId", + "gridBefore", + "gridAfter", + "wBefore", + "wAfter", + "cantSplit", + "trHeight", + "tblHeader", + "tblCellSpacing", + "jc", + "hidden", + "ins", + "del", + "trPrChange", + "conflictIns", + "conflictDel" + ] + ], + [ + "tcPr", + [ + "cnfStyle", + "tcW", + "gridSpan", + "hMerge", + "vMerge", + "tcBorders", + "shd", + "noWrap", + "tcMar", + "textDirection", + "tcFitText", + "vAlign", + "hideMark", + "headers", + "cellIns", + "cellDel", + "cellMerge", + "tcPrChange" + ] + ], + [ + "tblStylePr", + ["pPr", "rPr", "tblPr", "trPr", "tcPr"] + ], + [ + "sectPr", + [ + "headerReference", + "footerReference", + "footnotePr", + "endnotePr", + "type", + "pgSz", + "pgMar", + "paperSrc", + "pgBorders", + "lnNumType", + "pgNumType", + "cols", + "formProt", + "vAlign", + "noEndnote", + "titlePg", + "textDirection", + "bidi", + "rtlGutter", + "docGrid", + "printerSettings", + "sectPrChange" + ] + ], + [ + "tblBorders", + ["top", "left", "bottom", "right", "insideH", "insideV"] + ], + [ + "tcBorders", + [ + "top", + "left", + "bottom", + "right", + "insideH", + "insideV", + "tl2br", + "tr2bl" + ] + ], + [ + "tblCellMar", + ["top", "left", "start", "bottom", "right", "end"] + ], + [ + "tcMar", + ["top", "left", "start", "bottom", "right", "end"] + ], + [ + "font", + [ + "altName", + "panose1", + "charset", + "family", + "notTrueType", + "pitch", + "sig", + "embedRegular", + "embedBold", + "embedItalic", + "embedBoldItalic" + ] + ] +]); + +function directWordChildren(parent: XmlElement) { + return Array.from(parent.childNodes).filter( + (node): node is XmlElement => + node.nodeType === 1 && + (node as XmlElement).namespaceURI === WORD_NAMESPACE + ); +} + +function wordChildOrder(parent: XmlElement) { + if (parent.namespaceURI !== WORD_NAMESPACE) { + return undefined; + } + return wordprocessingChildOrders.get(parent.localName ?? ""); +} + +function orderedWordChildren(parent: XmlElement) { + const order = wordChildOrder(parent); + if (!order) { + return undefined; + } + const ranks = new Map(order.map((name, index) => [name, index])); + return { + order, + ranks, + children: directWordChildren(parent).filter((child) => + ranks.has(child.localName ?? "") + ) + }; +} + +function insertWordElementInOrder( + parent: XmlElement, + element: XmlElement +) { + const ordered = orderedWordChildren(parent); + const rank = ordered?.ranks.get(element.localName ?? ""); + if (!ordered || rank === undefined) { + parent.appendChild(element); + return; + } + const insertionPoint = ordered.children.find((child) => { + const childRank = ordered.ranks.get(child.localName ?? ""); + return childRank !== undefined && childRank > rank; + }); + parent.insertBefore(element, insertionPoint ?? null); +} + +export function normalizeWordprocessingElementOrder( + document: XmlDocument +) { + const elements = Array.from(document.getElementsByTagName("*")); + for (const parent of elements) { + const ordered = orderedWordChildren(parent); + if (!ordered || ordered.children.length < 2) { + continue; + } + const sorted = ordered.children + .map((child, index) => ({ child, index })) + .sort((left, right) => { + const leftRank = ordered.ranks.get( + left.child.localName ?? "" + )!; + const rightRank = ordered.ranks.get( + right.child.localName ?? "" + )!; + return leftRank - rightRank || left.index - right.index; + }) + .map(({ child }) => child); + if ( + sorted.every( + (child, index) => child === ordered.children[index] + ) + ) { + continue; + } + const anchor: XmlNode | null = + ordered.children.at(-1)?.nextSibling ?? null; + for (const child of ordered.children) { + parent.removeChild(child); + } + for (const child of sorted) { + parent.insertBefore(child, anchor); + } + } +} + +export function validateWordprocessingElementOrder( + document: XmlDocument, + partName: string +) { + const elements = Array.from(document.getElementsByTagName("*")); + for (const parent of elements) { + const ordered = orderedWordChildren(parent); + if (!ordered) { + continue; + } + let previousRank = -1; + let previousName = ""; + for (const child of ordered.children) { + const name = child.localName ?? ""; + const rank = ordered.ranks.get(name)!; + if (rank < previousRank) { + throw new Error( + `${partName} 的 w:${parent.localName} 子节点顺序无效:` + + `w:${name} 不得位于 w:${previousName} 之后` + ); + } + previousRank = rank; + previousName = name; + } + } +} + export function parseXmlPart( content: Uint8Array, partName: string @@ -42,6 +422,7 @@ export function parseXmlPart( } export function serializeXmlPart(document: XmlDocument) { + normalizeWordprocessingElementOrder(document); return encoder.encode( `\n${new XMLSerializer().serializeToString( document.documentElement! @@ -113,7 +494,11 @@ export function appendElement( : null; element.setAttributeNS(attributeNamespace, name, value); } - parent.appendChild(element); + if (namespace === WORD_NAMESPACE) { + insertWordElementInOrder(parent, element); + } else { + parent.appendChild(element); + } return element; } diff --git a/packages/docx-engine/src/pandoc-converter.ts b/packages/docx-engine/src/pandoc-converter.ts index 42d816f..fee7c79 100644 --- a/packages/docx-engine/src/pandoc-converter.ts +++ b/packages/docx-engine/src/pandoc-converter.ts @@ -1,3 +1,6 @@ +import { + randomUUID +} from "node:crypto"; import { mkdir, mkdtemp, @@ -15,6 +18,7 @@ import { type ExportConfig, type MarkdownDocumentMetadata, type PreparedDocxMedia, + type SemanticDocumentModel, type ThemeManifest } from "@md-to-pdf/core"; import type { DocxThemeTokenSet } from "@md-to-pdf/docx-theme-engine"; @@ -23,8 +27,10 @@ import { createDynamicReferenceDocx, type DynamicReferenceDocxResult } from "./reference-builder.js"; +import { finalizeGeneratedDocxStructure } from "./document-structure-transform.js"; import type { DynamicReferenceDocxOptions } from "./reference-options.js"; import { preparePandocMedia } from "./pandoc-media.js"; +import { createPandocStructurePlan } from "./pandoc-structure.js"; import { runPandocProcess, type PandocProcessRunner @@ -42,6 +48,8 @@ const MAXIMUM_PANDOC_STDERR_BYTES = 64 * 1024; const DEFAULT_REFERENCE_CACHE_SIZE = 32; const temporaryDirectoryPrefix = "md-to-pdf-docx-"; const mediaMapEnvironmentName = "MD_TO_PDF_DOCX_MEDIA_MAP"; +const structurePlanEnvironmentName = + "MD_TO_PDF_DOCX_STRUCTURE_PLAN"; const luaFilterUrl = new URL( "../assets/docx-media-filter.lua", import.meta.url @@ -60,6 +68,7 @@ export interface PandocDocxConversionInput { theme: ThemeManifest; themeTokens: DocxThemeTokenSet; metadata: MarkdownDocumentMetadata; + semanticDocument: SemanticDocumentModel; media: PreparedDocxMedia; } @@ -215,12 +224,18 @@ export class PandocDocxConverter { signal?: AbortSignal ): Promise { let preparedMedia; + let structurePlan; try { preparedMedia = preparePandocMedia(input.media); + structurePlan = createPandocStructurePlan( + input.semanticDocument, + input.themeTokens, + { markerSeed: randomUUID() } + ); } catch (error) { throw new PandocDocxConversionError( "DOCX_GENERATION_FAILED", - "DOCX 媒体映射无效", + "DOCX 媒体或语义结构映射无效", { cause: error } ); } @@ -251,6 +266,10 @@ export class PandocDocxConverter { "docx-media-filter.lua" ), mediaMap: path.join(temporaryDirectory, "media-map.json"), + structurePlan: path.join( + temporaryDirectory, + "structure-plan.json" + ), mediaDirectory: path.join(temporaryDirectory, "media"), dataDirectory: path.join(temporaryDirectory, "pandoc-data") }; @@ -267,6 +286,11 @@ export class PandocDocxConverter { JSON.stringify(preparedMedia.map), "utf8" ), + writeFile( + paths.structurePlan, + JSON.stringify(structurePlan), + "utf8" + ), ...preparedMedia.files.map((file) => writeFile( path.join(temporaryDirectory, file.relativePath), @@ -291,7 +315,8 @@ export class PandocDocxConverter { cwd: temporaryDirectory, env: { ...(this.options.environment ?? process.env), - [mediaMapEnvironmentName]: paths.mediaMap + [mediaMapEnvironmentName]: paths.mediaMap, + [structurePlanEnvironmentName]: paths.structurePlan }, timeoutMs: this.timeoutMs, maxStdoutBytes: MAXIMUM_PANDOC_STDOUT_BYTES, @@ -333,9 +358,19 @@ export class PandocDocxConverter { "DOCX 输出大小无效" ); } - const docx = await readFile(paths.output); + let docx: Uint8Array; let validation: DynamicReferenceValidation; try { + const pandocDocx = await readFile(paths.output); + const finalized = finalizeGeneratedDocxStructure( + pandocDocx, + structurePlan, + input.themeTokens + ); + docx = finalized.content; + if (docx.byteLength > this.maximumOutputBytes) { + throw new Error("DOCX 结构收口后的输出大小无效"); + } validation = validateGeneratedDocx(docx); } catch (error) { throw new PandocDocxConversionError( diff --git a/packages/docx-engine/src/pandoc-structure.ts b/packages/docx-engine/src/pandoc-structure.ts new file mode 100644 index 0000000..0ee96a5 --- /dev/null +++ b/packages/docx-engine/src/pandoc-structure.ts @@ -0,0 +1,240 @@ +import { + semanticDocumentModelSchema, + type SemanticDocumentGroupNode, + type SemanticDocumentModel, + type SemanticDocumentNode, + type SemanticDocumentRole, + type SemanticDocumentTextNode +} from "@md-to-pdf/core"; +import { + type DocxStyleSlotName, + type DocxThemeTokenSet +} from "@md-to-pdf/docx-theme-engine"; +import { + createDocxTokenSlotMap, + DOCX_SLOT_WORD_STYLE_BINDINGS +} from "./token-style-map.js"; + +export interface PandocStructureParagraph { + kind: "paragraph"; + styleId?: string | undefined; + segments: string[]; + separator: "space" | "tab"; +} + +export interface PandocStructureContainer { + kind: "container"; + styleId?: string | undefined; + slot?: DocxStyleSlotName | undefined; + startMarker: string; + endMarker: string; + blocks: PandocStructureBlock[]; +} + +export interface PandocStructureSectionBreak { + kind: "section-break"; + marker: string; + headerFooter: "none"; + pageNumber: "hidden"; + followingPageNumberStart: number; + verticalAlignment?: "top" | "center" | "bottom" | undefined; +} + +export type PandocStructureBlock = + | PandocStructureParagraph + | PandocStructureContainer + | PandocStructureSectionBreak; + +export interface PandocStructurePlan { + schemaVersion: 1; + titlePolicy: SemanticDocumentModel["titlePolicy"]; + prefix: PandocStructureBlock[]; + suffix: PandocStructureBlock[]; +} + +export interface CreatePandocStructurePlanOptions { + markerSeed?: string | undefined; +} + +const containerRoles = new Set([ + "official-masthead", + "official-signature", + "official-edition", + "briefing-masthead", + "project-report-cover", + "tender-cover" +]); + +function slotName( + role: SemanticDocumentRole +): DocxStyleSlotName | undefined { + return Object.prototype.hasOwnProperty.call( + DOCX_SLOT_WORD_STYLE_BINDINGS, + role + ) + ? (role as DocxStyleSlotName) + : undefined; +} + +function paragraphStyleId( + role: SemanticDocumentRole +): string | undefined { + const slot = slotName(role); + return slot + ? DOCX_SLOT_WORD_STYLE_BINDINGS[slot]?.find( + (binding) => binding.type === "paragraph" + )?.styleId + : undefined; +} + +function textValue(node: SemanticDocumentTextNode): string { + return `${node.label ?? ""}${node.text}`; +} + +function isHidden( + role: SemanticDocumentRole, + slots: ReturnType +): boolean { + const slot = slotName(role); + return slot ? slots.get(slot)?.style.hidden === true : false; +} + +function inlineSegments( + node: SemanticDocumentNode, + slots: ReturnType +): string[] { + if (isHidden(node.role, slots)) { + return []; + } + return node.kind === "text" + ? [textValue(node)] + : node.children.flatMap((child) => + inlineSegments(child, slots) + ); +} + +function projectTextNode( + node: SemanticDocumentTextNode +): PandocStructureParagraph { + const styleId = paragraphStyleId(node.role); + return { + kind: "paragraph", + ...(styleId ? { styleId } : {}), + segments: [textValue(node)], + separator: "space" + }; +} + +function projectGroupNode( + node: SemanticDocumentGroupNode, + slots: ReturnType, + nextMarker: (kind: string) => string +): PandocStructureBlock | undefined { + const styleId = paragraphStyleId(node.role); + if (containerRoles.has(node.role)) { + const blocks = node.children + .map((child) => projectNode(child, slots, nextMarker)) + .filter( + (block): block is PandocStructureBlock => + block !== undefined + ); + return blocks.length + ? { + kind: "container", + ...(styleId ? { styleId } : {}), + ...(slotName(node.role) + ? { slot: slotName(node.role) } + : {}), + startMarker: nextMarker("container-start"), + endMarker: nextMarker("container-end"), + blocks + } + : undefined; + } + const segments = node.children.flatMap((child) => + inlineSegments(child, slots) + ); + return segments.length + ? { + kind: "paragraph", + ...(styleId ? { styleId } : {}), + segments, + separator: segments.length > 1 ? "tab" : "space" + } + : undefined; +} + +function projectNode( + node: SemanticDocumentNode, + slots: ReturnType, + nextMarker: (kind: string) => string +): PandocStructureBlock | undefined { + if (isHidden(node.role, slots)) { + return undefined; + } + return node.kind === "text" + ? projectTextNode(node) + : projectGroupNode(node, slots, nextMarker); +} + +export function createPandocStructurePlan( + modelInput: SemanticDocumentModel, + tokensInput: DocxThemeTokenSet, + options: CreatePandocStructurePlanOptions = {} +): PandocStructurePlan { + const model = semanticDocumentModelSchema.parse(modelInput); + const slots = createDocxTokenSlotMap(tokensInput); + const markerSeed = (options.markerSeed ?? "structure") + .replace(/[^a-z0-9]/giu, "") + .slice(0, 64); + if (!markerSeed) { + throw new Error("DOCX 结构标记种子无效"); + } + let markerIndex = 0; + const nextMarker = (kind: string) => + `MD_TO_PDF_${kind.toUpperCase().replace(/-/gu, "_")}_${markerSeed}_${++markerIndex}`; + const prefix: PandocStructureBlock[] = []; + const suffix: PandocStructureBlock[] = []; + for (const region of model.regions) { + const blocks = region.nodes + .map((node) => projectNode(node, slots, nextMarker)) + .filter( + (block): block is PandocStructureBlock => + block !== undefined + ); + if (region.kind === "suffix") { + suffix.push(...blocks); + continue; + } + prefix.push(...blocks); + if ( + blocks.length > 0 && + region.section?.breakAfter === "next-page" + ) { + const firstContainer = blocks.find( + ( + block + ): block is PandocStructureContainer => + block.kind === "container" + ); + const verticalAlignment = firstContainer?.slot + ? slots.get(firstContainer.slot)?.style.verticalAlignment + : undefined; + prefix.push({ + kind: "section-break", + marker: nextMarker("section-break"), + headerFooter: region.section.headerFooter, + pageNumber: region.section.pageNumber, + followingPageNumberStart: + region.section.followingPageNumberStart, + ...(verticalAlignment ? { verticalAlignment } : {}) + }); + } + } + return { + schemaVersion: 1, + titlePolicy: model.titlePolicy, + prefix, + suffix + }; +} diff --git a/packages/docx-engine/src/reference-package.ts b/packages/docx-engine/src/reference-package.ts index 171a56d..a792917 100644 --- a/packages/docx-engine/src/reference-package.ts +++ b/packages/docx-engine/src/reference-package.ts @@ -1,4 +1,5 @@ import { createHash } from "node:crypto"; +import { MAXIMUM_DOCX_OUTPUT_BYTES } from "@md-to-pdf/core"; import { unzipSync, zipSync, type Zippable } from "fflate"; export const MAXIMUM_REFERENCE_DOCX_BYTES = 2 * 1024 * 1024; @@ -21,6 +22,13 @@ export const referenceDocxPackageLimits: DocxPackageLimits = { maximumUncompressedBytes: MAXIMUM_REFERENCE_UNCOMPRESSED_BYTES }; +export const generatedDocxPackageLimits: DocxPackageLimits = { + maximumBytes: MAXIMUM_DOCX_OUTPUT_BYTES, + maximumEntryCount: MAXIMUM_GENERATED_DOCX_ENTRY_COUNT, + maximumUncompressedBytes: + MAXIMUM_GENERATED_DOCX_UNCOMPRESSED_BYTES +}; + export const requiredReferenceDocxParts = [ "[Content_Types].xml", "_rels/.rels", @@ -216,8 +224,15 @@ export function readReferenceDocxPackage( return readDocxPackage(content, referenceDocxPackageLimits); } -export function writeReferenceDocxPackage( - entries: ReadonlyMap +export function readGeneratedDocxPackage( + content: Uint8Array +): ReferenceDocxPackage { + return readDocxPackage(content, generatedDocxPackageLimits); +} + +function writeDocxPackage( + entries: ReadonlyMap, + limits: DocxPackageLimits ) { const zippable: Zippable = {}; const mtime = new Date("1980-01-01T00:00:00.000Z"); @@ -231,6 +246,18 @@ export function writeReferenceDocxPackage( level: 6, mtime }); - readReferenceDocxPackage(result); + readDocxPackage(result, limits); return result; } + +export function writeReferenceDocxPackage( + entries: ReadonlyMap +) { + return writeDocxPackage(entries, referenceDocxPackageLimits); +} + +export function writeGeneratedDocxPackage( + entries: ReadonlyMap +) { + return writeDocxPackage(entries, generatedDocxPackageLimits); +} diff --git a/packages/docx-engine/src/styles-transform.ts b/packages/docx-engine/src/styles-transform.ts index 650adda..a09ba6e 100644 --- a/packages/docx-engine/src/styles-transform.ts +++ b/packages/docx-engine/src/styles-transform.ts @@ -378,7 +378,10 @@ function applyTokenParagraphStyle( if (token.alignment !== undefined) { removeDirectChildren(parent, WORD_NAMESPACE, "jc"); appendElement(parent, WORD_NAMESPACE, "w:jc", { - "w:val": token.alignment + "w:val": + token.alignment === "justify" + ? "both" + : token.alignment }); } if (token.backgroundColor !== undefined) { @@ -709,7 +712,13 @@ function applyTableAndCaption( WORD_NAMESPACE, "w:tblPr" ); - removeDirectChildren(tblPr, WORD_NAMESPACE, "tblCellMar", "tblBorders"); + removeDirectChildren( + tblPr, + WORD_NAMESPACE, + "tblW", + "tblCellMar", + "tblBorders" + ); const margins = appendElement( tblPr, WORD_NAMESPACE, @@ -863,15 +872,7 @@ function applyTableTokenStyles( WORD_NAMESPACE, "w:tblPr" ); - if (tableToken?.widthPercent !== undefined) { - removeDirectChildren(tblPr, WORD_NAMESPACE, "tblW"); - appendElement(tblPr, WORD_NAMESPACE, "w:tblW", { - "w:w": String( - Math.round(tableToken.widthPercent * 50) - ), - "w:type": "pct" - }); - } + removeDirectChildren(tblPr, WORD_NAMESPACE, "tblW"); const padding = cellToken?.paddingPt ?? tableToken?.paddingPt; if (padding) { removeDirectChildren(tblPr, WORD_NAMESPACE, "tblCellMar"); diff --git a/packages/docx-engine/src/validator.ts b/packages/docx-engine/src/validator.ts index 8dd5ba9..2256a87 100644 --- a/packages/docx-engine/src/validator.ts +++ b/packages/docx-engine/src/validator.ts @@ -7,6 +7,7 @@ import { WORD_NAMESPACE, directChildren, parseXmlPart, + validateWordprocessingElementOrder, type XmlElement } from "./ooxml.js"; import { @@ -262,13 +263,54 @@ function validateStyles(entries: ReadonlyMap) { ) { throw new Error("word/styles.xml 的根元素无效"); } - const styleIds = new Set( - Array.from( - root.getElementsByTagNameNS(WORD_NAMESPACE, "style") - ).map((element) => - element.getAttributeNS(WORD_NAMESPACE, "styleId") - ) - ); + const styleIds = new Set(); + for (const style of directChildren( + root, + WORD_NAMESPACE, + "style" + )) { + const styleId = + style.getAttributeNS(WORD_NAMESPACE, "styleId") ?? ""; + if (styleIds.has(styleId)) { + throw new Error( + `word/styles.xml 包含重复样式 ID:${styleId}` + ); + } + styleIds.add(styleId); + if ( + style.getAttributeNS(WORD_NAMESPACE, "type") === "table" + ) { + for (const properties of directChildren( + style, + WORD_NAMESPACE, + "tblPr" + )) { + if ( + directChildren( + properties, + WORD_NAMESPACE, + "tblW" + ).length > 0 + ) { + throw new Error( + `word/styles.xml 的表格样式不得声明 w:tblW:${styleId}` + ); + } + } + } + } + for (const alignment of Array.from( + root.getElementsByTagNameNS(WORD_NAMESPACE, "jc") + )) { + if ( + alignment.getAttributeNS(WORD_NAMESPACE, "val") === + "justify" + ) { + throw new Error( + "word/styles.xml 的两端对齐必须使用 w:jc=\"both\"" + ); + } + } for (const styleId of requiredStyleIds) { if (!styleIds.has(styleId)) { throw new Error(`word/styles.xml 缺少关键样式:${styleId}`); @@ -297,7 +339,8 @@ function validateDocx( for (const [partName, part] of reference.entries) { if (partName.endsWith(".xml") || partName.endsWith(".rels")) { - parseXmlPart(part, partName); + const document = parseXmlPart(part, partName); + validateWordprocessingElementOrder(document, partName); xmlPartCount += 1; } if (partName.endsWith(".rels")) { diff --git a/packages/docx-engine/tests/acceptance-validator.test.ts b/packages/docx-engine/tests/acceptance-validator.test.ts index a665832..d55acb0 100644 --- a/packages/docx-engine/tests/acceptance-validator.test.ts +++ b/packages/docx-engine/tests/acceptance-validator.test.ts @@ -32,7 +32,14 @@ function xml(value: string) { return encoder.encode(value); } -function createAcceptanceDocx(overrides: { altChunk?: boolean } = {}) { +function createAcceptanceDocx( + overrides: { + altChunk?: boolean; + duplicateText?: boolean; + internalMarker?: boolean; + invalidPropertyOrder?: boolean; + } = {} +) { return zipSync({ "[Content_Types].xml": xml( '' @@ -41,7 +48,7 @@ function createAcceptanceDocx(overrides: { altChunk?: boolean } = {}) { `` ), "word/document.xml": xml( - `可编辑正文x原生表格${overrides.altChunk ? '' : ""}` + `可编辑正文x${overrides.invalidPropertyOrder ? '' : ""}${overrides.duplicateText ? "可编辑正文" : ""}${overrides.internalMarker ? "MD_TO_PDF_INTERNAL" : ""}原生表格${overrides.altChunk ? '' : ""}` ), "word/styles.xml": xml( `${["Normal", "Heading1", "SourceCode", "Table", "Caption"].map((id) => ``).join("")}` @@ -53,7 +60,7 @@ function createAcceptanceDocx(overrides: { altChunk?: boolean } = {}) { '' ), "word/footer1.xml": xml( - ` PAGE \\* MERGEFORMAT ` + ` PAGE \\* MERGEFORMAT SECTIONPAGES \\* MERGEFORMAT ` ), "word/footnotes.xml": xml( `脚注` @@ -93,7 +100,17 @@ const expectation = { minimumPngImages: 1, requiredImageAltText: ["普通图片"], requiredStyleIds: ["Normal", "Heading1", "SourceCode", "Table"], - requirePageField: true + requiredParagraphStyleIds: ["Heading1"], + requirePageField: true, + minimumSections: 2, + requireFirstSectionWithoutHeaderFooter: true, + finalPageNumberStart: 1, + requireSectionPagesField: true, + minimumFullWidthTables: 1, + maximumTextOccurrences: { + 可编辑正文: 1 + }, + forbidInternalMarkers: true }; describe("DOCX 自动验收器", () => { @@ -112,6 +129,10 @@ describe("DOCX 自动验收器", () => { mathObjectCount: 1, pngImageCount: 1, pageFieldCount: 1, + sectionPageFieldCount: 1, + sectionCount: 2, + fullWidthTableCount: 1, + internalMarkerCount: 0, altChunkCount: 0 }); expect(report.checks).toEqual( @@ -131,4 +152,28 @@ describe("DOCX 自动验收器", () => { ) ).toThrow("altChunk"); }); + + it("拒绝重复文本和未清理的内部标记", () => { + expect(() => + inspectDocxAcceptance( + createAcceptanceDocx({ duplicateText: true }), + expectation + ) + ).toThrow("最多允许 1 次"); + expect(() => + inspectDocxAcceptance( + createAcceptanceDocx({ internalMarker: true }), + expectation + ) + ).toThrow("internal-markers"); + }); + + it("拒绝 WordprocessingML 属性子节点乱序", () => { + expect(() => + inspectDocxAcceptance( + createAcceptanceDocx({ invalidPropertyOrder: true }), + expectation + ) + ).toThrow("w:pPr 子节点顺序无效"); + }); }); diff --git a/packages/docx-engine/tests/document-structure-transform.test.ts b/packages/docx-engine/tests/document-structure-transform.test.ts new file mode 100644 index 0000000..3e0124f --- /dev/null +++ b/packages/docx-engine/tests/document-structure-transform.test.ts @@ -0,0 +1,251 @@ +import { describe, expect, it } from "vitest"; +import type { DocxThemeTokenSet } from "@md-to-pdf/docx-theme-engine"; +import { + finalizeGeneratedDocxStructure, + readGeneratedDocxPackage, + readReferenceDocxPackage, + writeGeneratedDocxPackage, + type PandocStructurePlan +} from "../src/index.js"; +import { createTestBaselineReference } from "./reference-test-fixture.js"; + +const encoder = new TextEncoder(); +const decoder = new TextDecoder(); +const word = + "http://schemas.openxmlformats.org/wordprocessingml/2006/main"; +const relationships = + "http://schemas.openxmlformats.org/officeDocument/2006/relationships"; + +function generatedFixture() { + const baseline = readReferenceDocxPackage( + createTestBaselineReference() + ); + const entries = new Map(baseline.entries); + entries.set( + "word/document.xml", + encoder.encode( + `` + + `CONTAINER_START` + + `可编辑封面` + + `CONTAINER_END` + + `SECTION_BREAK` + + `正文标题` + + `AB` + + `` + + `` + ) + ); + entries.set( + "word/footer1.xml", + encoder.encode( + ` NUMPAGES \\* MERGEFORMAT ` + ) + ); + return writeGeneratedDocxPackage(entries); +} + +const plan: PandocStructurePlan = { + schemaVersion: 1, + titlePolicy: { + metadataTitle: "suppress", + firstBodyHeading: "keep" + }, + prefix: [ + { + kind: "container", + styleId: "MdTenderCover", + slot: "tender-cover", + startMarker: "CONTAINER_START", + endMarker: "CONTAINER_END", + blocks: [] + }, + { + kind: "section-break", + marker: "SECTION_BREAK", + headerFooter: "none", + pageNumber: "hidden", + followingPageNumberStart: 1, + verticalAlignment: "center" + } + ], + suffix: [] +}; + +const tokens: DocxThemeTokenSet = { + schemaVersion: 1, + themeId: "test-theme", + themeFingerprint: "c".repeat(64), + mode: "auto-with-overrides", + basePreset: "tender", + slots: [ + { + slot: "tender-cover", + source: "computed-css", + confidence: "approximate", + style: { + fontCandidates: [], + backgroundColor: "#f5f5f5", + pageBreakAfter: true, + keepLines: true, + borders: { + top: { + widthPt: 1, + style: "single", + color: "#111111" + }, + right: { + widthPt: 1, + style: "single", + color: "#111111" + }, + bottom: { + widthPt: 1, + style: "single", + color: "#111111" + }, + left: { + widthPt: 1, + style: "single", + color: "#111111" + } + } + } + }, + { + slot: "table", + source: "computed-css", + confidence: "exact", + style: { + fontCandidates: [], + widthPercent: 100, + keepLines: true + } + }, + { + slot: "heading-1", + source: "computed-css", + confidence: "exact", + style: { + fontCandidates: [], + pageBreakAfter: true + } + } + ], + diagnostics: [] +}; + +describe("生成 DOCX 结构收口", () => { + it("生成真实分节、正文页码、节总页数和固定宽度表格", () => { + const result = finalizeGeneratedDocxStructure( + generatedFixture(), + plan, + tokens + ); + const entries = readGeneratedDocxPackage( + result.content + ).entries; + const documentXml = decoder.decode( + entries.get("word/document.xml")! + ); + const footerXml = decoder.decode( + entries.get("word/footer1.xml")! + ); + + expect(documentXml).not.toContain("CONTAINER_START"); + expect(documentXml).not.toContain("CONTAINER_END"); + expect(documentXml).not.toContain("SECTION_BREAK"); + expect( + documentXml.match(/)/gu) + ).toHaveLength(2); + expect(documentXml).toContain( + '' + ); + expect(documentXml).toContain( + '' + ); + expect(documentXml).toContain( + '' + ); + expect(documentXml).toContain( + '' + ); + expect(documentXml).toContain( + '' + ); + expect(documentXml).toContain(''); + expect(documentXml).toContain(''); + expect(documentXml).toContain(""); + expect(documentXml).toContain('w:fill="F5F5F5"'); + expect( + documentXml.match(//gu) + ).toHaveLength(1); + expect(footerXml).toContain("SECTIONPAGES"); + expect(footerXml).not.toContain(" NUMPAGES "); + expect(result.report).toEqual({ + containerCount: 1, + sectionCount: 1, + tableCount: 1, + pageBreakAfterCount: 1, + sectionPageFieldCount: 1 + }); + }); + + it("拒绝缺失或重复的内部结构标记", () => { + expect(() => + finalizeGeneratedDocxStructure( + generatedFixture(), + { + ...plan, + prefix: [ + { + ...plan.prefix[0]!, + startMarker: "MISSING" + }, + plan.prefix[1]! + ] + }, + tokens + ) + ).toThrow("结构标记数量无效"); + }); + + it("清理 Pandoc 追加的重复和降级样式", () => { + const source = readGeneratedDocxPackage( + generatedFixture() + ); + const entries = new Map(source.entries); + const styles = decoder + .decode(entries.get("word/styles.xml")!) + .replace( + "", + '' + + '' + + '' + + '' + + '' + + '' + + "" + ); + entries.set("word/styles.xml", encoder.encode(styles)); + + const result = finalizeGeneratedDocxStructure( + writeGeneratedDocxPackage(entries), + plan, + tokens + ); + const outputStyles = decoder.decode( + readGeneratedDocxPackage(result.content).entries.get( + "word/styles.xml" + )! + ); + + expect( + outputStyles.match(/w:styleId="MdTenderTitle"/gu) + ).toHaveLength(1); + expect(outputStyles).toContain( + '' + ); + expect(outputStyles).not.toContain('w:val="justify"'); + expect(outputStyles).not.toContain("` + + `${children}` + + "" + ), + "word/document.xml" + ); +} + +describe("OOXML 子节点顺序", () => { + it("按 WordprocessingML 顺序插入新增属性", () => { + const document = paragraphPropertiesDocument( + "" + ); + const properties = document.getElementsByTagNameNS( + WORD_NAMESPACE, + "pPr" + )[0]!; + + appendElement(properties, WORD_NAMESPACE, "w:pBdr"); + appendElement(properties, WORD_NAMESPACE, "w:keepNext"); + appendElement(properties, WORD_NAMESPACE, "w:spacing"); + + const xml = decoder.decode(serializeXmlPart(document)); + expect(xml.indexOf(" + validateWordprocessingElementOrder( + document, + "word/document.xml" + ) + ).not.toThrow(); + }); + + it("在序列化时规范化已有乱序属性", () => { + const document = paragraphPropertiesDocument( + "" + + "" + ); + + const serialized = serializeXmlPart(document); + const reparsed = parseXmlPart( + serialized, + "word/document.xml" + ); + expect(() => + validateWordprocessingElementOrder( + reparsed, + "word/document.xml" + ) + ).not.toThrow(); + + const xml = decoder.decode(serialized); + expect(xml.indexOf(" { + const document = paragraphPropertiesDocument( + "" + + "" + ); + + expect(() => + validateWordprocessingElementOrder( + document, + "word/document.xml" + ) + ).toThrow( + "word/document.xml 的 w:pPr 子节点顺序无效" + ); + }); + + it("规范化表格、单元格和边框顺序", () => { + const document = parseXmlPart( + encoder.encode( + `` + + "" + + "" + + "" + + "" + + "" + + "" + + "" + ), + "word/document.xml" + ); + + const serialized = serializeXmlPart(document); + const reparsed = parseXmlPart( + serialized, + "word/document.xml" + ); + expect(() => + validateWordprocessingElementOrder( + reparsed, + "word/document.xml" + ) + ).not.toThrow(); + }); + + it("规范化表格条件样式的属性顺序", () => { + const document = parseXmlPart( + encoder.encode( + `` + + '' + + '' + + "" + + "" + ), + "word/styles.xml" + ); + + const xml = decoder.decode(serializeXmlPart(document)); + expect(xml.indexOf(" { expect( options.env?.MD_TO_PDF_DOCX_MEDIA_MAP ).toContain("media-map.json"); + expect( + options.env?.MD_TO_PDF_DOCX_STRUCTURE_PLAN + ).toContain("structure-plan.json"); + const structurePlan = JSON.parse( + await import("node:fs/promises").then(({ readFile }) => + readFile( + options.env!.MD_TO_PDF_DOCX_STRUCTURE_PLAN!, + "utf8" + ) + ) + ); + expect(structurePlan.titlePolicy.metadataTitle).toBe( + "suppress" + ); await copyFile( argumentAfter(arguments_, "--reference-doc"), argumentAfter(arguments_, "--output") diff --git a/packages/docx-engine/tests/pandoc-structure.test.ts b/packages/docx-engine/tests/pandoc-structure.test.ts new file mode 100644 index 0000000..86d4d80 --- /dev/null +++ b/packages/docx-engine/tests/pandoc-structure.test.ts @@ -0,0 +1,239 @@ +import { describe, expect, it } from "vitest"; +import type { + SemanticDocumentModel, + SemanticDocumentNode +} from "@md-to-pdf/core"; +import type { + DocxResolvedStyleSlot, + DocxStyleSlotName, + DocxThemeTokenSet +} from "@md-to-pdf/docx-theme-engine"; +import { + createPandocStructurePlan, + type PandocStructureBlock +} from "../src/index.js"; + +function slot( + name: DocxStyleSlotName, + hidden = false +): DocxResolvedStyleSlot { + return { + slot: name, + source: "computed-css", + confidence: "exact", + style: { + fontCandidates: [], + ...(hidden ? { hidden: true } : {}) + } + }; +} + +function tokens( + slots: DocxResolvedStyleSlot[] = [] +): DocxThemeTokenSet { + return { + schemaVersion: 1, + themeId: "test-theme", + themeFingerprint: "c".repeat(64), + mode: "auto-with-overrides", + basePreset: "technical", + slots, + diagnostics: [] + }; +} + +function text( + role: Extract["role"], + value: string, + label?: string +): SemanticDocumentNode { + return { + kind: "text", + role, + text: value, + ...(label ? { label } : {}) + }; +} + +function paragraphSegments(block: PandocStructureBlock): string[] { + if (block.kind === "paragraph") { + return block.segments; + } + if (block.kind === "container") { + return block.blocks.flatMap(paragraphSegments); + } + return []; +} + +describe("Pandoc 语义结构投影", () => { + it("将公文分组投影为固定 Word 样式和可编辑段落", () => { + const model: SemanticDocumentModel = { + schemaVersion: 1, + profile: "official", + titlePolicy: { + metadataTitle: "suppress", + firstBodyHeading: "suppress" + }, + regions: [ + { + kind: "prefix", + nodes: [ + { + kind: "group", + role: "official-masthead", + children: [ + { + kind: "group", + role: "official-classification", + children: [ + text("official-secrecy", "秘密"), + text("official-urgency", "特急") + ] + }, + text("official-issuer", "示例单位"), + { + kind: "group", + role: "official-issue-row", + children: [ + text("official-number", "示例〔2026〕1号"), + text( + "official-signatory", + "张三", + "签发人:" + ) + ] + } + ] + }, + text("official-title", "关于开展工作的通知") + ] + } + ] + }; + + const plan = createPandocStructurePlan(model, tokens(), { + markerSeed: "official" + }); + + expect(plan.titlePolicy.firstBodyHeading).toBe("suppress"); + expect(plan.prefix[0]).toMatchObject({ + kind: "container", + styleId: "MdOfficialMasthead", + slot: "official-masthead", + startMarker: + "MD_TO_PDF_CONTAINER_START_official_1", + endMarker: "MD_TO_PDF_CONTAINER_END_official_2", + blocks: [ + { + kind: "paragraph", + styleId: "MdOfficialClassification", + segments: ["秘密", "特急"], + separator: "tab" + }, + { + kind: "paragraph", + styleId: "MdOfficialIssuer", + segments: ["示例单位"] + }, + { + kind: "paragraph", + styleId: "MdOfficialIssueRow", + segments: ["示例〔2026〕1号", "签发人:张三"], + separator: "tab" + } + ] + }); + expect(plan.prefix[1]).toMatchObject({ + kind: "paragraph", + styleId: "MdOfficialTitle", + segments: ["关于开展工作的通知"] + }); + }); + + it("按主题隐藏令牌过滤标书字段并保留封面分页", () => { + const model: SemanticDocumentModel = { + schemaVersion: 1, + profile: "tender", + titlePolicy: { + metadataTitle: "suppress", + firstBodyHeading: "suppress" + }, + regions: [ + { + kind: "cover", + section: { + headerFooter: "none", + pageNumber: "hidden", + breakAfter: "next-page", + followingPageNumberStart: 1 + }, + nodes: [ + { + kind: "group", + role: "tender-cover", + children: [ + text("tender-title", "投标文件"), + text("tender-bidder", "投标单位"), + text( + "tender-representative", + "授权代表" + ), + text("tender-date", "2026年7月") + ] + } + ] + } + ] + }; + + const plan = createPandocStructurePlan( + model, + tokens([ + slot("tender-bidder", true), + slot("tender-representative", true) + ]), + { markerSeed: "tender" } + ); + + expect(plan.prefix.at(-1)).toEqual({ + kind: "section-break", + marker: "MD_TO_PDF_SECTION_BREAK_tender_3", + headerFooter: "none", + pageNumber: "hidden", + followingPageNumberStart: 1 + }); + expect(plan.prefix[0]).toMatchObject({ + kind: "container", + styleId: "MdTenderCover" + }); + expect(paragraphSegments(plan.prefix[0]!)).toEqual([ + "投标文件", + "2026年7月" + ]); + }); + + it("没有结构区域时只传递标题策略", () => { + const plan = createPandocStructurePlan( + { + schemaVersion: 1, + titlePolicy: { + metadataTitle: "emit", + firstBodyHeading: "keep" + }, + regions: [] + }, + tokens(), + { markerSeed: "empty" } + ); + + expect(plan).toEqual({ + schemaVersion: 1, + titlePolicy: { + metadataTitle: "emit", + firstBodyHeading: "keep" + }, + prefix: [], + suffix: [] + }); + }); +}); diff --git a/packages/docx-engine/tests/reference-builder.test.ts b/packages/docx-engine/tests/reference-builder.test.ts index 1acebdd..daba5a9 100644 --- a/packages/docx-engine/tests/reference-builder.test.ts +++ b/packages/docx-engine/tests/reference-builder.test.ts @@ -229,7 +229,8 @@ describe("动态 reference.docx", () => { fontSizePt: 14, color: "#112233", lineSpacing: 1.8, - firstLineIndentPt: 28 + firstLineIndentPt: 28, + alignment: "justify" } }, { @@ -297,7 +298,9 @@ describe("动态 reference.docx", () => { expect(stylesXml).toContain('w:val="112233"'); expect(stylesXml).toContain('w:styleId="Heading1"'); expect(stylesXml).toContain('w:val="AA0000"'); - expect(stylesXml).toContain('w:w="5000" w:type="pct"'); + expect(stylesXml).toContain(''); + expect(stylesXml).not.toContain('w:val="justify"'); + expect(stylesXml).not.toContain(" { ); }); + it("拒绝重复样式 ID 和非法两端对齐枚举", () => { + const result = createDynamicReferenceDocx( + createBaselineReference(), + createOptions(defaultExportConfig) + ); + const entries = new Map( + Object.entries(unzipSync(result.content)) + ); + const styles = decoder.decode(entries.get("word/styles.xml")!); + + entries.set( + "word/styles.xml", + encoder.encode( + styles.replace( + "", + '' + + "" + ) + ) + ); + expect(() => + validateDynamicReferenceDocx( + writeReferenceDocxPackage(entries) + ) + ).toThrow(/重复样式 ID:Normal/u); + + entries.set( + "word/styles.xml", + encoder.encode( + styles.replace( + //u, + '' + ) + ) + ); + expect(() => + validateDynamicReferenceDocx( + writeReferenceDocxPackage(entries) + ) + ).toThrow(/两端对齐必须使用/u); + }); + it("最终 DOCX 使用媒体输出上限而非模板的 2 MiB 上限", () => { const result = createDynamicReferenceDocx( createBaselineReference(), diff --git a/packages/docx-theme-engine/src/normalizer.ts b/packages/docx-theme-engine/src/normalizer.ts index 9980803..3c767da 100644 --- a/packages/docx-theme-engine/src/normalizer.ts +++ b/packages/docx-theme-engine/src/normalizer.ts @@ -46,6 +46,35 @@ interface SlotNormalizationContext { approximate: boolean; } +function cssLengthToPx(value: string): number { + const valuePt = parseCssLengthToPt(value); + return valuePt === undefined ? 0 : valuePt / 0.75; +} + +function resolveDocumentContentWidthPx( + snapshot: DocxThemeStyleSnapshot +): number | undefined { + const computed = snapshot.slots.find( + (entry) => entry.slot === "document" + )?.computed; + if (!computed) { + return undefined; + } + const widthPt = parseCssLengthToPt(computed.width); + if (widthPt === undefined) { + return undefined; + } + const borderWidthPx = [computed.borderLeft, computed.borderRight] + .map((value) => parseCssBorder(value)?.widthPt ?? 0) + .reduce((total, valuePt) => total + valuePt / 0.75, 0); + const contentWidthPx = + widthPt / 0.75 - + cssLengthToPx(computed.paddingLeft) - + cssLengthToPx(computed.paddingRight) - + borderWidthPx; + return contentWidthPx > 0 ? contentWidthPx : widthPt / 0.75; +} + const paragraphKinds = new Set([ "document", "paragraph", @@ -719,14 +748,7 @@ export function resolveDocxThemeTokens( ): DocxThemeTokenSet { const snapshot = docxThemeStyleSnapshotSchema.parse(input.snapshot); const diagnostics: DocxThemeDiagnostic[] = []; - const documentWidth = snapshot.slots.find( - (entry) => entry.slot === "document" - )?.computed?.width; - const documentWidthPt = documentWidth - ? parseCssLengthToPt(documentWidth) - : undefined; - const documentWidthPx = - documentWidthPt === undefined ? undefined : documentWidthPt / 0.75; + const documentWidthPx = resolveDocumentContentWidthPx(snapshot); const snapshotBySlot = new Map( snapshot.slots.map((entry) => [entry.slot, entry]) ); diff --git a/packages/docx-theme-engine/tests/normalizer.test.ts b/packages/docx-theme-engine/tests/normalizer.test.ts index 23460b2..ddbbec3 100644 --- a/packages/docx-theme-engine/tests/normalizer.test.ts +++ b/packages/docx-theme-engine/tests/normalizer.test.ts @@ -178,6 +178,31 @@ describe("DOCX 主题令牌归一化", () => { expect(tokens.slots).toHaveLength(DOCX_STYLE_SLOT_NAMES.length); }); + it("按 border-box 文档的内容区宽度归一化全宽表格", () => { + const tokens = resolveDocxThemeTokens({ + snapshot: createSnapshot({ + document: { + width: "790px", + paddingLeft: "30px", + paddingRight: "30px", + borderLeft: "0px none rgb(17, 34, 51)", + borderRight: "0px none rgb(17, 34, 51)" + }, + table: { + width: "730px" + } + }), + config: { + mode: "auto", + basePreset: "general", + overrides: {}, + legacyPreset: false + } + }); + + expect(findSlot(tokens, "table").style.widthPercent).toBe(100); + }); + it("保留封面高度、垂直对齐、轮廓线和隐藏字段语义", () => { const tokens = resolveDocxThemeTokens({ snapshot: createSnapshot({