From fa4e088b35425377691e1a6d21a061f3e880ddd4 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 14:15:18 +0800 Subject: [PATCH 01/12] =?UTF-8?q?=E5=BB=BA=E7=AB=8B=20XLSX=20=E7=9A=84?= =?UTF-8?q?=E5=AE=89=E5=85=A8=E8=BA=AB=E4=BB=BD=E4=B8=8E=E8=A7=A3=E6=9E=90?= =?UTF-8?q?=E5=85=A5=E5=8F=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 先锁定 XLSX 的容器身份、包关系安全和默认 registry seam,为后续原生内容解析提供有界入口。 Constraint: 关系 XML 使用 defusedxml;不改变 DOCX/PPTX 包预算与公共 Markdown 返回契约。 Tested: 154 focused tests;Ruff check/format;ty check src tests;uv sync --all-groups --frozen;git diff --check。 --- .../2026-08-07-001-feat-xlsx-parsing-plan.md | 651 ++++++++++++++++++ pyproject.toml | 2 + src/opendocs/_models.py | 1 + src/opendocs/detection.py | 25 +- src/opendocs/parsers/office/package.py | 63 +- src/opendocs/parsers/registry.py | 2 + src/opendocs/parsers/xlsx/__init__.py | 25 + tests/test_detection.py | 83 ++- tests/test_models.py | 1 + tests/test_office_package.py | 159 ++++- tests/test_registry.py | 41 +- tests/xlsx_fixtures.py | 99 +++ uv.lock | 34 + 13 files changed, 1169 insertions(+), 17 deletions(-) create mode 100644 docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md create mode 100644 src/opendocs/parsers/xlsx/__init__.py create mode 100644 tests/xlsx_fixtures.py diff --git a/docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md b/docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md new file mode 100644 index 0000000..eb81947 --- /dev/null +++ b/docs/plans/2026-08-07-001-feat-xlsx-parsing-plan.md @@ -0,0 +1,651 @@ +--- +title: XLSX 解析 - Plan +type: feat +date: 2026-08-07 +topic: xlsx-parsing +artifact_contract: ce-unified-plan/v1 +artifact_readiness: implementation-ready +product_contract_source: ce-brainstorm +execution: code +--- + +# XLSX 解析 - Plan + +## Goal Capsule + +- **Objective:** 为 OpenDocs v0.2.0 交付 XLSX 到 Markdown 的文本语义保真解析,并让视觉增强失败时仍可获得确定性的原生结果。 +- **Authority order:** Product Contract 定义产品行为,Planning Contract 定义实现方式,Implementation Units 不得覆盖前两者。本计划仅在 XLSX 范围内取代 `docs/plans/2026-08-07-v0.2.0-release-plan.md` 的 A4 视觉禁用约束;v0.2.0 的其他工作仍由父计划管理。 +- **Execution profile:** 代码实施,测试先行;先锁定容器、wire 和输出契约,再接入第三方解析与视觉增强。 +- **Stop conditions:** 若实现需要新增公共返回类型、把工作表解释为页面、引入 Excel/LibreOffice 运行时,或无法在预检阶段约束峰值资源,则停止并回到计划评审。 +- **Tail ownership:** 实施需完成公开测试、静态检查、构建和独立 wheel smoke;真实 XLSX 只进入忽略的私有探索流程。 +- **Open blockers:** 无发布阻塞型规划问题。 + +--- + +## Product Contract + +### Summary + +OpenDocs 将从 path、bytes 和 binary stream 解析标准 XLSX 工作簿,并通过 `parse()` 与 `aparse()` 返回确定性的 Markdown。 +原生解析是工作表、文本、显示值、合并关系和图表数值的事实来源,视觉模型只补充图片与图表的视觉含义。 + +### Problem Frame + +OpenDocs 当前没有 XLSX 文档类型、检测、注册或解析能力,因此 Excel 工作簿仍是明确的不支持格式。 +XLSX 不只是二维单元格集合:一份工作簿还可能包含多个可见或隐藏工作表、合并与稀疏区域、公式缓存、数字格式、批注、文本框、超链接、页眉页脚、图片、图表、外部关系和声明范围异常。 +没有真实工作簿可以作为当前基线,因此本任务需要先用合成样本锁定公开行为,再以维护者提供的私有真实工作簿做发布前探索性验证。 + +### Risk Map + +| 风险 | 可能造成的问题 | 契约响应 | +| --- | --- | --- | +| 公式缓存缺失或过期 | 输出为空或与重新计算结果不同 | 由 R5 和公式不重算边界约束 | +| 货币、日期和自定义格式 | 原始数值可读但用户看到的文本错误 | 由 R5-R6 约束 | +| 隐藏、空白和多工作表 | 内容被遗漏或顺序改变 | 由 R3 约束 | +| 合并单元格和相离区域 | 表格关系丢失或被错误拼接 | 由 R4 约束 | +| 稀疏但声明范围巨大的工作表 | 无界时间或内存消耗 | 由 R11-R12 约束 | +| 浮动图片、图表和文本框 | 内容脱离原始工作表位置 | 由 R4、R8-R10 约束 | +| 外部链接和数据关系 | 发生未授权网络访问或结果不可复现 | 由 R7 和 Scope Boundaries 约束 | +| 恶意或异常 OOXML 容器 | ZIP bomb、路径逃逸或不受控解压 | 由 R2、R11-R12 约束 | +| 厂商扩展和嵌入对象 | 静默遗漏或模型猜测内容 | 由 R14 约束 | + +### Key Decisions + +- **Markdown 语义保真。** (session-settled: user-directed — chosen over pixel-level fidelity or dual output: the SDK keeps its Markdown result contract and only text-bearing semantics are required.) Governs R1, R4-R7. +- **保存的显示值优先。** (session-settled: user-directed — chosen over always emitting formulas or attaching every formula: readable workbook text is the primary result.) Governs R5. +- **全部工作表进入结果。** (session-settled: user-directed — chosen over skipping hidden sheets or adding an inclusion option: sheet content must not disappear because of visibility state.) Governs R3. +- **全部标准文本对象属于核心内容。** (session-settled: user-directed — chosen over cell-only or content-only extraction: text outside cells is still document content.) Governs R7. +- **图表原生数据优先。** (session-settled: user-directed — chosen over vision-only or native-only chart handling: numeric accuracy and visual meaning are both useful, but native values remain authoritative.) Governs R8-R9. +- **视觉增强失败不阻断原生结果。** (session-settled: user-directed — chosen over failing the document or adding a strict mode: deterministic native content remains useful without a model.) Governs R9-R10. +- **可靠文本核心作为发布硬门。** (session-settled: user-directed — chosen over broad text completeness or full-sheet visual review: the first XLSX release needs a bounded contract without silently losing supported content.) Governs R1-R15. +- **真实工作簿是探索性验证。** (session-settled: user-directed — chosen over a mandatory release gate or post-release-only validation: real evidence should inform the release without making every rare gap blocking.) Governs R15. + + +### How This Work Fits Together + +本计划只负责 v0.2.0 的 XLSX 解析产品契约;下面是当前理解的相邻工作关系,不构成已承诺路线图。 + +- **XLSX 资源边界** — Shares 本计划的安全与有界处理要求;具体阈值在本任务的技术规划中确定。 + - **跨格式 `max_pages` 语义** — Can proceed independently of XLSX 内容解析;不得把工作表伪装成页面。 +- **取消清理与对抗性 PDF 回归** — Can proceed independently of 本计划,仅共享 v0.2.0 发布门。 +- **v0.2.0 发布准备** — Depends on 本计划完成公开文档、构建产物和独立安装 smoke 的 XLSX 覆盖。 +- **Windows 探索性 smoke** — Can proceed independently of 本计划,且不是 XLSX 产品契约的一部分。 + +### Actors + +- A1. SDK 使用者向 OpenDocs 提交 XLSX,并消费 Markdown 与 warning。 +- A2. 视觉模型提供方在已配置且可用时补充图片和图表含义,不拥有原生数值的解释权。 +- A3. 维护者在发布前审阅私有真实工作簿的探索性结果,并判断发现是否违反核心契约。 + +### Requirements + +**格式识别与公共契约** + +- R1. OpenDocs 必须从 path、bytes 和 binary stream 接受合法 XLSX,并让 `parse()` 与 `aparse()` 返回相同契约的确定性 Markdown。 +- R2. OpenDocs 必须在无界处理前拒绝损坏、加密、伪装或超过资源边界的工作簿,并沿用现有类型化错误语义。 + +**工作簿结构与文本语义** + +- R3. 输出必须按源顺序识别全部工作表,以稳定标题标明工作表名称及 Visible、Hidden 或 Very Hidden 状态,空工作表不得使其他工作表失败。 +- R4. 每个工作表必须按行列顺序输出全部非空区域,保留合并跨度,并为区域与浮动对象提供稳定锚点;只有源工作簿提供表格语义时才能标记表头,不得仅凭首行位置猜测。 +- R5. 单元格必须输出保存的显示文本及常见货币、百分比、日期时间、千分位和小数位语义;公式缓存缺失时输出公式文本,不支持的自定义格式输出可读值并产生 warning。 +- R6. 解析器不得仅因样式存在而输出空单元格,也不得把字体、颜色、边框、尺寸或条件格式外观解释为必须还原的内容。 +- R7. 标准批注、备注、文本框、图表文字、超链接文字与 URL、页眉和页脚必须进入结果;外部 URL 与数据关系只保留引用,不发起访问。 + +**图表与视觉内容** + +- R8. 图表必须优先原生输出可确定获取的标题、分类、系列和数值,并保留其工作表位置。 +- R9. 视觉模型已配置时,内嵌图片必须进入视觉解析,需要解释趋势、标注或含义的图表必须获得视觉补充,但视觉结果不得覆盖或改写原生文本和数值。 +- R10. 视觉模型未配置、超时或失败时必须返回已完成的原生结果,并为每个未解析对象产生包含工作表和位置的 warning。 + +**安全、有界处理与确定性** + +- R11. 解析必须限制工作表数量、声明维度、访问与非空单元格数量、合并区域数量、对象与媒体数量、媒体大小及总输出字符数。 +- R12. 稀疏但声明范围巨大的工作表不得触发无界遍历,超限必须在昂贵解析或视觉调用前失败。 +- R13. 相同工作簿在等价输入形式和同步、异步 API 下必须保持工作表顺序、内容顺序、warning 分类和失败类型一致。 +- R14. 对无法可靠理解的厂商扩展、复杂绘图、SmartArt 或嵌入对象必须确定性跳过并产生可定位 warning,禁止静默丢失或猜测内容。 + +**验收证据** + +- R15. 自动化发布门必须使用合成 XLSX 覆盖全部核心契约;私有真实工作簿作为发布前探索性验证,仅当发现违反 R1-R14 的核心缺陷时阻断发布,罕见结构和视觉增强缺口可以记录后发布。 + +### Key Flows + +```mermaid +flowchart TB + A[XLSX input] --> B{Safe and within limits?} + B -->|no| C[Typed failure] + B -->|yes| D[Native sheets and text] + D --> E[Native chart data] + D --> F[Images and visual chart regions] + F --> G{Vision available?} + G -->|yes| H[Visual enrichment] + G -->|no or failed| I[Anchored warnings] + E --> J[Deterministic Markdown] + H --> J + I --> J +``` + +- F1. 正常解析 + - **Trigger:** A1 提交合法且未超限的 XLSX。 + - **Actors:** A1、A2。 + - **Steps:** 按 R1-R9 验证工作簿、提取全部工作表与原生内容、补充可用的视觉结果,并按源顺序合并。 + - **Outcome:** 返回确定性 Markdown 和必要 warning。 + - **Covers:** R1-R9、R11、R13。 +- F2. 视觉能力不可用 + - **Trigger:** 工作簿包含图片或需要视觉补充的图表,但模型未配置、超时或失败。 + - **Actors:** A1、A2。 + - **Steps:** 保留原生结果并按 R10 标记每个未完成的视觉对象。 + - **Outcome:** 解析成功返回,调用方可以定位缺失的视觉补充。 + - **Covers:** R9-R10、R13-R14。 +- F3. 非法或超限工作簿 + - **Trigger:** 输入损坏、加密、伪装或触发任一资源边界。 + - **Actors:** A1。 + - **Steps:** 在无界遍历和视觉调用前终止处理。 + - **Outcome:** 返回稳定的类型化错误,不返回误导性的部分成功结果。 + - **Covers:** R2、R11-R12。 +- F4. 私有真实工作簿探索 + - **Trigger:** A3 在发布前提供并审阅一份真实 XLSX 及其 Markdown。 + - **Actors:** A3。 + - **Steps:** 判断发现是否违反核心要求,并将非核心缺口记录为后续候选。 + - **Outcome:** 核心缺陷阻断发布,罕见结构或视觉增强缺口不自动阻断。 + - **Covers:** R15。 + +### Acceptance Examples + +- AE1. 全部工作表与状态 + - **Covers R3-R4.** + - **Given:** 工作簿依次包含可见、Hidden、Very Hidden 和空工作表。 + - **When:** 通过任一受支持输入形式解析。 + - **Then:** Markdown 按相同顺序输出四个工作表标题并标注状态,空表不影响其他内容。 +- AE2. 公式与显示格式 + - **Covers R5-R6.** + - **Given:** 单元格包含货币、百分比、日期、带缓存的公式、无缓存公式和仅样式空单元格。 + - **When:** 解析工作表。 + - **Then:** 输出保存的可读文本,缺少缓存的公式降级为公式文本,仅样式空单元格不制造内容。 +- AE3. 合并与相离区域 + - **Covers R4.** + - **Given:** 工作表包含横向和纵向合并单元格,以及被全空行列分隔的多个非空区域。 + - **When:** 解析工作表。 + - **Then:** 合并跨度和所有非空内容均保留,各区域以确定的行列顺序出现并可定位。 +- AE4. 工作表外围文本与链接 + - **Covers R7、R14.** + - **Given:** 工作表包含批注、备注、文本框、图表标签、超链接和页眉页脚,并引用外部 URL。 + - **When:** 解析工作簿。 + - **Then:** 可支持的文本和 URL 进入结果,OpenDocs 不访问外部地址,无法支持的对象产生可定位 warning。 +- AE5. 图表原生与视觉结果 + - **Covers R8-R10.** + - **Given:** 图表具有标题、分类、系列、数值和可由视觉模型识别的趋势标注。 + - **When:** 原生与视觉解析均成功。 + - **Then:** 原生数据按源值输出,视觉结果只补充趋势与含义,并保留图表锚点。 +- AE6. 视觉失败降级 + - **Covers R9-R10.** + - **Given:** 工作簿包含图片和图表,但视觉模型不可用或调用失败。 + - **When:** 解析工作簿。 + - **Then:** 原生文本和图表数据正常返回,每个未解析视觉对象都有工作表与位置 warning。 +- AE7. 对抗性和稀疏工作簿 + - **Covers R2、R11-R12.** + - **Given:** 工作簿包含异常容器关系、超量媒体或极大的声明范围但只有少量非空单元格。 + - **When:** 解析工作簿。 + - **Then:** 在无界遍历或视觉调用前以稳定类型化错误终止。 +- AE8. 等价 API 行为 + - **Covers R1、R13.** + - **Given:** 同一合成 XLSX 以 path、bytes 和 binary stream 输入。 + - **When:** 分别调用 `parse()` 与 `aparse()`。 + - **Then:** Markdown、warning 分类和失败类型满足等价契约。 +- AE9. 私有探索性验证 + - **Covers R15.** + - **Given:** 维护者提供真实 XLSX,并人工对照源文件审阅结果。 + - **When:** 发现差异。 + - **Then:** 违反核心要求的差异阻断发布,罕见结构或视觉增强缺口被记录但不自动阻断。 + +### Success Criteria + +- R1-R15 均有自动化合成样本或确定性检查覆盖,且现有支持格式没有输出顺序回归。 +- 每一次内容降级都能由返回结果或 warning 观察,不能以“解析成功”掩盖静默丢失。 +- 维护者可以只依赖本 Product Contract 判断一个真实工作簿发现属于核心缺陷还是非阻断增强缺口。 + +### Scope Boundaries + +- 不还原字体、字号、颜色、背景、边框、列宽、行高、条件格式外观或像素级版式。 +- 不访问外部 URL、链接工作簿、外部数据连接或远程资源。 +- 不支持旧版 `.xls`、宏工作簿 `.xlsm`、二进制 `.xlsb` 或其他表格格式。 +- 不重新计算公式,也不承诺缓存值反映工作簿最后保存之后的外部变化。 +- 不把整张工作表的视觉解析设为默认结果或发布硬门。 +- 不在本计划中改变结构化返回类型、依赖 extras、全局并发、CLI、Node.js SDK 或跨语言 schema。 +- 不把跨格式 `max_pages`、DOCX 结构上限、取消清理、PDF 防护或 Windows 支持纳入 XLSX 主动范围。 + +### Dependencies and Assumptions + +- 当前公共 API 的成功结果是 Markdown 字符串,XLSX 必须遵循相同契约。 +- 当前 Office 安全层和视觉管线可以提供复用基础,但 XLSX 仍需自己的容器识别、关系约束和资源边界。 +- 公式显示值来自工作簿保存的缓存,OpenDocs 不承担电子表格计算引擎职责。 +- 视觉模型是可选增强依赖,任何视觉结论都不能成为原生数值的替代来源。 +- 当前没有真实 XLSX 基线;维护者将在发布前提供私有工作簿做探索性审阅。 + +### Sources and Research + +- `docs/plans/2026-08-07-v0.2.0-release-plan.md`:v0.2.0 总体边界、XLSX 候选范围与相邻工作。 +- `docs/roadmap.md`:已发布能力和 v0.2.0 在路线图中的位置。 +- `src/opendocs/api.py`:当前 Markdown 返回契约与同步、异步共享路径。 +- `src/opendocs/parsers/office/package.py`:现有 Office 容器安全边界及 XLSX 需要扩展的基础。 +- `src/opendocs/markdown.py`:合并单元格可使用现有跨度表格语义表达。 +- `src/opendocs/parsers/office/parser.py` 与 `src/opendocs/parsers/office/pptx.py`:现有 Office 视觉合并和图表原生数据行为。 + +--- + +## Planning Contract + +Product Contract unchanged. + +### Key Technical Decisions + +- KTD1. **使用独立 XLSX 解析域和混合读取路径。** 新增私有 `XlsxParser`、XLSX 文档模型和严格 wire;不把工作表塞入基于页面的 `OfficeDocument`。容器与补充对象由受限 OOXML 流式读取,单元格、合并和 Excel 表格由 `openpyxl` 完整模式读取。该方案复用公共 parser/runtime/vision 契约,但不扩大 DOCX/PPTX 的模型。Governs R1-R4, R7-R14. (session-settled: user-approved — chosen over extending the page-oriented OfficeParser or using only one parser: XLSX needs sheet semantics plus low-level coverage that neither alternative provides.) +- KTD2. **先预检,后加载工作簿。** `openpyxl.load_workbook()` 之前必须完成 ZIP 成员、关系、XML 安全、工作表、维度、单元格、合并、字符串、对象和媒体预算检查。XLSX 扩展现有 Office 包验证白名单,但不绕过已有 2,048 成员、32 MiB 声明总量、4 MiB 单 XML、256 个媒体、16 MiB 单媒体、24 MiB 媒体总量和 100:1 压缩比限制。直接 OOXML 解析显式使用 `defusedxml` 并拒绝 DTD/实体;ZIP bomb 仍由包级预算处理。Governs R2, R11-R12. (session-settled: user-approved — chosen over loader-first validation: malformed dimensions and XML can consume resources before a high-level library returns control.) +- KTD3. **固定依赖为 `openpyxl>=3.1.5,<3.2` 与 `defusedxml>=0.7.1,<1`。** 使用 `read_only=False`、`data_only=False`、`rich_text=False`、`keep_links=False`。公式工作簿只加载一次;直接 OOXML sidecar 同时保留 `` 和保存的 `` 缓存,避免第二次完整加载。禁止依赖 `openpyxl` 私有 `_charts` 或 `_images` 作为核心事实来源。Governs R2, R5, R7-R9, R11. (session-settled: user-approved — chosen over read-only loading or dependency-free OOXML reimplementation: read-only omits required objects, while a full spreadsheet reader is beyond v0.2.0.) +- KTD4. **“保存的显示文本”采用有界格式子集。** 原生事实是保存的标量或公式缓存,解析器只对 `General`、布尔/错误值、整数、定点小数、千分位、百分比、常用直接货币符号、日期、时间、日期时间和 elapsed-time 执行确定性格式化,并尊重 1900/1904 日期系统。颜色、填充、条件段、会计占位、科学计数、分数和复杂本地化自定义格式不进入 v0.2.0 保真承诺;它们输出稳定原始值并产生 `xlsx_unsupported_number_format`。Governs R5-R6. +- KTD5. **公式缓存优先,缺失时回退公式。** 缓存节点存在时按 KTD4 输出;节点不存在时输出公式文本并产生 `xlsx_formula_cache_missing`。解析器不计算公式、不判断缓存是否过期,也不访问外部工作簿。Governs R5, R7. +- KTD6. **按语义坐标生成区域和对象槽位。** Excel 原生表格范围先占位,且只有它设置 `header_rows=1`。其余非空语义单元格与合并矩形按上下左右连通分量拆分,按左上角、右下角排序;无跨度使用 `TableBlock(header_rows=0)`,有跨度使用 `SpannedTableBlock`。每个 sheet、区域和浮动对象使用仅含工作表序号、A1 范围和对象序号的受控 Markdown 注释锚点,禁止把用户文本写入注释。Governs R3-R4, R13. +- KTD7. **所有 sheet-like 条目按工作簿关系顺序处理。** `workbook.xml` 与关系文件是 worksheet、chartsheet、名称、状态和目标 part 的权威来源;不依赖 `wb.worksheets` 推断全量顺序。空 sheet 仍输出标题。页眉页脚使用 odd/even/first、header/footer、left/center/right 的固定次序;普通对象按 `(row, column, kind_rank, source_ordinal)` 插入。Governs R3-R4, R7-R8, R13. +- KTD8. **文本对象走低层 OOXML 补充读取。** 经典 comments/notes、threaded comments 与 person 映射、DrawingML 文本框、图表标题/轴/数据标签、超链接显示文本与目标、页眉页脚均进入原生槽位。SmartArt、OLE、控件、VML 绘图文本和厂商扩展无法可靠读取时,按每个对象产生 `xlsx_unsupported_object`。URL 只允许安全转义后的 `http`、`https`、`mailto` 和工作簿内锚点形成链接;其他 scheme 与外部工作簿引用保留为纯文本并 warning,绝不访问。Governs R7, R14. +- KTD9. **图表视觉使用原生事实生成的语义预览。** ChartML 直接提取标题、轴/标签文本、系列名、分类、X/Y/数值、缓存和锚点;简单本地引用可从已提取单元格解析,外部或不支持引用保留文字并 warning。使用 Pillow 把这些权威事实渲染成规范化语义卡片,再让视觉模型补充趋势、关系和含义。该结果必须标记为视觉解释,不宣称还原 Excel 图表外观。Governs R8-R9. (session-settled: user-approved — chosen over Excel/LibreOffice pixel rendering or native-only chart output: external rendering adds a cross-platform runtime, while native-only output cannot provide the approved visual interpretation.) +- KTD10. **图片复用原始媒体,视觉按内容去重、按出现位置回放。** 图片使用包内原始 bytes、锚点和 alt text,并复用现有图片安全准备逻辑。同一 SHA-256 只调用一次模型,但结果或失败 warning 必须回放到每个出现位置。图表语义预览采用相同调度模型。Governs R9-R10, R13. +- KTD11. **XLSX 视觉错误采用 fail-open。** 模型未配置、认证/权限错误、无效请求、单对象超时、提供方失败和模型输出无效都返回原生结果,并按对象产生 `xlsx_vision_unavailable`、`xlsx_vision_timeout` 或 `xlsx_vision_failed`。只有调用方取消和整份文档超时沿用公共 API 的中断语义。Governs R9-R10, R13. (session-settled: user-approved — chosen over inheriting every Office fatal-vision branch: the Product Contract requires useful native output for all visual-provider failures.) +- KTD12. **资源限制保持私有,不改变 `ParseOptions`。** `max_pages` 对 XLSX 无效,且不得限制工作表。v0.2.0 使用下面的内部常量;资源特征测试只能在不突破包预算和 wire 预算的前提下收紧它们。Governs R1-R2, R11-R13. (session-settled: user-approved — chosen over adding public spreadsheet options or mapping sheets to pages: the release should preserve the public API and correct semantics.) + +### Resource Budget + +| 资源 | v0.2.0 上限 | 执行点 | 超限结果 | +| --- | ---: | --- | --- | +| Sheet-like 条目 | 128 | `workbook.xml` 预检 | `LimitExceededError` | +| 单表声明矩形 | 2,000,000 个坐标 | worksheet dimension 与实际坐标预检 | `LimitExceededError` | +| 全工作簿序列化 `` | 200,000 | worksheet XML 流式计数 | `LimitExceededError` | +| 非空语义单元格 | 50,000 | 类型与值解码时 | `LimitExceededError` | +| 最终物化网格坐标 | 200,000 | 区域、表格与合并布局前 | `LimitExceededError` | +| 合并范围 | 10,000 个且总 footprint 50,000 | mergeCells 预检 | `LimitExceededError` | +| Shared strings | 100,000 项且解码文本 1,000,000 字符 | sharedStrings 流式读取 | `LimitExceededError` | +| Excel 表格 | 1,024 个且总 footprint 200,000 | table part 预检 | `LimitExceededError` | +| 超链接与批注 | 合计 20,000 个 | 关系和 comments 预检 | `LimitExceededError` | +| 浮动绘图、图表、图片、文本框 | 合计 256 个 | drawing/chart 关系预检 | `LimitExceededError` | +| 图表缓存点 | 200,000 个 | ChartML 预检 | `LimitExceededError` | +| 原生解码文本 | 1,000,000 字符 | block 构建前累计 | `LimitExceededError` | +| Native worker inline / frame | 沿用 8 MiB / 12 MiB | 严格 wire 编解码 | 现有协议错误映射 | +| 最终 Markdown | 沿用 `max_output_chars`,默认 400,000 | 公共 renderer | 现有截断 warning | + +不支持格式类 warning 每个 code 保留前 20 条并附确定性汇总;视觉对象与无法支持的浮动对象因总量已受 256 限制,必须逐个保留可定位 warning。 + +### Warning and Failure Taxonomy + +| 场景 | 结果 | +| --- | --- | +| ZIP 损坏、必需 part/关系缺失、DTD/实体、非法 XML | `CorruptDocumentError` | +| 加密 ZIP 成员或 Office 加密容器 | 沿用检测层的类型化拒绝;进入 XLSX 包层后映射为 `CorruptDocumentError` | +| 任一包、结构、对象、字符或 wire 预算超限 | `LimitExceededError` | +| 公式无缓存 | 公式文本 + `xlsx_formula_cache_missing` | +| 数字格式超出 KTD4 | 稳定原始值 + `xlsx_unsupported_number_format` | +| 外部引用、危险 URL scheme 或无法解析的本地引用 | 纯文本引用 + `xlsx_external_reference` | +| 不支持的标准/厂商对象 | 跳过该对象 + `xlsx_unsupported_object` | +| 视觉未配置、超时或失败 | 原生结果 + KTD11 对应 warning | + +### High-Level Technical Design + +#### Component and Data Flow + +```mermaid +flowchart TB + S[Resolved XLSX source] --> D[Detection and package preflight] + D --> O[Bounded OOXML index] + D --> W[Full-mode openpyxl load] + O --> X[XLSX extractor] + W --> X + X --> N[Strict XLSX native document wire] + N --> M[Deterministic slot merge] + X --> V[Image and chart visual slots] + V --> P[Shared image preparation and vision dispatch] + P --> M + M --> B[Existing core blocks] + B --> R[Existing Markdown renderer] +``` + +OOXML index 负责信任边界、sheet 关系、公式缓存和不支持对象发现。`openpyxl` 只在包已证明有界后负责受支持的工作簿值对象。XLSX merge layer 只输出现有 core blocks 和受控 Markdown anchors。 + +#### Parse Sequence + +```mermaid +sequenceDiagram + participant API as parse/aparse + participant PKG as XLSX preflight + participant RT as Native worker + participant EXT as XLSX extractor + participant VIS as Vision dispatcher + participant MD as Markdown renderer + API->>PKG: detect and validate bounded OOXML + PKG->>RT: validated path and options + RT->>EXT: extract native sheets, values, objects, chart facts + EXT-->>API: strict native document plus visual slots + API->>VIS: deduplicated images and semantic chart previews + VIS-->>API: result or per-object failure + API->>MD: merged existing blocks and warnings + MD-->>API: deterministic Markdown +``` + +预检发生在第三方工作簿加载之前。视觉调用发生在全部原生事实已经成功提取之后,因此 KTD11 的 fail-open 不会产生部分原生文档。 + +#### Decision and Failure Flow + +```mermaid +flowchart TB + A[Input] --> B{XLSX identity matches?} + B -->|No| E[Existing typed detection error] + B -->|Yes| C{Package and structure within budgets?} + C -->|No| F[CorruptDocumentError or LimitExceededError] + C -->|Yes| D[Build complete native document] + D --> G{Visual slots exist?} + G -->|No| J[Render native Markdown] + G -->|Yes| H{Vision configured and succeeds?} + H -->|Yes| I[Append grounded visual interpretation] + H -->|No| K[Append anchored warning] + I --> J + K --> J +``` + +### Output Structure + +```text +src/opendocs/parsers/xlsx/ +├── __init__.py +├── extract.py +├── merge.py +├── models.py +├── parser.py +├── preflight.py +└── values.py +``` + +共享图片准备若需要抽取,只新增一个中立的私有 helper,例如 `src/opendocs/parsers/embedded_vision.py`。不得借 XLSX 引入 DOCX/PPTX 模型重构。 + +### System-Wide Impact + +| 表面 | 影响 | 约束 | +| --- | --- | --- | +| 公共 API | 新增可识别格式,不改签名与返回类型 | `parse()`/`aparse()`、输入形态和 warning 对等 | +| Detection / registry | 新增 `DocumentType.XLSX`、ZIP 身份和默认 parser | 扩展名不可信;容器身份优先 | +| Native runtime | 新增 XLSX 严格 wire 与 worker 路径 | 8 MiB inline、12 MiB frame 和取消清理不放宽 | +| Markdown | 复用 Heading、Table、SpannedTable、Paragraph、InlineLink、MarkdownBlock | 不新增公开 XLSX Markdown 方言 | +| Vision | 新增图片与语义图表预览请求 | 原生事实优先;按 digest 去重、按位置回放 | +| Packaging | 增加两个直接运行依赖 | wheel metadata、锁文件和隔离安装必须一致 | +| Release evidence | 增加公开合成门与私有真实工作簿协议 | 私有文件、输出和检查表不得提交 | + +### Sequencing + +```mermaid +flowchart LR + U1[U1 Public wiring and dependencies] --> U2[U2 Models and preflight] + U2 --> U3[U3 Values and regions] + U3 --> U4[U4 Text objects] + U3 --> U5[U5 Charts and media] + U4 --> U6[U6 Parser, vision and merge] + U5 --> U6 + U6 --> U7[U7 API, lifecycle and adversarial proof] + U7 --> U8[U8 Release integration and private validation] +``` + +每个 feature-bearing unit 先增加失败测试,再修改生产代码。U1-U2 固定输入、错误和 wire 边界;U3-U5 构建原生事实;U6 才接视觉;U7-U8 完成跨层与发布证据。 + +### Alternative Approaches Considered + +- **纯 `openpyxl`。** 拒绝,因为 read-only 模式缺失图表、图片和批注,完整模式也不覆盖 threaded comments、DrawingML 文本框和所有图表关系;私有对象字段不能承担稳定核心契约。 +- **纯手写 OOXML。** 拒绝,因为 v0.2.0 不应重写 Excel 单元格类型、样式索引、日期系统、表格和合并兼容层;低层解析只负责安全预检和高层库缺失的对象。 +- **调用 Excel 或 LibreOffice 渲染原图。** 拒绝,因为它扩大跨平台运行时、沙箱、进程清理和发布体积;KTD9 已用有界语义预览满足趋势补充。 +- **整张工作表视觉解析。** 拒绝,因为成本、确定性和大表资源风险与原生数据优先相冲突。 +- **只做单元格文本,不做外围对象。** 拒绝,因为它违反 R7,并会把正文之外的标准文本静默丢失。 + +### Risks and Mitigations + +| 风险 | 缓解 | 验证证据 | +| --- | --- | --- | +| `openpyxl` 完整模式内存放大 | loader 前执行 KTD2 与 Resource Budget;worker wire 保持既有上限 | 对抗性结构测试与 U7 资源特征测试 | +| Excel 显示格式复杂且本地化 | KTD4 固定硬支持子集;其余可读降级并 warning | 精确字符串断言覆盖货币、百分比、日期和负值 | +| 公式缓存不存在或过期 | 保留缓存节点存在性;缺失回退公式;不宣称新鲜度 | 直接 patch OOXML 的缓存有/无/空值测试 | +| 图表 API 与厂商扩展不稳定 | ChartML 作为事实来源;私有 `openpyxl` 字段仅可作非契约辅助 | 原生图表 fixture 与未知扩展 warning | +| 图表语义预览被误读为原图 | 输出明确标为视觉解释,模型 prompt 禁止改写原生值 | mock vision 断言 prompt、锚点与合并优先级 | +| URL 或外部关系触发访问 | `keep_links=False`,直接关系只保留文本,危险 scheme 不形成链接 | 网络调用禁用测试和链接转义测试 | +| warning 风暴掩盖输出 | 非对象 warning 有界聚合;对象总数先硬限 | 超量格式与 256 对象边界测试 | +| 无真实工作簿导致合成盲区 | 公开合成门覆盖契约;发布前维护者提供私有 XLSX 做探索 | U8 私有审阅清单,不生成可提交 baseline | + +### Dependencies and Prerequisites + +- `openpyxl` 与 `defusedxml` 进入直接 runtime dependencies 和 `uv.lock`;不新增 extra。 +- Pillow、native worker、Markdown renderer、warning 发射与 vision dispatcher 沿用现有依赖和生命周期。 +- 真实 XLSX 在实施完成后由维护者提供;缺少该文件不阻止公开合成开发,但阻止把真实兼容性写成已验证事实。 + +### Sources and Research + +- [`openpyxl` PyPI](https://pypi.org/project/openpyxl/):当前稳定版 3.1.5、Python 版本要求和官方 XML 安全提示。 +- [`openpyxl` tutorial](https://openpyxl.readthedocs.io/en/stable/tutorial.html):`data_only`、`read_only`、`rich_text`、`keep_links` 行为,以及 shapes 和 read-only 特性缺口。 +- [`openpyxl` optimized modes](https://openpyxl.readthedocs.io/en/stable/optimized.html):read-only 对声明维度的依赖和显式 close 责任。 +- [`openpyxl` comments](https://openpyxl.readthedocs.io/en/3.0/comments.html):经典批注仅保留文本/作者、格式与容器信息丢失,read-only 不支持批注。 +- [`defusedxml` PyPI](https://pypi.org/project/defusedxml/):XML entity/DTD/DoS 防护范围;它不替代 ZIP bomb 预算。 +- `src/opendocs/parsers/office/package.py`:现有 OOXML 包预算、关系和路径验证模式。 +- `src/opendocs/parsers/office/models.py`、`src/opendocs/parsers/office/parser.py`、`src/opendocs/parsers/office/merge.py`:严格 worker wire、视觉去重和按出现位置合并模式。 +- `src/opendocs/parsers/office/pptx.py`:原生图表标题、分类、系列和值的相邻实现模式。 +- `src/opendocs/markdown.py`、`tests/test_markdown.py`:合并跨度与受控 Markdown 注释的渲染模式。 +- AGENTS.md 指定的本地 `43x-agent` checkout 中,Office parser 仅作为行为参考;OpenDocs 必须独立实现,且不得复制私有文件、模型载荷或应用依赖。 + +--- + +## Implementation Units + +### U1. Public XLSX Identity, Dependencies, and Registration + +- **Goal:** 让合法 XLSX 进入现有公共解析主路径,并在第三方加载前复用 OOXML 包安全边界。 +- **Requirements:** R1-R2, R11-R13;KTD2-KTD3、KTD12;Covers F3 / AE7-AE8. +- **Dependencies:** 无。 +- **Files:** `pyproject.toml`, `uv.lock`, `src/opendocs/_models.py`, `src/opendocs/detection.py`, `src/opendocs/parsers/registry.py`, `src/opendocs/parsers/office/package.py`, `tests/test_models.py`, `tests/test_detection.py`, `tests/test_registry.py`, `tests/test_office_package.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 先以合成 ZIP fixture 锁定 `.xlsx`、无名 bytes/stream、后缀不匹配、缺失 workbook part、重复/加密成员和关系逃逸行为。 + 2. 新增 `DocumentType.XLSX`,以 `[Content_Types].xml`、根 relationship 和 `xl/workbook.xml` 共同确认身份。 + 3. 扩展 Office package validator 的 XLSX required parts 与允许关系根,不改变 DOCX/PPTX 预算和错误行为。 + 4. 加入 KTD3 依赖与默认 registry;parser 在 U6 前可用明确的未完成测试替身,不合入不能解析的默认注册状态。 +- **Execution note:** 从失败的 detection、package 和 registry 契约测试开始;该 unit 必须以完整可调用的最小 parser seam 收尾,避免中间提交破坏默认 registry。 +- **Patterns to follow:** `src/opendocs/detection.py` 的容器身份匹配,`src/opendocs/parsers/office/package.py` 的路径/关系验证,`src/opendocs/parsers/registry.py` 的默认注册。 +- **Test scenarios:** + 1. `.xlsx` path、无扩展 bytes、命名与无名 binary stream 包含合法 workbook 关系时都识别为 XLSX。 + 2. `.xlsx` 实际为 DOCX/PPTX、ZIP 缺少 `xl/workbook.xml`、content type 或根关系不匹配时返回现有类型化 mismatch/corrupt 语义。 + 3. XLSX 包含路径穿越、重复成员、加密 flag、悬空关系、外部 required root relationship、超限成员或压缩比时,在 parser 调用前失败。 + 4. DOCX/PPTX 的 required part 和包预算回归测试保持不变。 + 5. 构建默认 registry 时 XLSX 与现有格式各注册一次,缺少 runtime 的错误契约不变。 +- **Verification:** 所有合法输入到达 XLSX parser seam;所有伪装、损坏与包超限输入在第三方加载前以稳定类型失败;现有 Office 格式无注册或验证回归。 + +### U2. XLSX Wire Models and Structural Preflight + +- **Goal:** 建立不使用 page 语义的严格 XLSX native document,并在 `openpyxl` 前执行完整结构预算。 +- **Requirements:** R2-R4, R7-R8, R11-R14;KTD1-KTD2、KTD7、KTD12;Covers F3 / AE1、AE7. +- **Dependencies:** U1. +- **Files:** `src/opendocs/parsers/xlsx/__init__.py`, `src/opendocs/parsers/xlsx/models.py`, `src/opendocs/parsers/xlsx/preflight.py`, `tests/test_xlsx_models.py`, `tests/test_xlsx_preflight.py`, `tests/xlsx_fixtures.py`, `tests/test_runtime.py`. +- **Approach:** + 1. 定义不可变 `XlsxDocument`、`XlsxSheet`、原生槽位、图片/图表视觉槽位、数字 sheet index 和 A1 anchor;wire 只接受已知字段、tuple、受限 basename、SHA-256 和现有 Block。 + 2. 用 `defusedxml` 流式建立 OOXML index,解析 workbook/sheet/chartsheet 顺序、状态、part 目标、日期系统、shared strings、worksheet 元数据、drawing/chart/comment/table 关系。 + 3. 在返回 index 前执行 Resource Budget;维度、实际 cell 坐标、merge/table footprint、对象和字符计数都必须交叉验证。 + 4. 把恶意 XML、非法命名空间/关系和超限分别映射到 Planning Contract 的失败分类。 +- **Execution note:** 在引入完整工作簿加载前用 ZIP/XML patch fixture 穷举失败面;不要创建或提交二进制样本。 +- **Patterns to follow:** `src/opendocs/parsers/office/models.py` 的 dataclass/wire 校验,`src/opendocs/_native_protocol.py` 的 frame 预算,`src/opendocs/parsers/office/package.py` 的安全路径。 +- **Test scenarios:** + 1. worksheet、chartsheet、hidden、veryHidden 和空 sheet 按 relationship 顺序进入严格 wire,重复 source index、非法 anchor 或未知字段被拒绝。 + 2. 128 个 sheet 成功,129 个失败;声明矩形、序列化 cell、非空 cell、materialized grid、merge、shared string、table、object、chart cache 和文本预算逐项验证边界值与超一值。 + 3. 声明 `A1:XFD1048576` 但只有一个 cell 的稀疏表在 `openpyxl` 前失败,不触发矩形遍历。 + 4. DTD、内部/外部实体、畸形 XML、Zip Slip、悬空 drawing/chart/comment 关系返回 `CorruptDocumentError`。 + 5. 编码后的最大合法 wire 小于 12 MiB;超出 inline/container/frame 预算时沿用现有 runtime 失败映射。 +- **Verification:** 任一第三方加载都只接收已通过预检的包;XLSX wire 不含 page 概念,且所有可放大结构有明确边界测试。 + +### U3. Saved Values, Sheets, Tables, Regions, and Merges + +- **Goal:** 输出全部 sheet 的稳定原生网格内容,并准确处理常见显示格式、公式缓存、表格和合并跨度。 +- **Requirements:** R3-R6, R11-R13;KTD3-KTD7、KTD12;Covers F1 / AE1-AE3. +- **Dependencies:** U2. +- **Files:** `src/opendocs/parsers/xlsx/values.py`, `src/opendocs/parsers/xlsx/extract.py`, `tests/test_xlsx_values.py`, `tests/test_xlsx_extract.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 以 full-mode `openpyxl` 读取已预检 workbook,并由 OOXML formula/cache sidecar 覆盖公式显示选择。 + 2. 在 `values.py` 集中实现 KTD4-KTD5;格式 warning 按 code 和 sheet/coordinate 稳定聚合。 + 3. 先占用 Excel table 精确范围,再对剩余语义单元格和 merge footprint 做连通分量;物化 component bounding box 前再次扣减 grid 预算。 + 4. 空 sheet 只产生标题与 sheet anchor;style-only 空 cell 计入安全访问但不成为语义坐标。 + 5. 将无 merge 区域转换为 `TableBlock(header_rows=0)`,有 merge 区域转换为 `SpannedTableBlock`,并按 KTD6 排序。 +- **Execution note:** 公式 fixture 必须直接 patch worksheet XML 的 ``/``,不能假装 `openpyxl` 会计算公式。 +- **Patterns to follow:** `src/opendocs/parsers/office/pptx.py` 和 `src/opendocs/parsers/office/docx.py` 的 table block 构造,`src/opendocs/markdown.py` 的 span 渲染。 +- **Test scenarios:** + 1. Covers AE1. visible、hidden、veryHidden、空 worksheet 和 chartsheet 按源顺序输出固定标题、状态和安全 anchor。 + 2. Covers AE2. `$1,234.50`、`¥1,234`、`12.50%`、千分位、负数、1900/1904 日期、时间和 elapsed-time 产生精确稳定文本。 + 3. 带缓存公式输出格式化缓存;缓存缺失输出公式并 warning;缓存节点存在但值为空与真正缺失可区分;外部公式不发起访问。 + 4. scientific、fraction、accounting、条件/颜色段和复杂本地化格式输出稳定原始值并按坐标 warning;同 code 超过 20 条时保留确定性汇总。 + 5. Covers AE3. 横向/纵向合并、Excel table、无表头普通区域、多个相离区域和内部空 cell 保留全部非空内容与跨度,且顺序可重复。 + 6. 仅字体/颜色/边框/条件格式的空 cell 不产生 Markdown 内容;全空工作簿仍成功输出所有 sheet 标题。 + 7. component bounding box、table 或 merge footprint 在预检后因组合超预算时,在物化前返回 `LimitExceededError`。 +- **Verification:** 合成工作簿的 sheet、值、公式、格式、区域和 merge Markdown 与黄金字符串一致;重复解析及输入形态不改变顺序。 + +### U4. Comments, Text Boxes, Links, and Headers or Footers + +- **Goal:** 把单元格之外的标准文本对象纳入原生结果,并对外部关系和不支持对象提供可定位降级。 +- **Requirements:** R4, R7, R11, R13-R14;KTD7-KTD8、KTD12;Covers F1 / AE4. +- **Dependencies:** U3. +- **Files:** `src/opendocs/parsers/xlsx/extract.py`, `src/opendocs/parsers/xlsx/models.py`, `tests/test_xlsx_extract.py`, `tests/test_xlsx_models.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 从 OOXML index 读取 classic comments/notes、threaded comments/person、`xdr:sp/a:txBody` 文本框和 shape alt text,不依赖 `openpyxl` 的有限 comment/drawing 映射。 + 2. 单元格 hyperlink 用 `InlineLink` 表达安全目标;内部 anchor、外部 URL、外部 workbook 和危险 scheme 按 KTD8 分类。 + 3. 解析页眉页脚文本并去除字体/颜色控制码;页码、日期、时间、文件名和 sheet 名字段保留为命名占位符,图片字段 warning。 + 4. 生成对象槽位并按 KTD7 与 cell regions 合并;SmartArt、OLE、controls、VML 文本和 vendor extensions 逐对象 warning。 +- **Patterns to follow:** `src/opendocs/parsers/office/docx.py` 的 `InlineLink` 与关系处理,`src/opendocs/parsers/office/models.py` 的 source index 和 warning 模型。 +- **Test scenarios:** + 1. Covers AE4. 经典批注文本/作者、threaded comment/person、DrawingML 文本框、alt text 和页眉页脚都在正确 sheet/anchor 下出现。 + 2. safe HTTP/HTTPS/mailto 与工作簿内链接转义为 Markdown 链接;`javascript:`、`file:`、相对外部工作簿和远程数据关系只输出文本并 warning。 + 3. 测试期间禁止网络调用,包含外部 URL、externalLinks、data connections 的 workbook 仍零访问完成解析。 + 4. odd/even/first × header/footer × left/center/right 按固定顺序输出;格式控制码不泄漏,动态字段使用稳定占位符。 + 5. SmartArt、OLE、ActiveX/control、VML-only textbox 和未知 extension 各产生包含 sheet index、A1 anchor 或 object ordinal 的 warning。 + 6. 20,000 个 comment/hyperlink 在边界成功,超一值在对象解码前失败;原生文本预算仍限制总内容。 +- **Verification:** 所有承诺的非 cell 文本均可定位;任何外部引用无 I/O;不支持对象不会静默消失。 + +### U5. Native Chart Facts, Embedded Images, and Semantic Previews + +- **Goal:** 以原生 ChartML 和媒体为事实来源,构建可去重的图片与图表视觉任务。 +- **Requirements:** R4, R8-R10, R11, R13-R14;KTD8-KTD10、KTD12;Covers F1-F2 / AE5-AE6. +- **Dependencies:** U2-U4. +- **Files:** `src/opendocs/parsers/xlsx/extract.py`, `src/opendocs/parsers/xlsx/models.py`, `src/opendocs/parsers/xlsx/parser.py`, `src/opendocs/parsers/embedded_vision.py`, `tests/test_xlsx_extract.py`, `tests/test_xlsx_parser.py`, `tests/test_office_parser.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 从 drawing anchors 和 ChartML 提取 chart title、axis/data-label text、series names、categories、X/Y/values、cache、local formulas、alt text 和位置。 + 2. 解析已在 workbook 内的简单引用;外部、动态或不支持公式保留引用文本并产生 `xlsx_external_reference`,不得调用计算引擎或网络。 + 3. 把原生图表事实输出为标题与无表头数据 block,再用 Pillow 生成无样式保真承诺的语义卡片;视觉 prompt 只允许补充趋势、关系、标注和含义。 + 4. 提取原始图片 part、anchor 和 alt text,复用现有安全解码、尺寸、像素和 tile 预算。若抽取共享 helper,保持 Office 现有结果与 warning 不变。 + 5. 以内容 digest 构建 visual slot;相同图片或语义卡片只准备一次。 +- **Execution note:** 先锁定 native-only 图表与图片槽位,再加入语义卡片;视觉测试只使用 fake client,不调用真实模型。 +- **Patterns to follow:** `src/opendocs/parsers/office/pptx.py` 的原生图表 block,`src/opendocs/parsers/office/parser.py` 的 digest 去重,`src/opendocs/vision/images.py` 的图片防护。 +- **Test scenarios:** + 1. Covers AE5. line、bar、pie/doughnut 和 scatter 合成图表的标题、系列、分类、X/Y/值及 anchor 由 ChartML 稳定输出,视觉文本不能覆盖原生表。 + 2. chart cache、简单本地 range、缺少 cache、动态 named formula 和 external reference 按 KTD9 解析或 warning,无网络和公式计算。 + 3. 语义预览只包含原生事实与明确“视觉解释”标签;fake vision 返回趋势时结果位于对应 chart anchor 后。 + 4. 同一媒体在多个 sheet/anchor 出现时只产生一次 vision request,但结果或失败状态可回放到全部位置。 + 5. 图片格式伪装、解压 bomb、超尺寸、超像素、超 tile 和损坏媒体沿用图片安全错误或 warning,不扩大既有预算。 + 6. 256 个浮动对象在边界成功,257 个失败;200,000 个 chart cache point 在边界成功,超一值在 preview 前失败。 + 7. Office parser 回归测试证明共享 helper 抽取不改变 DOCX/PPTX request、结果顺序和 warning。 +- **Verification:** 原生图表事实可独立构成完整 native-only 输出;视觉任务均有受控输入、digest 和出现位置;不需要 Excel/LibreOffice。 + +### U6. Parser Orchestration, Fail-Open Vision, and Deterministic Merge + +- **Goal:** 把 native worker、视觉调度、warning 和 Markdown 合并成完整 `XlsxParser`,同时保持取消、超时和清理语义。 +- **Requirements:** R1, R3-R4, R8-R10, R13-R14;KTD1、KTD6-KTD12;Covers F1-F2 / AE5-AE6、AE8. +- **Dependencies:** U3-U5. +- **Files:** `src/opendocs/parsers/xlsx/parser.py`, `src/opendocs/parsers/xlsx/merge.py`, `src/opendocs/parsers/xlsx/models.py`, `src/opendocs/parsers/registry.py`, `tests/test_xlsx_parser.py`, `tests/test_xlsx_merge.py`, `tests/test_registry.py`. +- **Approach:** + 1. 在 native worker 中执行 KTD2-KTD8 的完整原生提取并返回 strict wire;主进程只处理受限视觉 artifact 和 block merge。 + 2. 以 digest 去重请求并按 source ordinal 排序;把每个 request 的 outcome 回放到 occurrence 集合。 + 3. 单独实现 KTD11 的 XLSX fail-open 分类;不修改 Office parser 对现有格式的异常策略。 + 4. `merge.py` 先输出 sheet heading/anchor,再按 KTD6-KTD7 排序原生和视觉槽位;原生 block 永远在同一对象的视觉解释之前。 + 5. 只有 native extraction 不完整、调用方取消或文档 deadline 才中断成功路径;模型失败不得制造 `NoUsableContentError`。 +- **Execution note:** 用 fake runtime/vision 先锁定异常矩阵,再接真实 native worker;同步超时和异步取消必须使用现有工作区生命周期。 +- **Patterns to follow:** `src/opendocs/parsers/office/parser.py` 的 worker/vision 编排,`src/opendocs/parsers/office/merge.py` 的 occurrence 回放和 warning,`src/opendocs/api.py` 的 deadline/cancellation。 +- **Test scenarios:** + 1. 无视觉对象、视觉未配置、全部视觉成功、部分成功和全部失败都返回相同原生 Markdown 主干。 + 2. Covers AE6. 未配置、认证、权限、无效请求、provider error、坏模型输出和单 request timeout 按对象产生 KTD11 warning,且不抛出文档失败。 + 3. 调用方 async cancellation 及时传播;sync document timeout 返回现有超时错误;两者不被 fail-open 捕获。 + 4. 重复 digest 只调用一次,结果与 warning 按每个 sheet/anchor occurrence 回放,顺序不受 future 完成顺序影响。 + 5. 原生 chart/table 与视觉解释同 anchor 时原生先出现;视觉不得删除、替换或重排原生 block。 + 6. 全空 workbook 仍有 sheet heading,只有无法支持对象的 workbook 返回 headings 与 warnings,而不是空成功或模型猜测。 + 7. XLSX registry 使用完整 parser;DOCX/PPTX/Image/PDF 视觉异常行为不受 KTD11 影响。 +- **Verification:** `XlsxParser` 在所有视觉状态下满足原生优先和确定顺序;用户取消、文档超时和 parser 失败边界清晰且无资源泄漏。 + +### U7. Public API Parity, Resource Characterization, and Adversarial Regression + +- **Goal:** 从公共 API 证明输入、同步/异步、warning、失败、资源和生命周期契约,并保护所有现有格式。 +- **Requirements:** R1-R2, R10-R15;KTD2、KTD11-KTD12;Covers F1-F3 / AE7-AE8. +- **Dependencies:** U6. +- **Files:** `tests/test_api_xlsx.py`, `tests/test_xlsx_preflight.py`, `tests/test_xlsx_parser.py`, `tests/test_api.py`, `tests/test_api_m2.py`, `tests/test_runtime.py`, `tests/test_markdown.py`, `tests/xlsx_fixtures.py`. +- **Approach:** + 1. 用同一合成工作簿参数化 path、bytes、named/unnamed binary stream 和 `parse()`/`aparse()`,对比 Markdown、warning code/顺序和错误类型。 + 2. 覆盖 repeated parse、`max_output_chars`、`max_pages=1`、vision 配置和输入后缀组合;明确 `max_pages` 不减少 sheet。 + 3. 对 Resource Budget 每项执行边界/超一测试,并验证失败发生在 `openpyxl` load、grid materialization 或 vision 前。 + 4. 用受控 worker fixture 验证正常、异常、async cancellation、sync timeout 和 hard termination 后 workspace/artifact 清理。 + 5. 运行现有全格式回归,确认新增 DocumentType、共享 helper 和 renderer 使用不改变旧输出。 +- **Execution note:** 先写公共集成失败测试,再补资源特征;不要把环境相关峰值 RSS 写成未经证实的发布承诺。 +- **Patterns to follow:** `tests/test_api_m2.py` 的 path/bytes/stream 对等,`tests/test_runtime.py` 的 worker 边界,现有 parser cancellation/cleanup 测试。 +- **Test scenarios:** + 1. Covers AE8. 六组输入/API 组合的 Markdown 字符串完全一致,warning code 与失败类型对等;warning 的 Python 发射位置保持公共契约。 + 2. 同一 workbook 连续解析至少三次,sheet、region、object 和 warning 顺序完全一致。 + 3. `max_pages=1` 的多 sheet workbook 仍输出全部 sheet;`max_output_chars` 继续按现有 renderer 截断并 warning。 + 4. ZIP bomb、巨维度、巨 merge、巨 shared strings、对象风暴、chart cache 风暴和 XML entity 在昂贵阶段前失败。 + 5. sync timeout、async cancellation、native crash、vision timeout 后,OpenDocs 临时目录与测试前基线一致;清理失败保留主异常并 warning。 + 6. TXT、Markdown、PDF、Image、DOCX 和 PPTX 的选定黄金输出、warning 和 registry 行为不变。 +- **Verification:** 公共 XLSX 契约在所有等价入口可复现;资源限制执行点有可观察证据;全套现有测试无回归。 + +### U8. Packaging, Documentation, Release Smoke, and Private Workbook Protocol + +- **Goal:** 让 wheel、发布文档和验收流程准确包含 XLSX,并建立不提交真实内容的私有探索门。 +- **Requirements:** R1-R2, R9-R15;KTD3、KTD9、KTD12;Covers F4 / AE9. +- **Dependencies:** U7. +- **Files:** `README.md`, `CHANGELOG.md`, `docs/roadmap.md`, `docs/plans/2026-08-07-v0.2.0-release-plan.md`, `scripts/release_smoke.py`, `scripts/check_release_artifacts.py`, `tests/test_package.py`, `tests/test_release_scripts.py`, `tests/test_acceptance_corpus.py`. +- **Approach:** + 1. 更新 README、roadmap、CHANGELOG 和父 v0.2.0 计划,明确支持 `.xlsx`、不支持 `.xls/.xlsm/.xlsb`、不下载 URL、公式/格式/视觉边界,以及本计划取代父计划 A4 的范围。 + 2. 更新 release artifact dependency set、wheel metadata 断言和隔离安装 smoke,验证 `openpyxl`、`defusedxml` 与 parser 都来自构建产物环境。 + 3. release smoke 动态生成最小 XLSX,覆盖多 sheet、货币/日期、merge、公式 fallback 和 native-only 输出;不把视觉 provider 作为安装 smoke 前提。 + 4. 维护者提供真实 XLSX 后,在 `tests/corpus/` 与 `tests/corpus.local.toml` 建立忽略的探索条目,人工对照内容、sheet、格式、对象、图表和 warning。 + 5. 只有明确违反 R1-R14 的核心缺陷阻断 v0.2.0;罕见厂商扩展或视觉增强不足记录为 follow-up。禁止提交真实工作簿、hash、模型输出或完成的检查表。 +- **Execution note:** 该 unit 以构建后隔离安装为首要证明;私有探索结果必须区分“已运行”和“待提供样本”。 +- **Patterns to follow:** `scripts/release_smoke.py`、`scripts/check_release_artifacts.py`、`tests/test_package.py` 的 v0.1.0 发布验证;AGENTS.md 的私有 corpus 边界。 +- **Test scenarios:** + 1. wheel/sdist metadata 的直接依赖与 `pyproject.toml`、`uv.lock`、release checker 期望完全一致。 + 2. 隔离环境只安装构建 wheel 后可以导入 XLSX parser,并解析动态生成的 native-only workbook。 + 3. release smoke 的多 sheet、货币、日期、merge 和公式 fallback 产生预期 Markdown,且不需要网络、LibreOffice、Excel 或视觉凭据。 + 4. README/roadmap/release plan 不再声称 XLSX 禁用视觉,也不把语义图表预览描述为源图像素保真。 + 5. Covers AE9. 私有工作簿未提供时门明确为 `not_run`;提供后人工结论只记录核心缺陷或非阻断 follow-up,不自动生成可提交 baseline。 +- **Verification:** 构建产物可独立解析 XLSX;依赖、文档和父计划无冲突;私有验证不泄漏内容且证据等级准确。 + +--- + +## Verification Contract + +| Gate | Command | Proves | Applies after | +| --- | --- | --- | --- | +| Targeted XLSX suite | `uv run --frozen pytest tests/test_xlsx_models.py tests/test_xlsx_preflight.py tests/test_xlsx_values.py tests/test_xlsx_extract.py tests/test_xlsx_merge.py tests/test_xlsx_parser.py tests/test_api_xlsx.py -q` | XLSX core behavior、limits、vision fallback 和 API parity | U2-U7 | +| Detection and Office regression | `uv run --frozen pytest tests/test_detection.py tests/test_registry.py tests/test_office_package.py tests/test_office_parser.py tests/test_markdown.py -q` | Shared detection/package/vision/renderer 无回归 | U1、U5-U7 | +| Package and release checks | `uv run --frozen pytest tests/test_package.py tests/test_release_scripts.py -q` | Dependency metadata、isolated import 和 release smoke | U8 | +| Public suite | `uv run --frozen pytest -q` | 全格式公开回归 | U7-U8 | +| Lint | `uv run --frozen ruff check .` | Imports、style 和 common defects | 每个 unit 收尾 | +| Format | `uv run --frozen ruff format --check .` | 100 字符等格式契约 | 每个 unit 收尾 | +| Type check | `uv run --frozen ty check src tests` | Strict models、wire 和 parser type consistency | U2-U8 | +| Build | `uv build` | Wheel 与 sdist 可生成 | U8 | +| Fresh lock install | `uv sync --all-groups --frozen` | 锁文件完整且依赖可安装 | U1、U8 | +| Private corpus | `uv run --frozen pytest tests/test_acceptance_corpus.py -q --corpus-dir=@local` | 维护者提供真实 XLSX 后的私有探索与既有 corpus 回归 | U8,可选且证据需单列 | + +Python 3.11、3.12 和 3.13 的 CI/隔离 smoke 必须覆盖依赖安装与最小 XLSX 解析。不得用 focused suite 代替完整公开 suite,也不得把未运行的私有或真实模型验证写成通过。 + +--- + +## Definition of Done + +- R1-R15 均由至少一个 U-ID 和一个自动化场景覆盖;AE1-AE8 是公开合成硬门,AE9 按真实样本是否提供报告 `not_run` 或审阅结论。 +- `DocumentType.XLSX`、检测、包安全、严格 wire、解析、registry 和 Markdown 主路径完整连通,path/bytes/stream 与 `parse()`/`aparse()` 对等。 +- 全部 worksheet/chartsheet、状态、空 sheet、区域、Excel tables、merge、支持格式、公式 fallback、标准文本对象、超链接和页眉页脚满足 Product Contract。 +- 原生图表事实和图片均可定位;视觉成功只追加解释,任一视觉失败返回原生结果和逐对象 warning。 +- 所有 Resource Budget 在昂贵加载、物化或视觉调用前执行,`max_pages` 不改变 sheet 数量,`max_output_chars` 沿用公共语义。 +- `openpyxl` 和 `defusedxml` 的依赖范围、锁文件、wheel metadata、release checker 与隔离安装一致。 +- Verification Contract 中除条件式 private corpus 外的 gate 全部通过;private gate 的未运行/通过/缺陷证据单独报告。 +- README、CHANGELOG、roadmap 与父 v0.2.0 计划准确描述 XLSX 支持和限制,不再保留相互冲突的视觉范围。 +- 未提交任何真实 XLSX、私有 hash、模型 payload、生成 Markdown 或完成检查表。 +- 最终 diff 不包含试验性 parser、废弃 wire、临时预览文件、调试输出或被放弃方案的死代码。 diff --git a/pyproject.toml b/pyproject.toml index 423cacf..032650a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -43,7 +43,9 @@ classifiers = [ "Topic :: Scientific/Engineering :: Artificial Intelligence", ] dependencies = [ + "defusedxml>=0.7.1,<1", "litellm>=1.93,<2", + "openpyxl>=3.1.5,<3.2", "pdfplumber>=0.11.10,<0.12", "pillow>=12.3,<13", "python-docx>=1.1.2,<2", diff --git a/src/opendocs/_models.py b/src/opendocs/_models.py index cc98bfd..b8a83a1 100644 --- a/src/opendocs/_models.py +++ b/src/opendocs/_models.py @@ -65,6 +65,7 @@ class DocumentType(StrEnum): IMAGE = "image" DOCX = "docx" PPTX = "pptx" + XLSX = "xlsx" class ListKind(StrEnum): diff --git a/src/opendocs/detection.py b/src/opendocs/detection.py index 6959b86..2142a3e 100644 --- a/src/opendocs/detection.py +++ b/src/opendocs/detection.py @@ -10,6 +10,7 @@ DocumentTypeMismatchError, UnsupportedDocumentError, ) +from opendocs.parsers.office.package import validate_office_package from opendocs.source import ResolvedSource _SUFFIX_TYPES = { @@ -23,6 +24,7 @@ ".webp": DocumentType.IMAGE, ".docx": DocumentType.DOCX, ".pptx": DocumentType.PPTX, + ".xlsx": DocumentType.XLSX, } _ZIP_PREFIXES = (b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08") _UTF8_CHUNK_SIZE = 4096 @@ -43,17 +45,26 @@ def _signature_type(signature: bytes) -> DocumentType | None: def _container_type(path: Path) -> DocumentType: try: with ZipFile(path) as archive: + found_docx = False found_pptx = False + found_xlsx = False for info in archive.infolist(): if info.filename == "word/document.xml": - return DocumentType.DOCX - if info.filename == "ppt/presentation.xml": + found_docx = True + elif info.filename == "ppt/presentation.xml": found_pptx = True + elif info.filename == "xl/workbook.xml": + found_xlsx = True except BadZipFile as error: raise CorruptDocumentError("ZIP-based document is corrupt") from error + if found_docx: + return DocumentType.DOCX if found_pptx: return DocumentType.PPTX + if found_xlsx: + validate_office_package(path, document_type=DocumentType.XLSX) + return DocumentType.XLSX raise UnsupportedDocumentError("ZIP container is neither DOCX nor PPTX") @@ -81,11 +92,17 @@ def detect_document_type(source: ResolvedSource) -> DocumentType: if suffix and suffix not in _SUFFIX_TYPES: raise UnsupportedDocumentError(f"unsupported document extension: {suffix}") + declared = _SUFFIX_TYPES.get(suffix) detected = _signature_type(signature) if detected is None and signature.startswith(_ZIP_PREFIXES): - detected = _container_type(source.path) + try: + detected = _container_type(source.path) + except UnsupportedDocumentError: + if declared is not DocumentType.XLSX: + raise + validate_office_package(source.path, document_type=DocumentType.XLSX) + detected = DocumentType.XLSX - declared = _SUFFIX_TYPES.get(suffix) if declared is DocumentType.TEXT or declared is DocumentType.MARKDOWN: if detected is not None: raise _mismatch(suffix, detected) diff --git a/src/opendocs/parsers/office/package.py b/src/opendocs/parsers/office/package.py index 4b0b523..86a38e1 100644 --- a/src/opendocs/parsers/office/package.py +++ b/src/opendocs/parsers/office/package.py @@ -9,6 +9,9 @@ from typing import TypeVar from zipfile import BadZipFile, ZipFile, ZipInfo +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException + from opendocs._models import DocumentType from opendocs.errors import CorruptDocumentError, LimitExceededError from opendocs.source import ParseWorkspace @@ -23,9 +26,18 @@ _REL_NS = "{http://schemas.openxmlformats.org/package/2006/relationships}" _RELATIONSHIP_TAG = f"{_REL_NS}Relationship" +_CONTENT_TYPES_NS = "{http://schemas.openxmlformats.org/package/2006/content-types}" +_CONTENT_TYPE_OVERRIDE_TAG = f"{_CONTENT_TYPES_NS}Override" +_OFFICE_DOCUMENT_RELATIONSHIP = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) +_XLSX_WORKBOOK_CONTENT_TYPE = ( + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" +) _REQUIRED_PARTS = { DocumentType.DOCX: "word/document.xml", DocumentType.PPTX: "ppt/presentation.xml", + DocumentType.XLSX: "xl/workbook.xml", } _MEDIA_SEGMENT = "media" _BINARY_SUFFIXES = { @@ -123,15 +135,32 @@ def _normalize_target(base_dir: str, target: str) -> str: return joined +def _parse_package_xml(data: bytes, *, message: str) -> ET.Element: + try: + return DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (DefusedXmlException, ET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _read_relationships(archive: ZipFile, rels_name: str) -> ET.Element: + try: + data = archive.read(rels_name) + except (KeyError, OSError) as error: + raise CorruptDocumentError("Office package relationships are corrupt") from error + return _parse_package_xml(data, message="Office package relationships are corrupt") + + def _parse_relationship_targets( archive: ZipFile, infos_by_name: dict[str, ZipInfo], rels_name: str, ) -> None: - try: - root = ET.fromstring(archive.read(rels_name)) - except (KeyError, OSError, ET.ParseError) as error: - raise CorruptDocumentError("Office package relationships are corrupt") from error + root = _read_relationships(archive, rels_name) base_dir = _rels_source_base(rels_name) for node in root.iter(_RELATIONSHIP_TAG): target = node.get("Target") @@ -152,23 +181,37 @@ def _required_root_target( if "_rels/.rels" not in infos_by_name: raise CorruptDocumentError("Office package root relationships are missing") required = _REQUIRED_PARTS[document_type] - try: - root = ET.fromstring(archive.read("_rels/.rels")) - except (OSError, ET.ParseError) as error: - raise CorruptDocumentError("Office package relationships are corrupt") from error + root = _read_relationships(archive, "_rels/.rels") for node in root.iter(_RELATIONSHIP_TAG): target = node.get("Target") if target is None or node.get("TargetMode") == "External": continue + if document_type is DocumentType.XLSX and node.get("Type") != _OFFICE_DOCUMENT_RELATIONSHIP: + continue if _normalize_target("", target) == required: return raise CorruptDocumentError("Office package required root relationship is missing") +def _required_xlsx_content_type(archive: ZipFile) -> None: + try: + data = archive.read("[Content_Types].xml") + except (KeyError, OSError) as error: + raise CorruptDocumentError("Office package content types part is corrupt") from error + root = _parse_package_xml(data, message="Office package content types part is corrupt") + for node in root.iter(_CONTENT_TYPE_OVERRIDE_TAG): + if ( + node.get("PartName") == "/xl/workbook.xml" + and node.get("ContentType") == _XLSX_WORKBOOK_CONTENT_TYPE + ): + return + raise CorruptDocumentError("Office package required main part content type is missing") + + def validate_office_package(path: Path, *, document_type: DocumentType) -> OfficePackageLayout: required_main_part = _REQUIRED_PARTS.get(document_type) if required_main_part is None: - raise ValueError("document_type must be DOCX or PPTX") + raise ValueError("document_type must be DOCX, PPTX, or XLSX") try: with ZipFile(path) as archive: infos = archive.infolist() @@ -210,6 +253,8 @@ def validate_office_package(path: Path, *, document_type: DocumentType) -> Offic raise CorruptDocumentError("Office package content types part is missing") if required_main_part not in infos_by_name: raise CorruptDocumentError("Office package required main part is missing") + if document_type is DocumentType.XLSX: + _required_xlsx_content_type(archive) _required_root_target(archive, infos_by_name, document_type) for rels_name in tuple(name for name in infos_by_name if name.endswith(".rels")): _parse_relationship_targets(archive, infos_by_name, rels_name) diff --git a/src/opendocs/parsers/registry.py b/src/opendocs/parsers/registry.py index 7e7ccb0..3088e6f 100644 --- a/src/opendocs/parsers/registry.py +++ b/src/opendocs/parsers/registry.py @@ -69,6 +69,7 @@ def build_default_registry( from opendocs.parsers.image import ImageParser from opendocs.parsers.office.parser import OfficeParser from opendocs.parsers.pdf.parser import PDFParser + from opendocs.parsers.xlsx import XlsxParser registry.register(DocumentType.IMAGE, ImageParser(runtime, vision, vision_config)) registry.register( @@ -83,4 +84,5 @@ def build_default_registry( DocumentType.PPTX, OfficeParser(DocumentType.PPTX, runtime, vision, vision_config, deadline=deadline), ) + registry.register(DocumentType.XLSX, XlsxParser()) return registry diff --git a/src/opendocs/parsers/xlsx/__init__.py b/src/opendocs/parsers/xlsx/__init__.py new file mode 100644 index 0000000..6da30ec --- /dev/null +++ b/src/opendocs/parsers/xlsx/__init__.py @@ -0,0 +1,25 @@ +from __future__ import annotations + +import asyncio + +from opendocs._models import DocumentType, ParsedDocument +from opendocs.errors import UnsupportedDocumentError +from opendocs.options import ParseOptions +from opendocs.parsers.office.package import validate_office_package +from opendocs.source import ResolvedSource + + +class XlsxParser: + async def parse( + self, + source: ResolvedSource, + *, + options: ParseOptions, + ) -> ParsedDocument: + del options + await asyncio.to_thread( + validate_office_package, + source.path, + document_type=DocumentType.XLSX, + ) + raise UnsupportedDocumentError("XLSX content parsing is not implemented in this release") diff --git a/tests/test_detection.py b/tests/test_detection.py index a8c5a0d..d5c0fab 100644 --- a/tests/test_detection.py +++ b/tests/test_detection.py @@ -1,4 +1,6 @@ +import io from pathlib import Path +from typing import Any from zipfile import ZIP_DEFLATED, ZipFile import pytest @@ -11,7 +13,8 @@ ) from opendocs._models import DocumentType from opendocs.detection import detect_document_type -from opendocs.source import ResolvedSource +from opendocs.source import ResolvedSource, Source, materialize_source +from tests.xlsx_fixtures import write_xlsx, xlsx_bytes def _resolved(path: Path, name: str | None = None) -> ResolvedSource: @@ -94,6 +97,84 @@ def test_container_detection_prefers_docx_when_both_markers_exist(tmp_path: Path assert detect_document_type(_resolved(path)) is DocumentType.DOCX +class _NamedBytesIO(io.BytesIO): + def __init__(self, data: bytes, name: str) -> None: + super().__init__(data) + self.name = name + + +@pytest.mark.asyncio +@pytest.mark.parametrize("source_kind", ["path", "bytes", "named_stream", "unnamed_stream"]) +async def test_detects_xlsx_across_all_public_input_shapes( + tmp_path: Path, + source_kind: str, +) -> None: + path = tmp_path / "workbook.xlsx" + content = xlsx_bytes() + sources: dict[str, Source] = { + "path": path, + "bytes": content, + "named_stream": _NamedBytesIO(content, str(path)), + "unnamed_stream": io.BytesIO(content), + } + path.write_bytes(content) + + async with materialize_source(sources[source_kind]) as resolved: + assert detect_document_type(resolved) is DocumentType.XLSX + + +@pytest.mark.parametrize( + "kwargs", + [ + {"include_workbook": False}, + {"include_content_types": False}, + {"workbook_content_type": "application/xml"}, + {"include_root_relationships": False}, + {"root_target": "xl/missing.xml"}, + ], +) +def test_xlsx_identity_requires_content_type_root_relationship_and_workbook_part( + tmp_path: Path, + kwargs: dict[str, Any], +) -> None: + path = tmp_path / "workbook.xlsx" + write_xlsx(path, **kwargs) + + with pytest.raises(CorruptDocumentError): + detect_document_type(_resolved(path, "workbook.xlsx")) + + +@pytest.mark.parametrize( + ("name", "member"), + [ + ("workbook.xlsx", "word/document.xml"), + ("workbook.xlsx", "ppt/presentation.xml"), + ], +) +def test_xlsx_suffix_cannot_disguise_other_office_containers( + tmp_path: Path, + name: str, + member: str, +) -> None: + path = tmp_path / "office" + with ZipFile(path, "w", ZIP_DEFLATED) as archive: + archive.writestr("[Content_Types].xml", "") + archive.writestr(member, "") + + with pytest.raises(DocumentTypeMismatchError): + detect_document_type(_resolved(path, name)) + + +def test_xlsx_container_cannot_disguise_itself_with_another_office_suffix( + tmp_path: Path, +) -> None: + path = tmp_path / "office" + write_xlsx(path) + + with pytest.raises(DocumentTypeMismatchError): + detect_document_type(_resolved(path, "disguised.docx")) + + def test_unnamed_utf8_bytes_default_to_text(tmp_path: Path) -> None: path = tmp_path / "source" path.write_bytes("中文内容".encode()) diff --git a/tests/test_models.py b/tests/test_models.py index 810ed06..557b237 100644 --- a/tests/test_models.py +++ b/tests/test_models.py @@ -35,6 +35,7 @@ def test_document_type_exposes_stable_values() -> None: assert DocumentType.IMAGE == "image" assert DocumentType.DOCX == "docx" assert DocumentType.PPTX == "pptx" + assert DocumentType.XLSX == "xlsx" def test_parsed_document_preserves_block_order_and_tuple_storage() -> None: diff --git a/tests/test_office_package.py b/tests/test_office_package.py index 1d4d892..c6540ef 100644 --- a/tests/test_office_package.py +++ b/tests/test_office_package.py @@ -2,8 +2,9 @@ import os import zipfile -from collections.abc import Callable +from collections.abc import Callable, Sequence from pathlib import Path +from typing import Any from zipfile import ZIP_DEFLATED, ZipFile import pytest @@ -23,9 +24,10 @@ validate_office_package, ) from opendocs.source import ParseWorkspace +from tests.xlsx_fixtures import minimal_xlsx_entries, write_xlsx -def _write_zip(path: Path, entries: list[tuple[str | zipfile.ZipInfo, bytes]]) -> None: +def _write_zip(path: Path, entries: Sequence[tuple[str | zipfile.ZipInfo, bytes]]) -> None: with ZipFile(path, "w", ZIP_DEFLATED) as archive: for name_or_info, data in entries: archive.writestr(name_or_info, data) @@ -87,6 +89,159 @@ def test_validate_office_package_accepts_minimal_docx_and_pptx(tmp_path: Path) - assert pptx_layout.main_part_name == "ppt/presentation.xml" +def test_validate_office_package_accepts_minimal_xlsx(tmp_path: Path) -> None: + xlsx = tmp_path / "sample.xlsx" + write_xlsx(xlsx) + + layout = validate_office_package(xlsx, document_type=DocumentType.XLSX) + + assert layout.main_part_name == "xl/workbook.xml" + + +@pytest.mark.parametrize( + "kwargs", + [ + {"include_content_types": False}, + {"workbook_content_type": "application/xml"}, + {"include_root_relationships": False}, + {"root_target": "xl/missing.xml"}, + {"root_target_mode": "External"}, + {"include_workbook": False}, + ], +) +def test_validate_xlsx_requires_internal_root_relationship_content_type_and_main_part( + tmp_path: Path, + kwargs: dict[str, Any], +) -> None: + path = tmp_path / "sample.xlsx" + write_xlsx(path, **kwargs) + + with pytest.raises(CorruptDocumentError): + validate_office_package(path, document_type=DocumentType.XLSX) + + +@pytest.mark.parametrize("relationship_kind", ["root", "ordinary"]) +@pytest.mark.parametrize( + "payload", + [ + b"""]> + + +""", + b"""]> + + +""", + b""" +""", + ], + ids=["internal-entity", "external-entity", "dtd"], +) +def test_relationship_xml_rejects_dtd_and_entities_with_typed_corruption( + tmp_path: Path, + relationship_kind: str, + payload: bytes, +) -> None: + path = tmp_path / "sample.xlsx" + entries = minimal_xlsx_entries() + if relationship_kind == "root": + entries = [(name, payload if name == "_rels/.rels" else data) for name, data in entries] + else: + entries.append(("xl/_rels/workbook.xml.rels", payload)) + _write_zip(path, entries) + + with pytest.raises(CorruptDocumentError, match="relationships are corrupt"): + validate_office_package(path, document_type=DocumentType.XLSX) + + +@pytest.mark.parametrize( + ("entries", "error_type", "message"), + [ + ( + [*minimal_xlsx_entries(), ("../escape.xml", b"")], + CorruptDocumentError, + "unsafe member", + ), + ( + [*minimal_xlsx_entries(), ("xl/workbook.xml", b"")], + CorruptDocumentError, + "duplicate", + ), + ( + [ + *minimal_xlsx_entries(), + ( + "xl/_rels/workbook.xml.rels", + b""" + +""", + ), + ], + CorruptDocumentError, + "target is missing", + ), + ], +) +def test_validate_xlsx_rejects_unsafe_duplicate_and_dangling_members( + tmp_path: Path, + entries: list[tuple[str, bytes]], + error_type: type[Exception], + message: str, +) -> None: + path = tmp_path / "sample.xlsx" + _write_zip(path, entries) + + with pytest.raises(error_type, match=message): + validate_office_package(path, document_type=DocumentType.XLSX) + + +def test_validate_xlsx_rejects_encrypted_members( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "sample.xlsx" + write_xlsx(path) + original_infolist = ZipFile.infolist + + def flagged_infolist(self: ZipFile) -> list[zipfile.ZipInfo]: + infos = original_infolist(self) + for info in infos: + if info.filename == "xl/workbook.xml": + info.flag_bits |= 0x1 + return infos + + monkeypatch.setattr("opendocs.parsers.office.package.ZipFile.infolist", flagged_infolist) + + with pytest.raises(CorruptDocumentError, match="encrypted"): + validate_office_package(path, document_type=DocumentType.XLSX) + + +def test_validate_xlsx_reuses_member_count_limit(tmp_path: Path) -> None: + path = tmp_path / "sample.xlsx" + _write_zip( + path, + [ + *minimal_xlsx_entries(), + *[(f"xl/parts/part-{index}.xml", b"") for index in range(MAX_ARCHIVE_MEMBERS)], + ], + ) + + with pytest.raises(LimitExceededError, match="member count"): + validate_office_package(path, document_type=DocumentType.XLSX) + + +def test_validate_xlsx_reuses_compression_ratio_limit(tmp_path: Path) -> None: + path = tmp_path / "sample.xlsx" + entries = [ + (name, b"x" * (2 * 1024 * 1024) if name == "xl/workbook.xml" else data) + for name, data in minimal_xlsx_entries() + ] + _write_zip(path, entries) + + with pytest.raises(LimitExceededError, match="compression ratio"): + validate_office_package(path, document_type=DocumentType.XLSX) + + def test_pptx_page_limit_is_checked_from_main_part_before_extraction(tmp_path: Path) -> None: path = tmp_path / "sample.pptx" _write_zip( diff --git a/tests/test_registry.py b/tests/test_registry.py index 1bbc37a..a60173d 100644 --- a/tests/test_registry.py +++ b/tests/test_registry.py @@ -1,14 +1,19 @@ from __future__ import annotations +from pathlib import Path from typing import Any, cast -from opendocs import ParseOptions, UnsupportedDocumentError +import pytest + +from opendocs import CorruptDocumentError, ParseOptions, UnsupportedDocumentError from opendocs._models import DocumentType, ParsedDocument, TextBlock from opendocs._runtime import ParserRuntime from opendocs.parsers.base import DocumentParser from opendocs.parsers.office.parser import OfficeParser from opendocs.parsers.registry import ParserRegistry, build_default_registry +from opendocs.parsers.xlsx import XlsxParser from opendocs.source import ParseWorkspace, ResolvedSource +from tests.xlsx_fixtures import write_xlsx class StubParser: @@ -146,7 +151,41 @@ def test_injected_default_registry_registers_all_core_types(tmp_path) -> None: assert registry.get(DocumentType.PDF) assert isinstance(registry.get(DocumentType.DOCX), OfficeParser) assert isinstance(registry.get(DocumentType.PPTX), OfficeParser) + assert isinstance(registry.get(DocumentType.XLSX), XlsxParser) finally: import asyncio asyncio.run(runtime.aclose()) + + +def test_default_registry_without_runtime_keeps_binary_formats_unavailable() -> None: + registry = build_default_registry() + + for document_type in ( + DocumentType.IMAGE, + DocumentType.PDF, + DocumentType.DOCX, + DocumentType.PPTX, + DocumentType.XLSX, + ): + with pytest.raises(UnsupportedDocumentError, match=document_type.value): + registry.get(document_type) + + +@pytest.mark.asyncio +async def test_xlsx_parser_seam_is_callable_and_prevalidates_the_package(tmp_path) -> None: + parser = XlsxParser() + valid = tmp_path / "valid.xlsx" + write_xlsx(valid) + + with pytest.raises(UnsupportedDocumentError, match="XLSX content parsing"): + await parser.parse(_resolved(valid), options=ParseOptions()) + + corrupt = tmp_path / "corrupt.xlsx" + corrupt.write_bytes(b"not-a-zip") + with pytest.raises(CorruptDocumentError): + await parser.parse(_resolved(corrupt), options=ParseOptions()) + + +def _resolved(path: Path) -> ResolvedSource: + return ResolvedSource(path=path, original_name="workbook.xlsx", owned=False) diff --git a/tests/xlsx_fixtures.py b/tests/xlsx_fixtures.py new file mode 100644 index 0000000..54477d9 --- /dev/null +++ b/tests/xlsx_fixtures.py @@ -0,0 +1,99 @@ +from __future__ import annotations + +import io +from pathlib import Path +from zipfile import ZIP_DEFLATED, ZipFile + +XLSX_CONTENT_TYPE = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" +OFFICE_DOCUMENT_RELATIONSHIP = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) + + +def minimal_xlsx_entries( + *, + include_content_types: bool = True, + include_root_relationships: bool = True, + include_workbook: bool = True, + workbook_content_type: str = XLSX_CONTENT_TYPE, + root_target: str = "xl/workbook.xml", + root_target_mode: str | None = None, +) -> list[tuple[str, bytes]]: + entries: list[tuple[str, bytes]] = [] + if include_content_types: + entries.append( + ( + "[Content_Types].xml", + f""" + + + +""".encode(), + ) + ) + if include_root_relationships: + target_mode = f' TargetMode="{root_target_mode}"' if root_target_mode else "" + entries.append( + ( + "_rels/.rels", + f""" + + + +""".encode(), + ) + ) + if include_workbook: + entries.append( + ( + "xl/workbook.xml", + b"", + ) + ) + return entries + + +def xlsx_bytes( + *, + include_content_types: bool = True, + include_root_relationships: bool = True, + include_workbook: bool = True, + workbook_content_type: str = XLSX_CONTENT_TYPE, + root_target: str = "xl/workbook.xml", + root_target_mode: str | None = None, +) -> bytes: + output = io.BytesIO() + with ZipFile(output, "w", ZIP_DEFLATED) as archive: + for name, data in minimal_xlsx_entries( + include_content_types=include_content_types, + include_root_relationships=include_root_relationships, + include_workbook=include_workbook, + workbook_content_type=workbook_content_type, + root_target=root_target, + root_target_mode=root_target_mode, + ): + archive.writestr(name, data) + return output.getvalue() + + +def write_xlsx( + path: Path, + *, + include_content_types: bool = True, + include_root_relationships: bool = True, + include_workbook: bool = True, + workbook_content_type: str = XLSX_CONTENT_TYPE, + root_target: str = "xl/workbook.xml", + root_target_mode: str | None = None, +) -> None: + path.write_bytes( + xlsx_bytes( + include_content_types=include_content_types, + include_root_relationships=include_root_relationships, + include_workbook=include_workbook, + workbook_content_type=workbook_content_type, + root_target=root_target, + root_target_mode=root_target_mode, + ) + ) diff --git a/uv.lock b/uv.lock index dabe0e5..c0261fc 100644 --- a/uv.lock +++ b/uv.lock @@ -431,6 +431,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/aa/50/a9caea39ad19c431c1a3f8a31114df65b260cdfe67786b6c7e7c040c4c44/cryptography-49.0.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:be9fcb48a55f023493482827d4f459bd263cc20efde64f204b97c123201850c6", size = 3783731, upload-time = "2026-06-12T20:02:43.319Z" }, ] +[[package]] +name = "defusedxml" +version = "0.7.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/0f/d5/c66da9b79e5bdb124974bfe172b4daf3c984ebd9c2a06e2b8a4dc7331c72/defusedxml-0.7.1.tar.gz", hash = "sha256:1bb3032db185915b62d7c6209c5a8792be6a32ab2fedacc84e01b52c51aa3e69", size = 75520, upload-time = "2021-03-08T10:59:26.269Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/07/6c/aa3f2f849e01cb6a001cd8554a88d4c77c5c1a31c95bdf1cf9301e6d9ef4/defusedxml-0.7.1-py2.py3-none-any.whl", hash = "sha256:a352e7e428770286cc899e2542b6cdaedb2b4953ff269a210103ec58f6198a61", size = 25604, upload-time = "2021-03-08T10:59:24.45Z" }, +] + [[package]] name = "distro" version = "1.9.0" @@ -440,6 +449,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/12/b3/231ffd4ab1fc9d679809f356cebee130ac7daa00d6d6f3206dd4fd137e9e/distro-1.9.0-py3-none-any.whl", hash = "sha256:7bffd925d65168f85027d8da9af6bddab658135b840670a223589bc0c8ef02b2", size = 20277, upload-time = "2023-12-24T09:54:30.421Z" }, ] +[[package]] +name = "et-xmlfile" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d3/38/af70d7ab1ae9d4da450eeec1fa3918940a5fafb9055e934af8d6eb0c2313/et_xmlfile-2.0.0.tar.gz", hash = "sha256:dab3f4764309081ce75662649be815c4c9081e88f0837825f90fd28317d4da54", size = 17234, upload-time = "2024-10-25T17:25:40.039Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c1/8b/5fe2cc11fee489817272089c4203e679c63b570a5aaeb18d852ae3cbba6a/et_xmlfile-2.0.0-py3-none-any.whl", hash = "sha256:7a91720bc756843502c3b7504c77b8fe44217c85c537d85037f0f536151b2caa", size = 18059, upload-time = "2024-10-25T17:25:39.051Z" }, +] + [[package]] name = "fastuuid" version = "0.14.0" @@ -1198,7 +1216,9 @@ name = "opendocs-sdk" version = "0.1.0" source = { editable = "." } dependencies = [ + { name = "defusedxml" }, { name = "litellm" }, + { name = "openpyxl" }, { name = "pdfplumber" }, { name = "pillow" }, { name = "python-docx" }, @@ -1215,7 +1235,9 @@ dev = [ [package.metadata] requires-dist = [ + { name = "defusedxml", specifier = ">=0.7.1,<1" }, { name = "litellm", specifier = ">=1.93,<2" }, + { name = "openpyxl", specifier = ">=3.1.5,<3.2" }, { name = "pdfplumber", specifier = ">=0.11.10,<0.12" }, { name = "pillow", specifier = ">=12.3,<13" }, { name = "python-docx", specifier = ">=1.1.2,<2" }, @@ -1230,6 +1252,18 @@ dev = [ { name = "ty", specifier = ">=0.0.63" }, ] +[[package]] +name = "openpyxl" +version = "3.1.5" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "et-xmlfile" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/3d/f9/88d94a75de065ea32619465d2f77b29a0469500e99012523b91cc4141cd1/openpyxl-3.1.5.tar.gz", hash = "sha256:cf0e3cf56142039133628b5acffe8ef0c12bc902d2aadd3e0fe5878dc08d1050", size = 186464, upload-time = "2024-06-28T14:03:44.161Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c0/da/977ded879c29cbd04de313843e76868e6e13408a94ed6b987245dc7c8506/openpyxl-3.1.5-py2.py3-none-any.whl", hash = "sha256:5282c12b107bffeef825f4617dc029afaf41d0ea60823bbb665ef3079dc79de2", size = 250910, upload-time = "2024-06-28T14:03:41.161Z" }, +] + [[package]] name = "packaging" version = "26.2" From ca7dd7b8eb0507b27f8c065638bd6e7e9320b528 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 14:56:35 +0800 Subject: [PATCH 02/12] =?UTF-8?q?=E5=9C=A8=E5=8A=A0=E8=BD=BD=E5=89=8D?= =?UTF-8?q?=E5=B0=81=E4=BD=8F=20XLSX=20=E7=BB=93=E6=9E=84=E6=94=BE?= =?UTF-8?q?=E5=A4=A7=E8=B7=AF=E5=BE=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 sheet-oriented 严格 wire 建模,并在 openpyxl 介入前完成 OOXML 关系、XML 安全和可放大集合的资源预检。 Constraint: 保持公共 Markdown API 与 8 MiB inline / 12 MiB frame 契约,不引入 page 语义。 Rejected: 依赖 full-mode loader 边加载边发现超限。 Confidence: high Scope-risk: 真实 full-mode 峰值 RSS 仍需 U7 资源特征验证。 Tested: uv run --frozen pytest tests/test_xlsx_models.py tests/test_xlsx_preflight.py tests/test_runtime.py tests/test_models.py tests/test_detection.py tests/test_registry.py tests/test_office_package.py -q Tested: uv run --frozen ruff check . Tested: uv run --frozen ruff format --check . Tested: uv run --frozen ty check src tests Not-tested: 完整 public suite、真实 XLSX、私有 corpus、build --- src/opendocs/parsers/xlsx/__init__.py | 10 +- src/opendocs/parsers/xlsx/models.py | 552 ++++++++ src/opendocs/parsers/xlsx/preflight.py | 1783 ++++++++++++++++++++++++ tests/native_worker_helpers.py | 32 + tests/test_runtime.py | 11 + tests/test_xlsx_models.py | 203 +++ tests/test_xlsx_preflight.py | 759 ++++++++++ tests/xlsx_fixtures.py | 99 ++ 8 files changed, 3442 insertions(+), 7 deletions(-) create mode 100644 src/opendocs/parsers/xlsx/models.py create mode 100644 src/opendocs/parsers/xlsx/preflight.py create mode 100644 tests/test_xlsx_models.py create mode 100644 tests/test_xlsx_preflight.py diff --git a/src/opendocs/parsers/xlsx/__init__.py b/src/opendocs/parsers/xlsx/__init__.py index 6da30ec..6d9f5ba 100644 --- a/src/opendocs/parsers/xlsx/__init__.py +++ b/src/opendocs/parsers/xlsx/__init__.py @@ -2,10 +2,10 @@ import asyncio -from opendocs._models import DocumentType, ParsedDocument +from opendocs._models import ParsedDocument from opendocs.errors import UnsupportedDocumentError from opendocs.options import ParseOptions -from opendocs.parsers.office.package import validate_office_package +from opendocs.parsers.xlsx.preflight import preflight_xlsx from opendocs.source import ResolvedSource @@ -17,9 +17,5 @@ async def parse( options: ParseOptions, ) -> ParsedDocument: del options - await asyncio.to_thread( - validate_office_package, - source.path, - document_type=DocumentType.XLSX, - ) + await asyncio.to_thread(preflight_xlsx, source.path) raise UnsupportedDocumentError("XLSX content parsing is not implemented in this release") diff --git a/src/opendocs/parsers/xlsx/models.py b/src/opendocs/parsers/xlsx/models.py new file mode 100644 index 0000000..1642d26 --- /dev/null +++ b/src/opendocs/parsers/xlsx/models.py @@ -0,0 +1,552 @@ +from __future__ import annotations + +import re +from dataclasses import dataclass, fields, is_dataclass +from enum import Enum, StrEnum +from functools import cache +from pathlib import Path +from typing import Any, TypeAlias, cast, get_type_hints + +import opendocs._models as core_models +from opendocs._models import Block, WarningRecord +from opendocs.errors import LimitExceededError + +MAX_NATIVE_WIRE_ESTIMATE = 8 * 1024 * 1024 + +_SHA256_RE = re.compile(r"^[0-9a-f]{64}$") +_A1_RE = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})(?::([A-Z]{1,3})([1-9][0-9]{0,6}))?$") +_DATACLASS_TYPE = "__xlsx_dataclass__" +_XLSX_MAX_COLUMN = 16_384 +_XLSX_MAX_ROW = 1_048_576 +_WIRE_NODE_OVERHEAD = 32 + + +def _require_int(name: str, value: object) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"{name} must be an int") + return value + + +def _require_string(name: str, value: object) -> str: + if not isinstance(value, str): + raise TypeError(f"{name} must be a str") + return value + + +def _require_non_empty_string(name: str, value: object) -> str: + normalized = _require_string(name, value) + if not normalized: + raise ValueError(f"{name} must not be empty") + if any(ord(character) < 32 or ord(character) == 127 for character in normalized): + raise ValueError(f"{name} must not contain control characters") + return normalized + + +def _require_sheet_name(value: object) -> str: + name = _require_non_empty_string("name", value) + if len(name) > 31 or any(character in "[]:*?/\\" for character in name): + raise ValueError("name must be a valid XLSX sheet name") + return name + + +def _require_optional_string(name: str, value: object) -> str | None: + if value is None: + return None + return _require_string(name, value) + + +def _require_tuple(name: str, value: object) -> tuple[object, ...]: + if not isinstance(value, tuple): + raise TypeError(f"{name} must be a tuple") + return value + + +def _column_number(label: str) -> int: + number = 0 + for character in label: + number = number * 26 + ord(character) - ord("A") + 1 + return number + + +def _require_anchor(name: str, value: object) -> str: + anchor = _require_string(name, value) + match = _A1_RE.fullmatch(anchor) + if match is None: + raise ValueError(f"{name} must be a canonical A1 anchor or range") + start_column = _column_number(match.group(1)) + start_row = int(match.group(2)) + end_column = _column_number(match.group(3) or match.group(1)) + end_row = int(match.group(4) or match.group(2)) + if ( + start_column > _XLSX_MAX_COLUMN + or end_column > _XLSX_MAX_COLUMN + or start_row > _XLSX_MAX_ROW + or end_row > _XLSX_MAX_ROW + or end_column < start_column + or end_row < start_row + ): + raise ValueError(f"{name} must be a canonical A1 anchor or range") + return anchor + + +def _require_basename(name: str, value: object) -> str: + artifact_name = _require_string(name, value) + candidate = Path(artifact_name) + windows_stem = artifact_name.split(".", 1)[0].rstrip(" ").upper() + windows_reserved = windows_stem in { + "CON", + "PRN", + "AUX", + "NUL", + "CONIN$", + "CONOUT$", + } or ( + len(windows_stem) == 4 + and windows_stem[:3] in {"COM", "LPT"} + and windows_stem[3] in "123456789¹²³" + ) + forbidden = '<>:"/\\|?*' + if ( + not artifact_name + or artifact_name[-1] in {" ", "."} + or any( + ord(character) < 32 or ord(character) == 127 or character in forbidden + for character in artifact_name + ) + or candidate.is_absolute() + or candidate.name != artifact_name + or artifact_name in {".", ".."} + or windows_reserved + ): + raise ValueError(f"{name} must be a portable non-empty basename") + return artifact_name + + +def _require_sha256(name: str, value: object) -> str: + digest = _require_string(name, value) + if not _SHA256_RE.fullmatch(digest): + raise ValueError(f"{name} must be a lowercase 64-character SHA-256 hex digest") + return digest + + +def _require_source_index(name: str, value: object) -> int: + source_index = _require_int(name, value) + if source_index < 0: + raise ValueError(f"{name} must be greater than or equal to zero") + return source_index + + +def _block_class_names() -> tuple[str, ...]: + return ( + "TextBlock", + "MarkdownBlock", + "TableBlock", + "InlineText", + "InlineLink", + "ParagraphBlock", + "HeadingBlock", + "ListItemBlock", + "SpannedTableCell", + "SpannedTableBlock", + ) + + +_MODEL_REGISTRY: dict[str, type[Any]] = {} +for _name in (*_block_class_names(), "WarningRecord"): + _class = getattr(core_models, _name, None) + if isinstance(_class, type) and is_dataclass(_class): + _MODEL_REGISTRY[_name] = _class + +_KNOWN_BLOCK_TYPES = tuple( + value for name, value in _MODEL_REGISTRY.items() if name != "WarningRecord" +) + + +def _require_blocks(value: object) -> tuple[Block, ...]: + blocks = _require_tuple("blocks", value) + if not blocks: + raise ValueError("blocks must contain at least one block") + for index, block in enumerate(blocks): + if not isinstance(block, _KNOWN_BLOCK_TYPES): + raise TypeError(f"blocks[{index}] is not a supported block type") + return cast(tuple[Block, ...], blocks) + + +class XlsxSheetKind(StrEnum): + WORKSHEET = "worksheet" + CHARTSHEET = "chartsheet" + + +class XlsxSheetState(StrEnum): + VISIBLE = "visible" + HIDDEN = "hidden" + VERY_HIDDEN = "veryHidden" + + +@dataclass(frozen=True, slots=True) +class XlsxNativeSlot: + source_index: int + anchor: str + blocks: tuple[Block, ...] + + def __post_init__(self) -> None: + object.__setattr__( + self, "source_index", _require_source_index("source_index", self.source_index) + ) + object.__setattr__(self, "anchor", _require_anchor("anchor", self.anchor)) + object.__setattr__(self, "blocks", _require_blocks(self.blocks)) + + +@dataclass(frozen=True, slots=True) +class XlsxImageSlot: + source_index: int + anchor: str + artifact_name: str + content_sha256: str + alt_text: str | None = None + + def __post_init__(self) -> None: + object.__setattr__( + self, "source_index", _require_source_index("source_index", self.source_index) + ) + object.__setattr__(self, "anchor", _require_anchor("anchor", self.anchor)) + object.__setattr__( + self, "artifact_name", _require_basename("artifact_name", self.artifact_name) + ) + object.__setattr__( + self, + "content_sha256", + _require_sha256("content_sha256", self.content_sha256), + ) + object.__setattr__(self, "alt_text", _require_optional_string("alt_text", self.alt_text)) + + +@dataclass(frozen=True, slots=True) +class XlsxChartSlot: + source_index: int + anchor: str + artifact_name: str + content_sha256: str + blocks: tuple[Block, ...] + + def __post_init__(self) -> None: + object.__setattr__( + self, "source_index", _require_source_index("source_index", self.source_index) + ) + object.__setattr__(self, "anchor", _require_anchor("anchor", self.anchor)) + object.__setattr__( + self, "artifact_name", _require_basename("artifact_name", self.artifact_name) + ) + object.__setattr__( + self, + "content_sha256", + _require_sha256("content_sha256", self.content_sha256), + ) + object.__setattr__(self, "blocks", _require_blocks(self.blocks)) + + +XlsxSlot: TypeAlias = XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot + + +@dataclass(frozen=True, slots=True) +class XlsxSheet: + sheet_index: int + name: str + kind: XlsxSheetKind + state: XlsxSheetState + slots: tuple[XlsxSlot, ...] + + def __post_init__(self) -> None: + sheet_index = _require_int("sheet_index", self.sheet_index) + if sheet_index <= 0: + raise ValueError("sheet_index must be greater than zero") + if not isinstance(self.kind, XlsxSheetKind): + raise TypeError("kind must be an XlsxSheetKind") + if not isinstance(self.state, XlsxSheetState): + raise TypeError("state must be an XlsxSheetState") + slots = _require_tuple("slots", self.slots) + seen: set[int] = set() + for index, slot in enumerate(slots): + if not isinstance(slot, XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot): + raise TypeError(f"slots[{index}] must be an XLSX slot") + if slot.source_index in seen: + raise ValueError("XLSX sheet source indexes must be unique") + seen.add(slot.source_index) + object.__setattr__(self, "sheet_index", sheet_index) + object.__setattr__(self, "name", _require_sheet_name(self.name)) + object.__setattr__(self, "slots", cast(tuple[XlsxSlot, ...], slots)) + + +@dataclass(frozen=True, slots=True) +class XlsxDocument: + sheets: tuple[XlsxSheet, ...] + warnings: tuple[WarningRecord, ...] = () + + def __post_init__(self) -> None: + sheets = _require_tuple("sheets", self.sheets) + seen: set[int] = set() + for position, sheet in enumerate(sheets, start=1): + if not isinstance(sheet, XlsxSheet): + raise TypeError(f"sheets[{position - 1}] must be an XlsxSheet") + if sheet.sheet_index in seen or sheet.sheet_index != position: + raise ValueError("XLSX sheet indexes must be unique and preserve source order") + seen.add(sheet.sheet_index) + warnings = _require_tuple("warnings", self.warnings) + for index, warning in enumerate(warnings): + if not isinstance(warning, WarningRecord): + raise TypeError(f"warnings[{index}] must be a WarningRecord") + object.__setattr__(self, "sheets", cast(tuple[XlsxSheet, ...], sheets)) + object.__setattr__(self, "warnings", cast(tuple[WarningRecord, ...], warnings)) + + +def _dataclass_to_wire(value: object) -> dict[str, object]: + return { + _DATACLASS_TYPE: type(value).__name__, + "fields": { + field.name: _value_to_wire(getattr(value, field.name)) + for field in fields(cast(Any, value)) + }, + } + + +def _value_to_wire(value: object) -> object: + if value is None or isinstance(value, bool | int | float | str): + return value + if isinstance(value, Enum): + return value.value + if isinstance(value, tuple): + return tuple(_value_to_wire(item) for item in value) + if is_dataclass(value) and type(value).__name__ in _MODEL_REGISTRY: + return _dataclass_to_wire(value) + raise TypeError(f"XLSX wire value type is not supported: {type(value).__name__}") + + +@cache +def _resolved_field_types(cls: type[Any]) -> dict[str, Any]: + return get_type_hints(cls) + + +def _restore_enum_field(value: object, field_type: object) -> object: + if not isinstance(field_type, type) or not issubclass(field_type, Enum): + return value + if isinstance(value, field_type): + return value + try: + return field_type(value) + except (TypeError, ValueError): + return value + + +def _decode_dataclass(value: dict[str, object]) -> object: + if set(value) != {_DATACLASS_TYPE, "fields"}: + raise ValueError("XLSX dataclass wire is invalid") + class_name = value[_DATACLASS_TYPE] + fields_value = value["fields"] + if not isinstance(class_name, str) or not isinstance(fields_value, dict): + raise ValueError("XLSX dataclass wire is invalid") + cls = _MODEL_REGISTRY.get(class_name) + if cls is None: + raise ValueError("XLSX dataclass type is invalid") + field_names = {field.name for field in fields(cast(Any, cls))} + typed_fields = cast(dict[str, object], fields_value) + if set(typed_fields) != field_names: + raise ValueError("XLSX dataclass wire is invalid") + field_types = _resolved_field_types(cls) + kwargs: dict[str, Any] = { + name: _restore_enum_field(_value_from_wire(item), field_types.get(name)) + for name, item in typed_fields.items() + } + return cls(**kwargs) + + +def _value_from_wire(value: object) -> object: + if value is None or isinstance(value, bool | int | float | str): + return value + if isinstance(value, tuple): + return tuple(_value_from_wire(item) for item in value) + if isinstance(value, dict) and _DATACLASS_TYPE in value: + return _decode_dataclass(cast(dict[str, object], value)) + raise ValueError("XLSX wire value is invalid") + + +def _slot_to_wire(slot: XlsxSlot) -> dict[str, object]: + common: dict[str, object] = { + "source_index": slot.source_index, + "anchor": slot.anchor, + } + if isinstance(slot, XlsxNativeSlot): + return { + "type": "xlsx_native_slot", + **common, + "blocks": tuple(_value_to_wire(block) for block in slot.blocks), + } + if isinstance(slot, XlsxImageSlot): + return { + "type": "xlsx_image_slot", + **common, + "artifact_name": slot.artifact_name, + "content_sha256": slot.content_sha256, + "alt_text": slot.alt_text, + } + return { + "type": "xlsx_chart_slot", + **common, + "artifact_name": slot.artifact_name, + "content_sha256": slot.content_sha256, + "blocks": tuple(_value_to_wire(block) for block in slot.blocks), + } + + +def _slot_from_wire(value: object) -> XlsxSlot: + if not isinstance(value, dict) or not isinstance(value.get("type"), str): + raise ValueError("XLSX slot wire is invalid") + payload = cast(dict[str, object], value) + kind = payload["type"] + if kind == "xlsx_native_slot": + if set(payload) != {"type", "source_index", "anchor", "blocks"}: + raise ValueError("XLSX native slot wire is invalid") + blocks_value = payload["blocks"] + if not isinstance(blocks_value, tuple): + raise ValueError("XLSX native slot blocks are invalid") + return XlsxNativeSlot( + source_index=_require_source_index("source_index", payload["source_index"]), + anchor=_require_anchor("anchor", payload["anchor"]), + blocks=tuple(cast(Block, _value_from_wire(block)) for block in blocks_value), + ) + if kind == "xlsx_image_slot": + if set(payload) != { + "type", + "source_index", + "anchor", + "artifact_name", + "content_sha256", + "alt_text", + }: + raise ValueError("XLSX image slot wire is invalid") + return XlsxImageSlot( + source_index=_require_source_index("source_index", payload["source_index"]), + anchor=_require_anchor("anchor", payload["anchor"]), + artifact_name=_require_basename("artifact_name", payload["artifact_name"]), + content_sha256=_require_sha256("content_sha256", payload["content_sha256"]), + alt_text=_require_optional_string("alt_text", payload["alt_text"]), + ) + if kind == "xlsx_chart_slot": + if set(payload) != { + "type", + "source_index", + "anchor", + "artifact_name", + "content_sha256", + "blocks", + }: + raise ValueError("XLSX chart slot wire is invalid") + blocks_value = payload["blocks"] + if not isinstance(blocks_value, tuple): + raise ValueError("XLSX chart slot blocks are invalid") + return XlsxChartSlot( + source_index=_require_source_index("source_index", payload["source_index"]), + anchor=_require_anchor("anchor", payload["anchor"]), + artifact_name=_require_basename("artifact_name", payload["artifact_name"]), + content_sha256=_require_sha256("content_sha256", payload["content_sha256"]), + blocks=tuple(cast(Block, _value_from_wire(block)) for block in blocks_value), + ) + raise ValueError("XLSX slot type is invalid") + + +def _primitive_wire_estimate(value: object) -> int: + if value is None or isinstance(value, bool | int | float): + return _WIRE_NODE_OVERHEAD + if isinstance(value, str): + return _WIRE_NODE_OVERHEAD + len(value) * 4 + if isinstance(value, Enum): + return _primitive_wire_estimate(value.value) + if isinstance(value, tuple): + return _WIRE_NODE_OVERHEAD + sum(_primitive_wire_estimate(item) for item in value) + if is_dataclass(value) and type(value).__name__ in _MODEL_REGISTRY: + estimate = _WIRE_NODE_OVERHEAD + estimate += _primitive_wire_estimate(_DATACLASS_TYPE) + estimate += _primitive_wire_estimate(type(value).__name__) + estimate += _primitive_wire_estimate("fields") + _WIRE_NODE_OVERHEAD + for field in fields(cast(Any, value)): + estimate += _primitive_wire_estimate(field.name) + estimate += _primitive_wire_estimate(getattr(value, field.name)) + return estimate + raise TypeError(f"XLSX wire value type is not supported: {type(value).__name__}") + + +def _wire_estimate(document: XlsxDocument) -> int: + estimate = 512 + for sheet in document.sheets: + estimate += 512 + len(sheet.name) * 4 + for slot in sheet.slots: + estimate += 512 + len(slot.anchor) * 4 + if isinstance(slot, XlsxNativeSlot | XlsxChartSlot): + estimate += sum(_primitive_wire_estimate(block) for block in slot.blocks) + if isinstance(slot, XlsxImageSlot | XlsxChartSlot): + estimate += 512 + len(slot.artifact_name) * 4 + len(slot.content_sha256) * 4 + if isinstance(slot, XlsxImageSlot) and slot.alt_text is not None: + estimate += len(slot.alt_text) * 4 + estimate += sum(_primitive_wire_estimate(warning) for warning in document.warnings) + return estimate + + +def document_to_wire(document: XlsxDocument) -> dict[str, object]: + if not isinstance(document, XlsxDocument): + raise TypeError("document must be an XlsxDocument") + if _wire_estimate(document) > MAX_NATIVE_WIRE_ESTIMATE: + raise LimitExceededError("XLSX native document exceeds the inline result budget") + return { + "type": "xlsx_document", + "sheets": tuple( + { + "type": "xlsx_sheet", + "sheet_index": sheet.sheet_index, + "name": sheet.name, + "kind": sheet.kind.value, + "state": sheet.state.value, + "slots": tuple(_slot_to_wire(slot) for slot in sheet.slots), + } + for sheet in document.sheets + ), + "warnings": tuple(_value_to_wire(warning) for warning in document.warnings), + } + + +def document_from_wire(value: object) -> XlsxDocument: + if not isinstance(value, dict) or value.get("type") != "xlsx_document": + raise ValueError("XLSX document wire is invalid") + payload = cast(dict[str, object], value) + if set(payload) != {"type", "sheets", "warnings"}: + raise ValueError("XLSX document wire is invalid") + sheets_value = payload["sheets"] + if not isinstance(sheets_value, tuple): + raise ValueError("XLSX document sheets are invalid") + sheets: list[XlsxSheet] = [] + for sheet_value in sheets_value: + if not isinstance(sheet_value, dict) or sheet_value.get("type") != "xlsx_sheet": + raise ValueError("XLSX sheet wire is invalid") + sheet_payload = cast(dict[str, object], sheet_value) + if set(sheet_payload) != {"type", "sheet_index", "name", "kind", "state", "slots"}: + raise ValueError("XLSX sheet wire is invalid") + slots_value = sheet_payload["slots"] + if not isinstance(slots_value, tuple): + raise ValueError("XLSX sheet slots are invalid") + try: + kind = XlsxSheetKind(sheet_payload["kind"]) + state = XlsxSheetState(sheet_payload["state"]) + except (TypeError, ValueError) as error: + raise ValueError("XLSX sheet kind or state is invalid") from error + sheets.append( + XlsxSheet( + sheet_index=_require_int("sheet_index", sheet_payload["sheet_index"]), + name=_require_sheet_name(sheet_payload["name"]), + kind=kind, + state=state, + slots=tuple(_slot_from_wire(slot) for slot in slots_value), + ) + ) + warnings_value = payload["warnings"] + if not isinstance(warnings_value, tuple): + raise ValueError("XLSX document warnings are invalid") + warnings = tuple(cast(WarningRecord, _value_from_wire(item)) for item in warnings_value) + return XlsxDocument(sheets=tuple(sheets), warnings=warnings) diff --git a/src/opendocs/parsers/xlsx/preflight.py b/src/opendocs/parsers/xlsx/preflight.py new file mode 100644 index 0000000..0b96cbd --- /dev/null +++ b/src/opendocs/parsers/xlsx/preflight.py @@ -0,0 +1,1783 @@ +from __future__ import annotations + +import posixpath +import re +from collections.abc import Iterator +from dataclasses import dataclass +from pathlib import Path, PurePosixPath +from typing import IO, Any +from zipfile import BadZipFile, ZipFile, ZipInfo + +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException + +from opendocs._models import DocumentType +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.office.package import validate_office_package +from opendocs.parsers.xlsx.models import XlsxSheetKind, XlsxSheetState + +MAX_SHEETS = 128 +MAX_DECLARED_CELLS = 2_000_000 +MAX_SERIALIZED_CELLS = 200_000 +MAX_NON_EMPTY_CELLS = 50_000 +MAX_MATERIALIZED_GRID_CELLS = 200_000 +MAX_MERGE_RANGES = 10_000 +MAX_MERGE_FOOTPRINT = 50_000 +MAX_SHARED_STRINGS = 100_000 +MAX_SHARED_STRING_CHARS = 1_000_000 +MAX_TABLES = 1_024 +MAX_TABLE_FOOTPRINT = 200_000 +MAX_TABLE_COLUMNS = 10_000 +MAX_HYPERLINKS_AND_COMMENTS = 20_000 +MAX_HYPERLINK_FOOTPRINT = 50_000 +MAX_DRAWING_OBJECTS = 256 +MAX_CHART_CACHE_POINTS = 200_000 +MAX_NATIVE_TEXT_CHARS = 1_000_000 +MAX_STYLE_RECORDS = 50_000 +MAX_NUMBER_FORMATS = 10_000 +MAX_FONTS = 10_000 +MAX_FILLS = 10_000 +MAX_BORDERS = 10_000 +MAX_CELL_STYLE_XFS = 10_000 +MAX_CELL_XFS = 10_000 +MAX_NAMED_CELL_STYLES = 1_000 +MAX_DXFS = 10_000 +MAX_TABLE_STYLES = 1_000 +MAX_CONDITIONAL_FORMATTING_RULES = 20_000 +MAX_CONDITIONAL_FORMATTING_RANGES = 10_000 +MAX_DATA_VALIDATIONS = 10_000 +MAX_DEFINED_NAMES = 10_000 +MAX_PIVOT_CACHES = 128 +MAX_PIVOT_TABLES = 128 +MAX_PIVOT_CACHE_RECORDS = 50_000 +MAX_PIVOT_ITEMS = 200_000 +MAX_CUSTOM_PROPERTIES = 1_000 +MAX_ROW_DIMENSIONS = 50_000 +MAX_COLUMN_DIMENSIONS = 10_000 +MAX_PAGE_BREAKS = 10_000 +MAX_SCENARIOS = 1_000 +MAX_SHEET_VIEWS = 256 +MAX_FILTER_ITEMS = 10_000 +MAX_WORKBOOK_VIEWS = 256 +MAX_EXTERNAL_REFERENCES = 128 +MAX_COMMENT_AUTHORS = 10_000 +MAX_XML_ELEMENTS_PER_PART = 200_000 +MAX_TOTAL_XML_ELEMENTS = 1_000_000 +MAX_PROJECTED_WIRE_BYTES = 8 * 1024 * 1024 +MAX_PROJECTED_WORKBOOK_BYTES = 96 * 1024 * 1024 + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +_WORKSHEET_RELATIONSHIP = f"{_OFFICE_REL_NS}/worksheet" +_CHARTSHEET_RELATIONSHIP = f"{_OFFICE_REL_NS}/chartsheet" +_SHARED_STRINGS_RELATIONSHIP = f"{_OFFICE_REL_NS}/sharedStrings" +_STYLES_RELATIONSHIP = f"{_OFFICE_REL_NS}/styles" +_DRAWING_RELATIONSHIP = f"{_OFFICE_REL_NS}/drawing" +_CHART_RELATIONSHIP = f"{_OFFICE_REL_NS}/chart" +_IMAGE_RELATIONSHIP = f"{_OFFICE_REL_NS}/image" +_COMMENTS_RELATIONSHIP = f"{_OFFICE_REL_NS}/comments" +_TABLE_RELATIONSHIP = f"{_OFFICE_REL_NS}/table" +_HYPERLINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/hyperlink" +_PIVOT_TABLE_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotTable" +_PIVOT_CACHE_DEFINITION_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotCacheDefinition" +_PIVOT_CACHE_RECORDS_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotCacheRecords" +_RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" +_RELATIONSHIP_TAG = f"{{{_PACKAGE_REL_NS}}}Relationship" +_A1_RANGE_RE = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})(?::([A-Z]{1,3})([1-9][0-9]{0,6}))?$") +_MAX_COLUMN = 16_384 +_MAX_ROW = 1_048_576 +_DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +_DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +_CHART_NS = "http://schemas.openxmlformats.org/drawingml/2006/chart" +_CUSTOM_PROPERTIES_NS = "http://schemas.openxmlformats.org/officeDocument/2006/custom-properties" + + +@dataclass(frozen=True, slots=True) +class XlsxUnsupportedObjectRef: + sheet_index: int + source_index: int + kind: str + relationship_id: str + target: str + + +@dataclass(frozen=True, slots=True) +class XlsxPreflightSheet: + sheet_index: int + name: str + kind: XlsxSheetKind + state: XlsxSheetState + part_name: str + declared_cells: int + serialized_cells: int + non_empty_cells: int + unsupported_objects: tuple[XlsxUnsupportedObjectRef, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxPreflight: + sheets: tuple[XlsxPreflightSheet, ...] + date_1904: bool + serialized_cells: int + non_empty_cells: int + native_text_chars: int + projected_wire_bytes: int + projected_workbook_bytes: int + usage: XlsxResourceUsage + + +@dataclass(frozen=True, slots=True) +class XlsxResourceUsage: + serialized_cells: int = 0 + non_empty_cells: int = 0 + materialized_grid_cells: int = 0 + merge_ranges: int = 0 + merge_footprint: int = 0 + shared_strings: int = 0 + shared_string_chars: int = 0 + tables: int = 0 + table_footprint: int = 0 + table_columns: int = 0 + hyperlinks_and_comments: int = 0 + hyperlink_footprint: int = 0 + drawing_objects: int = 0 + chart_cache_points: int = 0 + native_text_chars: int = 0 + style_records: int = 0 + number_formats: int = 0 + fonts: int = 0 + fills: int = 0 + borders: int = 0 + cell_style_xfs: int = 0 + cell_xfs: int = 0 + named_cell_styles: int = 0 + dxfs: int = 0 + table_styles: int = 0 + conditional_formatting_rules: int = 0 + conditional_formatting_ranges: int = 0 + data_validations: int = 0 + defined_names: int = 0 + pivot_caches: int = 0 + pivot_tables: int = 0 + pivot_cache_records: int = 0 + pivot_items: int = 0 + custom_properties: int = 0 + row_dimensions: int = 0 + column_dimensions: int = 0 + page_breaks: int = 0 + scenarios: int = 0 + sheet_views: int = 0 + filter_items: int = 0 + workbook_views: int = 0 + external_references: int = 0 + comment_authors: int = 0 + xml_elements: int = 0 + + +@dataclass(slots=True) +class _ResourceUsage: + serialized_cells: int = 0 + non_empty_cells: int = 0 + materialized_grid_cells: int = 0 + merge_ranges: int = 0 + merge_footprint: int = 0 + shared_strings: int = 0 + shared_string_chars: int = 0 + tables: int = 0 + table_footprint: int = 0 + table_columns: int = 0 + hyperlinks_and_comments: int = 0 + hyperlink_footprint: int = 0 + drawing_objects: int = 0 + chart_cache_points: int = 0 + native_text_chars: int = 0 + style_records: int = 0 + number_formats: int = 0 + fonts: int = 0 + fills: int = 0 + borders: int = 0 + cell_style_xfs: int = 0 + cell_xfs: int = 0 + named_cell_styles: int = 0 + dxfs: int = 0 + table_styles: int = 0 + conditional_formatting_rules: int = 0 + conditional_formatting_ranges: int = 0 + data_validations: int = 0 + defined_names: int = 0 + pivot_caches: int = 0 + pivot_tables: int = 0 + pivot_cache_records: int = 0 + pivot_items: int = 0 + custom_properties: int = 0 + row_dimensions: int = 0 + column_dimensions: int = 0 + page_breaks: int = 0 + scenarios: int = 0 + sheet_views: int = 0 + filter_items: int = 0 + workbook_views: int = 0 + external_references: int = 0 + comment_authors: int = 0 + xml_elements: int = 0 + + def freeze(self) -> XlsxResourceUsage: + return XlsxResourceUsage( + **{name: getattr(self, name) for name in XlsxResourceUsage.__dataclass_fields__} + ) + + +@dataclass(frozen=True, slots=True) +class _Relationship: + relationship_type: str + target: str + external: bool + + +@dataclass(frozen=True, slots=True) +class _WorksheetCounts: + declared_cells: int + serialized_cells: int + non_empty_cells: int + native_text_chars: int + + +def _increment( + usage: _ResourceUsage, + field_name: str, + amount: int, + *, + limit: int, + message: str, +) -> None: + if amount < 0: + raise ValueError("XLSX resource increments must be non-negative") + value = getattr(usage, field_name) + amount + if value > limit: + raise LimitExceededError(message) + setattr(usage, field_name, value) + + +def _add_native_text(usage: _ResourceUsage, characters: int) -> None: + usage.native_text_chars += characters + if usage.native_text_chars > MAX_NATIVE_TEXT_CHARS: + raise LimitExceededError("XLSX exceeds the native text limit") + + +def _relationship_for( + relationships: dict[str, _Relationship], + relationship_id: str | None, + *, + expected_type: str, +) -> _Relationship: + relationship = relationships.get(relationship_id or "") + if relationship is None or relationship.relationship_type != expected_type: + raise CorruptDocumentError("XLSX object relationship is invalid") + return relationship + + +def _record_unsupported_object( + unsupported_objects: list[XlsxUnsupportedObjectRef], + *, + sheet_index: int, + relationship_id: str, + relationship: _Relationship, +) -> None: + unsupported_objects.append( + XlsxUnsupportedObjectRef( + sheet_index=sheet_index, + source_index=len(unsupported_objects), + kind=relationship.relationship_type.rsplit("/", 1)[-1] or "unknown", + relationship_id=relationship_id, + target=relationship.target, + ) + ) + + +def _column_number(label: str) -> int: + number = 0 + for character in label: + number = number * 26 + ord(character) - ord("A") + 1 + return number + + +def _parse_a1_range(value: str, *, message: str) -> tuple[int, int, int, int]: + match = _A1_RANGE_RE.fullmatch(value) + if match is None: + raise CorruptDocumentError(message) + start_column = _column_number(match.group(1)) + start_row = int(match.group(2)) + end_column = _column_number(match.group(3) or match.group(1)) + end_row = int(match.group(4) or match.group(2)) + if ( + start_column > _MAX_COLUMN + or end_column > _MAX_COLUMN + or start_row > _MAX_ROW + or end_row > _MAX_ROW + or end_column < start_column + or end_row < start_row + ): + raise CorruptDocumentError(message) + return start_column, start_row, end_column, end_row + + +def _area(bounds: tuple[int, int, int, int]) -> int: + start_column, start_row, end_column, end_row = bounds + return (end_column - start_column + 1) * (end_row - start_row + 1) + + +def _xml_events(stream: IO[bytes], *, message: str) -> Iterator[tuple[str, Any]]: + try: + yield from DefusedET.iterparse( + stream, + events=("start", "end"), + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _preflight_all_xml_parts( + archive: ZipFile, + infos: dict[str, ZipInfo], + usage: _ResourceUsage, +) -> None: + for part_name in sorted(infos): + if not ( + part_name.endswith(".xml") + or part_name.endswith(".rels") + or part_name == "[Content_Types].xml" + ): + continue + part_elements = 0 + try: + with archive.open(part_name) as stream: + for event, element in _xml_events( + stream, + message=f"XLSX XML part is corrupt: {part_name}", + ): + if event != "end": + continue + part_elements += 1 + if part_elements > MAX_XML_ELEMENTS_PER_PART: + raise LimitExceededError("XLSX XML part exceeds the element limit") + usage.xml_elements += 1 + if usage.xml_elements > MAX_TOTAL_XML_ELEMENTS: + raise LimitExceededError("XLSX exceeds the aggregate XML element limit") + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError(f"XLSX XML part is corrupt: {part_name}") from error + + +def _safe_relationship_target(source_part: str, target: str) -> str: + if not target or target.startswith(("/", "\\")) or "\\" in target: + raise CorruptDocumentError("XLSX relationship target is invalid") + first_parts = PurePosixPath(target).parts[:1] + if first_parts and ":" in first_parts[0]: + raise CorruptDocumentError("XLSX relationship target is invalid") + base = PurePosixPath(source_part).parent.as_posix() + normalized = posixpath.normpath(posixpath.join(base, target)) + if normalized in {"", ".", ".."} or normalized.startswith(("../", "/")): + raise CorruptDocumentError("XLSX relationship target is invalid") + return normalized + + +def _relationships_part(source_part: str) -> str: + path = PurePosixPath(source_part) + return (path.parent / "_rels" / f"{path.name}.rels").as_posix() + + +def _read_relationships( + archive: ZipFile, + infos: dict[str, ZipInfo], + source_part: str, + *, + required: bool, +) -> dict[str, _Relationship]: + relationships_part = _relationships_part(source_part) + if relationships_part not in infos: + if required: + raise CorruptDocumentError("XLSX relationships part is missing") + return {} + relationships: dict[str, _Relationship] = {} + root_seen = False + try: + with archive.open(relationships_part) as stream: + for event, element in _xml_events(stream, message="XLSX relationships part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_PACKAGE_REL_NS}}}Relationships": + raise CorruptDocumentError("XLSX relationships namespace is invalid") + if ( + event == "end" + and isinstance(element.tag, str) + and element.tag.rsplit("}", 1)[-1] == "Relationship" + and element.tag != _RELATIONSHIP_TAG + ): + raise CorruptDocumentError("XLSX relationships namespace is invalid") + if event != "end" or element.tag != _RELATIONSHIP_TAG: + continue + relationship_id = element.get("Id") + relationship_type = element.get("Type") + target = element.get("Target") + target_mode = element.get("TargetMode") + if ( + not relationship_id + or not relationship_type + or not target + or target_mode not in {None, "External"} + ): + raise CorruptDocumentError("XLSX relationship is malformed") + if relationship_id in relationships: + raise CorruptDocumentError("XLSX relationship identifiers must be unique") + external = target_mode == "External" + normalized_target = ( + target if external else _safe_relationship_target(source_part, target) + ) + if not external and normalized_target not in infos: + raise CorruptDocumentError("XLSX relationship target is missing") + relationships[relationship_id] = _Relationship( + relationship_type=relationship_type, + target=normalized_target, + external=external, + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX relationships part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX relationships part is corrupt") + return relationships + + +def _parse_workbook( + archive: ZipFile, + infos: dict[str, ZipInfo], + usage: _ResourceUsage, +) -> tuple[ + tuple[tuple[str, XlsxSheetKind, XlsxSheetState, str], ...], + bool, + tuple[str, ...], + str | None, + str | None, +]: + relationships = _read_relationships( + archive, + infos, + "xl/workbook.xml", + required="xl/_rels/workbook.xml.rels" in infos, + ) + sheet_entries: list[tuple[str, XlsxSheetKind, XlsxSheetState, str]] = [] + sheet_ids: set[int] = set() + names: set[str] = set() + pivot_cache_targets: list[str] = [] + date_1904 = False + root_seen = False + try: + with archive.open("xl/workbook.xml") as stream: + for event, element in _xml_events(stream, message="XLSX workbook part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}workbook": + raise CorruptDocumentError("XLSX workbook namespace is invalid") + if event != "end": + continue + if element.tag == f"{{{_SPREADSHEET_NS}}}workbookPr": + raw_date = element.get("date1904") + if raw_date not in {None, "0", "1", "false", "true"}: + raise CorruptDocumentError("XLSX workbook date system is invalid") + date_1904 = raw_date in {"1", "true"} + elif element.tag == f"{{{_SPREADSHEET_NS}}}definedName": + _increment( + usage, + "defined_names", + 1, + limit=MAX_DEFINED_NAMES, + message="XLSX exceeds the defined name limit", + ) + _add_native_text(usage, len(element.text or "")) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}workbookView": + _increment( + usage, + "workbook_views", + 1, + limit=MAX_WORKBOOK_VIEWS, + message="XLSX exceeds the workbook view limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}externalReference": + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + if element.get(_RELATIONSHIP_ID) not in relationships: + raise CorruptDocumentError("XLSX external reference is invalid") + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}pivotCache": + _increment( + usage, + "pivot_caches", + 1, + limit=MAX_PIVOT_CACHES, + message="XLSX exceeds the pivot cache limit", + ) + relationship = _relationship_for( + relationships, + element.get(_RELATIONSHIP_ID), + expected_type=_PIVOT_CACHE_DEFINITION_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX pivot cache relationship is invalid") + pivot_cache_targets.append(relationship.target) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}sheet": + if len(sheet_entries) >= MAX_SHEETS: + raise LimitExceededError("XLSX exceeds the sheet count limit") + name = element.get("name") + relationship_id = element.get(_RELATIONSHIP_ID) + raw_sheet_id = element.get("sheetId") + raw_state = element.get("state", XlsxSheetState.VISIBLE.value) + if not name or not relationship_id or not raw_sheet_id: + raise CorruptDocumentError("XLSX workbook sheet entry is malformed") + if ( + len(name) > 31 + or any(ord(character) < 32 for character in name) + or any(character in "[]:*?/\\" for character in name) + ): + raise CorruptDocumentError("XLSX workbook sheet name is invalid") + try: + sheet_id = int(raw_sheet_id) + state = XlsxSheetState(raw_state) + except ValueError as error: + raise CorruptDocumentError( + "XLSX workbook sheet entry is malformed" + ) from error + if sheet_id <= 0 or sheet_id in sheet_ids or name.casefold() in names: + raise CorruptDocumentError("XLSX workbook sheet entries must be unique") + relationship = relationships.get(relationship_id) + if relationship is None or relationship.external: + raise CorruptDocumentError("XLSX workbook sheet relationship is invalid") + if relationship.relationship_type == _WORKSHEET_RELATIONSHIP: + kind = XlsxSheetKind.WORKSHEET + elif relationship.relationship_type == _CHARTSHEET_RELATIONSHIP: + kind = XlsxSheetKind.CHARTSHEET + else: + raise CorruptDocumentError( + "XLSX workbook sheet relationship type is invalid" + ) + sheet_ids.add(sheet_id) + names.add(name.casefold()) + sheet_entries.append((name, kind, state, relationship.target)) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX workbook part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX workbook part is corrupt") + if sheet_entries and not relationships: + raise CorruptDocumentError("XLSX relationships part is missing") + if any( + relationship.external + for relationship in relationships.values() + if relationship.relationship_type + in { + _SHARED_STRINGS_RELATIONSHIP, + _STYLES_RELATIONSHIP, + } + ): + raise CorruptDocumentError("XLSX workbook singleton relationship is external") + shared_string_targets = tuple( + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _SHARED_STRINGS_RELATIONSHIP + and not relationship.external + ) + style_targets = tuple( + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _STYLES_RELATIONSHIP and not relationship.external + ) + if len(shared_string_targets) > 1 or len(style_targets) > 1: + raise CorruptDocumentError("XLSX workbook singleton relationships are duplicated") + shared_strings_part = ( + shared_string_targets[0] + if shared_string_targets + else "xl/sharedStrings.xml" + if "xl/sharedStrings.xml" in infos + else None + ) + styles_part = ( + style_targets[0] if style_targets else "xl/styles.xml" if "xl/styles.xml" in infos else None + ) + return ( + tuple(sheet_entries), + date_1904, + tuple(pivot_cache_targets), + shared_strings_part, + styles_part, + ) + + +def _preflight_shared_strings( + archive: ZipFile, + part_name: str | None, + usage: _ResourceUsage, +) -> None: + if part_name is None: + return + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events( + stream, message="XLSX shared strings part is corrupt" + ): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}sst": + raise CorruptDocumentError("XLSX shared strings namespace is invalid") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}si": + _increment( + usage, + "shared_strings", + 1, + limit=MAX_SHARED_STRINGS, + message="XLSX exceeds the shared string item limit", + ) + characters = sum( + len(node.text or "") for node in element.iter(f"{{{_SPREADSHEET_NS}}}t") + ) + _increment( + usage, + "shared_string_chars", + characters, + limit=MAX_SHARED_STRING_CHARS, + message="XLSX exceeds the shared string text limit", + ) + _add_native_text(usage, characters) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX shared strings part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX shared strings part is corrupt") + + +def _preflight_styles( + archive: ZipFile, + part_name: str | None, + usage: _ResourceUsage, +) -> None: + if part_name is None: + return + containers = { + "numFmts": ("number_formats", MAX_NUMBER_FORMATS, "number format"), + "fonts": ("fonts", MAX_FONTS, "font"), + "fills": ("fills", MAX_FILLS, "fill"), + "borders": ("borders", MAX_BORDERS, "border"), + "cellStyleXfs": ("cell_style_xfs", MAX_CELL_STYLE_XFS, "cellStyleXfs"), + "cellXfs": ("cell_xfs", MAX_CELL_XFS, "cellXfs"), + "cellStyles": ("named_cell_styles", MAX_NAMED_CELL_STYLES, "named style"), + "dxfs": ("dxfs", MAX_DXFS, "dxf"), + "tableStyles": ("table_styles", MAX_TABLE_STYLES, "table style"), + } + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX stylesheet part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}styleSheet": + raise CorruptDocumentError("XLSX stylesheet namespace is invalid") + if event != "end" or not isinstance(element.tag, str): + continue + local_name = element.tag.rsplit("}", 1)[-1] + details = containers.get(local_name) + if details is None or not element.tag.startswith(f"{{{_SPREADSHEET_NS}}}"): + continue + field_name, limit, label = details + count = len(element) + _increment( + usage, + field_name, + count, + limit=limit, + message=f"XLSX exceeds the {label} limit", + ) + _increment( + usage, + "style_records", + count, + limit=MAX_STYLE_RECORDS, + message="XLSX exceeds the aggregate style record limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX stylesheet part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX stylesheet part is corrupt") + + +def _preflight_custom_properties( + archive: ZipFile, + infos: dict[str, ZipInfo], + usage: _ResourceUsage, +) -> None: + part_name = "docProps/custom.xml" + if part_name not in infos: + return + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX custom properties are corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_CUSTOM_PROPERTIES_NS}}}Properties": + raise CorruptDocumentError("XLSX custom properties namespace is invalid") + if event == "end" and element.tag == f"{{{_CUSTOM_PROPERTIES_NS}}}property": + _increment( + usage, + "custom_properties", + 1, + limit=MAX_CUSTOM_PROPERTIES, + message="XLSX exceeds the custom property limit", + ) + characters = sum(len(node.text or "") for node in element.iter()) + _add_native_text(usage, characters) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX custom properties are corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX custom properties are corrupt") + + +def _preflight_table( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> int: + root_seen = False + table_reference: str | None = None + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX table part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}table": + raise CorruptDocumentError("XLSX table namespace is invalid") + table_reference = element.get("ref") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}tableColumn": + _increment( + usage, + "table_columns", + 1, + limit=MAX_TABLE_COLUMNS, + message="XLSX exceeds the table column limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX table part is corrupt") from error + if not root_seen or table_reference is None: + raise CorruptDocumentError("XLSX table part is corrupt") + footprint = _area(_parse_a1_range(table_reference, message="XLSX table range is invalid")) + _increment( + usage, + "table_footprint", + footprint, + limit=MAX_TABLE_FOOTPRINT, + message="XLSX exceeds the table footprint limit", + ) + return footprint + + +def _preflight_comments( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX comments part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}comments": + raise CorruptDocumentError("XLSX comments namespace is invalid") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}comment": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX comment anchor is invalid") + bounds = _parse_a1_range(reference, message="XLSX comment anchor is invalid") + if _area(bounds) != 1: + raise CorruptDocumentError("XLSX comment anchor is invalid") + _increment( + usage, + "hyperlinks_and_comments", + 1, + limit=MAX_HYPERLINKS_AND_COMMENTS, + message="XLSX exceeds the hyperlink and comment limit", + ) + characters = sum( + len(node.text or "") for node in element.iter(f"{{{_SPREADSHEET_NS}}}t") + ) + _add_native_text(usage, characters) + element.clear() + elif event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}author": + _increment( + usage, + "comment_authors", + 1, + limit=MAX_COMMENT_AUTHORS, + message="XLSX exceeds the comment author limit", + ) + _add_native_text(usage, len(element.text or "")) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX comments part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX comments part is corrupt") + + +def _preflight_chart( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX chart part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_CHART_NS}}}chartSpace": + raise CorruptDocumentError("XLSX chart namespace is invalid") + if event != "end": + continue + if element.tag == f"{{{_CHART_NS}}}pt": + _increment( + usage, + "chart_cache_points", + 1, + limit=MAX_CHART_CACHE_POINTS, + message="XLSX exceeds the chart cache point limit", + ) + elif element.tag == f"{{{_CHART_NS}}}ptCount": + raw_count = element.get("val") + try: + declared_count = int(raw_count or "") + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache count is invalid") from error + if declared_count < 0 or declared_count > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + elif element.tag in { + f"{{{_CHART_NS}}}v", + f"{{{_CHART_NS}}}f", + f"{{{_DRAWING_MAIN_NS}}}t", + }: + _add_native_text(usage, len(element.text or "")) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX chart part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX chart part is corrupt") + + +def _require_anchor_index(value: str | None, *, maximum: int) -> int: + try: + index = int(value or "") + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if index < 0 or index >= maximum: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + return index + + +def _validate_drawing_anchor(anchor: Any) -> None: + local_name = anchor.tag.rsplit("}", 1)[-1] + if local_name in {"oneCellAnchor", "twoCellAnchor"}: + starts = anchor.findall(f"{{{_DRAWING_NS}}}from") + if len(starts) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + markers = starts + if local_name == "twoCellAnchor": + ends = anchor.findall(f"{{{_DRAWING_NS}}}to") + if len(ends) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + markers = [*markers, *ends] + for marker in markers: + _require_anchor_index( + marker.findtext(f"{{{_DRAWING_NS}}}col"), + maximum=_MAX_COLUMN, + ) + _require_anchor_index( + marker.findtext(f"{{{_DRAWING_NS}}}row"), + maximum=_MAX_ROW, + ) + if local_name == "oneCellAnchor": + extents = anchor.findall(f"{{{_DRAWING_NS}}}ext") + if len(extents) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + try: + cx = int(extents[0].get("cx", "")) + cy = int(extents[0].get("cy", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if cx <= 0 or cy <= 0: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + elif local_name == "absoluteAnchor": + positions = anchor.findall(f"{{{_DRAWING_NS}}}pos") + extents = anchor.findall(f"{{{_DRAWING_NS}}}ext") + if len(positions) != 1 or len(extents) != 1: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + try: + values = tuple( + int(value) + for value in ( + positions[0].get("x", ""), + positions[0].get("y", ""), + extents[0].get("cx", ""), + extents[0].get("cy", ""), + ) + ) + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if values[0] < 0 or values[1] < 0 or values[2] <= 0 or values[3] <= 0: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + else: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + + +def _preflight_drawing( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, + visited_charts: set[str], + sheet_index: int, + unsupported_objects: list[XlsxUnsupportedObjectRef], +) -> None: + relationships = _read_relationships( + archive, + infos, + part_name, + required=_relationships_part(part_name) in infos, + ) + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX drawing part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_DRAWING_NS}}}wsDr": + raise CorruptDocumentError("XLSX drawing namespace is invalid") + if event != "end" or element.tag not in { + f"{{{_DRAWING_NS}}}oneCellAnchor", + f"{{{_DRAWING_NS}}}twoCellAnchor", + f"{{{_DRAWING_NS}}}absoluteAnchor", + }: + continue + _validate_drawing_anchor(element) + _increment( + usage, + "drawing_objects", + 1, + limit=MAX_DRAWING_OBJECTS, + message="XLSX exceeds the drawing object limit", + ) + _add_native_text( + usage, + sum(len(node.text or "") for node in element.iter(f"{{{_DRAWING_MAIN_NS}}}t")), + ) + for node in element.iter(): + for attribute_name, relationship_id in node.attrib.items(): + if not attribute_name.startswith(f"{{{_OFFICE_REL_NS}}}"): + continue + relationship = relationships.get(relationship_id) + if relationship is None or relationship.external: + raise CorruptDocumentError("XLSX drawing relationship is invalid") + local_attribute = attribute_name.rsplit("}", 1)[-1] + local_tag = node.tag.rsplit("}", 1)[-1] + expected_type = None + if local_tag == "chart" and local_attribute == "id": + expected_type = _CHART_RELATIONSHIP + elif local_tag == "blip" and local_attribute in {"embed", "link"}: + expected_type = _IMAGE_RELATIONSHIP + if ( + expected_type is not None + and relationship.relationship_type != expected_type + ): + raise CorruptDocumentError("XLSX drawing relationship type is invalid") + if ( + expected_type == _CHART_RELATIONSHIP + and relationship.target not in visited_charts + ): + visited_charts.add(relationship.target) + _preflight_chart(archive, relationship.target, usage) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX drawing part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX drawing part is corrupt") + known_relationship_types = { + _CHART_RELATIONSHIP, + _IMAGE_RELATIONSHIP, + _HYPERLINK_RELATIONSHIP, + } + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type not in known_relationship_types: + _add_native_text(usage, len(relationship.target)) + _record_unsupported_object( + unsupported_objects, + sheet_index=sheet_index, + relationship_id=relationship_id, + relationship=relationship, + ) + + +def _preflight_pivot_cache( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, +) -> None: + relationships = _read_relationships( + archive, + infos, + part_name, + required=False, + ) + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX pivot cache part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}pivotCacheDefinition": + raise CorruptDocumentError("XLSX pivot cache namespace is invalid") + if event == "end" and element.tag in { + f"{{{_SPREADSHEET_NS}}}cacheField", + f"{{{_SPREADSHEET_NS}}}s", + f"{{{_SPREADSHEET_NS}}}n", + f"{{{_SPREADSHEET_NS}}}d", + f"{{{_SPREADSHEET_NS}}}b", + f"{{{_SPREADSHEET_NS}}}e", + f"{{{_SPREADSHEET_NS}}}m", + }: + _increment( + usage, + "pivot_items", + 1, + limit=MAX_PIVOT_ITEMS, + message="XLSX exceeds the pivot item limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX pivot cache part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX pivot cache part is corrupt") + for relationship in relationships.values(): + if relationship.relationship_type != _PIVOT_CACHE_RECORDS_RELATIONSHIP: + continue + _preflight_pivot_records(archive, relationship.target, usage) + + +def _preflight_pivot_records( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX pivot records are corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}pivotCacheRecords": + raise CorruptDocumentError("XLSX pivot records namespace is invalid") + if event == "end" and element.tag == f"{{{_SPREADSHEET_NS}}}r": + _increment( + usage, + "pivot_cache_records", + 1, + limit=MAX_PIVOT_CACHE_RECORDS, + message="XLSX exceeds the pivot cache record limit", + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX pivot records are corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX pivot records are corrupt") + + +def _worksheet_counts( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, + visited_drawings: set[str], + visited_charts: set[str], + sheet_index: int, + unsupported_objects: list[XlsxUnsupportedObjectRef], +) -> _WorksheetCounts: + relationships = _read_relationships( + archive, + infos, + part_name, + required=False, + ) + declared_bounds: tuple[int, int, int, int] | None = None + serialized_cells = 0 + non_empty_cells = 0 + native_text_chars = 0 + actual_columns: list[int] = [] + actual_rows: list[int] = [] + seen_cells: set[tuple[int, int]] = set() + drawing_targets: list[str] = [] + table_targets: list[str] = [] + pivot_targets: list[str] = [] + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX worksheet part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}worksheet": + raise CorruptDocumentError("XLSX worksheet namespace is invalid") + if event != "end": + continue + if element.tag == f"{{{_SPREADSHEET_NS}}}dimension": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX worksheet dimension is invalid") + declared_bounds = _parse_a1_range( + reference, + message="XLSX worksheet dimension is invalid", + ) + if _area(declared_bounds) > MAX_DECLARED_CELLS: + raise LimitExceededError( + "XLSX worksheet exceeds the declared dimension limit" + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}c": + serialized_cells += 1 + if serialized_cells > MAX_SERIALIZED_CELLS: + raise LimitExceededError("XLSX exceeds the serialized cell limit") + reference = element.get("r") + if reference is None: + raise CorruptDocumentError("XLSX cell coordinate is missing") + column, row, end_column, end_row = _parse_a1_range( + reference, + message="XLSX cell coordinate is invalid", + ) + if column != end_column or row != end_row: + raise CorruptDocumentError("XLSX cell coordinate is invalid") + coordinate = (row, column) + if coordinate in seen_cells: + raise CorruptDocumentError("XLSX cell coordinates must be unique") + seen_cells.add(coordinate) + actual_columns.append(column) + actual_rows.append(row) + value_nodes = tuple( + child + for child in element.iter() + if child is not element + and child.tag + in { + f"{{{_SPREADSHEET_NS}}}v", + f"{{{_SPREADSHEET_NS}}}f", + f"{{{_SPREADSHEET_NS}}}t", + } + ) + if any((node.text or "") != "" for node in value_nodes): + non_empty_cells += 1 + if non_empty_cells > MAX_NON_EMPTY_CELLS: + raise LimitExceededError("XLSX exceeds the non-empty cell limit") + native_text_chars += sum(len(node.text or "") for node in value_nodes) + if native_text_chars > MAX_NATIVE_TEXT_CHARS: + raise LimitExceededError("XLSX exceeds the native text limit") + if element.get("t") == "s": + value_node = element.find(f"{{{_SPREADSHEET_NS}}}v") + try: + shared_string_index = int( + value_node.text if value_node is not None else "" + ) + except (TypeError, ValueError) as error: + raise CorruptDocumentError( + "XLSX shared string index is invalid" + ) from error + if not 0 <= shared_string_index < usage.shared_strings: + raise CorruptDocumentError("XLSX shared string index is invalid") + raw_style = element.get("s") + if raw_style is not None: + try: + style_index = int(raw_style) + except ValueError as error: + raise CorruptDocumentError( + "XLSX cell style index is invalid" + ) from error + if style_index < 0 or style_index >= usage.cell_xfs: + raise CorruptDocumentError("XLSX cell style index is invalid") + _increment( + usage, + "materialized_grid_cells", + 1, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}mergeCell": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX merge range is invalid") + footprint = _area( + _parse_a1_range(reference, message="XLSX merge range is invalid") + ) + _increment( + usage, + "merge_ranges", + 1, + limit=MAX_MERGE_RANGES, + message="XLSX exceeds the merge range limit", + ) + _increment( + usage, + "merge_footprint", + footprint, + limit=MAX_MERGE_FOOTPRINT, + message="XLSX exceeds the merge footprint limit", + ) + _increment( + usage, + "materialized_grid_cells", + footprint, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}cfRule": + _increment( + usage, + "conditional_formatting_rules", + 1, + limit=MAX_CONDITIONAL_FORMATTING_RULES, + message="XLSX exceeds the conditional formatting rule limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}conditionalFormatting": + _increment( + usage, + "conditional_formatting_ranges", + 1, + limit=MAX_CONDITIONAL_FORMATTING_RANGES, + message="XLSX exceeds the conditional formatting range limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}dataValidation": + _increment( + usage, + "data_validations", + 1, + limit=MAX_DATA_VALIDATIONS, + message="XLSX exceeds the data validation limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}row": + raw_row = element.get("r") + if raw_row is not None: + try: + row_number = int(raw_row) + except ValueError as error: + raise CorruptDocumentError("XLSX row index is invalid") from error + if not 1 <= row_number <= _MAX_ROW: + raise CorruptDocumentError("XLSX row index is invalid") + if set(element.attrib) - {"r", "spans"}: + _increment( + usage, + "row_dimensions", + 1, + limit=MAX_ROW_DIMENSIONS, + message="XLSX exceeds the row dimension limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}col": + try: + minimum_column = int(element.get("min", "")) + maximum_column = int(element.get("max", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX column range is invalid") from error + if ( + not 1 <= minimum_column <= _MAX_COLUMN + or not minimum_column <= maximum_column <= _MAX_COLUMN + ): + raise CorruptDocumentError("XLSX column range is invalid") + _increment( + usage, + "column_dimensions", + 1, + limit=MAX_COLUMN_DIMENSIONS, + message="XLSX exceeds the column dimension limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}brk": + _increment( + usage, + "page_breaks", + 1, + limit=MAX_PAGE_BREAKS, + message="XLSX exceeds the page break limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}scenario": + _increment( + usage, + "scenarios", + 1, + limit=MAX_SCENARIOS, + message="XLSX exceeds the scenario limit", + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}sheetView": + _increment( + usage, + "sheet_views", + 1, + limit=MAX_SHEET_VIEWS, + message="XLSX exceeds the sheet view limit", + ) + element.clear() + elif element.tag in { + f"{{{_SPREADSHEET_NS}}}filterColumn", + f"{{{_SPREADSHEET_NS}}}customFilter", + f"{{{_SPREADSHEET_NS}}}filter", + }: + _increment( + usage, + "filter_items", + 1, + limit=MAX_FILTER_ITEMS, + message="XLSX exceeds the filter item limit", + ) + element.clear() + elif element.tag in { + f"{{{_SPREADSHEET_NS}}}oddHeader", + f"{{{_SPREADSHEET_NS}}}oddFooter", + f"{{{_SPREADSHEET_NS}}}evenHeader", + f"{{{_SPREADSHEET_NS}}}evenFooter", + f"{{{_SPREADSHEET_NS}}}firstHeader", + f"{{{_SPREADSHEET_NS}}}firstFooter", + }: + _add_native_text(usage, len(element.text or "")) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}hyperlink": + reference = element.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX hyperlink anchor is invalid") + footprint = _area( + _parse_a1_range(reference, message="XLSX hyperlink anchor is invalid") + ) + _increment( + usage, + "hyperlinks_and_comments", + 1, + limit=MAX_HYPERLINKS_AND_COMMENTS, + message="XLSX exceeds the hyperlink and comment limit", + ) + _increment( + usage, + "hyperlink_footprint", + footprint, + limit=MAX_HYPERLINK_FOOTPRINT, + message="XLSX exceeds the hyperlink footprint limit", + ) + _increment( + usage, + "materialized_grid_cells", + footprint, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + relationship_id = element.get(_RELATIONSHIP_ID) + if relationship_id is not None: + relationship = _relationship_for( + relationships, + relationship_id, + expected_type=_HYPERLINK_RELATIONSHIP, + ) + _add_native_text(usage, len(relationship.target)) + _add_native_text( + usage, + sum( + len(element.get(attribute, "")) + for attribute in ("display", "location", "tooltip") + ), + ) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}drawing": + relationship = _relationship_for( + relationships, + element.get(_RELATIONSHIP_ID), + expected_type=_DRAWING_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX drawing relationship is invalid") + drawing_targets.append(relationship.target) + element.clear() + elif element.tag == f"{{{_SPREADSHEET_NS}}}tablePart": + relationship = _relationship_for( + relationships, + element.get(_RELATIONSHIP_ID), + expected_type=_TABLE_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX table relationship is invalid") + _increment( + usage, + "tables", + 1, + limit=MAX_TABLES, + message="XLSX exceeds the table count limit", + ) + table_targets.append(relationship.target) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX worksheet part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX worksheet part is corrupt") + for relationship in relationships.values(): + if relationship.relationship_type == _COMMENTS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX comments relationship is invalid") + _preflight_comments(archive, relationship.target, usage) + elif relationship.relationship_type == _PIVOT_TABLE_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX pivot table relationship is invalid") + pivot_targets.append(relationship.target) + for target in table_targets: + footprint = _preflight_table(archive, target, usage) + _increment( + usage, + "materialized_grid_cells", + footprint, + limit=MAX_MATERIALIZED_GRID_CELLS, + message="XLSX exceeds the materialized grid limit", + ) + for target in drawing_targets: + if target not in visited_drawings: + visited_drawings.add(target) + _preflight_drawing( + archive, + infos, + target, + usage, + visited_charts, + sheet_index, + unsupported_objects, + ) + for target in pivot_targets: + _increment( + usage, + "pivot_tables", + 1, + limit=MAX_PIVOT_TABLES, + message="XLSX exceeds the pivot table limit", + ) + _preflight_xml_root( + archive, + target, + expected_tag=f"{{{_SPREADSHEET_NS}}}pivotTableDefinition", + message="XLSX pivot table part is corrupt", + ) + known_relationship_types = { + _DRAWING_RELATIONSHIP, + _COMMENTS_RELATIONSHIP, + _TABLE_RELATIONSHIP, + _HYPERLINK_RELATIONSHIP, + _PIVOT_TABLE_RELATIONSHIP, + } + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type in known_relationship_types: + continue + _increment( + usage, + "drawing_objects", + 1, + limit=MAX_DRAWING_OBJECTS, + message="XLSX exceeds the drawing object limit", + ) + _add_native_text(usage, len(relationship.target)) + _record_unsupported_object( + unsupported_objects, + sheet_index=sheet_index, + relationship_id=relationship_id, + relationship=relationship, + ) + if actual_columns: + actual_bounds = ( + min(actual_columns), + min(actual_rows), + max(actual_columns), + max(actual_rows), + ) + if _area(actual_bounds) > MAX_DECLARED_CELLS: + raise LimitExceededError("XLSX worksheet actual coordinates exceed the resource budget") + if declared_bounds is not None: + left, top, right, bottom = declared_bounds + actual_left, actual_top, actual_right, actual_bottom = actual_bounds + if ( + actual_left < left + or actual_top < top + or actual_right > right + or actual_bottom > bottom + ): + raise CorruptDocumentError("XLSX cells fall outside the declared dimension") + return _WorksheetCounts( + declared_cells=_area(declared_bounds) if declared_bounds is not None else 0, + serialized_cells=serialized_cells, + non_empty_cells=non_empty_cells, + native_text_chars=native_text_chars, + ) + + +def _preflight_xml_root( + archive: ZipFile, + part_name: str, + *, + expected_tag: str, + message: str, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message=message): + if not root_seen and event == "start": + root_seen = True + if element.tag != expected_tag: + raise CorruptDocumentError(message) + if event == "end": + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError(message) from error + if not root_seen: + raise CorruptDocumentError(message) + + +def _chartsheet_preflight( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, + usage: _ResourceUsage, + visited_drawings: set[str], + visited_charts: set[str], + sheet_index: int, + unsupported_objects: list[XlsxUnsupportedObjectRef], +) -> None: + relationships = _read_relationships(archive, infos, part_name, required=False) + drawing_ids: list[str] = [] + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX chartsheet part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}chartsheet": + raise CorruptDocumentError("XLSX chartsheet namespace is invalid") + if event == "end": + if element.tag == f"{{{_SPREADSHEET_NS}}}drawing": + relationship_id = element.get(_RELATIONSHIP_ID) + if relationship_id is None: + raise CorruptDocumentError("XLSX chartsheet drawing is invalid") + drawing_ids.append(relationship_id) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX chartsheet part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX chartsheet part is corrupt") + for relationship_id in drawing_ids: + relationship = _relationship_for( + relationships, + relationship_id, + expected_type=_DRAWING_RELATIONSHIP, + ) + if relationship.external: + raise CorruptDocumentError("XLSX chartsheet drawing is invalid") + if relationship.target not in visited_drawings: + visited_drawings.add(relationship.target) + _preflight_drawing( + archive, + infos, + relationship.target, + usage, + visited_charts, + sheet_index, + unsupported_objects, + ) + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type == _DRAWING_RELATIONSHIP: + continue + _increment( + usage, + "drawing_objects", + 1, + limit=MAX_DRAWING_OBJECTS, + message="XLSX exceeds the drawing object limit", + ) + _add_native_text(usage, len(relationship.target)) + _record_unsupported_object( + unsupported_objects, + sheet_index=sheet_index, + relationship_id=relationship_id, + relationship=relationship, + ) + + +def _projected_budgets( + *, + sheet_count: int, + usage: _ResourceUsage, +) -> tuple[int, int]: + projected_wire = ( + sheet_count * 1_024 + + usage.serialized_cells * 256 + + usage.non_empty_cells * 1_024 + + usage.materialized_grid_cells * 128 + + usage.drawing_objects * 2_048 + + usage.chart_cache_points * 512 + + usage.hyperlinks_and_comments * 1_024 + + usage.native_text_chars * 8 + ) + projected_workbook = ( + sheet_count * 16_384 + + usage.serialized_cells * 512 + + usage.materialized_grid_cells * 256 + + usage.shared_strings * 128 + + usage.style_records * 1_024 + + usage.table_columns * 512 + + usage.conditional_formatting_rules * 1_024 + + usage.conditional_formatting_ranges * 512 + + usage.data_validations * 1_024 + + usage.defined_names * 512 + + usage.pivot_caches * 16_384 + + usage.pivot_tables * 16_384 + + usage.pivot_cache_records * 512 + + usage.pivot_items * 256 + + usage.custom_properties * 1_024 + + usage.row_dimensions * 512 + + usage.column_dimensions * 512 + + usage.page_breaks * 256 + + usage.scenarios * 2_048 + + usage.sheet_views * 2_048 + + usage.filter_items * 512 + + usage.workbook_views * 2_048 + + usage.external_references * 2_048 + + usage.comment_authors * 256 + + usage.drawing_objects * 32_768 + + usage.chart_cache_points * 512 + + usage.native_text_chars * 4 + + usage.xml_elements * 128 + ) + if projected_wire > MAX_PROJECTED_WIRE_BYTES: + raise LimitExceededError("XLSX projected native wire exceeds the inline result budget") + if projected_workbook > MAX_PROJECTED_WORKBOOK_BYTES: + raise LimitExceededError("XLSX projected workbook exceeds the resource budget") + return projected_wire, projected_workbook + + +def preflight_xlsx(path: Path) -> XlsxPreflight: + validate_office_package(path, document_type=DocumentType.XLSX) + try: + with ZipFile(path) as archive: + infos = {info.filename: info for info in archive.infolist()} + usage = _ResourceUsage() + _preflight_all_xml_parts(archive, infos, usage) + ( + sheet_entries, + date_1904, + pivot_cache_targets, + shared_strings_part, + styles_part, + ) = _parse_workbook( + archive, + infos, + usage, + ) + _preflight_shared_strings(archive, shared_strings_part, usage) + _preflight_styles(archive, styles_part, usage) + _preflight_custom_properties(archive, infos, usage) + for target in pivot_cache_targets: + _preflight_pivot_cache(archive, infos, target, usage) + sheets: list[XlsxPreflightSheet] = [] + serialized_cells = 0 + non_empty_cells = 0 + visited_drawings: set[str] = set() + visited_charts: set[str] = set() + for sheet_index, (name, kind, state, part_name) in enumerate( + sheet_entries, + start=1, + ): + unsupported_objects: list[XlsxUnsupportedObjectRef] = [] + if kind is XlsxSheetKind.WORKSHEET: + counts = _worksheet_counts( + archive, + infos, + part_name, + usage, + visited_drawings, + visited_charts, + sheet_index, + unsupported_objects, + ) + else: + _chartsheet_preflight( + archive, + infos, + part_name, + usage, + visited_drawings, + visited_charts, + sheet_index, + unsupported_objects, + ) + counts = _WorksheetCounts(0, 0, 0, 0) + serialized_cells += counts.serialized_cells + non_empty_cells += counts.non_empty_cells + if serialized_cells > MAX_SERIALIZED_CELLS: + raise LimitExceededError("XLSX exceeds the serialized cell limit") + if non_empty_cells > MAX_NON_EMPTY_CELLS: + raise LimitExceededError("XLSX exceeds the non-empty cell limit") + usage.serialized_cells = serialized_cells + usage.non_empty_cells = non_empty_cells + _add_native_text(usage, counts.native_text_chars) + sheets.append( + XlsxPreflightSheet( + sheet_index=sheet_index, + name=name, + kind=kind, + state=state, + part_name=part_name, + declared_cells=counts.declared_cells, + serialized_cells=counts.serialized_cells, + non_empty_cells=counts.non_empty_cells, + unsupported_objects=tuple(unsupported_objects), + ) + ) + projected_wire, projected_workbook = _projected_budgets( + sheet_count=len(sheets), + usage=usage, + ) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + return XlsxPreflight( + sheets=tuple(sheets), + date_1904=date_1904, + serialized_cells=serialized_cells, + non_empty_cells=non_empty_cells, + native_text_chars=usage.native_text_chars, + projected_wire_bytes=projected_wire, + projected_workbook_bytes=projected_workbook, + usage=usage.freeze(), + ) diff --git a/tests/native_worker_helpers.py b/tests/native_worker_helpers.py index 868e543..0d4e4e0 100644 --- a/tests/native_worker_helpers.py +++ b/tests/native_worker_helpers.py @@ -52,3 +52,35 @@ def noisy_echo(value: object) -> object: def make_bytes(size: int) -> bytes: return b"x" * size + + +def make_xlsx_wire(block_count: int) -> dict[str, object]: + from opendocs._models import TextBlock + from opendocs.parsers.xlsx.models import ( + XlsxDocument, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, + document_to_wire, + ) + + return document_to_wire( + XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot( + source_index=0, + anchor="A1", + blocks=tuple(TextBlock("x" * 200) for _ in range(block_count)), + ), + ), + ), + ) + ) + ) diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 08764a0..8a1c0fe 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -21,6 +21,7 @@ dependency_versions, echo, make_bytes, + make_xlsx_wire, noisy_echo, raise_corrupt, raise_limit, @@ -214,6 +215,16 @@ async def test_oversized_child_result_maps_to_runtime_dependency_error() -> None await worker.aclose() +@pytest.mark.asyncio +async def test_xlsx_wire_budget_maps_to_limit_before_protocol_failure() -> None: + worker = NativeWorker() + try: + with pytest.raises(LimitExceededError, match="inline result budget"): + await worker.run(make_xlsx_wire, 8_000) + finally: + await worker.aclose() + + def test_frame_limit_is_stricter_than_large_artifact_inputs() -> None: assert MAX_FRAME_BYTES <= 16 * 1024 * 1024 with pytest.raises(ValueError, match="inline values exceed"): diff --git a/tests/test_xlsx_models.py b/tests/test_xlsx_models.py new file mode 100644 index 0000000..765fdad --- /dev/null +++ b/tests/test_xlsx_models.py @@ -0,0 +1,203 @@ +from __future__ import annotations + +from dataclasses import FrozenInstanceError +from typing import Any, cast + +import pytest + +from opendocs._models import PageBreakBlock, TableBlock, TextBlock, WarningRecord +from opendocs._native_protocol import MAX_FRAME_BYTES, encode_message +from opendocs.errors import LimitExceededError +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, + document_from_wire, + document_to_wire, +) + + +def _document() -> XlsxDocument: + return XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Visible", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot( + source_index=0, + anchor="A1:B2", + blocks=( + TextBlock("alpha"), + TableBlock((("head", None), ("value", "tail")), header_rows=0), + ), + ), + XlsxImageSlot( + source_index=1, + anchor="D4", + artifact_name="xlsx-image-1.png", + content_sha256="a" * 64, + alt_text="diagram", + ), + XlsxChartSlot( + source_index=2, + anchor="F5:J20", + artifact_name="xlsx-chart-1.png", + content_sha256="b" * 64, + blocks=(TextBlock("Series: 1, 2, 3"),), + ), + ), + ), + XlsxSheet( + sheet_index=2, + name="Chart", + kind=XlsxSheetKind.CHARTSHEET, + state=XlsxSheetState.VERY_HIDDEN, + slots=(), + ), + ), + warnings=(WarningRecord(code="kept", message="warning"),), + ) + + +def test_xlsx_document_wire_round_trip_is_sheet_oriented_and_strict() -> None: + document = _document() + + wire = document_to_wire(document) + restored = document_from_wire(wire) + + assert restored == document + assert "pages" not in repr(wire) + assert "page_number" not in repr(wire) + assert isinstance(restored.sheets, tuple) + assert isinstance(restored.sheets[0].slots, tuple) + + +def test_xlsx_models_are_frozen_and_require_tuple_collections() -> None: + document = _document() + + with pytest.raises(FrozenInstanceError): + document.sheets[0].__setattr__("slots", ()) + with pytest.raises(TypeError, match="sheets"): + XlsxDocument(sheets=cast(Any, [])) + with pytest.raises(TypeError, match="slots"): + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=cast(Any, []), + ) + with pytest.raises(TypeError, match="supported block"): + XlsxNativeSlot(source_index=0, anchor="A1", blocks=(cast(Any, PageBreakBlock(1)),)) + + +@pytest.mark.parametrize("anchor", ["a1", "$A$1", "A0", "XFE1", "A1048577", "B2:A1"]) +def test_xlsx_slots_reject_noncanonical_or_out_of_range_a1_anchors(anchor: str) -> None: + with pytest.raises(ValueError, match="anchor"): + XlsxNativeSlot(source_index=0, anchor=anchor, blocks=(TextBlock("value"),)) + + +def test_xlsx_models_reject_duplicate_indexes_and_unsafe_artifacts() -> None: + sheet = XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot(source_index=0, anchor="A1", blocks=(TextBlock("one"),)), + XlsxNativeSlot(source_index=1, anchor="A2", blocks=(TextBlock("two"),)), + ), + ) + with pytest.raises(ValueError, match="sheet indexes"): + XlsxDocument(sheets=(sheet, sheet)) + with pytest.raises(ValueError, match="sheet name"): + XlsxSheet( + sheet_index=1, + name="Bad/Name", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=(), + ) + with pytest.raises(ValueError, match="source indexes"): + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=(sheet.slots[0], sheet.slots[0]), + ) + with pytest.raises(ValueError, match="artifact_name"): + XlsxImageSlot( + source_index=0, + anchor="A1", + artifact_name="../escape.png", + content_sha256="a" * 64, + ) + with pytest.raises(ValueError, match="content_sha256"): + XlsxImageSlot( + source_index=0, + anchor="A1", + artifact_name="image.png", + content_sha256="short", + ) + + +def test_xlsx_wire_rejects_unknown_fields_and_non_tuple_collections() -> None: + wire = document_to_wire(_document()) + wire["page"] = 1 + with pytest.raises(ValueError, match="XLSX document wire"): + document_from_wire(wire) + + wire = document_to_wire(_document()) + wire["sheets"] = list(cast(tuple[object, ...], wire["sheets"])) + with pytest.raises(ValueError, match="sheets"): + document_from_wire(wire) + + +def test_xlsx_wire_estimator_rejects_result_before_protocol_encoding() -> None: + blocks = tuple(TextBlock("x" * 200) for _ in range(8_000)) + document = XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=(XlsxNativeSlot(source_index=0, anchor="A1", blocks=blocks),), + ), + ) + ) + + with pytest.raises(LimitExceededError, match="inline result budget"): + document_to_wire(document) + + +def test_xlsx_wire_estimator_keeps_accepted_payload_below_frame_limit() -> None: + document = XlsxDocument( + sheets=( + XlsxSheet( + sheet_index=1, + name="Sheet", + kind=XlsxSheetKind.WORKSHEET, + state=XlsxSheetState.VISIBLE, + slots=( + XlsxNativeSlot( + source_index=0, + anchor="A1", + blocks=tuple(TextBlock("x" * 200) for _ in range(7_156)), + ), + ), + ), + ) + ) + + encoded = encode_message({"version": 1, "value": document_to_wire(document)}) + + assert len(encoded) - 8 < MAX_FRAME_BYTES diff --git a/tests/test_xlsx_preflight.py b/tests/test_xlsx_preflight.py new file mode 100644 index 0000000..b390129 --- /dev/null +++ b/tests/test_xlsx_preflight.py @@ -0,0 +1,759 @@ +from __future__ import annotations + +from pathlib import Path +from zipfile import ZIP_DEFLATED, ZipFile + +import pytest + +import opendocs.parsers.xlsx.preflight as preflight_module +from opendocs.errors import CorruptDocumentError, LimitExceededError, UnsupportedDocumentError +from opendocs.options import ParseOptions +from opendocs.parsers.xlsx import XlsxParser +from opendocs.parsers.xlsx.models import XlsxSheetKind, XlsxSheetState +from opendocs.parsers.xlsx.preflight import MAX_SHEETS, preflight_xlsx +from opendocs.source import ResolvedSource +from tests.xlsx_fixtures import rewrite_xlsx, write_structured_xlsx + +SHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" + + +def _one_sheet(path: Path, *, cells: tuple[str, ...] = ()) -> None: + write_structured_xlsx( + path, + sheets=(("Sheet", "worksheet", "visible", "A1:B2", cells),), + ) + + +def _worksheet(body: str) -> bytes: + return (f'{body}').encode() + + +def _relationships(*items: tuple[str, str, str]) -> bytes: + values = "".join( + f'' + for relationship_id, relationship_type, target in items + ) + return f'{values}'.encode() + + +def test_preflight_preserves_worksheet_chartsheet_state_and_empty_sheet_order( + tmp_path: Path, +) -> None: + path = tmp_path / "ordered.xlsx" + write_structured_xlsx( + path, + sheets=( + ("Visible", "worksheet", "visible", "A1", ("A1",)), + ("Hidden", "worksheet", "hidden", None, ()), + ("Very Hidden", "worksheet", "veryHidden", "C3", ("C3",)), + ("Chart", "chartsheet", "visible", None, ()), + ), + ) + + result = preflight_xlsx(path) + + assert [(sheet.sheet_index, sheet.name) for sheet in result.sheets] == [ + (1, "Visible"), + (2, "Hidden"), + (3, "Very Hidden"), + (4, "Chart"), + ] + assert [sheet.kind for sheet in result.sheets] == [ + XlsxSheetKind.WORKSHEET, + XlsxSheetKind.WORKSHEET, + XlsxSheetKind.WORKSHEET, + XlsxSheetKind.CHARTSHEET, + ] + assert [sheet.state for sheet in result.sheets] == [ + XlsxSheetState.VISIBLE, + XlsxSheetState.HIDDEN, + XlsxSheetState.VERY_HIDDEN, + XlsxSheetState.VISIBLE, + ] + assert result.serialized_cells == 2 + + +def test_preflight_accepts_128_sheets_and_rejects_129(tmp_path: Path) -> None: + accepted = tmp_path / "accepted.xlsx" + rejected = tmp_path / "rejected.xlsx" + sheets = tuple( + (f"Sheet {index}", "worksheet", "visible", None, ()) for index in range(1, MAX_SHEETS + 1) + ) + write_structured_xlsx(accepted, sheets=sheets) + write_structured_xlsx( + rejected, + sheets=(*sheets, ("One too many", "worksheet", "visible", None, ())), + ) + + assert len(preflight_xlsx(accepted).sheets) == MAX_SHEETS + with pytest.raises(LimitExceededError, match="sheet count"): + preflight_xlsx(rejected) + + +def test_sparse_full_grid_dimension_fails_before_any_loader(tmp_path: Path) -> None: + path = tmp_path / "sparse.xlsx" + write_structured_xlsx( + path, + sheets=(("Sparse", "worksheet", "visible", "A1:XFD1048576", ("A1",)),), + ) + + with pytest.raises(LimitExceededError, match="declared dimension"): + preflight_xlsx(path) + + +@pytest.mark.asyncio +async def test_parser_seam_preflights_limits_and_does_not_map_max_pages_to_sheets( + tmp_path: Path, +) -> None: + parser = XlsxParser() + accepted = tmp_path / "two-sheets.xlsx" + rejected = tmp_path / "too-many-sheets.xlsx" + write_structured_xlsx( + accepted, + sheets=( + ("One", "worksheet", "visible", None, ()), + ("Two", "worksheet", "visible", None, ()), + ), + ) + write_structured_xlsx( + rejected, + sheets=tuple( + (f"S{index}", "worksheet", "visible", None, ()) for index in range(MAX_SHEETS + 1) + ), + ) + + with pytest.raises(UnsupportedDocumentError, match="content parsing"): + await parser.parse( + ResolvedSource(accepted, "two-sheets.xlsx", False), + options=ParseOptions(max_pages=1), + ) + with pytest.raises(LimitExceededError, match="sheet count"): + await parser.parse( + ResolvedSource(rejected, "too-many-sheets.xlsx", False), + options=ParseOptions(max_pages=1), + ) + + +@pytest.mark.parametrize( + ("limit_name", "body", "message"), + [ + ( + "MAX_SERIALIZED_CELLS", + '', + "serialized cell", + ), + ( + "MAX_NON_EMPTY_CELLS", + '' + '12', + "non-empty cell", + ), + ( + "MAX_MERGE_RANGES", + '' + '', + "merge range", + ), + ( + "MAX_MERGE_FOOTPRINT", + '', + "merge footprint", + ), + ( + "MAX_CONDITIONAL_FORMATTING_RULES", + '' + '' + '', + "conditional formatting", + ), + ( + "MAX_DATA_VALIDATIONS", + '' + '' + '', + "data validation", + ), + ( + "MAX_ROW_DIMENSIONS", + '', + "row dimension", + ), + ( + "MAX_COLUMN_DIMENSIONS", + '' + '', + "column dimension", + ), + ( + "MAX_PAGE_BREAKS", + '' + '', + "page break", + ), + ( + "MAX_SCENARIOS", + '' + '', + "scenario", + ), + ], +) +def test_worksheet_loader_collections_are_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + limit_name: str, + body: str, + message: str, +) -> None: + path = tmp_path / f"{limit_name}.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, limit_name, 1) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": _worksheet(body)}) + + with pytest.raises(LimitExceededError, match=message): + preflight_xlsx(path) + + +def test_loader_collection_boundary_values_are_accepted(tmp_path: Path) -> None: + path = tmp_path / "loader-boundaries.xlsx" + _one_sheet(path) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + '' + '' + '' + '' + '' + '' + ) + }, + ) + + usage = preflight_xlsx(path).usage + + assert usage.serialized_cells == 1 + assert usage.non_empty_cells == 1 + assert usage.merge_ranges == 1 + assert usage.merge_footprint == 2 + assert usage.conditional_formatting_rules == 1 + assert usage.data_validations == 1 + assert usage.row_dimensions == 1 + assert usage.column_dimensions == 1 + assert usage.page_breaks == 1 + assert usage.scenarios == 1 + + +def test_shared_string_item_and_text_budgets_have_boundary_and_overflow( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "strings.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_SHARED_STRINGS", 2) + monkeypatch.setattr(preflight_module, "MAX_SHARED_STRING_CHARS", 3) + rewrite_xlsx( + path, + { + "xl/sharedStrings.xml": ( + f'abc' + ).encode() + }, + ) + assert preflight_xlsx(path).usage.shared_strings == 2 + + rewrite_xlsx( + path, + {"xl/sharedStrings.xml": (f'abcd').encode()}, + ) + with pytest.raises(LimitExceededError, match="shared string text"): + preflight_xlsx(path) + + +@pytest.mark.parametrize( + ("element", "limit_name", "message"), + [ + ("font", "MAX_FONTS", "font"), + ("fill", "MAX_FILLS", "fill"), + ("border", "MAX_BORDERS", "border"), + ("xf", "MAX_CELL_XFS", "cellXfs"), + ("dxf", "MAX_DXFS", "dxf"), + ], +) +def test_stylesheet_loader_collections_are_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + element: str, + limit_name: str, + message: str, +) -> None: + path = tmp_path / f"{element}.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, limit_name, 1) + container = { + "font": "fonts", + "fill": "fills", + "border": "borders", + "xf": "cellXfs", + "dxf": "dxfs", + }[element] + rewrite_xlsx( + path, + { + "xl/styles.xml": ( + f'<{container} count="2">' + f"<{element}/><{element}/>" + ).encode() + }, + ) + + with pytest.raises(LimitExceededError, match=message): + preflight_xlsx(path) + + +def test_stylesheet_boundary_values_are_counted(tmp_path: Path) -> None: + path = tmp_path / "styles-boundary.xlsx" + _one_sheet(path) + rewrite_xlsx( + path, + { + "xl/styles.xml": ( + f'' + "" + "" + "" + ).encode() + }, + ) + + usage = preflight_xlsx(path).usage + + assert usage.style_records == 8 + assert usage.cell_xfs == 1 + assert usage.dxfs == 1 + + +def test_defined_names_and_custom_properties_are_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "names.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_DEFINED_NAMES", 1) + one_name = ( + f'' + '' + 'Sheet!$A$1' + ).encode() + rewrite_xlsx(path, {"xl/workbook.xml": one_name}) + assert preflight_xlsx(path).usage.defined_names == 1 + + rewrite_xlsx( + path, + { + "xl/workbook.xml": ( + f'' + '' + 'Sheet!$A$1' + 'Sheet!$A$2' + "" + ).encode() + }, + ) + with pytest.raises(LimitExceededError, match="defined name"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_DEFINED_NAMES", 10) + monkeypatch.setattr(preflight_module, "MAX_CUSTOM_PROPERTIES", 1) + rewrite_xlsx( + path, + { + "docProps/custom.xml": ( + b'' + ) + }, + ) + assert preflight_xlsx(path).usage.custom_properties == 1 + + rewrite_xlsx( + path, + { + "docProps/custom.xml": ( + b'' + ) + }, + ) + with pytest.raises(LimitExceededError, match="custom propert"): + preflight_xlsx(path) + + +def test_table_count_and_footprint_are_bounded_before_loader( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "table.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_TABLES", 1) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdT1", f"{OFFICE_REL_NS}/table", "../tables/table1.xml"), + ("rIdT2", f"{OFFICE_REL_NS}/table", "../tables/table2.xml"), + ), + "xl/tables/table1.xml": f''.encode(), + "xl/tables/table2.xml": f'
'.encode(), + }, + ) + with pytest.raises(LimitExceededError, match="table count"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_TABLES", 2) + monkeypatch.setattr(preflight_module, "MAX_TABLE_FOOTPRINT", 8) + assert preflight_xlsx(path).usage.tables == 2 + + monkeypatch.setattr(preflight_module, "MAX_TABLE_FOOTPRINT", 7) + with pytest.raises(LimitExceededError, match="table footprint"): + preflight_xlsx(path) + + +def test_drawing_object_chart_cache_and_anchor_are_preflighted( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "drawing.xlsx" + _one_sheet(path) + drawing_rel = f"{OFFICE_REL_NS}/drawing" + chart_rel = f"{OFFICE_REL_NS}/chart" + drawing_ns = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" + chart_ns = "http://schemas.openxmlformats.org/drawingml/2006/chart" + drawing = ( + f'' + "00" + '' + "" + ).encode() + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdD1", drawing_rel, "../drawings/drawing1.xml"), + ), + "xl/drawings/drawing1.xml": drawing, + "xl/drawings/_rels/drawing1.xml.rels": _relationships( + ("rIdC1", chart_rel, "../charts/chart1.xml"), + ), + "xl/charts/chart1.xml": ( + f'1' + '2' + ).encode(), + }, + ) + monkeypatch.setattr(preflight_module, "MAX_CHART_CACHE_POINTS", 1) + with pytest.raises(LimitExceededError, match="chart cache"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_CHART_CACHE_POINTS", 2) + assert preflight_xlsx(path).usage.drawing_objects == 1 + + monkeypatch.setattr(preflight_module, "MAX_DRAWING_OBJECTS", 0) + with pytest.raises(LimitExceededError, match="drawing object"): + preflight_xlsx(path) + monkeypatch.setattr(preflight_module, "MAX_DRAWING_OBJECTS", 256) + + invalid = drawing.replace(b"0", b"16384") + rewrite_xlsx(path, {"xl/drawings/drawing1.xml": invalid}) + with pytest.raises(CorruptDocumentError, match="drawing anchor"): + preflight_xlsx(path) + + +def test_hyperlink_comment_and_relationship_references_are_bounded_and_resolved( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "links.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 1) + one_link = _worksheet( + '' + '' + ) + one_link_relationship = ( + f'' + f'' + ).encode() + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": one_link, + "xl/worksheets/_rels/sheet1.xml.rels": one_link_relationship, + }, + ) + assert preflight_xlsx(path).usage.hyperlinks_and_comments == 1 + + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + '' + "" + ), + "xl/worksheets/_rels/sheet1.xml.rels": ( + f'' + f'' + f'' + ).encode(), + }, + ) + with pytest.raises(LimitExceededError, match="hyperlink and comment"): + preflight_xlsx(path) + + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + }, + ) + with pytest.raises(CorruptDocumentError, match="relationship"): + preflight_xlsx(path) + + +def test_pivot_cache_collections_are_bounded( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = tmp_path / "pivot.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_PIVOT_CACHES", 1) + one_cache_workbook = ( + f'' + '' + '' + ).encode() + one_cache_relationships = _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ("rIdP1", f"{OFFICE_REL_NS}/pivotCacheDefinition", "pivotCache/a.xml"), + ) + rewrite_xlsx( + path, + { + "xl/workbook.xml": one_cache_workbook, + "xl/_rels/workbook.xml.rels": one_cache_relationships, + "xl/pivotCache/a.xml": (f'').encode(), + }, + ) + assert preflight_xlsx(path).usage.pivot_caches == 1 + + rewrite_xlsx( + path, + { + "xl/workbook.xml": ( + f'' + '' + '' + "" + ).encode(), + "xl/_rels/workbook.xml.rels": _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ("rIdP1", f"{OFFICE_REL_NS}/pivotCacheDefinition", "pivotCache/a.xml"), + ("rIdP2", f"{OFFICE_REL_NS}/pivotCacheDefinition", "pivotCache/b.xml"), + ), + "xl/pivotCache/a.xml": b"", + "xl/pivotCache/b.xml": b"", + }, + ) + with pytest.raises(LimitExceededError, match="pivot cache"): + preflight_xlsx(path) + + +def test_comment_and_pivot_record_limits_apply_before_loader( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "comments-pivot.xlsx" + _one_sheet(path) + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 1) + rewrite_xlsx( + path, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdC1", f"{OFFICE_REL_NS}/comments", "../comments1.xml"), + ), + "xl/comments1.xml": ( + f'A' + 'one' + "" + ).encode(), + }, + ) + assert preflight_xlsx(path).usage.hyperlinks_and_comments == 1 + + rewrite_xlsx( + path, + { + "xl/comments1.xml": ( + f'A' + 'one' + 'two' + ).encode() + }, + ) + with pytest.raises(LimitExceededError, match="hyperlink and comment"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 20_000) + monkeypatch.setattr(preflight_module, "MAX_PIVOT_CACHE_RECORDS", 1) + rewrite_xlsx( + path, + { + "xl/comments1.xml": None, + "xl/worksheets/_rels/sheet1.xml.rels": None, + "xl/pivotCache/cache.xml": f''.encode(), + "xl/pivotCache/_rels/cache.xml.rels": _relationships( + ( + "rIdR1", + f"{OFFICE_REL_NS}/pivotCacheRecords", + "records.xml", + ), + ), + "xl/pivotCache/records.xml": ( + f'' + ).encode(), + "xl/workbook.xml": ( + f'' + '' + '' + ).encode(), + "xl/_rels/workbook.xml.rels": _relationships( + ("rId1", f"{OFFICE_REL_NS}/worksheet", "worksheets/sheet1.xml"), + ( + "rIdP1", + f"{OFFICE_REL_NS}/pivotCacheDefinition", + "pivotCache/cache.xml", + ), + ), + }, + ) + with pytest.raises(LimitExceededError, match="pivot cache record"): + preflight_xlsx(path) + + +def test_unknown_relationship_objects_are_deterministically_locatable(tmp_path: Path) -> None: + path = tmp_path / "unknown.xlsx" + _one_sheet(path) + rewrite_xlsx( + path, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdOle", f"{OFFICE_REL_NS}/oleObject", "../embeddings/ole1.bin"), + ), + "xl/embeddings/ole1.bin": b"opaque", + }, + ) + + unsupported = preflight_xlsx(path).sheets[0].unsupported_objects + + assert [ + (item.source_index, item.kind, item.relationship_id, item.target) for item in unsupported + ] == [(0, "oleObject", "rIdOle", "xl/embeddings/ole1.bin")] + + +@pytest.mark.parametrize( + "xml", + [ + b']>', + b'', + b'', + ], +) +def test_worksheet_dtd_entity_namespace_and_malformed_xml_are_typed_corruption( + tmp_path: Path, + xml: bytes, +) -> None: + path = tmp_path / "corrupt.xlsx" + _one_sheet(path) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": xml}) + + with pytest.raises(CorruptDocumentError): + preflight_xlsx(path) + + +def test_zip_slip_is_rejected_by_preflight(tmp_path: Path) -> None: + path = tmp_path / "zip-slip.xlsx" + _one_sheet(path) + with ZipFile(path, "a", ZIP_DEFLATED) as archive: + archive.writestr("../escape.xml", b"") + + with pytest.raises(CorruptDocumentError, match="unsafe member"): + preflight_xlsx(path) + + +def test_projected_wire_budget_is_a_stricter_success_boundary_than_grid_limit( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "wire.xlsx" + _one_sheet(path, cells=("A1", "B1")) + assert preflight_module.MAX_MATERIALIZED_GRID_CELLS == 200_000 + monkeypatch.setattr(preflight_module, "MAX_PROJECTED_WIRE_BYTES", 2_000) + + with pytest.raises(LimitExceededError, match="inline result budget"): + preflight_xlsx(path) + + +def test_projected_workbook_budget_bounds_full_mode_loader_work( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "projected-workbook.xlsx" + _one_sheet(path, cells=("A1",)) + monkeypatch.setattr(preflight_module, "MAX_PROJECTED_WORKBOOK_BYTES", 1) + + with pytest.raises(LimitExceededError, match="projected workbook"): + preflight_xlsx(path) + + +def test_materialized_grid_and_native_text_have_independent_outer_limits( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "grid-text.xlsx" + _one_sheet(path, cells=("A1", "B1")) + monkeypatch.setattr(preflight_module, "MAX_MATERIALIZED_GRID_CELLS", 1) + with pytest.raises(LimitExceededError, match="materialized grid"): + preflight_xlsx(path) + + monkeypatch.setattr(preflight_module, "MAX_MATERIALIZED_GRID_CELLS", 200_000) + monkeypatch.setattr(preflight_module, "MAX_NATIVE_TEXT_CHARS", 3) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '1234' + ) + }, + ) + with pytest.raises(LimitExceededError, match="native text"): + preflight_xlsx(path) diff --git a/tests/xlsx_fixtures.py b/tests/xlsx_fixtures.py index 54477d9..586b3d7 100644 --- a/tests/xlsx_fixtures.py +++ b/tests/xlsx_fixtures.py @@ -2,6 +2,7 @@ import io from pathlib import Path +from typing import Literal from zipfile import ZIP_DEFLATED, ZipFile XLSX_CONTENT_TYPE = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" @@ -97,3 +98,101 @@ def write_xlsx( root_target_mode=root_target_mode, ) ) + + +def write_structured_xlsx( + path: Path, + *, + sheets: tuple[ + tuple[ + str, + Literal["worksheet", "chartsheet"], + Literal["visible", "hidden", "veryHidden"], + str | None, + tuple[str, ...], + ], + ..., + ], +) -> None: + workbook_sheets: list[str] = [] + relationships: list[str] = [] + entries = minimal_xlsx_entries(include_workbook=False) + for index, (name, kind, state, dimension, cells) in enumerate(sheets, start=1): + relationship_id = f"rId{index}" + workbook_sheets.append( + f'' + ) + folder = "worksheets" if kind == "worksheet" else "chartsheets" + relationship_type = ( + f"http://schemas.openxmlformats.org/officeDocument/2006/relationships/{kind}" + ) + relationships.append( + f'' + ) + if kind == "worksheet": + dimension_xml = f'' if dimension else "" + cell_xml = "".join(f'1' for cell in cells) + entries.append( + ( + f"xl/{folder}/sheet{index}.xml", + ( + '' + '' + f"{dimension_xml}{cell_xml}" + "" + ).encode(), + ) + ) + else: + entries.append( + ( + f"xl/{folder}/sheet{index}.xml", + b'', + ) + ) + entries.extend( + [ + ( + "xl/workbook.xml", + ( + '' + '' + f"{''.join(workbook_sheets)}" + "" + ).encode(), + ), + ( + "xl/_rels/workbook.xml.rels", + ( + '' + '{"".join(relationships)}' + ).encode(), + ), + ] + ) + with ZipFile(path, "w", ZIP_DEFLATED) as archive: + for name, data in entries: + archive.writestr(name, data) + + +def rewrite_xlsx( + path: Path, + replacements: dict[str, bytes | None], +) -> None: + with ZipFile(path) as archive: + entries = {info.filename: archive.read(info) for info in archive.infolist()} + for name, data in replacements.items(): + if data is None: + entries.pop(name, None) + else: + entries[name] = data + with ZipFile(path, "w", ZIP_DEFLATED) as archive: + for name, data in entries.items(): + archive.writestr(name, data) From df30cdea3904a218557793492f24112b6054fb32 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 15:23:29 +0800 Subject: [PATCH 03/12] =?UTF-8?q?=E4=BF=9D=E7=95=99=20XLSX=20=E7=9A=84?= =?UTF-8?q?=E4=BF=9D=E5=AD=98=E5=80=BC=E4=B8=8E=E8=A1=A8=E6=A0=BC=E8=AF=AD?= =?UTF-8?q?=E4=B9=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 在安全预检后单次加载工作簿,用 OOXML sidecar 保留公式缓存与回退,并按源顺序生成表格、区域和合并跨度。 Constraint: 不计算公式、不下载外部引用、不把字体颜色或页面语义写入结果。 Rejected: 通过 data_only 双加载或首行猜测表头。 Confidence: high Scope-risk: 真实工作簿兼容与 full-mode RSS 仍待 U7 私有/资源门验证。 Tested: uv run --frozen pytest tests/test_xlsx_values.py tests/test_xlsx_extract.py tests/test_xlsx_models.py tests/test_xlsx_preflight.py tests/test_office_package.py tests/test_runtime.py -q Tested: uv run --frozen ruff check . Tested: uv run --frozen ruff format --check . Tested: uv run --frozen ty check src tests Not-tested: 完整 public suite、build、真实 XLSX、私有 corpus --- src/opendocs/parsers/office/package.py | 15 +- src/opendocs/parsers/xlsx/extract.py | 703 +++++++++++++++++++++++++ src/opendocs/parsers/xlsx/preflight.py | 31 +- src/opendocs/parsers/xlsx/values.py | 363 +++++++++++++ tests/test_office_package.py | 78 +++ tests/test_xlsx_extract.py | 365 +++++++++++++ tests/test_xlsx_preflight.py | 22 + tests/test_xlsx_values.py | 88 ++++ 8 files changed, 1653 insertions(+), 12 deletions(-) create mode 100644 src/opendocs/parsers/xlsx/extract.py create mode 100644 src/opendocs/parsers/xlsx/values.py create mode 100644 tests/test_xlsx_extract.py create mode 100644 tests/test_xlsx_values.py diff --git a/src/opendocs/parsers/office/package.py b/src/opendocs/parsers/office/package.py index 86a38e1..10a423e 100644 --- a/src/opendocs/parsers/office/package.py +++ b/src/opendocs/parsers/office/package.py @@ -121,16 +121,19 @@ def _rels_source_base(name: str) -> str: def _normalize_target(base_dir: str, target: str) -> str: - if not target or target.startswith(("/", "\\")) or "\\" in target: + if not target or target.startswith(("//", "\\")) or "\\" in target: raise CorruptDocumentError("Office package relationship target is invalid") - if ":" in PurePosixPath(target).parts[:1]: + package_absolute = target.startswith("/") + candidate = target[1:] if package_absolute else target + first_parts = PurePosixPath(candidate).parts[:1] + if not candidate or (first_parts and ":" in first_parts[0]): raise CorruptDocumentError("Office package relationship target is invalid") joined = ( - posixpath.normpath(posixpath.join(base_dir, target)) - if base_dir - else posixpath.normpath(target) + posixpath.normpath(candidate) + if package_absolute or not base_dir + else posixpath.normpath(posixpath.join(base_dir, candidate)) ) - if joined.startswith("../") or joined == ".." or joined.startswith("/"): + if joined in {"", ".", ".."} or joined.startswith(("../", "/")): raise CorruptDocumentError("Office package relationship target is invalid") return joined diff --git a/src/opendocs/parsers/xlsx/extract.py b/src/opendocs/parsers/xlsx/extract.py new file mode 100644 index 0000000..eec82e2 --- /dev/null +++ b/src/opendocs/parsers/xlsx/extract.py @@ -0,0 +1,703 @@ +from __future__ import annotations + +import re +from dataclasses import dataclass +from datetime import datetime +from decimal import Decimal, InvalidOperation +from pathlib import Path +from typing import Any +from zipfile import BadZipFile, ZipFile + +import openpyxl +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException +from openpyxl.utils.cell import coordinate_to_tuple, get_column_letter, range_boundaries + +from opendocs._models import ( + Block, + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + SpannedTableBlock, + SpannedTableCell, + TableBlock, + WarningRecord, +) +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.xlsx.models import ( + XlsxDocument, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, +) +from opendocs.parsers.xlsx.preflight import ( + MAX_MATERIALIZED_GRID_CELLS as PREFLIGHT_MAX_MATERIALIZED_GRID_CELLS, +) +from opendocs.parsers.xlsx.preflight import XlsxPreflight, XlsxPreflightSheet +from opendocs.parsers.xlsx.values import format_saved_value + +MAX_MATERIALIZED_GRID_CELLS = PREFLIGHT_MAX_MATERIALIZED_GRID_CELLS + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_CELL_TAG = f"{{{_SPREADSHEET_NS}}}c" +_FORMULA_TAG = f"{{{_SPREADSHEET_NS}}}f" +_VALUE_TAG = f"{{{_SPREADSHEET_NS}}}v" +_WARNING_LIMIT_PER_CODE = 20 +_EXTERNAL_FORMULA_REFERENCE = re.compile(r"\[[^\]\r\n]+\][^!\r\n]*!") + +_Coordinate = tuple[int, int] +_Bounds = tuple[int, int, int, int] + + +@dataclass(frozen=True, slots=True) +class _FormulaRecord: + formula_type: str + text: str | None + reference: str | None + attributes: tuple[tuple[str, str], ...] + + +@dataclass(frozen=True, slots=True) +class _CellRecord: + coordinate: str + cell_type: str | None + formula: _FormulaRecord | None + cache_present: bool + cached_value: str | None + + +@dataclass(frozen=True, slots=True, order=True) +class _WarningEvent: + code: str + sheet_index: int + row: int + column: int + sheet_name: str + detail: str + + +class _WarningCollector: + def __init__(self) -> None: + self._events: set[_WarningEvent] = set() + + def add( + self, + code: str, + *, + sheet: XlsxPreflightSheet, + coordinate: str, + detail: str, + ) -> None: + row, column = coordinate_to_tuple(coordinate) + self._events.add( + _WarningEvent( + code=code, + sheet_index=sheet.sheet_index, + row=row, + column=column, + sheet_name=sheet.name, + detail=detail, + ) + ) + + def freeze(self) -> tuple[WarningRecord, ...]: + grouped: dict[str, list[_WarningEvent]] = {} + for event in sorted(self._events): + grouped.setdefault(event.code, []).append(event) + warnings: list[WarningRecord] = [] + for code in sorted(grouped): + events = grouped[code] + for event in events[:_WARNING_LIMIT_PER_CODE]: + coordinate = f"{get_column_letter(event.column)}{event.row}" + warnings.append( + WarningRecord( + code=code, + message=f"{event.sheet_name}!{coordinate}: {event.detail}", + ) + ) + suppressed = len(events) - _WARNING_LIMIT_PER_CODE + if suppressed > 0: + warnings.append( + WarningRecord( + code=code, + message=f"{suppressed} additional {code} warnings suppressed", + ) + ) + return tuple(warnings) + + +@dataclass(frozen=True, slots=True) +class _MergeSpec: + bounds: _Bounds + + +@dataclass(frozen=True, slots=True) +class _RegionSpec: + bounds: _Bounds + header_rows: int + merges: tuple[_MergeSpec, ...] + kind: str + + +@dataclass(slots=True) +class _MaterializationBudget: + used: int = 0 + + def consume(self, cells: int) -> None: + self.used += cells + if self.used > MAX_MATERIALIZED_GRID_CELLS: + raise LimitExceededError("XLSX exceeds the materialized grid limit") + + +def _safe_worksheet_xml(data: bytes, *, part_name: str) -> Any: + try: + root = DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(f"XLSX worksheet part is corrupt: {part_name}") from error + if root.tag != f"{{{_SPREADSHEET_NS}}}worksheet": + raise CorruptDocumentError(f"XLSX worksheet part is corrupt: {part_name}") + return root + + +def _formula_record(element: Any) -> _FormulaRecord: + attributes = tuple(sorted((str(name), str(value)) for name, value in element.attrib.items())) + return _FormulaRecord( + formula_type=element.get("t", "normal"), + text=element.text if element.text not in {None, ""} else None, + reference=element.get("ref"), + attributes=attributes, + ) + + +def _worksheet_sidecar(data: bytes, *, part_name: str) -> tuple[_CellRecord, ...]: + root = _safe_worksheet_xml(data, part_name=part_name) + records: list[_CellRecord] = [] + for element in root.iter(_CELL_TAG): + coordinate = element.get("r") + if coordinate is None: + raise CorruptDocumentError("XLSX cell coordinate is missing") + formula_node = element.find(_FORMULA_TAG) + value_node = element.find(_VALUE_TAG) + records.append( + _CellRecord( + coordinate=coordinate, + cell_type=element.get("t"), + formula=_formula_record(formula_node) if formula_node is not None else None, + cache_present=value_node is not None, + cached_value=value_node.text if value_node is not None else None, + ) + ) + return tuple(records) + + +def _read_sidecars( + path: Path, + preflight: XlsxPreflight, +) -> dict[int, tuple[_CellRecord, ...]]: + records: dict[int, tuple[_CellRecord, ...]] = {} + try: + with ZipFile(path) as archive: + for sheet in preflight.sheets: + if sheet.kind is XlsxSheetKind.CHARTSHEET: + records[sheet.sheet_index] = () + continue + try: + data = archive.read(sheet.part_name) + except KeyError as error: + raise CorruptDocumentError("XLSX worksheet part is missing") from error + records[sheet.sheet_index] = _worksheet_sidecar( + data, + part_name=sheet.part_name, + ) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + return records + + +def _decode_cached_value(record: _CellRecord) -> object: + value = record.cached_value + if value is None: + return None + if record.cell_type == "b": + if value not in {"0", "1"}: + raise CorruptDocumentError("XLSX formula boolean cache is invalid") + return value == "1" + if record.cell_type in {"e", "str", "inlineStr"}: + return value + if record.cell_type == "d": + try: + return datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError as error: + raise CorruptDocumentError("XLSX formula date cache is invalid") from error + try: + return Decimal(value) + except InvalidOperation as error: + raise CorruptDocumentError("XLSX formula numeric cache is invalid") from error + + +def _loaded_formula_text(value: object) -> str | None: + if isinstance(value, str): + return value if value.startswith("=") else f"={value}" + text = getattr(value, "text", None) + if isinstance(text, str) and text: + return text if text.startswith("=") else f"={text}" + return None + + +def _literal_formula_text(record: _CellRecord, loaded_value: object) -> str | None: + if record.formula is not None and record.formula.text: + return ( + record.formula.text + if record.formula.text.startswith("=") + else f"={record.formula.text}" + ) + return _loaded_formula_text(loaded_value) + + +def _references_external_workbook(record: _CellRecord, loaded_value: object) -> bool: + formula_text = _literal_formula_text(record, loaded_value) + return formula_text is not None and _EXTERNAL_FORMULA_REFERENCE.search(formula_text) is not None + + +def _record_format_warning( + warning_collector: _WarningCollector, + *, + sheet: XlsxPreflightSheet, + coordinate: str, + warning: str | None, +) -> None: + if warning is None: + return + warning_collector.add( + "xlsx_unsupported_number_format", + sheet=sheet, + coordinate=coordinate, + detail=warning, + ) + + +def _special_formula_text( + record: _CellRecord, + loaded_value: object, + *, + saved_value: str, +) -> str | None: + formula = record.formula + if formula is None: + return None + reference = formula.reference or record.coordinate + if formula.formula_type == "array": + expression = _literal_formula_text(record, loaded_value) or "(expression unavailable)" + rendered = f"Array/spill formula {reference}: {expression}" + if record.cache_present and saved_value: + rendered += f"; saved value: {saved_value}" + return rendered + if formula.formula_type != "dataTable": + return None + parameters = [ + f"{name}={value}" for name, value in formula.attributes if name not in {"t", "ref"} + ] + suffix = f" ({', '.join(parameters)})" if parameters else "" + rendered = f"Data-table formula {reference}{suffix}" + if record.cache_present and saved_value: + rendered += f"; saved value: {saved_value}" + return rendered + + +def _formula_text( + record: _CellRecord, + loaded_value: object, + *, + number_format: str, + epoch: datetime, + conditional_number_format: bool, + sheet: XlsxPreflightSheet, + warnings: _WarningCollector, +) -> str: + if _references_external_workbook(record, loaded_value): + warnings.add( + "xlsx_external_reference", + sheet=sheet, + coordinate=record.coordinate, + detail="external workbook formula was preserved without access", + ) + saved_value = "" + if record.cache_present: + formatted = format_saved_value( + _decode_cached_value(record), + number_format, + epoch=epoch, + conditional_number_format=conditional_number_format, + ) + saved_value = formatted.text + _record_format_warning( + warnings, + sheet=sheet, + coordinate=record.coordinate, + warning=formatted.warning, + ) + + special = _special_formula_text(record, loaded_value, saved_value=saved_value) + if special is not None: + if record.formula is not None and record.formula.formula_type == "dataTable": + warnings.add( + "xlsx_data_table_formula", + sheet=sheet, + coordinate=record.coordinate, + detail="data-table formula has no portable literal expression", + ) + if not record.cache_present: + warnings.add( + "xlsx_formula_cache_missing", + sheet=sheet, + coordinate=record.coordinate, + detail="formula has no saved cache; formula text was preserved", + ) + return special + if record.cache_present: + return saved_value + warnings.add( + "xlsx_formula_cache_missing", + sheet=sheet, + coordinate=record.coordinate, + detail="formula has no saved cache; formula text was preserved", + ) + return _literal_formula_text(record, loaded_value) or "Formula expression unavailable" + + +def _conditional_number_format_ranges(worksheet: Any) -> tuple[Any, ...]: + ranges: list[Any] = [] + for conditional in worksheet.conditional_formatting: + rules = worksheet.conditional_formatting[conditional] + changes_number_format = False + for rule in rules: + dxf = getattr(rule, "dxf", None) + if dxf is not None: + changes_number_format |= getattr(dxf, "numFmt", None) is not None + elif getattr(rule, "dxfId", None) is not None: + changes_number_format = True + if changes_number_format: + ranges.append(conditional.sqref) + return tuple(ranges) + + +def _has_conditional_number_format(coordinate: str, ranges: tuple[Any, ...]) -> bool: + return any(coordinate in cell_range for cell_range in ranges) + + +def _cell_texts( + worksheet: Any, + records: tuple[_CellRecord, ...], + *, + epoch: datetime, + sheet: XlsxPreflightSheet, + warnings: _WarningCollector, +) -> tuple[dict[_Coordinate, str], set[_Coordinate]]: + texts: dict[_Coordinate, str] = {} + semantic: set[_Coordinate] = set() + conditional_ranges = _conditional_number_format_ranges(worksheet) + for record in records: + cell = worksheet[record.coordinate] + conditional_number_format = _has_conditional_number_format( + record.coordinate, + conditional_ranges, + ) + if record.formula is not None: + text = _formula_text( + record, + cell.value, + number_format=cell.number_format, + epoch=epoch, + conditional_number_format=conditional_number_format, + sheet=sheet, + warnings=warnings, + ) + else: + formatted = format_saved_value( + cell.value, + cell.number_format, + epoch=epoch, + conditional_number_format=conditional_number_format, + ) + text = formatted.text + _record_format_warning( + warnings, + sheet=sheet, + coordinate=record.coordinate, + warning=formatted.warning, + ) + coordinate = coordinate_to_tuple(record.coordinate) + texts[coordinate] = text + if text != "": + semantic.add(coordinate) + return texts, semantic + + +def _area(bounds: _Bounds) -> int: + minimum_column, minimum_row, maximum_column, maximum_row = bounds + return (maximum_column - minimum_column + 1) * (maximum_row - minimum_row + 1) + + +def _coordinates(bounds: _Bounds) -> set[_Coordinate]: + minimum_column, minimum_row, maximum_column, maximum_row = bounds + return { + (row, column) + for row in range(minimum_row, maximum_row + 1) + for column in range(minimum_column, maximum_column + 1) + } + + +def _anchor(bounds: _Bounds) -> str: + minimum_column, minimum_row, maximum_column, maximum_row = bounds + start = f"{get_column_letter(minimum_column)}{minimum_row}" + end = f"{get_column_letter(maximum_column)}{maximum_row}" + return start if start == end else f"{start}:{end}" + + +def _merge_specs(worksheet: Any) -> tuple[_MergeSpec, ...]: + return tuple( + sorted( + ( + _MergeSpec(range_boundaries(str(cell_range))) + for cell_range in worksheet.merged_cells.ranges + ), + key=lambda item: item.bounds, + ) + ) + + +def _table_specs(worksheet: Any, merges: tuple[_MergeSpec, ...]) -> tuple[_RegionSpec, ...]: + tables = sorted(worksheet.tables.values(), key=lambda table: range_boundaries(table.ref)) + specs: list[_RegionSpec] = [] + for table in tables: + bounds = range_boundaries(table.ref) + table_merges = tuple(merge for merge in merges if _bounds_within(merge.bounds, bounds)) + specs.append( + _RegionSpec( + bounds=bounds, + header_rows=1 if (table.headerRowCount or 0) > 0 else 0, + merges=table_merges, + kind="table", + ) + ) + return tuple(specs) + + +def _bounds_within(inner: _Bounds, outer: _Bounds) -> bool: + inner_left, inner_top, inner_right, inner_bottom = inner + outer_left, outer_top, outer_right, outer_bottom = outer + return ( + outer_left <= inner_left <= inner_right <= outer_right + and outer_top <= inner_top <= inner_bottom <= outer_bottom + ) + + +def _component_specs( + semantic: set[_Coordinate], + merges: tuple[_MergeSpec, ...], + occupied: set[_Coordinate], +) -> tuple[_RegionSpec, ...]: + available = set(semantic) - occupied + available_merges: list[_MergeSpec] = [] + for merge in merges: + footprint = _coordinates(merge.bounds) - occupied + if not footprint: + continue + available.update(footprint) + available_merges.append(merge) + + remaining = set(available) + specs: list[_RegionSpec] = [] + for seed in sorted(available): + if seed not in remaining: + continue + stack = [seed] + remaining.remove(seed) + component: set[_Coordinate] = set() + while stack: + row, column = stack.pop() + component.add((row, column)) + for neighbor in ( + (row - 1, column), + (row + 1, column), + (row, column - 1), + (row, column + 1), + ): + if neighbor in remaining: + remaining.remove(neighbor) + stack.append(neighbor) + rows = [row for row, _ in component] + columns = [column for _, column in component] + bounds = (min(columns), min(rows), max(columns), max(rows)) + component_merges = tuple( + merge for merge in available_merges if _bounds_within(merge.bounds, bounds) + ) + specs.append(_RegionSpec(bounds, 0, component_merges, "region")) + return tuple(sorted(specs, key=lambda item: item.bounds)) + + +def _region_specs( + worksheet: Any, + semantic: set[_Coordinate], +) -> tuple[_RegionSpec, ...]: + merges = _merge_specs(worksheet) + table_specs = _table_specs(worksheet, merges) + occupied: set[_Coordinate] = set() + for table in table_specs: + occupied.update(_coordinates(table.bounds)) + component_specs = _component_specs(semantic, merges, occupied) + return tuple(sorted((*table_specs, *component_specs), key=lambda item: item.bounds)) + + +def _region_block( + spec: _RegionSpec, texts: dict[_Coordinate, str] +) -> TableBlock | SpannedTableBlock: + minimum_column, minimum_row, maximum_column, maximum_row = spec.bounds + row_count = maximum_row - minimum_row + 1 + column_count = maximum_column - minimum_column + 1 + if not spec.merges: + return TableBlock( + tuple( + tuple( + texts.get((row, column), "") + for column in range(minimum_column, maximum_column + 1) + ) + for row in range(minimum_row, maximum_row + 1) + ), + spec.header_rows, + ) + + merge_by_origin = {(merge.bounds[1], merge.bounds[0]): merge for merge in spec.merges} + covered: set[_Coordinate] = set() + for merge in spec.merges: + covered.update(_coordinates(merge.bounds)) + cells: list[SpannedTableCell] = [] + for row in range(minimum_row, maximum_row + 1): + for column in range(minimum_column, maximum_column + 1): + merge = merge_by_origin.get((row, column)) + if merge is not None: + left, top, right, bottom = merge.bounds + cells.append( + SpannedTableCell( + row - minimum_row, + column - minimum_column, + bottom - top + 1, + right - left + 1, + texts.get((row, column), ""), + ) + ) + elif (row, column) in covered: + continue + else: + cells.append( + SpannedTableCell( + row - minimum_row, + column - minimum_column, + 1, + 1, + texts.get((row, column), ""), + ) + ) + return SpannedTableBlock(row_count, column_count, tuple(cells), spec.header_rows) + + +def _sheet_prelude(sheet: XlsxPreflightSheet) -> XlsxNativeSlot: + return XlsxNativeSlot( + source_index=0, + anchor="A1", + blocks=( + MarkdownBlock(f""), + HeadingBlock(1, (InlineText(sheet.name),)), + ParagraphBlock((InlineText(f"Sheet state: {sheet.state.value}"),)), + ), + ) + + +def _sheet_slots( + worksheet: Any, + records: tuple[_CellRecord, ...], + *, + epoch: datetime, + sheet: XlsxPreflightSheet, + warnings: _WarningCollector, + budget: _MaterializationBudget, +) -> tuple[XlsxNativeSlot, ...]: + texts, semantic = _cell_texts( + worksheet, + records, + epoch=epoch, + sheet=sheet, + warnings=warnings, + ) + slots: list[XlsxNativeSlot] = [_sheet_prelude(sheet)] + for source_index, spec in enumerate(_region_specs(worksheet, semantic), start=1): + budget.consume(_area(spec.bounds)) + anchor = _anchor(spec.bounds) + comment = ( + f"" + ) + block: Block = _region_block(spec, texts) + slots.append( + XlsxNativeSlot( + source_index=source_index, + anchor=anchor, + blocks=(MarkdownBlock(comment), block), + ) + ) + return tuple(slots) + + +def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: + if not isinstance(path, Path): + raise TypeError("path must be a Path") + if not isinstance(preflight, XlsxPreflight): + raise TypeError("preflight must be an XlsxPreflight") + sidecars = _read_sidecars(path, preflight) + warnings = _WarningCollector() + workbook = openpyxl.load_workbook( + path, + read_only=False, + data_only=False, + rich_text=False, + keep_links=False, + ) + try: + worksheets = {worksheet.title: worksheet for worksheet in workbook.worksheets} + sheets: list[XlsxSheet] = [] + budget = _MaterializationBudget() + for sheet in preflight.sheets: + if sheet.kind is XlsxSheetKind.CHARTSHEET: + slots = (_sheet_prelude(sheet),) + else: + worksheet = worksheets.get(sheet.name) + if worksheet is None: + raise CorruptDocumentError("XLSX worksheet is missing after full-mode load") + slots = _sheet_slots( + worksheet, + sidecars[sheet.sheet_index], + epoch=workbook.epoch, + sheet=sheet, + warnings=warnings, + budget=budget, + ) + sheets.append( + XlsxSheet( + sheet_index=sheet.sheet_index, + name=sheet.name, + kind=sheet.kind, + state=sheet.state, + slots=slots, + ) + ) + finally: + workbook.close() + return XlsxDocument(sheets=tuple(sheets), warnings=warnings.freeze()) diff --git a/src/opendocs/parsers/xlsx/preflight.py b/src/opendocs/parsers/xlsx/preflight.py index 0b96cbd..d6e966b 100644 --- a/src/opendocs/parsers/xlsx/preflight.py +++ b/src/opendocs/parsers/xlsx/preflight.py @@ -373,13 +373,21 @@ def _preflight_all_xml_parts( def _safe_relationship_target(source_part: str, target: str) -> str: - if not target or target.startswith(("/", "\\")) or "\\" in target: + if not target or target.startswith(("//", "\\")) or "\\" in target: raise CorruptDocumentError("XLSX relationship target is invalid") - first_parts = PurePosixPath(target).parts[:1] + package_absolute = target.startswith("/") + candidate = target[1:] if package_absolute else target + if not candidate: + raise CorruptDocumentError("XLSX relationship target is invalid") + first_parts = PurePosixPath(candidate).parts[:1] if first_parts and ":" in first_parts[0]: raise CorruptDocumentError("XLSX relationship target is invalid") base = PurePosixPath(source_part).parent.as_posix() - normalized = posixpath.normpath(posixpath.join(base, target)) + normalized = ( + posixpath.normpath(candidate) + if package_absolute + else posixpath.normpath(posixpath.join(base, candidate)) + ) if normalized in {"", ".", ".."} or normalized.startswith(("../", "/")): raise CorruptDocumentError("XLSX relationship target is invalid") return normalized @@ -895,7 +903,7 @@ def _require_anchor_index(value: str | None, *, maximum: int) -> int: return index -def _validate_drawing_anchor(anchor: Any) -> None: +def _validate_drawing_anchor(anchor: Any, *, allow_zero_absolute_extent: bool) -> None: local_name = anchor.tag.rsplit("}", 1)[-1] if local_name in {"oneCellAnchor", "twoCellAnchor"}: starts = anchor.findall(f"{{{_DRAWING_NS}}}from") @@ -944,7 +952,12 @@ def _validate_drawing_anchor(anchor: Any) -> None: ) except ValueError as error: raise CorruptDocumentError("XLSX drawing anchor is invalid") from error - if values[0] < 0 or values[1] < 0 or values[2] <= 0 or values[3] <= 0: + invalid_extent = ( + values[2] < 0 or values[3] < 0 + if allow_zero_absolute_extent + else values[2] <= 0 or values[3] <= 0 + ) + if values[0] < 0 or values[1] < 0 or invalid_extent: raise CorruptDocumentError("XLSX drawing anchor is invalid") else: raise CorruptDocumentError("XLSX drawing anchor is invalid") @@ -958,6 +971,8 @@ def _preflight_drawing( visited_charts: set[str], sheet_index: int, unsupported_objects: list[XlsxUnsupportedObjectRef], + *, + allow_zero_absolute_extent: bool = False, ) -> None: relationships = _read_relationships( archive, @@ -979,7 +994,10 @@ def _preflight_drawing( f"{{{_DRAWING_NS}}}absoluteAnchor", }: continue - _validate_drawing_anchor(element) + _validate_drawing_anchor( + element, + allow_zero_absolute_extent=allow_zero_absolute_extent, + ) _increment( usage, "drawing_objects", @@ -1613,6 +1631,7 @@ def _chartsheet_preflight( visited_charts, sheet_index, unsupported_objects, + allow_zero_absolute_extent=True, ) for relationship_id, relationship in relationships.items(): if relationship.relationship_type == _DRAWING_RELATIONSHIP: diff --git a/src/opendocs/parsers/xlsx/values.py b/src/opendocs/parsers/xlsx/values.py new file mode 100644 index 0000000..7c2942a --- /dev/null +++ b/src/opendocs/parsers/xlsx/values.py @@ -0,0 +1,363 @@ +from __future__ import annotations + +import math +import re +from dataclasses import dataclass +from datetime import date, datetime, time, timedelta +from decimal import ROUND_HALF_UP, Decimal, InvalidOperation + +from openpyxl.styles.numbers import is_date_format, is_timedelta_format +from openpyxl.utils.datetime import from_excel + +_CURRENCY_SYMBOLS = frozenset({"$", "€", "£", "¥"}) +_BRACKET_TOKEN = re.compile(r"\[([^]]+)]") +_SCIENTIFIC_FORMAT = re.compile(r"[0#?](?:\.[0#?]+)?[Ee][+-]?[0#?]+") +_FRACTION_FORMAT = re.compile(r"(?:^|[^A-Za-z])[0#?]+(?:\s+[0#?]+)?/[0#?]+") +_LOCALE_CURRENCY = re.compile(r"^\$([^\]-]*)-[0-9A-Fa-f]+$") +_QUOTED_LITERAL = re.compile(r'"([^"]*)"') + + +@dataclass(frozen=True, slots=True) +class FormattedSavedValue: + text: str + warning: str | None = None + + +def _stable_decimal(value: Decimal) -> str: + if not value.is_finite(): + return str(value) + rendered = format(value, "f") + if "." in rendered: + rendered = rendered.rstrip("0").rstrip(".") + return rendered or "0" + + +def stable_raw_value(value: object) -> str: + if value is None: + return "" + if isinstance(value, bool): + return "TRUE" if value else "FALSE" + if isinstance(value, datetime): + return value.isoformat(sep=" ", timespec="seconds") + if isinstance(value, date): + return value.isoformat() + if isinstance(value, time): + return value.isoformat(timespec="seconds") + if isinstance(value, timedelta): + return _stable_decimal(Decimal(str(value.total_seconds()))) + if isinstance(value, Decimal): + return _stable_decimal(value) + if isinstance(value, int): + return str(value) + if isinstance(value, float): + if not math.isfinite(value): + return str(value) + return _stable_decimal(Decimal(str(value))) + return str(value) + + +def _split_sections(number_format: str) -> tuple[str, ...]: + sections: list[str] = [] + current: list[str] = [] + quoted = False + bracket_depth = 0 + escaped = False + for character in number_format: + if escaped: + current.append(character) + escaped = False + elif character == "\\": + current.append(character) + escaped = True + elif character == '"': + current.append(character) + quoted = not quoted + elif not quoted and character == "[": + bracket_depth += 1 + current.append(character) + elif not quoted and character == "]": + bracket_depth = max(0, bracket_depth - 1) + current.append(character) + elif not quoted and bracket_depth == 0 and character == ";": + sections.append("".join(current)) + current = [] + else: + current.append(character) + sections.append("".join(current)) + return tuple(sections) + + +def _has_unsupported_token(number_format: str) -> bool: + if _SCIENTIFIC_FORMAT.search(number_format) or _FRACTION_FORMAT.search(number_format): + return True + for literal in _QUOTED_LITERAL.findall(number_format): + if any(character not in {*_CURRENCY_SYMBOLS, " ", "-", "(", ")"} for character in literal): + return True + for match in _BRACKET_TOKEN.finditer(number_format): + token = match.group(1) + if token.casefold() in {"h", "hh", "m", "mm", "s", "ss"}: + continue + currency = _LOCALE_CURRENCY.fullmatch(token) + if currency is not None and currency.group(1) in _CURRENCY_SYMBOLS: + continue + return True + remaining = _QUOTED_LITERAL.sub("", number_format) + remaining = _BRACKET_TOKEN.sub("", remaining) + remaining = remaining.casefold().replace("am/pm", "") + index = 0 + while index < len(remaining): + character = remaining[index] + if character in {"_", "*"}: + index += 2 + continue + if character == "\\" and index + 1 < len(remaining): + if remaining[index + 1].isalnum(): + return True + index += 2 + continue + if character.isalpha() and character not in "ymdhs": + return True + index += 1 + return False + + +def _clean_section(section: str) -> str: + cleaned: list[str] = [] + index = 0 + while index < len(section): + character = section[index] + if character in {"_", "*"}: + index += 2 + continue + if character == "\\" and index + 1 < len(section): + cleaned.append(section[index + 1]) + index += 2 + continue + if character == '"': + end = section.find('"', index + 1) + if end == -1: + return section + cleaned.append(section[index + 1 : end]) + index = end + 1 + continue + if character == "[": + end = section.find("]", index + 1) + if end == -1: + return section + token = section[index + 1 : end] + currency = _LOCALE_CURRENCY.fullmatch(token) + cleaned.append(currency.group(1) if currency is not None else f"[{token}]") + index = end + 1 + continue + cleaned.append(character) + index += 1 + return "".join(cleaned).strip() + + +def _as_decimal(value: object) -> Decimal | None: + if isinstance(value, bool): + return None + if isinstance(value, Decimal): + return value + if isinstance(value, int): + return Decimal(value) + if isinstance(value, float) and math.isfinite(value): + return Decimal(str(value)) + return None + + +def _selected_numeric_section(sections: tuple[str, ...], value: Decimal) -> tuple[str, bool]: + if value < 0: + return (sections[1] if len(sections) > 1 else sections[0], True) + if value == 0 and len(sections) > 2: + return sections[2], False + return sections[0], False + + +def _decimal_places(pattern: str) -> tuple[int, int]: + if "." not in pattern: + return 0, 0 + suffix = pattern.split(".", 1)[1] + placeholders: list[str] = [] + for character in suffix: + if character in "0#?": + placeholders.append(character) + elif placeholders: + break + return placeholders.count("0"), len(placeholders) + + +def _format_number(value: Decimal, number_format: str) -> str | None: + sections = _split_sections(number_format) + section, negative = _selected_numeric_section(sections, value) + pattern = _clean_section(section) + if not any(character in "0#?" for character in pattern): + if value == 0 and "-" in pattern: + symbol = next((item for item in pattern if item in _CURRENCY_SYMBOLS), "") + return f"{symbol}-" + return None + + percent_count = pattern.count("%") + magnitude = abs(value) * (Decimal(100) ** percent_count) + minimum_decimals, maximum_decimals = _decimal_places(pattern) + if maximum_decimals: + quantum = Decimal(1).scaleb(-maximum_decimals) + magnitude = magnitude.quantize(quantum, rounding=ROUND_HALF_UP) + rendered = f"{magnitude:.{maximum_decimals}f}" + integer, fraction = rendered.split(".", 1) + if maximum_decimals > minimum_decimals: + fraction = fraction.rstrip("0") + fraction += "0" * max(0, minimum_decimals - len(fraction)) + rendered = integer + (f".{fraction}" if fraction else "") + else: + rendered = str(magnitude.quantize(Decimal(1), rounding=ROUND_HALF_UP)) + + integer, separator, fraction = rendered.partition(".") + integer_pattern = pattern.split(".", 1)[0] + if "," in integer_pattern: + integer = f"{int(integer):,}" + rendered = integer + (separator + fraction if separator else "") + + currency = next((symbol for symbol in _CURRENCY_SYMBOLS if symbol in pattern), "") + first_placeholder = min( + (pattern.find(character) for character in "0#?" if character in pattern), + default=0, + ) + if currency: + rendered = ( + f"{currency}{rendered}" + if pattern.find(currency) <= first_placeholder + else f"{rendered}{currency}" + ) + if percent_count: + rendered += "%" * percent_count + if negative: + rendered = f"({rendered})" if "(" in pattern and ")" in pattern else f"-{rendered}" + return rendered + + +def _time_fraction_digits(number_format: str) -> int: + match = re.search(r"s{1,2}\.([0#]+)", number_format, flags=re.IGNORECASE) + return len(match.group(1)) if match is not None else 0 + + +def _render_clock(value: time, number_format: str) -> str: + lowered = number_format.casefold() + include_seconds = "s" in lowered + twelve_hour = "am/pm" in lowered + raw_hour = value.hour % 12 or 12 if twelve_hour else value.hour + hour_token = re.search(r"(? str: + total_seconds = Decimal(str(value.total_seconds())) + digits = _time_fraction_digits(number_format) + quantum = Decimal(1).scaleb(-digits) if digits else Decimal(1) + total_seconds = total_seconds.quantize(quantum, rounding=ROUND_HALF_UP) + negative = total_seconds < 0 + total_seconds = abs(total_seconds) + whole_seconds = int(total_seconds) + fraction = total_seconds - whole_seconds + lowered = number_format.casefold() + if "[h]" in lowered or "[hh]" in lowered: + total_hours = whole_seconds // 3600 + rendered_hours = f"{total_hours:02d}" if "[hh]" in lowered else str(total_hours) + rendered = f"{rendered_hours}:{whole_seconds % 3600 // 60:02d}" + if "s" in lowered: + rendered += f":{whole_seconds % 60:02d}" + elif "[m]" in lowered or "[mm]" in lowered: + total_minutes = whole_seconds // 60 + rendered = f"{total_minutes:02d}" if "[mm]" in lowered else str(total_minutes) + if "s" in lowered: + rendered += f":{whole_seconds % 60:02d}" + else: + rendered = str(whole_seconds) + if digits: + rendered += f".{str(fraction)[2:]:0<{digits}}"[: digits + 1] + return f"-{rendered}" if negative else rendered + + +def _format_date_value(value: object, number_format: str, epoch: datetime) -> str | None: + converted = value + decimal = _as_decimal(value) + try: + if decimal is not None: + converted = from_excel( + float(decimal), + epoch, + timedelta=is_timedelta_format(number_format), + ) + except (OverflowError, ValueError): + return None + if isinstance(converted, timedelta): + return _render_elapsed(converted, number_format) + if isinstance(converted, datetime): + lowered = number_format.casefold() + has_date = "y" in lowered or "d" in lowered + has_time = "h" in lowered or "s" in lowered or not has_date + if has_date and has_time: + return ( + f"{converted.date().isoformat()} {_render_clock(converted.time(), number_format)}" + ) + if has_date: + return converted.date().isoformat() + return _render_clock(converted.time(), number_format) + if isinstance(converted, date): + return converted.isoformat() + if isinstance(converted, time): + return _render_clock(converted, number_format) + return None + + +def format_saved_value( + value: object, + number_format: str, + *, + epoch: datetime, + conditional_number_format: bool = False, +) -> FormattedSavedValue: + raw = stable_raw_value(value) + if value is None or isinstance(value, bool | str): + return FormattedSavedValue(raw) + if conditional_number_format: + return FormattedSavedValue(raw, "unsupported number format") + if number_format.casefold() == "general": + return FormattedSavedValue(raw) + if _has_unsupported_token(number_format): + return FormattedSavedValue(raw, "unsupported number format") + if is_date_format(number_format): + rendered_date = _format_date_value(value, number_format, epoch) + if rendered_date is not None: + return FormattedSavedValue(rendered_date) + return FormattedSavedValue(raw, "unsupported number format") + decimal = _as_decimal(value) + if decimal is None: + return FormattedSavedValue(raw) + try: + rendered_number = _format_number(decimal, number_format) + except (InvalidOperation, ValueError): + rendered_number = None + if rendered_number is None: + return FormattedSavedValue(raw, "unsupported number format") + return FormattedSavedValue(rendered_number) diff --git a/tests/test_office_package.py b/tests/test_office_package.py index c6540ef..f6127a9 100644 --- a/tests/test_office_package.py +++ b/tests/test_office_package.py @@ -98,6 +98,84 @@ def test_validate_office_package_accepts_minimal_xlsx(tmp_path: Path) -> None: assert layout.main_part_name == "xl/workbook.xml" +def test_validate_office_package_accepts_package_absolute_relationship_target( + tmp_path: Path, +) -> None: + xlsx = tmp_path / "absolute-target.xlsx" + entries = minimal_xlsx_entries(include_workbook=False) + entries.extend( + [ + ( + "xl/workbook.xml", + b'', + ), + ( + "xl/_rels/workbook.xml.rels", + b'', + ), + ( + "xl/worksheets/sheet1.xml", + b'', + ), + ] + ) + _write_zip(xlsx, entries) + + layout = validate_office_package(xlsx, document_type=DocumentType.XLSX) + + assert layout.main_part_name == "xl/workbook.xml" + + +@pytest.mark.parametrize( + "target", + [ + "//server/xl/worksheets/sheet1.xml", + "\\xl\\worksheets\\sheet1.xml", + "https://example.com/sheet1.xml", + "C:/xl/worksheets/sheet1.xml", + "/../xl/worksheets/sheet1.xml", + "", + ], +) +def test_validate_office_package_rejects_unsafe_package_absolute_relationship_targets( + tmp_path: Path, + target: str, +) -> None: + xlsx = tmp_path / "unsafe-absolute-target.xlsx" + entries = minimal_xlsx_entries(include_workbook=False) + entries.extend( + [ + ( + "xl/workbook.xml", + b'', + ), + ( + "xl/_rels/workbook.xml.rels", + ( + '' + ).encode(), + ), + ( + "xl/worksheets/sheet1.xml", + b'', + ), + ] + ) + _write_zip(xlsx, entries) + + with pytest.raises(CorruptDocumentError, match="relationship target"): + validate_office_package(xlsx, document_type=DocumentType.XLSX) + + @pytest.mark.parametrize( "kwargs", [ diff --git a/tests/test_xlsx_extract.py b/tests/test_xlsx_extract.py new file mode 100644 index 0000000..e513aee --- /dev/null +++ b/tests/test_xlsx_extract.py @@ -0,0 +1,365 @@ +from __future__ import annotations + +from pathlib import Path +from zipfile import ZipFile + +import pytest +from openpyxl import Workbook +from openpyxl.chart import BarChart, Reference +from openpyxl.formatting.rule import Rule +from openpyxl.styles import Font +from openpyxl.styles.differential import DifferentialStyle +from openpyxl.styles.numbers import NumberFormat +from openpyxl.worksheet.table import Table + +import opendocs.parsers.xlsx.extract as extract_module +from opendocs._models import ( + DocumentType, + HeadingBlock, + MarkdownBlock, + ParagraphBlock, + ParsedDocument, + SpannedTableBlock, + TableBlock, +) +from opendocs.errors import LimitExceededError +from opendocs.markdown import render_markdown +from opendocs.parsers.xlsx.extract import extract_xlsx +from opendocs.parsers.xlsx.models import XlsxNativeSlot, XlsxSheet, XlsxSheetKind, XlsxSheetState +from opendocs.parsers.xlsx.preflight import preflight_xlsx +from tests.xlsx_fixtures import rewrite_xlsx + + +def _save_ordered_workbook(path: Path) -> None: + workbook = Workbook() + visible = workbook.active + visible.title = "Visible" + visible["A1"] = "kept" + hidden = workbook.create_sheet("Hidden") + hidden.sheet_state = "hidden" + very_hidden = workbook.create_sheet("Very Hidden") + very_hidden.sheet_state = "veryHidden" + chart = BarChart() + chart.add_data(Reference(visible, min_col=1, min_row=1, max_row=1)) + chart_sheet = workbook.create_chartsheet("Chart") + chart_sheet.add_chart(chart) + workbook.save(path) + + +def _native_slots(document_sheet: XlsxSheet) -> tuple[XlsxNativeSlot, ...]: + return tuple(slot for slot in document_sheet.slots if isinstance(slot, XlsxNativeSlot)) + + +def test_extract_preserves_all_sheet_like_entries_states_and_empty_sheets( + tmp_path: Path, +) -> None: + path = tmp_path / "ordered.xlsx" + _save_ordered_workbook(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + assert [ + (sheet.sheet_index, sheet.name, sheet.kind, sheet.state) for sheet in document.sheets + ] == [ + (1, "Visible", XlsxSheetKind.WORKSHEET, XlsxSheetState.VISIBLE), + (2, "Hidden", XlsxSheetKind.WORKSHEET, XlsxSheetState.HIDDEN), + (3, "Very Hidden", XlsxSheetKind.WORKSHEET, XlsxSheetState.VERY_HIDDEN), + (4, "Chart", XlsxSheetKind.CHARTSHEET, XlsxSheetState.VISIBLE), + ] + for expected_index, sheet in enumerate(document.sheets, start=1): + prelude = _native_slots(sheet)[0] + assert prelude.anchor == "A1" + assert prelude.source_index == 0 + assert isinstance(prelude.blocks[0], MarkdownBlock) + assert prelude.blocks[0].markdown == f"" + assert sheet.name not in prelude.blocks[0].markdown + assert isinstance(prelude.blocks[1], HeadingBlock) + assert isinstance(prelude.blocks[2], ParagraphBlock) + assert len(_native_slots(document.sheets[1])) == 1 + assert len(_native_slots(document.sheets[2])) == 1 + assert len(_native_slots(document.sheets[3])) == 1 + + +def test_extract_builds_tables_regions_merges_and_ignores_style_only_cells( + tmp_path: Path, +) -> None: + path = tmp_path / "regions.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Regions" + sheet.append(("Name", "Amount")) + sheet.append(("A", 1)) + sheet.append(("B", 2)) + table = Table(displayName="Ledger", ref="A1:B3") + table.headerRowCount = 1 + sheet.add_table(table) + sheet["D1"] = "Merged" + sheet.merge_cells("D1:E1") + sheet["D2"] = "left" + sheet["E2"] = "right" + sheet["H1"] = "first" + sheet["H3"] = "second" + sheet["J1"].font = Font(bold=True, color="FF0000") + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + slots = _native_slots(document.sheets[0]) + assert [slot.anchor for slot in slots] == ["A1", "A1:B3", "D1:E2", "H1", "H3"] + assert [slot.source_index for slot in slots] == list(range(5)) + assert isinstance(slots[1].blocks[1], TableBlock) + assert slots[1].blocks[1].header_rows == 1 + assert slots[1].blocks[1].grid == (("Name", "Amount"), ("A", "1"), ("B", "2")) + assert isinstance(slots[2].blocks[1], SpannedTableBlock) + assert slots[2].blocks[1].header_rows == 0 + assert [(cell.row_span, cell.column_span, cell.text) for cell in slots[2].blocks[1].cells] == [ + (1, 2, "Merged"), + (1, 1, "left"), + (1, 1, "right"), + ] + assert isinstance(slots[3].blocks[1], TableBlock) + assert slots[3].blocks[1].header_rows == 0 + assert all( + "Regions" not in block.markdown + for slot in slots + for block in slot.blocks + if isinstance(block, MarkdownBlock) + ) + assert all(slot.anchor != "J1" for slot in slots) + + +def test_excel_table_header_row_count_zero_is_not_promoted_to_header(tmp_path: Path) -> None: + path = tmp_path / "headerless.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = "one" + sheet["A2"] = "two" + table = Table(displayName="Headerless", ref="A1:A2") + table.headerRowCount = 0 + sheet.add_table(table) + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + table_block = _native_slots(document.sheets[0])[1].blocks[1] + assert isinstance(table_block, TableBlock) + assert table_block.header_rows == 0 + + +def test_extracted_native_blocks_render_to_stable_markdown(tmp_path: Path) -> None: + path = tmp_path / "golden.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Ledger" + sheet.append(("Item", "Amount")) + sheet.append(("Book", 12.5)) + sheet["B2"].number_format = "$0.00" + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + blocks = tuple( + block + for extracted_sheet in document.sheets + for slot in extracted_sheet.slots + if isinstance(slot, XlsxNativeSlot) + for block in slot.blocks + ) + rendered = render_markdown( + ParsedDocument(DocumentType.XLSX, blocks, document.warnings), + max_output_chars=10_000, + ) + + assert rendered.markdown == ( + "\n\n" + "# Ledger\n\n" + "Sheet state: visible\n\n" + "\n\n" + "
\n" + "\n" + "\n" + "\n" + "\n" + "
ItemAmount
Book$12.50
\n" + ) + + +def test_extract_rechecks_component_bounding_box_before_materializing( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "bounded.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = "a" + sheet["B1"] = "b" + sheet["B2"] = "c" + workbook.save(path) + index = preflight_xlsx(path) + monkeypatch.setattr(extract_module, "MAX_MATERIALIZED_GRID_CELLS", 3) + + with pytest.raises(LimitExceededError, match="materialized grid"): + extract_xlsx(path, index) + + +def test_extract_rechecks_materialized_grid_across_all_sheets( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "bounded-workbook.xlsx" + workbook = Workbook() + first = workbook.active + first.append(("a", "b")) + second = workbook.create_sheet("Second") + second.append(("c", "d")) + workbook.save(path) + index = preflight_xlsx(path) + monkeypatch.setattr(extract_module, "MAX_MATERIALIZED_GRID_CELLS", 3) + + with pytest.raises(LimitExceededError, match="materialized grid"): + extract_xlsx(path, index) + + +def test_extract_aggregates_number_format_warnings_after_twenty_coordinates( + tmp_path: Path, +) -> None: + path = tmp_path / "warnings.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Warnings" + for row in range(1, 23): + cell = sheet.cell(row, 1, row) + cell.number_format = "0.00E+00" + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_unsupported_number_format" + ] + assert len(warnings) == 21 + assert warnings[0].message.startswith("Warnings!A1:") + assert warnings[19].message.startswith("Warnings!A20:") + assert warnings[20].message == "2 additional xlsx_unsupported_number_format warnings suppressed" + + +def test_extract_falls_back_when_conditional_rule_can_change_number_format( + tmp_path: Path, +) -> None: + path = tmp_path / "conditional-number-format.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = 1234.5 + sheet.conditional_formatting.add( + "A1", + Rule( + type="expression", + formula=("1",), + dxf=DifferentialStyle(numFmt=NumberFormat(numFmtId=164, formatCode="$0.00")), + ), + ) + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + region = _native_slots(document.sheets[0])[1] + assert isinstance(region.blocks[1], TableBlock) + assert region.blocks[1].grid == (("1234.5",),) + assert [warning.code for warning in document.warnings] == ["xlsx_unsupported_number_format"] + + +def test_extract_loads_full_mode_openpyxl_once_with_links_disabled( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "once.xlsx" + workbook = Workbook() + workbook.active["A1"] = "value" + workbook.save(path) + index = preflight_xlsx(path) + calls: list[tuple[Path, dict[str, object]]] = [] + original = extract_module.openpyxl.load_workbook + + def recording_loader(filename: Path, **kwargs: object) -> object: + calls.append((filename, kwargs)) + return original(filename, **kwargs) + + monkeypatch.setattr(extract_module.openpyxl, "load_workbook", recording_loader) + + extract_xlsx(path, index) + + assert calls == [ + ( + path, + { + "read_only": False, + "data_only": False, + "rich_text": False, + "keep_links": False, + }, + ) + ] + + +def test_formula_sidecar_prefers_cache_and_distinguishes_missing_empty_and_special_formulas( + tmp_path: Path, +) -> None: + path = tmp_path / "formulas.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Formulas" + for row in range(1, 9): + sheet.cell(row, 1, row) + sheet.cell(row, 2, f"=A{row}*2") + sheet["B1"].number_format = "$#,##0.00" + workbook.save(path) + with ZipFile(path) as archive: + xml = archive.read("xl/worksheets/sheet1.xml") + replacements = { + b'A1*2': b'A1*22', + b'A2*2': b'A2*2', + b'A3*2': b'A3*2', + b'A4*2': b'[1]Sheet1!A1', + b'A5*2': ( + b'SUM(A5:A6)10' + ), + b'A6*2': ( + b'' + ), + b'A7*2': ( + b'A7*2' + ), + b'A8*2': b'', + } + for old, new in replacements.items(): + assert old in xml + xml = xml.replace(old, new) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": xml}) + + document = extract_xlsx(path, preflight_xlsx(path)) + + text_by_anchor = { + slot.anchor: "\n".join(cell for row in slot.blocks[1].grid for cell in row) + for slot in _native_slots(document.sheets[0])[1:] + if isinstance(slot.blocks[1], TableBlock) + } + all_text = "\n".join(text_by_anchor.values()) + assert "$2.00" in all_text + assert "=A2*2" in all_text + assert "=A3*2" not in all_text + assert "=[1]Sheet1!A1" in all_text + assert "Array/spill formula B5:C5: =SUM(A5:A6); saved value: 10" in all_text + assert "Data-table formula B6:C7 (dt2D=1, r1=A1, r2=A2)" in all_text + assert "=A7*2" in all_text + assert "=A8*2" in all_text + codes = [warning.code for warning in document.warnings] + assert codes.count("xlsx_formula_cache_missing") == 4 + assert codes.count("xlsx_data_table_formula") == 1 + assert codes.count("xlsx_external_reference") == 1 + + +def test_repeated_extraction_is_deterministic(tmp_path: Path) -> None: + path = tmp_path / "repeat.xlsx" + _save_ordered_workbook(path) + index = preflight_xlsx(path) + + assert extract_xlsx(path, index) == extract_xlsx(path, index) diff --git a/tests/test_xlsx_preflight.py b/tests/test_xlsx_preflight.py index b390129..d3486a5 100644 --- a/tests/test_xlsx_preflight.py +++ b/tests/test_xlsx_preflight.py @@ -4,6 +4,8 @@ from zipfile import ZIP_DEFLATED, ZipFile import pytest +from openpyxl import Workbook +from openpyxl.chart import BarChart, Reference import opendocs.parsers.xlsx.preflight as preflight_module from opendocs.errors import CorruptDocumentError, LimitExceededError, UnsupportedDocumentError @@ -75,6 +77,26 @@ def test_preflight_preserves_worksheet_chartsheet_state_and_empty_sheet_order( assert result.serialized_cells == 2 +def test_preflight_accepts_openpyxl_package_absolute_relationship_targets( + tmp_path: Path, +) -> None: + path = tmp_path / "openpyxl.xlsx" + workbook = Workbook() + workbook.active["A1"] = "value" + chart = BarChart() + chart.add_data(Reference(workbook.active, min_col=1, min_row=1, max_row=1)) + chart_sheet = workbook.create_chartsheet("Chart") + chart_sheet.add_chart(chart) + workbook.save(path) + + result = preflight_xlsx(path) + + assert [(sheet.name, sheet.kind) for sheet in result.sheets] == [ + ("Sheet", XlsxSheetKind.WORKSHEET), + ("Chart", XlsxSheetKind.CHARTSHEET), + ] + + def test_preflight_accepts_128_sheets_and_rejects_129(tmp_path: Path) -> None: accepted = tmp_path / "accepted.xlsx" rejected = tmp_path / "rejected.xlsx" diff --git a/tests/test_xlsx_values.py b/tests/test_xlsx_values.py new file mode 100644 index 0000000..e2c6214 --- /dev/null +++ b/tests/test_xlsx_values.py @@ -0,0 +1,88 @@ +from __future__ import annotations + +from datetime import datetime, time +from decimal import Decimal + +import pytest +from openpyxl.utils.datetime import MAC_EPOCH, WINDOWS_EPOCH + +from opendocs.parsers.xlsx.values import format_saved_value + + +@pytest.mark.parametrize( + ("value", "number_format", "expected"), + [ + (True, "General", "TRUE"), + ("#DIV/0!", "General", "#DIV/0!"), + (Decimal("1234"), "0", "1234"), + (Decimal("1234.5"), "0.00", "1234.50"), + (Decimal("1234.5"), "#,##0.00", "1,234.50"), + (Decimal("0.125"), "0.00%", "12.50%"), + (Decimal("1234.5"), "$#,##0.00", "$1,234.50"), + (Decimal("1234"), "¥#,##0", "¥1,234"), + (Decimal("-1234.5"), "#,##0.00;(#,##0.00)", "(1,234.50)"), + ( + Decimal("-1234.5"), + '_(€* #,##0.00_);_(€* (#,##0.00);_(€* "-"??_);_(@_)', + "(€1,234.50)", + ), + ], +) +def test_format_saved_value_supports_core_and_accounting_formats( + value: object, + number_format: str, + expected: str, +) -> None: + result = format_saved_value(value, number_format, epoch=WINDOWS_EPOCH) + + assert result.text == expected + assert result.warning is None + + +def test_format_saved_value_supports_both_date_systems_and_elapsed_time() -> None: + windows = format_saved_value(43831, "yyyy-mm-dd", epoch=WINDOWS_EPOCH) + mac = format_saved_value(42369, "yyyy-mm-dd", epoch=MAC_EPOCH) + timestamp = format_saved_value( + datetime(2020, 1, 2, 3, 4, 5), + "yyyy-mm-dd hh:mm:ss", + epoch=WINDOWS_EPOCH, + ) + elapsed = format_saved_value(1.5, "[h]:mm:ss", epoch=WINDOWS_EPOCH) + + assert windows.text == "2020-01-01" + assert mac.text == "2020-01-01" + assert timestamp.text == "2020-01-02 03:04:05" + assert elapsed.text == "36:00:00" + assert not any(item.warning for item in (windows, mac, timestamp, elapsed)) + + +def test_format_saved_value_supports_common_twelve_hour_time() -> None: + result = format_saved_value( + time(15, 4, 5), + "h:mm:ss AM/PM", + epoch=WINDOWS_EPOCH, + ) + + assert result.text == "3:04:05 PM" + assert result.warning is None + + +@pytest.mark.parametrize( + "number_format", + [ + "0.00E+00", + "# ?/?", + "[Red]0.00", + "[$-409]d-mmm-yy", + "[>100]0;0", + '0.00" kg"', + "0.00\\k", + ], +) +def test_format_saved_value_falls_back_for_unsupported_number_formats( + number_format: str, +) -> None: + result = format_saved_value(1234.5, number_format, epoch=WINDOWS_EPOCH) + + assert result.text == "1234.5" + assert result.warning == "unsupported number format" From 28a7ac3802293d6679bcce66c84a0763783555a2 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 15:44:14 +0800 Subject: [PATCH 04/12] =?UTF-8?q?=E8=AE=A9=20XLSX=20=E5=A4=96=E5=9B=B4?= =?UTF-8?q?=E6=96=87=E6=9C=AC=E4=B8=8D=E5=86=8D=E9=9D=99=E9=BB=98=E4=B8=A2?= =?UTF-8?q?=E5=A4=B1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 从安全 OOXML 读取批注、threaded comment、文本框、链接和页眉页脚,并把外部引用与不支持对象转成可定位降级。 Constraint: URL 只保留引用且零网络访问;所有对象保持 sheet/A1/ordinal 的稳定顺序。 Rejected: 依赖 openpyxl 不完整的 drawing/comment 映射或静默跳过扩展对象。 Confidence: high Scope-risk: 真实 threaded-comments 文件仍待私有 corpus 验证。 Tested: uv run --frozen pytest tests/test_xlsx_models.py tests/test_xlsx_values.py tests/test_xlsx_extract.py tests/test_xlsx_preflight.py tests/test_office_package.py tests/test_runtime.py -q Tested: uv run --frozen ruff check . Tested: uv run --frozen ruff format --check . Tested: uv run --frozen ty check src tests Not-tested: U8 文档/依赖白名单修复前全套仍有 3 个已知发布契约失败;真实 XLSX、私有 corpus --- src/opendocs/parsers/xlsx/extract.py | 120 +++- src/opendocs/parsers/xlsx/preflight.py | 218 ++++++- src/opendocs/parsers/xlsx/text_objects.py | 691 ++++++++++++++++++++++ tests/test_xlsx_extract.py | 442 +++++++++++++- tests/test_xlsx_models.py | 12 +- tests/test_xlsx_preflight.py | 92 +++ 6 files changed, 1563 insertions(+), 12 deletions(-) create mode 100644 src/opendocs/parsers/xlsx/text_objects.py diff --git a/src/opendocs/parsers/xlsx/extract.py b/src/opendocs/parsers/xlsx/extract.py index eec82e2..50649ba 100644 --- a/src/opendocs/parsers/xlsx/extract.py +++ b/src/opendocs/parsers/xlsx/extract.py @@ -35,6 +35,11 @@ MAX_MATERIALIZED_GRID_CELLS as PREFLIGHT_MAX_MATERIALIZED_GRID_CELLS, ) from opendocs.parsers.xlsx.preflight import XlsxPreflight, XlsxPreflightSheet +from opendocs.parsers.xlsx.text_objects import ( + XlsxTextObject, + read_xlsx_text_objects, + text_object_blocks, +) from opendocs.parsers.xlsx.values import format_saved_value MAX_MATERIALIZED_GRID_CELLS = PREFLIGHT_MAX_MATERIALIZED_GRID_CELLS @@ -140,6 +145,17 @@ class _RegionSpec: kind: str +@dataclass(frozen=True, slots=True) +class _SlotCandidate: + row: int + column: int + kind_rank: int + source_ordinal: int + anchor: str + region: _RegionSpec | None = None + text_object: XlsxTextObject | None = None + + @dataclass(slots=True) class _MaterializationBudget: used: int = 0 @@ -629,6 +645,7 @@ def _sheet_slots( sheet: XlsxPreflightSheet, warnings: _WarningCollector, budget: _MaterializationBudget, + text_objects: tuple[XlsxTextObject, ...], ) -> tuple[XlsxNativeSlot, ...]: texts, semantic = _cell_texts( worksheet, @@ -637,20 +654,91 @@ def _sheet_slots( sheet=sheet, warnings=warnings, ) - slots: list[XlsxNativeSlot] = [_sheet_prelude(sheet)] - for source_index, spec in enumerate(_region_specs(worksheet, semantic), start=1): + candidates: list[_SlotCandidate] = [] + for ordinal, spec in enumerate(_region_specs(worksheet, semantic), start=1): budget.consume(_area(spec.bounds)) anchor = _anchor(spec.bounds) - comment = ( - f"" + candidates.append( + _SlotCandidate( + row=spec.bounds[1], + column=spec.bounds[0], + kind_rank=0, + source_ordinal=ordinal, + anchor=anchor, + region=spec, + ) ) - block: Block = _region_block(spec, texts) + candidates.extend( + _SlotCandidate( + row=item.row, + column=item.column, + kind_rank=item.kind_rank, + source_ordinal=item.source_ordinal, + anchor=item.anchor, + text_object=item, + ) + for item in text_objects + ) + slots: list[XlsxNativeSlot] = [_sheet_prelude(sheet)] + for source_index, candidate in enumerate( + sorted( + candidates, + key=lambda item: (item.row, item.column, item.kind_rank, item.source_ordinal), + ), + start=1, + ): + if candidate.region is not None: + spec = candidate.region + comment = ( + f"" + ) + block: Block = _region_block(spec, texts) + blocks = (MarkdownBlock(comment), block) + elif candidate.text_object is not None: + blocks = text_object_blocks( + candidate.text_object, + fallback_label=texts.get((candidate.row, candidate.column), ""), + object_index=source_index, + ) + else: + raise AssertionError("XLSX slot candidate is invalid") slots.append( XlsxNativeSlot( source_index=source_index, - anchor=anchor, - blocks=(MarkdownBlock(comment), block), + anchor=candidate.anchor, + blocks=blocks, + ) + ) + return tuple(slots) + + +def _non_worksheet_slots( + sheet: XlsxPreflightSheet, + text_objects: tuple[XlsxTextObject, ...], +) -> tuple[XlsxNativeSlot, ...]: + slots = [_sheet_prelude(sheet)] + for source_index, item in enumerate( + sorted( + text_objects, + key=lambda value: ( + value.row, + value.column, + value.kind_rank, + value.source_ordinal, + ), + ), + start=1, + ): + slots.append( + XlsxNativeSlot( + source_index=source_index, + anchor=item.anchor, + blocks=text_object_blocks( + item, + fallback_label="", + object_index=source_index, + ), ) ) return tuple(slots) @@ -662,7 +750,19 @@ def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: if not isinstance(preflight, XlsxPreflight): raise TypeError("preflight must be an XlsxPreflight") sidecars = _read_sidecars(path, preflight) + text_objects = read_xlsx_text_objects(path, preflight) warnings = _WarningCollector() + for warning in text_objects.warnings: + sheet = preflight.sheets[warning.sheet_index - 1] + warnings.add( + warning.code, + sheet=sheet, + coordinate=warning.anchor.split(":", 1)[0], + detail=( + f"sheet={warning.sheet_index} anchor={warning.anchor} " + f"object={warning.object_ordinal}: {warning.detail}" + ), + ) workbook = openpyxl.load_workbook( path, read_only=False, @@ -675,8 +775,9 @@ def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: sheets: list[XlsxSheet] = [] budget = _MaterializationBudget() for sheet in preflight.sheets: + sheet_text_objects = text_objects.by_sheet[sheet.sheet_index - 1] if sheet.kind is XlsxSheetKind.CHARTSHEET: - slots = (_sheet_prelude(sheet),) + slots = _non_worksheet_slots(sheet, sheet_text_objects) else: worksheet = worksheets.get(sheet.name) if worksheet is None: @@ -688,6 +789,7 @@ def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: sheet=sheet, warnings=warnings, budget=budget, + text_objects=sheet_text_objects, ) sheets.append( XlsxSheet( diff --git a/src/opendocs/parsers/xlsx/preflight.py b/src/opendocs/parsers/xlsx/preflight.py index d6e966b..5f30496 100644 --- a/src/opendocs/parsers/xlsx/preflight.py +++ b/src/opendocs/parsers/xlsx/preflight.py @@ -82,6 +82,12 @@ _PIVOT_TABLE_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotTable" _PIVOT_CACHE_DEFINITION_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotCacheDefinition" _PIVOT_CACHE_RECORDS_RELATIONSHIP = f"{_OFFICE_REL_NS}/pivotCacheRecords" +_EXTERNAL_LINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLink" +_EXTERNAL_LINK_PATH_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLinkPath" +_CONNECTIONS_RELATIONSHIP = f"{_OFFICE_REL_NS}/connections" +_THREADED_REL_NS = "http://schemas.microsoft.com/office/2017/10/relationships" +_THREADED_COMMENTS_RELATIONSHIP = f"{_THREADED_REL_NS}/threadedComment" +_PERSON_RELATIONSHIP = f"{_THREADED_REL_NS}/person" _RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" _RELATIONSHIP_TAG = f"{{{_PACKAGE_REL_NS}}}Relationship" _A1_RANGE_RE = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})(?::([A-Z]{1,3})([1-9][0-9]{0,6}))?$") @@ -91,6 +97,7 @@ _DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" _CHART_NS = "http://schemas.openxmlformats.org/drawingml/2006/chart" _CUSTOM_PROPERTIES_NS = "http://schemas.openxmlformats.org/officeDocument/2006/custom-properties" +_THREADED_COMMENTS_NS = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" @dataclass(frozen=True, slots=True) @@ -460,6 +467,140 @@ def _read_relationships( return relationships +def _preflight_persons( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX persons part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_THREADED_COMMENTS_NS}}}personList": + raise CorruptDocumentError("XLSX persons namespace is invalid") + if event == "end" and element.tag == f"{{{_THREADED_COMMENTS_NS}}}person": + if not element.get("id") or not element.get("displayName"): + raise CorruptDocumentError("XLSX person entry is invalid") + _increment( + usage, + "comment_authors", + 1, + limit=MAX_COMMENT_AUTHORS, + message="XLSX exceeds the comment author limit", + ) + _add_native_text( + usage, + sum( + len(element.get(attribute, "")) + for attribute in ("displayName", "userId", "providerId") + ), + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX persons part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX persons part is corrupt") + + +def _preflight_connections( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events(stream, message="XLSX connections part is corrupt"): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_SPREADSHEET_NS}}}connections": + raise CorruptDocumentError("XLSX connections namespace is invalid") + if event != "end" or element.tag != f"{{{_SPREADSHEET_NS}}}connection": + continue + references = tuple( + element.get(attribute, "") + for attribute in ("sourceFile", "odcFile", "connectionFile") + if element.get(attribute) + ) + for reference in references: + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + _add_native_text(usage, len(reference)) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX connections part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX connections part is corrupt") + + +def _preflight_workbook_text_relationships( + archive: ZipFile, + infos: dict[str, ZipInfo], + relationships: dict[str, _Relationship], + usage: _ResourceUsage, + referenced_external_ids: set[str], +) -> None: + person_targets = tuple( + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP and not relationship.external + ) + if ( + any( + relationship.external + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP + ) + or len(person_targets) > 1 + ): + raise CorruptDocumentError("XLSX persons relationship is invalid") + for target in person_targets: + _preflight_persons(archive, target, usage) + + for relationship_id, relationship in relationships.items(): + if relationship.relationship_type == _CONNECTIONS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX connections relationship is invalid") + _preflight_connections(archive, relationship.target, usage) + elif relationship.relationship_type == _EXTERNAL_LINK_RELATIONSHIP: + if relationship.external: + if relationship_id not in referenced_external_ids: + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + _add_native_text(usage, len(relationship.target)) + continue + nested = _read_relationships( + archive, + infos, + relationship.target, + required=_relationships_part(relationship.target) in infos, + ) + for nested_relationship in nested.values(): + if nested_relationship.relationship_type != _EXTERNAL_LINK_PATH_RELATIONSHIP: + continue + if relationship_id not in referenced_external_ids: + _increment( + usage, + "external_references", + 1, + limit=MAX_EXTERNAL_REFERENCES, + message="XLSX exceeds the external reference limit", + ) + _add_native_text(usage, len(nested_relationship.target)) + + def _parse_workbook( archive: ZipFile, infos: dict[str, ZipInfo], @@ -481,6 +622,7 @@ def _parse_workbook( sheet_ids: set[int] = set() names: set[str] = set() pivot_cache_targets: list[str] = [] + referenced_external_ids: set[str] = set() date_1904 = False root_seen = False try: @@ -524,8 +666,15 @@ def _parse_workbook( limit=MAX_EXTERNAL_REFERENCES, message="XLSX exceeds the external reference limit", ) - if element.get(_RELATIONSHIP_ID) not in relationships: + relationship_id = element.get(_RELATIONSHIP_ID) + relationship = relationships.get(relationship_id or "") + if ( + relationship_id is None + or relationship is None + or relationship.relationship_type != _EXTERNAL_LINK_RELATIONSHIP + ): raise CorruptDocumentError("XLSX external reference is invalid") + referenced_external_ids.add(relationship_id) element.clear() elif element.tag == f"{{{_SPREADSHEET_NS}}}pivotCache": _increment( @@ -612,6 +761,13 @@ def _parse_workbook( ) if len(shared_string_targets) > 1 or len(style_targets) > 1: raise CorruptDocumentError("XLSX workbook singleton relationships are duplicated") + _preflight_workbook_text_relationships( + archive, + infos, + relationships, + usage, + referenced_external_ids, + ) shared_strings_part = ( shared_string_targets[0] if shared_string_targets @@ -849,6 +1005,53 @@ def _preflight_comments( raise CorruptDocumentError("XLSX comments part is corrupt") +def _preflight_threaded_comments( + archive: ZipFile, + part_name: str, + usage: _ResourceUsage, +) -> None: + root_seen = False + try: + with archive.open(part_name) as stream: + for event, element in _xml_events( + stream, + message="XLSX threaded comments part is corrupt", + ): + if not root_seen and event == "start": + root_seen = True + if element.tag != f"{{{_THREADED_COMMENTS_NS}}}ThreadedComments": + raise CorruptDocumentError("XLSX threaded comments namespace is invalid") + if event == "end" and element.tag == f"{{{_THREADED_COMMENTS_NS}}}threadedComment": + reference = element.get("ref") + if reference is None or not element.get("personId"): + raise CorruptDocumentError("XLSX threaded comment is invalid") + bounds = _parse_a1_range( + reference, + message="XLSX threaded comment anchor is invalid", + ) + if _area(bounds) != 1: + raise CorruptDocumentError("XLSX threaded comment anchor is invalid") + _increment( + usage, + "hyperlinks_and_comments", + 1, + limit=MAX_HYPERLINKS_AND_COMMENTS, + message="XLSX exceeds the hyperlink and comment limit", + ) + _add_native_text( + usage, + sum( + len(node.text or "") + for node in element.iter(f"{{{_THREADED_COMMENTS_NS}}}text") + ), + ) + element.clear() + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX threaded comments part is corrupt") from error + if not root_seen: + raise CorruptDocumentError("XLSX threaded comments part is corrupt") + + def _preflight_chart( archive: ZipFile, part_name: str, @@ -1009,6 +1212,14 @@ def _preflight_drawing( usage, sum(len(node.text or "") for node in element.iter(f"{{{_DRAWING_MAIN_NS}}}t")), ) + _add_native_text( + usage, + sum( + len(node.get(attribute, "")) + for node in element.iter(f"{{{_DRAWING_NS}}}cNvPr") + for attribute in ("name", "descr", "title") + ), + ) for node in element.iter(): for attribute_name, relationship_id in node.attrib.items(): if not attribute_name.startswith(f"{{{_OFFICE_REL_NS}}}"): @@ -1469,6 +1680,10 @@ def _worksheet_counts( if relationship.external: raise CorruptDocumentError("XLSX comments relationship is invalid") _preflight_comments(archive, relationship.target, usage) + elif relationship.relationship_type == _THREADED_COMMENTS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX threaded comments relationship is invalid") + _preflight_threaded_comments(archive, relationship.target, usage) elif relationship.relationship_type == _PIVOT_TABLE_RELATIONSHIP: if relationship.external: raise CorruptDocumentError("XLSX pivot table relationship is invalid") @@ -1511,6 +1726,7 @@ def _worksheet_counts( known_relationship_types = { _DRAWING_RELATIONSHIP, _COMMENTS_RELATIONSHIP, + _THREADED_COMMENTS_RELATIONSHIP, _TABLE_RELATIONSHIP, _HYPERLINK_RELATIONSHIP, _PIVOT_TABLE_RELATIONSHIP, diff --git a/src/opendocs/parsers/xlsx/text_objects.py b/src/opendocs/parsers/xlsx/text_objects.py new file mode 100644 index 0000000..3148865 --- /dev/null +++ b/src/opendocs/parsers/xlsx/text_objects.py @@ -0,0 +1,691 @@ +from __future__ import annotations + +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any +from urllib.parse import quote, urlsplit +from zipfile import BadZipFile, ZipFile, ZipInfo + +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException +from openpyxl.utils.cell import get_column_letter, range_boundaries + +from opendocs._models import Block, InlineLink, InlineText, MarkdownBlock, ParagraphBlock +from opendocs.errors import CorruptDocumentError +from opendocs.parsers.xlsx.preflight import ( + XlsxPreflight, + XlsxPreflightSheet, + _read_relationships, + _relationships_part, +) + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +_DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +_THREADED_REL_NS = "http://schemas.microsoft.com/office/2017/10/relationships" +_THREADED_COMMENTS_NS = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" + +_COMMENTS_RELATIONSHIP = f"{_OFFICE_REL_NS}/comments" +_DRAWING_RELATIONSHIP = f"{_OFFICE_REL_NS}/drawing" +_HYPERLINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/hyperlink" +_EXTERNAL_LINK_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLink" +_EXTERNAL_LINK_PATH_RELATIONSHIP = f"{_OFFICE_REL_NS}/externalLinkPath" +_CONNECTIONS_RELATIONSHIP = f"{_OFFICE_REL_NS}/connections" +_THREADED_COMMENTS_RELATIONSHIP = f"{_THREADED_REL_NS}/threadedComment" +_PERSON_RELATIONSHIP = f"{_THREADED_REL_NS}/person" +_RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" + +_KIND_HYPERLINK = 10 +_KIND_COMMENT = 20 +_KIND_TEXT_BOX = 30 +_KIND_HEADER_FOOTER = 40 +_KIND_EXTERNAL_REFERENCE = 50 +_SAFE_LINK_SCHEMES = {"http", "https", "mailto"} +_UNSUPPORTED_KIND = re.compile(r"[^A-Za-z0-9_.-]+") +_HEADER_TAGS = ( + ("oddHeader", "Odd header"), + ("oddFooter", "Odd footer"), + ("evenHeader", "Even header"), + ("evenFooter", "Even footer"), + ("firstHeader", "First header"), + ("firstFooter", "First footer"), +) +_HEADER_SECTIONS = (("L", "left"), ("C", "center"), ("R", "right")) +_HEADER_FIELDS = { + "P": "{page}", + "N": "{pages}", + "D": "{date}", + "T": "{time}", + "F": "{file}", + "Z": "{path}", + "A": "{sheet}", +} +_HEADER_NAMED_FIELDS = { + "Page": "{page}", + "Pages": "{pages}", + "Date": "{date}", + "Time": "{time}", + "File": "{file}", + "Path": "{path}", + "Tab": "{sheet}", +} +_HEADER_FORMAT_CODES = {"B", "I", "U", "E", "S", "X", "Y", "O", "H"} + + +@dataclass(frozen=True, slots=True) +class XlsxTextObject: + sheet_index: int + anchor: str + row: int + column: int + kind_rank: int + source_ordinal: int + paragraphs: tuple[str, ...] = () + link_label: str | None = None + link_target: str | None = None + safe_link: bool = False + + +@dataclass(frozen=True, slots=True) +class XlsxTextWarning: + code: str + sheet_index: int + anchor: str + object_ordinal: int + detail: str + + +@dataclass(frozen=True, slots=True) +class XlsxTextObjects: + by_sheet: tuple[tuple[XlsxTextObject, ...], ...] + warnings: tuple[XlsxTextWarning, ...] + + +@dataclass(slots=True) +class _SheetCollector: + sheet: XlsxPreflightSheet + objects: list[XlsxTextObject] + warnings: list[XlsxTextWarning] + next_ordinal: int = 1 + + def add_object( + self, + *, + anchor: str, + kind_rank: int, + paragraphs: tuple[str, ...] = (), + link_label: str | None = None, + link_target: str | None = None, + safe_link: bool = False, + ) -> int: + row, column = _top_left(anchor) + ordinal = self.next_ordinal + self.next_ordinal += 1 + self.objects.append( + XlsxTextObject( + sheet_index=self.sheet.sheet_index, + anchor=anchor, + row=row, + column=column, + kind_rank=kind_rank, + source_ordinal=ordinal, + paragraphs=paragraphs, + link_label=link_label, + link_target=link_target, + safe_link=safe_link, + ) + ) + return ordinal + + def add_warning( + self, + code: str, + *, + anchor: str, + detail: str, + object_ordinal: int | None = None, + ) -> None: + ordinal = object_ordinal + if ordinal is None: + ordinal = self.next_ordinal + self.next_ordinal += 1 + else: + self.next_ordinal = max(self.next_ordinal, ordinal + 1) + self.warnings.append( + XlsxTextWarning( + code=code, + sheet_index=self.sheet.sheet_index, + anchor=anchor, + object_ordinal=ordinal, + detail=detail, + ) + ) + + +def _safe_root(archive: ZipFile, part_name: str, *, message: str) -> Any: + try: + data = archive.read(part_name) + return DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (KeyError, OSError, DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _top_left(anchor: str) -> tuple[int, int]: + try: + minimum_column, minimum_row, _, _ = range_boundaries(anchor) + except ValueError as error: + raise CorruptDocumentError("XLSX object anchor is invalid") from error + return minimum_row, minimum_column + + +def _relationship_index( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, +) -> dict[str, Any]: + return _read_relationships( + archive, + infos, + part_name, + required=_relationships_part(part_name) in infos, + ) + + +def _relationship( + relationships: dict[str, Any], + relationship_id: str | None, + *, + expected_type: str, +) -> Any: + relationship = relationships.get(relationship_id or "") + if relationship is None or relationship.relationship_type != expected_type: + raise CorruptDocumentError("XLSX object relationship is invalid") + return relationship + + +def _person_names( + archive: ZipFile, + infos: dict[str, ZipInfo], +) -> dict[str, str]: + relationships = _relationship_index(archive, infos, "xl/workbook.xml") + person_targets = [ + relationship.target + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP and not relationship.external + ] + if ( + any( + relationship.external + for relationship in relationships.values() + if relationship.relationship_type == _PERSON_RELATIONSHIP + ) + or len(person_targets) > 1 + ): + raise CorruptDocumentError("XLSX persons relationship is invalid") + if not person_targets: + return {} + root = _safe_root(archive, person_targets[0], message="XLSX persons part is corrupt") + if root.tag != f"{{{_THREADED_COMMENTS_NS}}}personList": + raise CorruptDocumentError("XLSX persons namespace is invalid") + people: dict[str, str] = {} + for person in root.findall(f"{{{_THREADED_COMMENTS_NS}}}person"): + person_id = person.get("id") + display_name = person.get("displayName") + if not person_id or not display_name or person_id in people: + raise CorruptDocumentError("XLSX person entry is invalid") + people[person_id] = display_name + return people + + +def _classic_comments( + archive: ZipFile, + relationship: Any, + collector: _SheetCollector, +) -> None: + if relationship.external: + raise CorruptDocumentError("XLSX comments relationship is invalid") + root = _safe_root(archive, relationship.target, message="XLSX comments part is corrupt") + if root.tag != f"{{{_SPREADSHEET_NS}}}comments": + raise CorruptDocumentError("XLSX comments namespace is invalid") + authors_node = root.find(f"{{{_SPREADSHEET_NS}}}authors") + authors = ( + tuple(node.text or "" for node in authors_node.findall(f"{{{_SPREADSHEET_NS}}}author")) + if authors_node is not None + else () + ) + comment_list = root.find(f"{{{_SPREADSHEET_NS}}}commentList") + if comment_list is None: + raise CorruptDocumentError("XLSX comments part is corrupt") + for comment in comment_list.findall(f"{{{_SPREADSHEET_NS}}}comment"): + reference = comment.get("ref") + try: + author_index = int(comment.get("authorId", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX comment author is invalid") from error + if reference is None or not 0 <= author_index < len(authors): + raise CorruptDocumentError("XLSX comment author or anchor is invalid") + _top_left(reference) + text = "".join(node.text or "" for node in comment.iter(f"{{{_SPREADSHEET_NS}}}t")) + author = authors[author_index] + prefix = f"Comment by {author}: " if author else "Comment: " + collector.add_object( + anchor=reference, + kind_rank=_KIND_COMMENT, + paragraphs=(f"{prefix}{text}",), + ) + + +def _threaded_comments( + archive: ZipFile, + relationship: Any, + people: dict[str, str], + collector: _SheetCollector, +) -> None: + if relationship.external: + raise CorruptDocumentError("XLSX threaded comments relationship is invalid") + root = _safe_root( + archive, + relationship.target, + message="XLSX threaded comments part is corrupt", + ) + if root.tag != f"{{{_THREADED_COMMENTS_NS}}}ThreadedComments": + raise CorruptDocumentError("XLSX threaded comments namespace is invalid") + for comment in root.findall(f"{{{_THREADED_COMMENTS_NS}}}threadedComment"): + reference = comment.get("ref") + person_id = comment.get("personId") + if reference is None or not person_id or person_id not in people: + raise CorruptDocumentError("XLSX threaded comment person or anchor is invalid") + _top_left(reference) + text = "".join(node.text or "" for node in comment.iter(f"{{{_THREADED_COMMENTS_NS}}}text")) + collector.add_object( + anchor=reference, + kind_rank=_KIND_COMMENT, + paragraphs=(f"Threaded comment by {people[person_id]}: {text}",), + ) + + +def _safe_link_target(target: str) -> bool: + if target.startswith("#"): + return not any(character.isspace() for character in target) + if any(character.isspace() or ord(character) < 32 for character in target): + return False + return urlsplit(target).scheme.lower() in _SAFE_LINK_SCHEMES + + +def _internal_target(location: str) -> str: + return f"#{_quoted_location(location)}" + + +def _quoted_location(location: str) -> str: + return quote(location, safe="!$&'()*+,-./:;=@_~?") + + +def _hyperlinks( + worksheet_root: Any, + relationships: dict[str, Any], + collector: _SheetCollector, +) -> None: + for hyperlink in worksheet_root.iter(f"{{{_SPREADSHEET_NS}}}hyperlink"): + reference = hyperlink.get("ref") + if reference is None: + raise CorruptDocumentError("XLSX hyperlink anchor is invalid") + _top_left(reference) + relationship_id = hyperlink.get(_RELATIONSHIP_ID) + location = hyperlink.get("location") or "" + target = "" + if relationship_id is not None: + target = _relationship( + relationships, + relationship_id, + expected_type=_HYPERLINK_RELATIONSHIP, + ).target + if target and location: + target = f"{target}#{_quoted_location(location)}" + elif location: + target = location if location.startswith("[") else _internal_target(location) + if not target: + raise CorruptDocumentError("XLSX hyperlink target is missing") + safe = _safe_link_target(target) + ordinal = collector.add_object( + anchor=reference, + kind_rank=_KIND_HYPERLINK, + link_label=hyperlink.get("display"), + link_target=target, + safe_link=safe, + ) + if not safe: + collector.add_warning( + "xlsx_external_reference", + anchor=reference, + object_ordinal=ordinal, + detail="hyperlink target was preserved as plain text without access", + ) + + +def _drawing_anchor(anchor: Any) -> str: + local_name = anchor.tag.rsplit("}", 1)[-1] + if local_name == "absoluteAnchor": + return "A1" + start = anchor.find(f"{{{_DRAWING_NS}}}from") + if start is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + + def coordinate(marker: Any) -> tuple[int, int]: + try: + column = int(marker.findtext(f"{{{_DRAWING_NS}}}col", "")) + 1 + row = int(marker.findtext(f"{{{_DRAWING_NS}}}row", "")) + 1 + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if not 1 <= column <= 16_384 or not 1 <= row <= 1_048_576: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + return row, column + + start_row, start_column = coordinate(start) + start_address = f"{get_column_letter(start_column)}{start_row}" + if local_name == "oneCellAnchor": + return start_address + end = anchor.find(f"{{{_DRAWING_NS}}}to") + if end is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_row, end_column = coordinate(end) + if end_row < start_row or end_column < start_column: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_address = f"{get_column_letter(end_column)}{end_row}" + return start_address if start_address == end_address else f"{start_address}:{end_address}" + + +def _drawing_text_boxes( + archive: ZipFile, + relationship: Any, + collector: _SheetCollector, +) -> None: + if relationship.external: + raise CorruptDocumentError("XLSX drawing relationship is invalid") + root = _safe_root(archive, relationship.target, message="XLSX drawing part is corrupt") + if root.tag != f"{{{_DRAWING_NS}}}wsDr": + raise CorruptDocumentError("XLSX drawing namespace is invalid") + for anchor in root: + if anchor.tag not in { + f"{{{_DRAWING_NS}}}oneCellAnchor", + f"{{{_DRAWING_NS}}}twoCellAnchor", + f"{{{_DRAWING_NS}}}absoluteAnchor", + }: + continue + address = _drawing_anchor(anchor) + for shape in anchor.iter(f"{{{_DRAWING_NS}}}sp"): + paragraphs: list[str] = [] + text_paragraphs = [ + "".join(node.text or "" for node in paragraph.iter(f"{{{_DRAWING_MAIN_NS}}}t")) + for paragraph in shape.iter(f"{{{_DRAWING_MAIN_NS}}}p") + ] + text = "\n".join(value for value in text_paragraphs if value) + if text: + paragraphs.append(f"Text box: {text}") + metadata = shape.find(f".//{{{_DRAWING_NS}}}cNvPr") + if metadata is not None: + for attribute, label in ( + ("name", "Shape name"), + ("descr", "Alt text"), + ("title", "Alt title"), + ): + value = metadata.get(attribute) + if value: + paragraphs.append(f"{label}: {value}") + if paragraphs: + collector.add_object( + anchor=address, + kind_rank=_KIND_TEXT_BOX, + paragraphs=tuple(paragraphs), + ) + + +def _split_header_footer(raw: str) -> tuple[dict[str, str], set[str]]: + sections = {"L": [], "C": [], "R": []} + image_sections: set[str] = set() + current = "C" + index = 0 + while index < len(raw): + if raw[index] != "&": + sections[current].append(raw[index]) + index += 1 + continue + if index + 1 >= len(raw): + sections[current].append("&") + break + code = raw[index + 1] + index += 2 + if code == "&": + sections[current].append("&") + elif code in sections: + current = code + elif code == '"': + closing = raw.find('"', index) + index = len(raw) if closing < 0 else closing + 1 + elif code == "K": + index = min(index + 6, len(raw)) + elif code == "[": + closing = raw.find("]", index) + if closing < 0: + sections[current].append("&[") + continue + field = raw[index:closing] + index = closing + 1 + if field == "Picture": + image_sections.add(current) + elif field in _HEADER_NAMED_FIELDS: + sections[current].append(_HEADER_NAMED_FIELDS[field]) + else: + sections[current].append(f"&[{field}]") + elif code.isdigit(): + while index < len(raw) and raw[index].isdigit(): + index += 1 + elif code in _HEADER_FIELDS: + sections[current].append(_HEADER_FIELDS[code]) + elif code == "G": + image_sections.add(current) + elif code in _HEADER_FORMAT_CODES: + continue + else: + sections[current].append(f"&{code}") + return ( + {section: "".join(value).strip() for section, value in sections.items()}, + image_sections, + ) + + +def _headers_and_footers(worksheet_root: Any, collector: _SheetCollector) -> None: + for tag, label in _HEADER_TAGS: + element = worksheet_root.find(f".//{{{_SPREADSHEET_NS}}}{tag}") + if element is None or element.text is None: + continue + sections, image_sections = _split_header_footer(element.text) + for section, section_label in _HEADER_SECTIONS: + value = sections[section] + ordinal: int | None = None + if value: + ordinal = collector.add_object( + anchor="A1", + kind_rank=_KIND_HEADER_FOOTER, + paragraphs=(f"{label} {section_label}: {value}",), + ) + if section in image_sections: + collector.add_warning( + "xlsx_unsupported_object", + anchor="A1", + object_ordinal=ordinal, + detail="header/footer image field was skipped", + ) + + +def _visible_vml_text(archive: ZipFile, part_name: str) -> bool: + root = _safe_root(archive, part_name, message="XLSX VML drawing part is corrupt") + return any( + "".join(node.itertext()).strip() + for node in root.iter() + if isinstance(node.tag, str) and node.tag.rsplit("}", 1)[-1] == "textbox" + ) + + +def _unsupported_objects( + archive: ZipFile, + worksheet_root: Any, + collector: _SheetCollector, +) -> None: + for reference in collector.sheet.unsupported_objects: + if reference.kind == "vmlDrawing" and not _visible_vml_text(archive, reference.target): + continue + kind = _UNSUPPORTED_KIND.sub("_", reference.kind).strip("_") or "unknown" + collector.add_warning( + "xlsx_unsupported_object", + anchor="A1", + object_ordinal=reference.source_index + 1, + detail=f"unsupported {kind} object was skipped", + ) + for _extension in worksheet_root.iter(f"{{{_SPREADSHEET_NS}}}ext"): + collector.add_warning( + "xlsx_unsupported_object", + anchor="A1", + detail="unsupported vendor extension was skipped", + ) + + +def _external_reference( + collector: _SheetCollector, + target: str, + *, + detail: str, +) -> None: + ordinal = collector.add_object( + anchor="A1", + kind_rank=_KIND_EXTERNAL_REFERENCE, + paragraphs=(f"External reference: {target}",), + ) + collector.add_warning( + "xlsx_external_reference", + anchor="A1", + object_ordinal=ordinal, + detail=detail, + ) + + +def _workbook_external_references( + archive: ZipFile, + infos: dict[str, ZipInfo], + collector: _SheetCollector, +) -> None: + relationships = _relationship_index(archive, infos, "xl/workbook.xml") + for relationship in relationships.values(): + if relationship.relationship_type == _EXTERNAL_LINK_RELATIONSHIP: + if relationship.external: + _external_reference( + collector, + relationship.target, + detail="external workbook reference was preserved without access", + ) + continue + nested = _relationship_index(archive, infos, relationship.target) + for nested_relationship in nested.values(): + if nested_relationship.relationship_type == _EXTERNAL_LINK_PATH_RELATIONSHIP: + _external_reference( + collector, + nested_relationship.target, + detail="external workbook reference was preserved without access", + ) + elif relationship.relationship_type == _CONNECTIONS_RELATIONSHIP: + if relationship.external: + raise CorruptDocumentError("XLSX connections relationship is invalid") + root = _safe_root( + archive, + relationship.target, + message="XLSX connections part is corrupt", + ) + if root.tag != f"{{{_SPREADSHEET_NS}}}connections": + raise CorruptDocumentError("XLSX connections namespace is invalid") + for connection in root.findall(f"{{{_SPREADSHEET_NS}}}connection"): + for attribute in ("sourceFile", "odcFile", "connectionFile"): + target = connection.get(attribute) + if target: + _external_reference( + collector, + target, + detail="external data reference was preserved without access", + ) + + +def read_xlsx_text_objects(path: Path, preflight: XlsxPreflight) -> XlsxTextObjects: + collectors = { + sheet.sheet_index: _SheetCollector(sheet=sheet, objects=[], warnings=[]) + for sheet in preflight.sheets + } + try: + with ZipFile(path) as archive: + infos = {info.filename: info for info in archive.infolist()} + people = _person_names(archive, infos) + for sheet in preflight.sheets: + collector = collectors[sheet.sheet_index] + worksheet_root = _safe_root( + archive, + sheet.part_name, + message="XLSX worksheet part is corrupt", + ) + if sheet.kind.value != "worksheet": + _unsupported_objects(archive, worksheet_root, collector) + continue + if worksheet_root.tag != f"{{{_SPREADSHEET_NS}}}worksheet": + raise CorruptDocumentError("XLSX worksheet namespace is invalid") + relationships = _relationship_index(archive, infos, sheet.part_name) + _hyperlinks(worksheet_root, relationships, collector) + for relationship in relationships.values(): + if relationship.relationship_type == _COMMENTS_RELATIONSHIP: + _classic_comments(archive, relationship, collector) + elif relationship.relationship_type == _THREADED_COMMENTS_RELATIONSHIP: + _threaded_comments(archive, relationship, people, collector) + for drawing in worksheet_root.iter(f"{{{_SPREADSHEET_NS}}}drawing"): + relationship = _relationship( + relationships, + drawing.get(_RELATIONSHIP_ID), + expected_type=_DRAWING_RELATIONSHIP, + ) + _drawing_text_boxes(archive, relationship, collector) + _headers_and_footers(worksheet_root, collector) + _unsupported_objects(archive, worksheet_root, collector) + if preflight.sheets: + _workbook_external_references(archive, infos, collectors[1]) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + warnings = tuple( + warning for sheet in preflight.sheets for warning in collectors[sheet.sheet_index].warnings + ) + return XlsxTextObjects( + by_sheet=tuple(tuple(collectors[sheet.sheet_index].objects) for sheet in preflight.sheets), + warnings=warnings, + ) + + +def text_object_blocks( + item: XlsxTextObject, + *, + fallback_label: str, + object_index: int, +) -> tuple[Block, ...]: + marker = MarkdownBlock( + f"" + ) + if item.link_target is not None: + label = item.link_label or fallback_label or item.link_target + inline = ( + InlineLink(label, item.link_target) + if item.safe_link + else InlineText(label if label == item.link_target else f"{label} ({item.link_target})") + ) + return marker, ParagraphBlock((inline,)) + return (marker, *(ParagraphBlock((InlineText(text),)) for text in item.paragraphs)) diff --git a/tests/test_xlsx_extract.py b/tests/test_xlsx_extract.py index e513aee..e073cd4 100644 --- a/tests/test_xlsx_extract.py +++ b/tests/test_xlsx_extract.py @@ -1,11 +1,14 @@ from __future__ import annotations +import socket +import urllib.request from pathlib import Path from zipfile import ZipFile import pytest from openpyxl import Workbook from openpyxl.chart import BarChart, Reference +from openpyxl.comments import Comment from openpyxl.formatting.rule import Rule from openpyxl.styles import Font from openpyxl.styles.differential import DifferentialStyle @@ -16,19 +19,68 @@ from opendocs._models import ( DocumentType, HeadingBlock, + InlineLink, + InlineText, MarkdownBlock, ParagraphBlock, ParsedDocument, SpannedTableBlock, TableBlock, ) -from opendocs.errors import LimitExceededError +from opendocs.errors import CorruptDocumentError, LimitExceededError from opendocs.markdown import render_markdown from opendocs.parsers.xlsx.extract import extract_xlsx from opendocs.parsers.xlsx.models import XlsxNativeSlot, XlsxSheet, XlsxSheetKind, XlsxSheetState from opendocs.parsers.xlsx.preflight import preflight_xlsx from tests.xlsx_fixtures import rewrite_xlsx +SHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +THREADED_REL_NS = "http://schemas.microsoft.com/office/2017/10/relationships" +THREADED_NS = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" + + +def _append_before(xml: bytes, closing_tag: bytes, body: str) -> bytes: + assert closing_tag in xml + return xml.replace(closing_tag, body.encode() + closing_tag) + + +def _with_relationship_namespace(xml: bytes) -> bytes: + if b"xmlns:r=" in xml: + return xml + return xml.replace( + b" bytes: + values = "".join( + f'' + for relationship_id, relationship_type, target, external in items + ) + return f'{values}'.encode() + + +def _target_mode(external: bool) -> str: + return ' TargetMode="External"' if external else "" + + +def _paragraph_text(slot: XlsxNativeSlot) -> str: + return "\n".join( + "".join( + inline.label if isinstance(inline, InlineLink) else inline.text + for inline in block.inlines + ) + for block in slot.blocks + if isinstance(block, ParagraphBlock) + ) + def _save_ordered_workbook(path: Path) -> None: workbook = Workbook() @@ -363,3 +415,391 @@ def test_repeated_extraction_is_deterministic(tmp_path: Path) -> None: index = preflight_xlsx(path) assert extract_xlsx(path, index) == extract_xlsx(path, index) + + +def test_extracts_comments_threaded_text_boxes_links_and_headers_in_stable_order( + tmp_path: Path, +) -> None: + path = tmp_path / "text-objects.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Objects" + sheet.append(("HTTP [docs]", "HTTPS", "Mail", "Jump")) + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + workbook_rels = archive.read("xl/_rels/workbook.xml.rels") + worksheet_xml = _append_before( + worksheet_xml, + b"", + ( + '' + '' + '' + '' + '' + "" + "&LOdd header left && kept &P &N" + "&C&D &T&R&"Arial"&KFF0000&12&B" + "Odd header right &F &Z &A &G" + "&LOdd footer left&COdd footer center&ROdd footer right" + "" + "&LEven header left&CEven header center&REven header right" + "" + "&LEven footer left&CEven footer center&REven footer right" + "" + "&LFirst header left&CFirst header center&RFirst header right" + "" + "&LFirst footer left&CFirst footer center&RFirst footer right" + "" + "" + ), + ) + workbook_rels = _append_before( + workbook_rels, + b"
", + ( + f'' + ), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdHttp", f"{OFFICE_REL_NS}/hyperlink", "http://example.test/a(b)", True), + ("rIdHttps", f"{OFFICE_REL_NS}/hyperlink", "https://example.test/two", True), + ("rIdMail", f"{OFFICE_REL_NS}/hyperlink", "mailto:team@example.test", True), + ("rIdComment", f"{OFFICE_REL_NS}/comments", "../comments1.xml", False), + ( + "rIdThreaded", + f"{THREADED_REL_NS}/threadedComment", + "../threadedComments/threadedComment1.xml", + False, + ), + ("rIdDrawing", f"{OFFICE_REL_NS}/drawing", "../drawings/drawing1.xml", False), + ), + "xl/comments1.xml": ( + f'Alice' + 'Classic note' + "" + ).encode(), + "xl/threadedComments/threadedComment1.xml": ( + f'Threaded reply' + "" + ).encode(), + "xl/persons/person.xml": ( + f'' + "" + ).encode(), + "xl/_rels/workbook.xml.rels": workbook_rels, + "xl/drawings/drawing1.xml": ( + f'' + "41" + '' + '' + "Box & text" + "" + "" + ).encode(), + }, + ) + + document = extract_xlsx(path, preflight_xlsx(path)) + + slots = _native_slots(document.sheets[0]) + assert [slot.source_index for slot in slots] == list(range(len(slots))) + assert slots[0].blocks[0] == MarkdownBlock("") + links = [ + inline + for slot in slots + for block in slot.blocks + if isinstance(block, ParagraphBlock) + for inline in block.inlines + if isinstance(inline, InlineLink) + ] + assert [(link.label, link.target) for link in links] == [ + ("HTTP [docs]", "http://example.test/a(b)"), + ("HTTPS", "https://example.test/two"), + ("Mail", "mailto:team@example.test"), + ("Jump", "#Objects!A1"), + ] + all_text = [_paragraph_text(slot) for slot in slots] + assert "Comment by Alice: Classic note" in all_text + assert "Threaded comment by Bob: Threaded reply" in all_text + assert any( + text == "Text box: Box & text\nShape name: Callout 1\nAlt text: Shape details\n" + "Alt title: Shape title" + for text in all_text + ) + header_text = [text for text in all_text if text.startswith(("Odd ", "Even ", "First "))] + assert header_text == [ + "Odd header left: Odd header left & kept {page} {pages}", + "Odd header center: {date} {time}", + "Odd header right: Odd header right {file} {path} {sheet}", + "Odd footer left: Odd footer left", + "Odd footer center: Odd footer center", + "Odd footer right: Odd footer right", + "Even header left: Even header left", + "Even header center: Even header center", + "Even header right: Even header right", + "Even footer left: Even footer left", + "Even footer center: Even footer center", + "Even footer right: Even footer right", + "First header left: First header left", + "First header center: First header center", + "First header right: First header right", + "First footer left: First footer left", + "First footer center: First footer center", + "First footer right: First footer right", + ] + assert all("Arial" not in text and "FF0000" not in text for text in header_text) + assert [warning.code for warning in document.warnings] == ["xlsx_unsupported_object"] + assert "sheet=1 anchor=A1" in document.warnings[0].message + assert "header/footer image field" in document.warnings[0].message + assert all( + "Objects" not in block.markdown + for slot in slots[1:] + for block in slot.blocks + if isinstance(block, MarkdownBlock) + ) + rendered = render_markdown( + ParsedDocument( + DocumentType.XLSX, + tuple(block for slot in slots for block in slot.blocks), + document.warnings, + ), + max_output_chars=100_000, + ) + assert "[HTTP \\[docs\\]](http://example.test/a\\(b\\))" in rendered.markdown + + +def test_standard_note_vml_presentation_is_not_reported_as_lost_text(tmp_path: Path) -> None: + path = tmp_path / "standard-note.xlsx" + workbook = Workbook() + workbook.active["B2"].comment = Comment("A standard note", "Reviewer") + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + paragraphs = [_paragraph_text(slot) for slot in _native_slots(document.sheets[0])] + assert "Comment by Reviewer: A standard note" in paragraphs + assert not any( + warning.code == "xlsx_unsupported_object" and "vmlDrawing" in warning.message + for warning in document.warnings + ) + + +def test_unsafe_hyperlinks_and_remote_data_are_plain_text_and_never_accessed( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "external.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.append(("Script", "File", "Book")) + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + workbook_rels = archive.read("xl/_rels/workbook.xml.rels") + worksheet_xml = _append_before( + worksheet_xml, + b"", + ( + '' + '' + '' + ), + ) + workbook_rels = _append_before( + workbook_rels, + b"", + ( + f'' + f'' + ), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdScript", f"{OFFICE_REL_NS}/hyperlink", "javascript:alert(1)", True), + ("rIdFile", f"{OFFICE_REL_NS}/hyperlink", "file:///tmp/private.xlsx", True), + ("rIdBook", f"{OFFICE_REL_NS}/hyperlink", "../other.xlsx", True), + ), + "xl/_rels/workbook.xml.rels": workbook_rels, + "xl/externalLinks/externalLink1.xml": (f'').encode(), + "xl/externalLinks/_rels/externalLink1.xml.rels": _relationships( + ( + "rIdPath", + f"{OFFICE_REL_NS}/externalLinkPath", + "https://remote.example.test/book.xlsx", + True, + ), + ), + "xl/connections.xml": ( + f'' + "" + ).encode(), + }, + ) + + def deny_network(*args: object, **kwargs: object) -> None: + raise AssertionError(f"network access attempted: {args!r} {kwargs!r}") + + monkeypatch.setattr(socket, "create_connection", deny_network) + monkeypatch.setattr(urllib.request, "urlopen", deny_network) + for module_name in ("httpx", "requests"): + module = pytest.importorskip(module_name) + monkeypatch.setattr(module, "get", deny_network) + + document = extract_xlsx(path, preflight_xlsx(path)) + + paragraphs = [ + block + for slot in _native_slots(document.sheets[0]) + for block in slot.blocks + if isinstance(block, ParagraphBlock) + ] + assert not any( + isinstance(inline, InlineLink) for paragraph in paragraphs for inline in paragraph.inlines + ) + plain = "\n".join( + inline.text + for paragraph in paragraphs + for inline in paragraph.inlines + if isinstance(inline, InlineText) + ) + assert "Script (javascript:alert(1))" in plain + assert "File (file:///tmp/private.xlsx)" in plain + assert "Book (../other.xlsx)" in plain + assert "https://remote.example.test/book.xlsx" in plain + assert "https://remote.example.test/data.csv" in plain + external_warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_external_reference" + ] + assert len(external_warnings) == 5 + assert all("sheet=1" in warning.message for warning in external_warnings) + + +def test_unsupported_xlsx_objects_are_locatable_and_aggregated(tmp_path: Path) -> None: + path = tmp_path / "unsupported.xlsx" + workbook = Workbook() + workbook.active["A1"] = "kept" + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + worksheet_xml = _append_before( + worksheet_xml, + b"", + '', + ) + unsupported = ( + ("rIdSmartArt", f"{OFFICE_REL_NS}/diagramData", "../diagrams/data1.xml", False), + ("rIdOle", f"{OFFICE_REL_NS}/oleObject", "../embeddings/ole1.bin", False), + ("rIdActiveX", f"{OFFICE_REL_NS}/activeXControl", "../activeX/activeX1.xml", False), + ("rIdControl", f"{OFFICE_REL_NS}/control", "../controls/control1.xml", False), + ("rIdVml", f"{OFFICE_REL_NS}/vmlDrawing", "../drawings/vmlDrawing1.vml", False), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships(*unsupported), + "xl/diagrams/data1.xml": b'', + "xl/embeddings/ole1.bin": b"ole", + "xl/activeX/activeX1.xml": b'', + "xl/controls/control1.xml": b'', + "xl/drawings/vmlDrawing1.vml": ( + b'' + b"VML text" + ), + }, + ) + + document = extract_xlsx(path, preflight_xlsx(path)) + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_unsupported_object" + ] + assert len(warnings) == 6 + assert all("sheet=1" in warning.message for warning in warnings) + assert all("object=" in warning.message for warning in warnings) + assert any("diagramData" in warning.message for warning in warnings) + assert any("oleObject" in warning.message for warning in warnings) + assert any("activeXControl" in warning.message for warning in warnings) + assert any("control" in warning.message for warning in warnings) + assert any("vmlDrawing" in warning.message for warning in warnings) + assert any("vendor extension" in warning.message for warning in warnings) + + +def test_unsupported_object_warnings_keep_twenty_then_summarize(tmp_path: Path) -> None: + path = tmp_path / "many-unsupported.xlsx" + workbook = Workbook() + workbook.active["A1"] = "kept" + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = archive.read("xl/worksheets/sheet1.xml") + extensions = "".join( + f'' for index in range(22) + ) + worksheet_xml = _append_before( + worksheet_xml, + b"", + f"{extensions}", + ) + rewrite_xlsx(path, {"xl/worksheets/sheet1.xml": worksheet_xml}) + + document = extract_xlsx(path, preflight_xlsx(path)) + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_unsupported_object" + ] + assert len(warnings) == 21 + assert warnings[-1].message == "2 additional xlsx_unsupported_object warnings suppressed" + + +@pytest.mark.parametrize("relationship_id", ["missing", "wrong-type"]) +def test_hyperlink_relationship_must_exist_with_the_expected_type( + tmp_path: Path, + relationship_id: str, +) -> None: + path = tmp_path / f"bad-link-{relationship_id}.xlsx" + workbook = Workbook() + workbook.active["A1"] = "link" + workbook.save(path) + with ZipFile(path) as archive: + worksheet_xml = _with_relationship_namespace(archive.read("xl/worksheets/sheet1.xml")) + worksheet_xml = _append_before( + worksheet_xml, + b"", + f'', + ) + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": worksheet_xml, + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ( + "wrong-type", + f"{OFFICE_REL_NS}/comments", + "../comments1.xml", + False, + ), + ), + "xl/comments1.xml": ( + f'' + ).encode(), + }, + ) + + with pytest.raises(CorruptDocumentError, match="relationship"): + preflight_xlsx(path) diff --git a/tests/test_xlsx_models.py b/tests/test_xlsx_models.py index 765fdad..488c999 100644 --- a/tests/test_xlsx_models.py +++ b/tests/test_xlsx_models.py @@ -5,7 +5,14 @@ import pytest -from opendocs._models import PageBreakBlock, TableBlock, TextBlock, WarningRecord +from opendocs._models import ( + InlineLink, + PageBreakBlock, + ParagraphBlock, + TableBlock, + TextBlock, + WarningRecord, +) from opendocs._native_protocol import MAX_FRAME_BYTES, encode_message from opendocs.errors import LimitExceededError from opendocs.parsers.xlsx.models import ( @@ -35,6 +42,9 @@ def _document() -> XlsxDocument: anchor="A1:B2", blocks=( TextBlock("alpha"), + ParagraphBlock( + (InlineLink("docs [safe]", "https://example.test/a(b)"),) + ), TableBlock((("head", None), ("value", "tail")), header_rows=0), ), ), diff --git a/tests/test_xlsx_preflight.py b/tests/test_xlsx_preflight.py index d3486a5..d96208b 100644 --- a/tests/test_xlsx_preflight.py +++ b/tests/test_xlsx_preflight.py @@ -701,6 +701,98 @@ def test_unknown_relationship_objects_are_deterministically_locatable(tmp_path: ] == [(0, "oleObject", "rIdOle", "xl/embeddings/ole1.bin")] +def test_threaded_comments_and_person_text_are_preflighted_before_extraction( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "threaded-comments.xlsx" + _one_sheet(path) + threaded_relationship = ( + "http://schemas.microsoft.com/office/2017/10/relationships/threadedComment" + ) + person_relationship = "http://schemas.microsoft.com/office/2017/10/relationships/person" + threaded_namespace = "http://schemas.microsoft.com/office/spreadsheetml/2018/threadedcomments" + with ZipFile(path) as archive: + workbook_relationships = archive.read("xl/_rels/workbook.xml.rels") + workbook_relationships = workbook_relationships.replace( + b"", + ( + f'' + ).encode(), + ) + rewrite_xlsx( + path, + { + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ( + "rIdThreaded", + threaded_relationship, + "../threadedComments/threadedComment1.xml", + ), + ), + "xl/threadedComments/threadedComment1.xml": ( + f'' + 'hello' + "" + ).encode(), + "xl/persons/person.xml": ( + f'' + ).encode(), + "xl/_rels/workbook.xml.rels": workbook_relationships, + }, + ) + + result = preflight_xlsx(path) + + assert result.usage.hyperlinks_and_comments == 1 + assert result.native_text_chars >= len("helloReviewer") + assert result.sheets[0].unsupported_objects == () + + monkeypatch.setattr(preflight_module, "MAX_HYPERLINKS_AND_COMMENTS", 0) + with pytest.raises(LimitExceededError, match="hyperlink and comment"): + preflight_xlsx(path) + + +def test_drawing_shape_metadata_is_included_in_native_text_budget( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "drawing-text-budget.xlsx" + _one_sheet(path) + drawing_ns = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" + drawing_main_ns = "http://schemas.openxmlformats.org/drawingml/2006/main" + rewrite_xlsx( + path, + { + "xl/worksheets/sheet1.xml": _worksheet( + '' + ), + "xl/worksheets/_rels/sheet1.xml.rels": _relationships( + ("rIdDrawing", f"{OFFICE_REL_NS}/drawing", "../drawings/drawing1.xml"), + ), + "xl/drawings/drawing1.xml": ( + f'' + "00" + '' + '' + "Text" + "" + "" + ).encode(), + }, + ) + + result = preflight_xlsx(path) + expected = len("NamedDescriptionTitleText") + assert result.native_text_chars >= expected + + monkeypatch.setattr(preflight_module, "MAX_NATIVE_TEXT_CHARS", expected - 1) + with pytest.raises(LimitExceededError, match="native text"): + preflight_xlsx(path) + + @pytest.mark.parametrize( "xml", [ From 6fa44604e3e64f8d1a825345c2c6ffc4fe014514 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 16:11:17 +0800 Subject: [PATCH 05/12] =?UTF-8?q?=E4=B8=BA=20XLSX=20=E5=9B=BE=E8=A1=A8?= =?UTF-8?q?=E4=BF=9D=E7=95=99=E5=8E=9F=E7=94=9F=E4=BA=8B=E5=AE=9E=E5=B9=B6?= =?UTF-8?q?=E6=8E=A5=E9=80=9A=E8=A7=86=E8=A7=89=E5=85=A5=E5=8F=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Constraint: 图表数值以 OOXML 原生事实为准,视觉只补充趋势、标注和含义。 Rejected: 不使用 openpyxl 私有图表对象,也不把语义卡冒充 Excel 原始外观。 Confidence: high Scope-risk: 视觉调度、结果合并与公共 API 由后续编排单元完成。 Tested: 189 focused tests; ruff; format; ty; diff-check. Not-tested: real XLSX workbooks or live vision providers. --- src/opendocs/parsers/xlsx/extract.py | 97 ++- src/opendocs/parsers/xlsx/media.py | 977 +++++++++++++++++++++++++++ src/opendocs/parsers/xlsx/models.py | 41 +- tests/test_xlsx_extract.py | 535 ++++++++++++++- tests/test_xlsx_models.py | 5 + tests/test_xlsx_parser.py | 67 ++ 6 files changed, 1701 insertions(+), 21 deletions(-) create mode 100644 src/opendocs/parsers/xlsx/media.py create mode 100644 tests/test_xlsx_parser.py diff --git a/src/opendocs/parsers/xlsx/extract.py b/src/opendocs/parsers/xlsx/extract.py index 50649ba..f371c79 100644 --- a/src/opendocs/parsers/xlsx/extract.py +++ b/src/opendocs/parsers/xlsx/extract.py @@ -25,8 +25,11 @@ WarningRecord, ) from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.xlsx.media import XlsxVisualOccurrence, read_xlsx_visual_objects from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, XlsxDocument, + XlsxImageSlot, XlsxNativeSlot, XlsxSheet, XlsxSheetKind, @@ -639,21 +642,13 @@ def _sheet_prelude(sheet: XlsxPreflightSheet) -> XlsxNativeSlot: def _sheet_slots( worksheet: Any, - records: tuple[_CellRecord, ...], *, - epoch: datetime, sheet: XlsxPreflightSheet, - warnings: _WarningCollector, budget: _MaterializationBudget, text_objects: tuple[XlsxTextObject, ...], + texts: dict[_Coordinate, str], + semantic: set[_Coordinate], ) -> tuple[XlsxNativeSlot, ...]: - texts, semantic = _cell_texts( - worksheet, - records, - epoch=epoch, - sheet=sheet, - warnings=warnings, - ) candidates: list[_SlotCandidate] = [] for ordinal, spec in enumerate(_region_specs(worksheet, semantic), start=1): budget.consume(_area(spec.bounds)) @@ -744,11 +739,47 @@ def _non_worksheet_slots( return tuple(slots) -def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: +def _visual_slot( + occurrence: XlsxVisualOccurrence, + *, + source_index: int, +) -> XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot: + if occurrence.artifact_name is None or occurrence.content_sha256 is None: + return XlsxNativeSlot(source_index, occurrence.anchor, occurrence.blocks) + if occurrence.kind == "image": + return XlsxImageSlot( + source_index=source_index, + anchor=occurrence.anchor, + artifact_name=occurrence.artifact_name, + content_sha256=occurrence.content_sha256, + alt_text=occurrence.alt_text, + object_name=occurrence.object_name, + title=occurrence.title, + ) + return XlsxChartSlot( + source_index=source_index, + anchor=occurrence.anchor, + artifact_name=occurrence.artifact_name, + content_sha256=occurrence.content_sha256, + blocks=occurrence.blocks, + alt_text=occurrence.alt_text, + object_name=occurrence.object_name, + title=occurrence.title, + ) + + +def extract_xlsx( + path: Path, + preflight: XlsxPreflight, + *, + artifact_dir: Path | None = None, +) -> XlsxDocument: if not isinstance(path, Path): raise TypeError("path must be a Path") if not isinstance(preflight, XlsxPreflight): raise TypeError("preflight must be an XlsxPreflight") + if artifact_dir is not None and not isinstance(artifact_dir, Path): + raise TypeError("artifact_dir must be a Path or None") sidecars = _read_sidecars(path, preflight) text_objects = read_xlsx_text_objects(path, preflight) warnings = _WarningCollector() @@ -772,6 +803,32 @@ def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: ) try: worksheets = {worksheet.title: worksheet for worksheet in workbook.worksheets} + sheet_values: dict[int, tuple[dict[_Coordinate, str], set[_Coordinate]]] = {} + workbook_values: dict[str, dict[str, str]] = {} + for sheet in preflight.sheets: + if sheet.kind is XlsxSheetKind.CHARTSHEET: + workbook_values[sheet.name] = {} + continue + worksheet = worksheets.get(sheet.name) + if worksheet is None: + raise CorruptDocumentError("XLSX worksheet is missing after full-mode load") + texts, semantic = _cell_texts( + worksheet, + sidecars[sheet.sheet_index], + epoch=workbook.epoch, + sheet=sheet, + warnings=warnings, + ) + sheet_values[sheet.sheet_index] = (texts, semantic) + workbook_values[sheet.name] = { + f"{get_column_letter(column)}{row}": text for (row, column), text in texts.items() + } + visual_objects = read_xlsx_visual_objects( + path, + preflight, + workbook_values, + artifact_dir=artifact_dir, + ) sheets: list[XlsxSheet] = [] budget = _MaterializationBudget() for sheet in preflight.sheets: @@ -782,15 +839,24 @@ def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: worksheet = worksheets.get(sheet.name) if worksheet is None: raise CorruptDocumentError("XLSX worksheet is missing after full-mode load") + texts, semantic = sheet_values[sheet.sheet_index] slots = _sheet_slots( worksheet, - sidecars[sheet.sheet_index], - epoch=workbook.epoch, sheet=sheet, - warnings=warnings, budget=budget, text_objects=sheet_text_objects, + texts=texts, + semantic=semantic, ) + slots = ( + *slots, + *( + _visual_slot(occurrence, source_index=len(slots) + offset) + for offset, occurrence in enumerate( + visual_objects.by_sheet[sheet.sheet_index - 1] + ) + ), + ) sheets.append( XlsxSheet( sheet_index=sheet.sheet_index, @@ -802,4 +868,5 @@ def extract_xlsx(path: Path, preflight: XlsxPreflight) -> XlsxDocument: ) finally: workbook.close() - return XlsxDocument(sheets=tuple(sheets), warnings=warnings.freeze()) + combined_warnings = (*warnings.freeze(), *visual_objects.warnings) + return XlsxDocument(sheets=tuple(sheets), warnings=combined_warnings) diff --git a/src/opendocs/parsers/xlsx/media.py b/src/opendocs/parsers/xlsx/media.py new file mode 100644 index 0000000..59c014d --- /dev/null +++ b/src/opendocs/parsers/xlsx/media.py @@ -0,0 +1,977 @@ +from __future__ import annotations + +import hashlib +import io +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Literal +from zipfile import BadZipFile, ZipFile, ZipInfo + +from defusedxml import ElementTree as DefusedET +from defusedxml.common import DefusedXmlException +from openpyxl.utils.cell import get_column_letter, range_boundaries +from PIL import Image, ImageDraw, ImageFont + +from opendocs._models import ( + Block, + HeadingBlock, + InlineText, + ParagraphBlock, + TableBlock, + WarningRecord, +) +from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.parsers.xlsx.models import XlsxChartSlot, XlsxDocument, XlsxImageSlot +from opendocs.parsers.xlsx.preflight import ( + MAX_CHART_CACHE_POINTS, + MAX_DRAWING_OBJECTS, + XlsxPreflight, + XlsxPreflightSheet, + _read_relationships, + _relationships_part, +) +from opendocs.vision.base import VisionRequest, VisionRequestKind +from opendocs.vision.images import PreparedImage, prepare_image + +_SPREADSHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_OFFICE_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_DRAWING_NS = "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing" +_DRAWING_MAIN_NS = "http://schemas.openxmlformats.org/drawingml/2006/main" +_CHART_NS = "http://schemas.openxmlformats.org/drawingml/2006/chart" +_DRAWING_RELATIONSHIP = f"{_OFFICE_REL_NS}/drawing" +_CHART_RELATIONSHIP = f"{_OFFICE_REL_NS}/chart" +_IMAGE_RELATIONSHIP = f"{_OFFICE_REL_NS}/image" +_RELATIONSHIP_ID = f"{{{_OFFICE_REL_NS}}}id" +_RELATIONSHIP_EMBED = f"{{{_OFFICE_REL_NS}}}embed" + +_CHART_TYPES = { + "lineChart": "line", + "barChart": "bar", + "pieChart": "pie", + "doughnutChart": "doughnut", + "scatterChart": "scatter", +} +_AXIS_TYPES = {"catAx", "valAx", "dateAx", "serAx"} +_LOCAL_RANGE = re.compile( + r"^(?P'(?:[^']|'')+'|[^'!\[\]]+)!" + r"(?P\$?[A-Z]{1,3}\$?[1-9][0-9]{0,6})" + r"(?::(?P\$?[A-Z]{1,3}\$?[1-9][0-9]{0,6}))?$", + re.IGNORECASE, +) +_SAFE_ARTIFACT_SUFFIX = re.compile(r"^\.[A-Za-z0-9]{1,8}$") +_PREVIEW_WIDTH = 1_280 +_PREVIEW_PADDING = 40 +_PREVIEW_LINE_HEIGHT = 26 +_PREVIEW_MAX_LINES = 80 +_PREVIEW_MAX_LINE_CHARS = 180 +_PREVIEW_SERIES_POINTS = 24 + +XLSX_CHART_VISION_PROMPT = ( + "这是由 XLSX 原生图表事实生成的语义卡片, 不是 Excel 外观还原。" + "仅补充可由卡片支持的趋势、关系、标注和含义; 不要改写原生数值, " + "不要猜测缺失数据, 并将结论明确标记为 '视觉解释'。" +) +XLSX_IMAGE_VISION_PROMPT = ( + "仅解释图片中可见的文字、标注、关系和含义; 涉及趋势时只描述可见证据, " + "不要猜测未显示的内容, 并将结论明确标记为 '视觉解释'。" +) + + +@dataclass(frozen=True, slots=True) +class XlsxChartSeriesFacts: + name: str + categories: tuple[str, ...] + values: tuple[str, ...] + x_values: tuple[str, ...] + y_values: tuple[str, ...] + formulas: tuple[str, ...] + unresolved_formulas: tuple[str, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxChartFacts: + chart_type: str + title: str + axis_titles: tuple[str, ...] + axis_labels: tuple[str, ...] + data_labels: tuple[str, ...] + formulas: tuple[str, ...] + unresolved_formulas: tuple[str, ...] + series: tuple[XlsxChartSeriesFacts, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxVisualOccurrence: + kind: Literal["chart", "image"] + anchor: str + blocks: tuple[Block, ...] + artifact_name: str | None + content_sha256: str | None + alt_text: str | None + object_name: str | None + title: str | None + + +@dataclass(frozen=True, slots=True) +class XlsxVisualObjects: + by_sheet: tuple[tuple[XlsxVisualOccurrence, ...], ...] + warnings: tuple[WarningRecord, ...] + + +@dataclass(frozen=True, slots=True) +class XlsxVisualRequest: + digest: str + image_path: Path + prompt: str + source_index: int + kind: VisionRequestKind + + def to_vision_request(self) -> VisionRequest: + return VisionRequest( + image_path=self.image_path, + prompt=self.prompt, + source_index=self.source_index, + kind=self.kind, + ) + + +@dataclass(frozen=True, slots=True) +class _OccurrenceWarning: + code: str + anchor: str + detail: str + + +class _UnsupportedChartType(Exception): + pass + + +def _safe_root(archive: ZipFile, part_name: str, *, message: str) -> Any: + try: + data = archive.read(part_name) + return DefusedET.fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except (KeyError, OSError, DefusedXmlException, DefusedET.ParseError) as error: + raise CorruptDocumentError(message) from error + + +def _local_name(element: Any) -> str: + return str(element.tag).rsplit("}", 1)[-1] + + +def _relationship_index( + archive: ZipFile, + infos: dict[str, ZipInfo], + part_name: str, +) -> dict[str, Any]: + return _read_relationships( + archive, + infos, + part_name, + required=_relationships_part(part_name) in infos, + ) + + +def _relationship( + relationships: dict[str, Any], + relationship_id: str | None, + *, + expected_type: str, +) -> Any: + relationship = relationships.get(relationship_id or "") + if ( + relationship is None + or relationship.external + or relationship.relationship_type != expected_type + ): + raise CorruptDocumentError("XLSX drawing relationship is invalid") + return relationship + + +def _marker_coordinate(marker: Any) -> tuple[int, int]: + try: + column = int(marker.findtext(f"{{{_DRAWING_NS}}}col", "")) + 1 + row = int(marker.findtext(f"{{{_DRAWING_NS}}}row", "")) + 1 + except ValueError as error: + raise CorruptDocumentError("XLSX drawing anchor is invalid") from error + if not 1 <= column <= 16_384 or not 1 <= row <= 1_048_576: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + return row, column + + +def _drawing_anchor(element: Any) -> str: + kind = _local_name(element) + if kind == "absoluteAnchor": + return "A1" + start = element.find(f"{{{_DRAWING_NS}}}from") + if start is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + start_row, start_column = _marker_coordinate(start) + start_text = f"{get_column_letter(start_column)}{start_row}" + if kind == "oneCellAnchor": + return start_text + end = element.find(f"{{{_DRAWING_NS}}}to") + if kind != "twoCellAnchor" or end is None: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_row, end_column = _marker_coordinate(end) + if end_row < start_row or end_column < start_column: + raise CorruptDocumentError("XLSX drawing anchor is invalid") + end_text = f"{get_column_letter(end_column)}{end_row}" + return start_text if start_text == end_text else f"{start_text}:{end_text}" + + +def _metadata(anchor: Any) -> tuple[str | None, str | None, str | None]: + node = next(anchor.iter(f"{{{_DRAWING_NS}}}cNvPr"), None) + if node is None: + return None, None, None + return node.get("name"), node.get("descr"), node.get("title") + + +def _plain_text(element: Any | None) -> str: + if element is None: + return "" + rich = "".join(node.text or "" for node in element.iter(f"{{{_DRAWING_MAIN_NS}}}t")) + if rich: + return rich.strip() + values = "".join(node.text or "" for node in element.iter(f"{{{_CHART_NS}}}v")) + return values.strip() + + +def _cache_values(reference: Any) -> tuple[str, ...] | None: + cache = next( + ( + child + for child in reference + if _local_name(child) in {"strCache", "numCache", "multiLvlStrCache"} + ), + None, + ) + if cache is None: + return None + if _local_name(cache) == "multiLvlStrCache": + levels: list[dict[int, str]] = [] + for level in cache.findall(f"{{{_CHART_NS}}}lvl"): + indexed_level: dict[int, str] = {} + for point in level.findall(f"{{{_CHART_NS}}}pt"): + try: + index = int(point.get("idx", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache index is invalid") from error + if index < 0 or index in indexed_level: + raise CorruptDocumentError("XLSX chart cache index is invalid") + indexed_level[index] = point.findtext(f"{{{_CHART_NS}}}v", "") + levels.append(indexed_level) + declared_node = cache.find(f"{{{_CHART_NS}}}ptCount") + declared = max( + (max(level, default=-1) + 1 for level in levels), + default=0, + ) + if declared_node is not None: + try: + declared = int(declared_node.get("val", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache count is invalid") from error + if declared < 0 or declared > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + return tuple( + " / ".join(value for level in levels if (value := level.get(index, ""))) + for index in range(declared) + ) + indexed: dict[int, str] = {} + for point in cache.iter(f"{{{_CHART_NS}}}pt"): + try: + index = int(point.get("idx", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache index is invalid") from error + if index < 0 or index in indexed: + raise CorruptDocumentError("XLSX chart cache index is invalid") + indexed[index] = point.findtext(f"{{{_CHART_NS}}}v", "") + declared_node = next(cache.iter(f"{{{_CHART_NS}}}ptCount"), None) + declared = max(indexed, default=-1) + 1 + if declared_node is not None: + try: + declared = int(declared_node.get("val", "")) + except ValueError as error: + raise CorruptDocumentError("XLSX chart cache count is invalid") from error + if declared < 0 or declared > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + return tuple(indexed.get(index, "") for index in range(declared)) + + +def _unquote_sheet(value: str) -> str: + if value.startswith("'") and value.endswith("'"): + return value[1:-1].replace("''", "'") + return value + + +def _resolve_local_formula( + formula: str, + workbook_values: dict[str, dict[str, str]], +) -> tuple[str, ...] | None: + match = _LOCAL_RANGE.fullmatch(formula.removeprefix("=")) + if match is None: + return None + sheet_name = _unquote_sheet(match.group("sheet")) + values = workbook_values.get(sheet_name) + if values is None: + return None + start = match.group("start").replace("$", "").upper() + end = (match.group("end") or start).replace("$", "").upper() + try: + minimum_column, minimum_row, maximum_column, maximum_row = range_boundaries( + f"{start}:{end}" + ) + except ValueError: + return None + return tuple( + values.get(f"{get_column_letter(column)}{row}", "") + for row in range(minimum_row, maximum_row + 1) + for column in range(minimum_column, maximum_column + 1) + ) + + +def _reference_values( + container: Any | None, + workbook_values: dict[str, dict[str, str]], +) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]: + if container is None: + return (), (), () + reference = next( + ( + node + for node in container.iter() + if _local_name(node) in {"strRef", "numRef", "multiLvlStrRef"} + ), + None, + ) + if reference is None: + direct = container.find(f"{{{_CHART_NS}}}v") + return ((direct.text or "",), (), ()) if direct is not None else ((), (), ()) + formula = reference.findtext(f"{{{_CHART_NS}}}f", "").strip() + formulas = (formula,) if formula else () + resolved = _resolve_local_formula(formula, workbook_values) if formula else None + cached = _cache_values(reference) + if cached is not None: + return cached, formulas, () if resolved is not None else formulas + if resolved is not None: + return resolved, formulas, () + return (), formulas, formulas + + +def _series_facts( + element: Any, + *, + chart_type: str, + workbook_values: dict[str, dict[str, str]], +) -> XlsxChartSeriesFacts: + name_values, name_formulas, name_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}tx"), + workbook_values, + ) + categories, category_formulas, category_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}cat"), + workbook_values, + ) + values, value_formulas, value_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}val"), + workbook_values, + ) + x_values, x_formulas, x_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}xVal"), + workbook_values, + ) + y_values, y_formulas, y_unresolved = _reference_values( + element.find(f"{{{_CHART_NS}}}yVal"), + workbook_values, + ) + if chart_type == "scatter": + categories = () + values = () + formulas = tuple( + dict.fromkeys( + (*name_formulas, *category_formulas, *value_formulas, *x_formulas, *y_formulas) + ) + ) + unresolved = tuple( + dict.fromkeys( + ( + *name_unresolved, + *category_unresolved, + *value_unresolved, + *x_unresolved, + *y_unresolved, + ) + ) + ) + return XlsxChartSeriesFacts( + name=next((value for value in name_values if value), "Series"), + categories=categories, + values=values, + x_values=x_values, + y_values=y_values, + formulas=formulas, + unresolved_formulas=unresolved, + ) + + +def _data_label_facts(chart: Any) -> tuple[str, ...]: + facts: list[str] = [] + labels = chart.find(f"{{{_CHART_NS}}}dLbls") + if labels is None: + return () + flag_labels = { + "showLegendKey": "legend key", + "showVal": "value", + "showCatName": "category name", + "showSerName": "series name", + "showPercent": "percentage", + "showBubbleSize": "bubble size", + } + for name, label in flag_labels.items(): + node = labels.find(f"{{{_CHART_NS}}}{name}") + if node is not None and node.get("val", "1") not in {"0", "false", "False"}: + facts.append(label) + for item in labels.findall(f"{{{_CHART_NS}}}dLbl"): + text = _plain_text(item.find(f"{{{_CHART_NS}}}tx")) + if text: + facts.append(text) + return tuple(dict.fromkeys(facts)) + + +def _chart_facts( + root: Any, + workbook_values: dict[str, dict[str, str]], +) -> XlsxChartFacts: + if root.tag != f"{{{_CHART_NS}}}chartSpace": + raise CorruptDocumentError("XLSX chart namespace is invalid") + chart = root.find(f"{{{_CHART_NS}}}chart") + plot_area = chart.find(f"{{{_CHART_NS}}}plotArea") if chart is not None else None + if chart is None or plot_area is None: + raise CorruptDocumentError("XLSX chart part is corrupt") + chart_node = next( + (child for child in plot_area if _local_name(child) in _CHART_TYPES), + None, + ) + if chart_node is None: + raise _UnsupportedChartType + chart_type = _CHART_TYPES[_local_name(chart_node)] + chart_title = chart.find(f"{{{_CHART_NS}}}title") + title_values, title_formulas, title_unresolved = _reference_values( + chart_title.find(f"{{{_CHART_NS}}}tx") if chart_title is not None else None, + workbook_values, + ) + title = next((value for value in title_values if value), _plain_text(chart_title)) + axis_titles: list[str] = [] + axis_labels: list[str] = [] + axis_formulas: list[str] = [] + axis_unresolved: list[str] = [] + for axis in plot_area: + if _local_name(axis) not in _AXIS_TYPES: + continue + axis_title_node = axis.find(f"{{{_CHART_NS}}}title") + axis_title_values, formulas, unresolved = _reference_values( + axis_title_node.find(f"{{{_CHART_NS}}}tx") if axis_title_node is not None else None, + workbook_values, + ) + axis_title = next( + (value for value in axis_title_values if value), + _plain_text(axis_title_node), + ) + if axis_title: + axis_titles.append(axis_title) + axis_formulas.extend(formulas) + axis_unresolved.extend(unresolved) + label_position = axis.find(f"{{{_CHART_NS}}}tickLblPos") + if label_position is not None and label_position.get("val"): + axis_labels.append(label_position.get("val", "")) + return XlsxChartFacts( + chart_type=chart_type, + title=title, + axis_titles=tuple(axis_titles), + axis_labels=tuple(axis_labels), + data_labels=_data_label_facts(chart_node), + formulas=tuple(dict.fromkeys((*title_formulas, *axis_formulas))), + unresolved_formulas=tuple(dict.fromkeys((*title_unresolved, *axis_unresolved))), + series=tuple( + _series_facts(series, chart_type=chart_type, workbook_values=workbook_values) + for series in chart_node.findall(f"{{{_CHART_NS}}}ser") + ), + ) + + +def _paragraph(text: str) -> ParagraphBlock: + return ParagraphBlock((InlineText(text),)) + + +def chart_fact_blocks(facts: XlsxChartFacts) -> tuple[Block, ...]: + blocks: list[Block] = [ + HeadingBlock(2, (InlineText(facts.title or f"{facts.chart_type.title()} chart"),)), + _paragraph(f"Chart type: {facts.chart_type}"), + ] + if facts.axis_titles: + blocks.append(_paragraph(f"Axis titles: {'; '.join(facts.axis_titles)}")) + if facts.axis_labels: + blocks.append(_paragraph(f"Axis label positions: {'; '.join(facts.axis_labels)}")) + if facts.data_labels: + blocks.append(_paragraph(f"Data labels: {'; '.join(facts.data_labels)}")) + series_names = tuple(dict.fromkeys(series.name for series in facts.series)) + if series_names: + blocks.append(_paragraph(f"Series names: {'; '.join(series_names)}")) + formulas = tuple( + dict.fromkeys( + (*facts.formulas, *(formula for item in facts.series for formula in item.formulas)) + ) + ) + if formulas: + blocks.append(_paragraph(f"Local/formula references: {'; '.join(formulas)}")) + unresolved = tuple( + dict.fromkeys( + ( + *facts.unresolved_formulas, + *(formula for item in facts.series for formula in item.unresolved_formulas), + ) + ) + ) + if unresolved: + blocks.append( + _paragraph(f"References preserved without evaluation: {'; '.join(unresolved)}") + ) + rows: list[tuple[str, ...]] = [] + for series in facts.series: + if facts.chart_type == "scatter": + width = max(len(series.x_values), len(series.y_values)) + rows.extend( + ( + "Series", + series.name, + "X", + series.x_values[index] if index < len(series.x_values) else "", + "Y", + series.y_values[index] if index < len(series.y_values) else "", + ) + for index in range(width) + ) + else: + width = max(len(series.categories), len(series.values)) + rows.extend( + ( + "Series", + series.name, + "Category", + series.categories[index] if index < len(series.categories) else "", + "Value", + series.values[index] if index < len(series.values) else "", + ) + for index in range(width) + ) + if rows: + blocks.append(TableBlock(tuple(rows), header_rows=0)) + elif facts.series: + blocks.append(_paragraph("Chart series have no saved cache or resolvable local values.")) + else: + blocks.append(_paragraph("Chart has no readable series.")) + return tuple(blocks) + + +def _sample_pairs( + left: tuple[str, ...], + right: tuple[str, ...], +) -> tuple[tuple[str, str], ...]: + count = max(len(left), len(right)) + if count == 0: + return () + if count <= _PREVIEW_SERIES_POINTS: + indexes = range(count) + else: + indexes = tuple( + round(index * (count - 1) / (_PREVIEW_SERIES_POINTS - 1)) + for index in range(_PREVIEW_SERIES_POINTS) + ) + return tuple( + ( + left[index] if index < len(left) else "", + right[index] if index < len(right) else "", + ) + for index in indexes + ) + + +def _preview_lines(facts: XlsxChartFacts) -> tuple[str, ...]: + lines = [ + "视觉解释输入 / 非 Excel 外观还原", + f"Chart type: {facts.chart_type}", + f"Title: {facts.title or '(none)'}", + ] + if facts.axis_titles: + lines.append(f"Axis titles: {'; '.join(facts.axis_titles)}") + if facts.data_labels: + lines.append(f"Data labels: {'; '.join(facts.data_labels)}") + if facts.formulas: + lines.append(f"Chart references: {'; '.join(facts.formulas)}") + for series in facts.series: + lines.append(f"Series: {series.name}") + if facts.chart_type == "scatter": + count = max(len(series.x_values), len(series.y_values)) + pairs = _sample_pairs(series.x_values, series.y_values) + lines.append( + f"Points (sampled across {count}): " + "; ".join(f"({x}, {y})" for x, y in pairs) + ) + else: + count = max(len(series.categories), len(series.values)) + pairs = _sample_pairs(series.categories, series.values) + lines.append( + f"Values (sampled across {count}): " + + "; ".join(f"{category}={value}" for category, value in pairs) + ) + if series.formulas: + lines.append(f"References: {'; '.join(series.formulas)}") + normalized = [line[:_PREVIEW_MAX_LINE_CHARS] for line in lines[:_PREVIEW_MAX_LINES]] + if len(lines) > _PREVIEW_MAX_LINES: + normalized.append("(additional native facts omitted from preview only)") + return tuple(normalized) + + +def render_chart_semantic_preview(facts: XlsxChartFacts) -> bytes: + lines = _preview_lines(facts) + height = _PREVIEW_PADDING * 2 + _PREVIEW_LINE_HEIGHT * len(lines) + image = Image.new("RGB", (_PREVIEW_WIDTH, max(160, height)), "white") + try: + draw = ImageDraw.Draw(image) + font = ImageFont.load_default(size=18) + for index, line in enumerate(lines): + draw.text( + (_PREVIEW_PADDING, _PREVIEW_PADDING + index * _PREVIEW_LINE_HEIGHT), + line, + fill="black", + font=font, + ) + output = io.BytesIO() + image.save(output, format="PNG", optimize=False) + return output.getvalue() + finally: + image.close() + + +def _write_artifact(directory: Path, name: str, data: bytes) -> None: + directory.mkdir(parents=True, exist_ok=True) + root = directory.resolve() + path = directory / name + if path.resolve().parent != root: + raise ValueError("XLSX visual artifact escapes the artifact directory") + if path.exists(): + if path.read_bytes() != data: + raise CorruptDocumentError("XLSX visual artifact digest collision") + return + path.write_bytes(data) + + +def _chart_occurrence( + archive: ZipFile, + relationship: Any, + *, + anchor: str, + metadata: tuple[str | None, str | None, str | None], + workbook_values: dict[str, dict[str, str]], + artifact_dir: Path | None, + facts_cache: dict[str, XlsxChartFacts], +) -> tuple[XlsxVisualOccurrence, tuple[str, ...]]: + facts = facts_cache.get(relationship.target) + if facts is None: + root = _safe_root(archive, relationship.target, message="XLSX chart part is corrupt") + facts = _chart_facts(root, workbook_values) + facts_cache[relationship.target] = facts + blocks = chart_fact_blocks(facts) + artifact_name: str | None = None + digest: str | None = None + if artifact_dir is not None: + preview = render_chart_semantic_preview(facts) + digest = hashlib.sha256(preview).hexdigest() + artifact_name = f"xlsx-chart-{digest}.png" + _write_artifact(artifact_dir, artifact_name, preview) + object_name, alt_text, title = metadata + warnings = tuple( + dict.fromkeys( + ( + *facts.unresolved_formulas, + *(formula for series in facts.series for formula in series.unresolved_formulas), + ) + ) + ) + return ( + XlsxVisualOccurrence( + kind="chart", + anchor=anchor, + blocks=blocks, + artifact_name=artifact_name, + content_sha256=digest, + alt_text=alt_text, + object_name=object_name, + title=title, + ), + warnings, + ) + + +def _image_occurrence( + archive: ZipFile, + relationship: Any, + *, + anchor: str, + metadata: tuple[str | None, str | None, str | None], + artifact_dir: Path | None, +) -> XlsxVisualOccurrence: + try: + data = archive.read(relationship.target) + except (KeyError, OSError) as error: + raise CorruptDocumentError("XLSX embedded image part is corrupt") from error + digest = hashlib.sha256(data).hexdigest() + suffix = Path(relationship.target).suffix.lower() + if _SAFE_ARTIFACT_SUFFIX.fullmatch(suffix) is None: + suffix = ".bin" + artifact_name = f"xlsx-media-{digest}{suffix}" + if artifact_dir is not None: + _write_artifact(artifact_dir, artifact_name, data) + object_name, alt_text, title = metadata + description = tuple( + f"{label}: {value}" + for label, value in ( + ("Image name", object_name), + ("Image description", alt_text), + ("Image title", title), + ) + if value + ) + blocks: tuple[Block, ...] = ( + HeadingBlock(3, (InlineText("Embedded image"),)), + *(_paragraph(value) for value in description), + ) + return XlsxVisualOccurrence( + kind="image", + anchor=anchor, + blocks=blocks, + artifact_name=artifact_name if artifact_dir is not None else None, + content_sha256=digest if artifact_dir is not None else None, + alt_text=alt_text, + object_name=object_name, + title=title, + ) + + +def _sheet_drawings( + archive: ZipFile, + infos: dict[str, ZipInfo], + sheet: XlsxPreflightSheet, +) -> tuple[str, ...]: + root = _safe_root(archive, sheet.part_name, message="XLSX sheet part is corrupt") + relationships = _relationship_index(archive, infos, sheet.part_name) + targets: list[str] = [] + for node in root.iter(f"{{{_SPREADSHEET_NS}}}drawing"): + relationship = _relationship( + relationships, + node.get(_RELATIONSHIP_ID), + expected_type=_DRAWING_RELATIONSHIP, + ) + targets.append(relationship.target) + return tuple(targets) + + +def _drawing_occurrences( + archive: ZipFile, + infos: dict[str, ZipInfo], + drawing_part: str, + *, + workbook_values: dict[str, dict[str, str]], + artifact_dir: Path | None, + chart_facts_cache: dict[str, XlsxChartFacts], +) -> tuple[tuple[XlsxVisualOccurrence, ...], tuple[_OccurrenceWarning, ...]]: + root = _safe_root(archive, drawing_part, message="XLSX drawing part is corrupt") + if root.tag != f"{{{_DRAWING_NS}}}wsDr": + raise CorruptDocumentError("XLSX drawing namespace is invalid") + relationships = _relationship_index(archive, infos, drawing_part) + occurrences: list[XlsxVisualOccurrence] = [] + warnings: list[_OccurrenceWarning] = [] + for anchor_node in root: + if _local_name(anchor_node) not in {"oneCellAnchor", "twoCellAnchor", "absoluteAnchor"}: + continue + anchor = _drawing_anchor(anchor_node) + metadata = _metadata(anchor_node) + chart_node = next(anchor_node.iter(f"{{{_CHART_NS}}}chart"), None) + image_node = next(anchor_node.iter(f"{{{_DRAWING_MAIN_NS}}}blip"), None) + if chart_node is not None: + relationship = _relationship( + relationships, + chart_node.get(_RELATIONSHIP_ID), + expected_type=_CHART_RELATIONSHIP, + ) + try: + occurrence, formulas = _chart_occurrence( + archive, + relationship, + anchor=anchor, + metadata=metadata, + workbook_values=workbook_values, + artifact_dir=artifact_dir, + facts_cache=chart_facts_cache, + ) + except _UnsupportedChartType: + warnings.append( + _OccurrenceWarning( + "xlsx_unsupported_object", + anchor, + "unsupported chart type was skipped without guessing", + ) + ) + continue + occurrences.append(occurrence) + warnings.extend( + _OccurrenceWarning( + "xlsx_external_reference", + anchor, + f"chart reference was preserved without access: {formula}", + ) + for formula in formulas + ) + elif image_node is not None: + relationship = _relationship( + relationships, + image_node.get(_RELATIONSHIP_EMBED), + expected_type=_IMAGE_RELATIONSHIP, + ) + occurrences.append( + _image_occurrence( + archive, + relationship, + anchor=anchor, + metadata=metadata, + artifact_dir=artifact_dir, + ) + ) + return tuple(occurrences), tuple(warnings) + + +def _visual_warning( + code: str, + sheet: XlsxPreflightSheet, + anchor: str, + detail: str, +) -> WarningRecord: + return WarningRecord( + code=code, + message=f"{sheet.name}!{anchor}: {detail}", + ) + + +def read_xlsx_visual_objects( + path: Path, + preflight: XlsxPreflight, + workbook_values: dict[str, dict[str, str]], + *, + artifact_dir: Path | None, +) -> XlsxVisualObjects: + if preflight.usage.drawing_objects > MAX_DRAWING_OBJECTS: + raise LimitExceededError("XLSX exceeds the drawing object limit") + if preflight.usage.chart_cache_points > MAX_CHART_CACHE_POINTS: + raise LimitExceededError("XLSX exceeds the chart cache point limit") + by_sheet: list[tuple[XlsxVisualOccurrence, ...]] = [] + warnings: list[WarningRecord] = [] + try: + with ZipFile(path) as archive: + infos = {info.filename: info for info in archive.infolist()} + chart_facts_cache: dict[str, XlsxChartFacts] = {} + for sheet in preflight.sheets: + occurrences: list[XlsxVisualOccurrence] = [] + for drawing_part in _sheet_drawings(archive, infos, sheet): + drawing_items, drawing_warnings = _drawing_occurrences( + archive, + infos, + drawing_part, + workbook_values=workbook_values, + artifact_dir=artifact_dir, + chart_facts_cache=chart_facts_cache, + ) + occurrences.extend(drawing_items) + for warning in drawing_warnings: + warnings.append( + _visual_warning( + warning.code, + sheet, + warning.anchor, + warning.detail, + ) + ) + if artifact_dir is None: + warnings.extend( + _visual_warning( + "xlsx_visual_artifact_unavailable", + sheet, + occurrence.anchor, + f"{occurrence.kind} visual artifact requires an explicit directory", + ) + for occurrence in occurrences + ) + by_sheet.append(tuple(occurrences)) + except BadZipFile as error: + raise CorruptDocumentError("XLSX package is corrupt") from error + except OSError as error: + raise CorruptDocumentError("XLSX package could not be read") from error + return XlsxVisualObjects(tuple(by_sheet), tuple(warnings)) + + +def _artifact_path(artifact_dir: Path, artifact_name: str) -> Path: + root = artifact_dir.resolve() + path = artifact_dir / artifact_name + if path.resolve().parent != root: + raise ValueError("XLSX visual artifact escapes the artifact directory") + return path + + +def build_xlsx_visual_requests( + document: XlsxDocument, + artifact_dir: Path, +) -> tuple[XlsxVisualRequest, ...]: + requests: list[XlsxVisualRequest] = [] + seen: set[tuple[str, str]] = set() + for sheet in document.sheets: + for slot in sorted(sheet.slots, key=lambda item: item.source_index): + if not isinstance(slot, XlsxImageSlot | XlsxChartSlot): + continue + slot_kind = "chart" if isinstance(slot, XlsxChartSlot) else "image" + key = (slot_kind, slot.content_sha256) + if key in seen: + continue + seen.add(key) + requests.append( + XlsxVisualRequest( + digest=slot.content_sha256, + image_path=_artifact_path(artifact_dir, slot.artifact_name), + prompt=( + XLSX_CHART_VISION_PROMPT + if isinstance(slot, XlsxChartSlot) + else XLSX_IMAGE_VISION_PROMPT + ), + source_index=len(requests), + kind=VisionRequestKind.PROSE, + ) + ) + return tuple(requests) + + +def prepare_xlsx_visual_artifact( + slot: XlsxImageSlot | XlsxChartSlot, + artifact_dir: Path, + output_directory: Path, + output_stem: str, +) -> PreparedImage: + return prepare_image( + _artifact_path(artifact_dir, slot.artifact_name), + output_directory, + output_stem, + slot.artifact_name, + "embedded", + None, + ) diff --git a/src/opendocs/parsers/xlsx/models.py b/src/opendocs/parsers/xlsx/models.py index 1642d26..d61c5ce 100644 --- a/src/opendocs/parsers/xlsx/models.py +++ b/src/opendocs/parsers/xlsx/models.py @@ -204,6 +204,8 @@ class XlsxImageSlot: artifact_name: str content_sha256: str alt_text: str | None = None + object_name: str | None = None + title: str | None = None def __post_init__(self) -> None: object.__setattr__( @@ -219,6 +221,12 @@ def __post_init__(self) -> None: _require_sha256("content_sha256", self.content_sha256), ) object.__setattr__(self, "alt_text", _require_optional_string("alt_text", self.alt_text)) + object.__setattr__( + self, + "object_name", + _require_optional_string("object_name", self.object_name), + ) + object.__setattr__(self, "title", _require_optional_string("title", self.title)) @dataclass(frozen=True, slots=True) @@ -228,6 +236,9 @@ class XlsxChartSlot: artifact_name: str content_sha256: str blocks: tuple[Block, ...] + alt_text: str | None = None + object_name: str | None = None + title: str | None = None def __post_init__(self) -> None: object.__setattr__( @@ -243,6 +254,13 @@ def __post_init__(self) -> None: _require_sha256("content_sha256", self.content_sha256), ) object.__setattr__(self, "blocks", _require_blocks(self.blocks)) + object.__setattr__(self, "alt_text", _require_optional_string("alt_text", self.alt_text)) + object.__setattr__( + self, + "object_name", + _require_optional_string("object_name", self.object_name), + ) + object.__setattr__(self, "title", _require_optional_string("title", self.title)) XlsxSlot: TypeAlias = XlsxNativeSlot | XlsxImageSlot | XlsxChartSlot @@ -387,6 +405,8 @@ def _slot_to_wire(slot: XlsxSlot) -> dict[str, object]: "artifact_name": slot.artifact_name, "content_sha256": slot.content_sha256, "alt_text": slot.alt_text, + "object_name": slot.object_name, + "title": slot.title, } return { "type": "xlsx_chart_slot", @@ -394,6 +414,9 @@ def _slot_to_wire(slot: XlsxSlot) -> dict[str, object]: "artifact_name": slot.artifact_name, "content_sha256": slot.content_sha256, "blocks": tuple(_value_to_wire(block) for block in slot.blocks), + "alt_text": slot.alt_text, + "object_name": slot.object_name, + "title": slot.title, } @@ -421,6 +444,8 @@ def _slot_from_wire(value: object) -> XlsxSlot: "artifact_name", "content_sha256", "alt_text", + "object_name", + "title", }: raise ValueError("XLSX image slot wire is invalid") return XlsxImageSlot( @@ -429,6 +454,8 @@ def _slot_from_wire(value: object) -> XlsxSlot: artifact_name=_require_basename("artifact_name", payload["artifact_name"]), content_sha256=_require_sha256("content_sha256", payload["content_sha256"]), alt_text=_require_optional_string("alt_text", payload["alt_text"]), + object_name=_require_optional_string("object_name", payload["object_name"]), + title=_require_optional_string("title", payload["title"]), ) if kind == "xlsx_chart_slot": if set(payload) != { @@ -438,6 +465,9 @@ def _slot_from_wire(value: object) -> XlsxSlot: "artifact_name", "content_sha256", "blocks", + "alt_text", + "object_name", + "title", }: raise ValueError("XLSX chart slot wire is invalid") blocks_value = payload["blocks"] @@ -449,6 +479,9 @@ def _slot_from_wire(value: object) -> XlsxSlot: artifact_name=_require_basename("artifact_name", payload["artifact_name"]), content_sha256=_require_sha256("content_sha256", payload["content_sha256"]), blocks=tuple(cast(Block, _value_from_wire(block)) for block in blocks_value), + alt_text=_require_optional_string("alt_text", payload["alt_text"]), + object_name=_require_optional_string("object_name", payload["object_name"]), + title=_require_optional_string("title", payload["title"]), ) raise ValueError("XLSX slot type is invalid") @@ -484,8 +517,12 @@ def _wire_estimate(document: XlsxDocument) -> int: estimate += sum(_primitive_wire_estimate(block) for block in slot.blocks) if isinstance(slot, XlsxImageSlot | XlsxChartSlot): estimate += 512 + len(slot.artifact_name) * 4 + len(slot.content_sha256) * 4 - if isinstance(slot, XlsxImageSlot) and slot.alt_text is not None: - estimate += len(slot.alt_text) * 4 + if isinstance(slot, XlsxImageSlot | XlsxChartSlot): + estimate += sum( + len(value) * 4 + for value in (slot.alt_text, slot.object_name, slot.title) + if value is not None + ) estimate += sum(_primitive_wire_estimate(warning) for warning in document.warnings) return estimate diff --git a/tests/test_xlsx_extract.py b/tests/test_xlsx_extract.py index e073cd4..10e4991 100644 --- a/tests/test_xlsx_extract.py +++ b/tests/test_xlsx_extract.py @@ -1,21 +1,37 @@ from __future__ import annotations +import hashlib import socket import urllib.request +from copy import deepcopy +from dataclasses import replace from pathlib import Path +from typing import Any, cast from zipfile import ZipFile import pytest from openpyxl import Workbook -from openpyxl.chart import BarChart, Reference +from openpyxl.chart import ( + BarChart, + DoughnutChart, + LineChart, + PieChart, + Reference, + ScatterChart, + Series, +) +from openpyxl.chart.label import DataLabelList from openpyxl.comments import Comment +from openpyxl.drawing.image import Image from openpyxl.formatting.rule import Rule from openpyxl.styles import Font from openpyxl.styles.differential import DifferentialStyle from openpyxl.styles.numbers import NumberFormat from openpyxl.worksheet.table import Table +from PIL import Image as PILImage import opendocs.parsers.xlsx.extract as extract_module +import opendocs.parsers.xlsx.media as media_module from opendocs._models import ( DocumentType, HeadingBlock, @@ -27,10 +43,22 @@ SpannedTableBlock, TableBlock, ) -from opendocs.errors import CorruptDocumentError, LimitExceededError +from opendocs.errors import CorruptDocumentError, DocumentTypeMismatchError, LimitExceededError from opendocs.markdown import render_markdown from opendocs.parsers.xlsx.extract import extract_xlsx -from opendocs.parsers.xlsx.models import XlsxNativeSlot, XlsxSheet, XlsxSheetKind, XlsxSheetState +from opendocs.parsers.xlsx.media import ( + XLSX_CHART_VISION_PROMPT, + build_xlsx_visual_requests, + prepare_xlsx_visual_artifact, +) +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, +) from opendocs.parsers.xlsx.preflight import preflight_xlsx from tests.xlsx_fixtures import rewrite_xlsx @@ -102,6 +130,505 @@ def _native_slots(document_sheet: XlsxSheet) -> tuple[XlsxNativeSlot, ...]: return tuple(slot for slot in document_sheet.slots if isinstance(slot, XlsxNativeSlot)) +def _visual_slots(document_sheet: XlsxSheet) -> tuple[XlsxImageSlot | XlsxChartSlot, ...]: + return tuple( + slot for slot in document_sheet.slots if isinstance(slot, XlsxImageSlot | XlsxChartSlot) + ) + + +def _write_png(path: Path, *, color: str = "navy") -> bytes: + image = PILImage.new("RGB", (32, 16), color) + try: + image.save(path, format="PNG") + finally: + image.close() + return path.read_bytes() + + +def test_extracts_native_chart_facts_and_raw_image_artifacts(tmp_path: Path) -> None: + path = tmp_path / "visuals.xlsx" + source_image = tmp_path / "source.png" + image_bytes = _write_png(source_image) + workbook = Workbook() + sheet = workbook.active + sheet.title = "Sales Data" + for row in (("Month", "Revenue"), ("Jan", 10), ("Feb", 15), ("Mar", 12)): + sheet.append(row) + chart = LineChart() + chart.title = "Revenue trend" + cast(Any, chart.x_axis).title = "Month" + cast(Any, chart.y_axis).title = "USD" + chart.dataLabels = DataLabelList(showVal=True) + chart.add_data(Reference(sheet, min_col=2, min_row=1, max_row=4), titles_from_data=True) + chart.set_categories(Reference(sheet, min_col=1, min_row=2, max_row=4)) + sheet.add_chart(chart, "D2") + embedded = Image(source_image) + sheet.add_image(embedded, "K3") + workbook.save(path) + with ZipFile(path) as archive: + drawing_xml = archive.read("xl/drawings/drawing1.xml") + assert b'name="Chart 1"' in drawing_xml + assert b'name="Image 2" descr="Picture"' in drawing_xml + drawing_xml = drawing_xml.replace( + b'name="Chart 1"', + b'name="Chart 1" descr="chart description" title="chart title"', + ).replace( + b'name="Image 2" descr="Picture"', + b'name="Image 2" descr="image description" title="image title"', + ) + rewrite_xlsx(path, {"xl/drawings/drawing1.xml": drawing_xml}) + artifacts = tmp_path / "artifacts" + + document = extract_xlsx(path, preflight_xlsx(path), artifact_dir=artifacts) + + slots = _visual_slots(document.sheets[0]) + chart_slot = next(slot for slot in slots if isinstance(slot, XlsxChartSlot)) + image_slot = next(slot for slot in slots if isinstance(slot, XlsxImageSlot)) + assert chart_slot.anchor == "D2" + assert image_slot.anchor == "K3" + assert image_slot.content_sha256 == hashlib.sha256(image_bytes).hexdigest() + assert (chart_slot.object_name, chart_slot.alt_text, chart_slot.title) == ( + "Chart 1", + "chart description", + "chart title", + ) + assert (image_slot.object_name, image_slot.alt_text, image_slot.title) == ( + "Image 2", + "image description", + "image title", + ) + assert (artifacts / image_slot.artifact_name).read_bytes() == image_bytes + chart_bytes = (artifacts / chart_slot.artifact_name).read_bytes() + assert chart_bytes.startswith(b"\x89PNG\r\n\x1a\n") + assert chart_slot.content_sha256 == hashlib.sha256(chart_bytes).hexdigest() + headings = [block for block in chart_slot.blocks if isinstance(block, HeadingBlock)] + tables = [block for block in chart_slot.blocks if isinstance(block, TableBlock)] + paragraphs = [block for block in chart_slot.blocks if isinstance(block, ParagraphBlock)] + assert headings[0].inlines == (InlineText("Revenue trend"),) + assert tables == [ + TableBlock( + ( + ("Series", "Revenue", "Category", "Jan", "Value", "10"), + ("Series", "Revenue", "Category", "Feb", "Value", "15"), + ("Series", "Revenue", "Category", "Mar", "Value", "12"), + ), + header_rows=0, + ) + ] + paragraph_text = "\n".join( + "".join(inline.text for inline in block.inlines if isinstance(inline, InlineText)) + for block in paragraphs + ) + assert "Axis titles: Month; USD" in paragraph_text + assert "Data labels: value" in paragraph_text + assert "'Sales Data'!$A$2:$A$4" in paragraph_text + assert "'Sales Data'!$B$2:$B$4" in paragraph_text + + +def test_chart_cache_wins_and_unsupported_references_are_preserved_without_access( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + path = tmp_path / "chart-references.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Quoted Sheet" + for row in (("Label", "Value"), ("local", 1), ("ignored", 2)): + sheet.append(row) + chart = BarChart() + chart.add_data(Reference(sheet, min_col=2, min_row=1, max_row=3), titles_from_data=True) + chart.set_categories(Reference(sheet, min_col=1, min_row=2, max_row=3)) + sheet.add_chart(chart, "D4") + workbook.save(path) + with ZipFile(path) as archive: + chart_xml = archive.read("xl/charts/chart1.xml") + chart_xml = chart_xml.replace( + b"'Quoted Sheet'!B1", + ( + b"'Quoted Sheet'!B1" + b'Cached series' + ), + ) + chart_xml = chart_xml.replace( + b"'Quoted Sheet'!$A$2:$A$3", + b"[Book.xlsx]Quoted Sheet!$A$2:$A$3", + ) + chart_xml = chart_xml.replace( + b"'Quoted Sheet'!$B$2:$B$3", + b"DynamicName", + ) + rewrite_xlsx(path, {"xl/charts/chart1.xml": chart_xml}) + + def forbidden_network(*args: object, **kwargs: object) -> object: + del args, kwargs + raise AssertionError("chart references must not access the network") + + monkeypatch.setattr(socket, "create_connection", forbidden_network) + monkeypatch.setattr(urllib.request, "urlopen", forbidden_network) + document = extract_xlsx( + path, + preflight_xlsx(path), + artifact_dir=tmp_path / "artifacts", + ) + + chart_slot = next(slot for slot in document.sheets[0].slots if isinstance(slot, XlsxChartSlot)) + assert ( + any( + isinstance(block, HeadingBlock) and block.inlines == (InlineText("Cached series"),) + for block in chart_slot.blocks + ) + is False + ) + rendered_facts = repr(chart_slot.blocks) + assert "Cached series" in rendered_facts + assert "[Book.xlsx]Quoted Sheet!$A$2:$A$3" in rendered_facts + assert "DynamicName" in rendered_facts + assert [warning.code for warning in document.warnings].count("xlsx_external_reference") == 2 + + +def test_multiple_chart_reference_warnings_stay_bound_to_their_occurrence( + tmp_path: Path, +) -> None: + path = tmp_path / "chart-warning-occurrences.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Data" + sheet.append(("Category", "Value")) + sheet.append(("A", 1)) + chart = BarChart() + chart.add_data(Reference(sheet, min_col=2, min_row=1, max_row=2), titles_from_data=True) + chart.set_categories(Reference(sheet, min_col=1, min_row=2, max_row=2)) + sheet.add_chart(chart, "D2") + sheet.add_chart(deepcopy(chart), "D20") + workbook.save(path) + replacements: dict[str, bytes | None] = {} + with ZipFile(path) as archive: + for index, formula in ((1, b"DynamicOne"), (2, b"DynamicTwo")): + part = f"xl/charts/chart{index}.xml" + xml = archive.read(part) + original = b"'Data'!$B$2" + assert original in xml + replacements[part] = xml.replace( + original, + b"" + formula + b"", + ) + rewrite_xlsx(path, replacements) + + document = extract_xlsx(path, preflight_xlsx(path), artifact_dir=tmp_path / "artifacts") + + warnings = [ + warning for warning in document.warnings if warning.code == "xlsx_external_reference" + ] + assert [warning.message for warning in warnings] == [ + "Data!D2: chart reference was preserved without access: DynamicOne", + "Data!D20: chart reference was preserved without access: DynamicTwo", + ] + + +def test_semantic_chart_previews_and_requests_are_occurrence_independent( + tmp_path: Path, +) -> None: + path = tmp_path / "duplicate-charts.xlsx" + workbook = Workbook() + first = workbook.active + first.title = "Data" + first.append(("X", "Y")) + first.append((1, 3)) + first.append((2, 5)) + chart = ScatterChart() + chart.title = "Relationship" + cast(Any, chart.x_axis).title = "X" + cast(Any, chart.y_axis).title = "Y" + chart.series.append( + Series( + Reference(first, min_col=2, min_row=2, max_row=3), + Reference(first, min_col=1, min_row=2, max_row=3), + title="points", + ) + ) + first.add_chart(chart, "D2") + second = workbook.create_sheet("Other") + second.add_chart(deepcopy(chart), "H8") + workbook.save(path) + with ZipFile(path) as archive: + second_drawing = archive.read("xl/drawings/drawing2.xml") + assert b'name="Chart 1"' in second_drawing + rewrite_xlsx( + path, + { + "xl/drawings/drawing2.xml": second_drawing.replace( + b'name="Chart 1"', + b'name="Different occurrence" descr="different alt"', + ) + }, + ) + artifacts = tmp_path / "artifacts" + + document = extract_xlsx(path, preflight_xlsx(path), artifact_dir=artifacts) + chart_slots = tuple( + slot for sheet in document.sheets for slot in sheet.slots if isinstance(slot, XlsxChartSlot) + ) + requests = build_xlsx_visual_requests(document, artifacts) + + assert [(slot.anchor, slot.content_sha256) for slot in chart_slots] == [ + ("D2", chart_slots[0].content_sha256), + ("H8", chart_slots[0].content_sha256), + ] + assert chart_slots[0].artifact_name == chart_slots[1].artifact_name + assert len(requests) == 1 + assert requests[0].digest == chart_slots[0].content_sha256 + assert requests[0].prompt == XLSX_CHART_VISION_PROMPT + assert "趋势" in requests[0].prompt + assert "关系" in requests[0].prompt + assert "标注" in requests[0].prompt + assert "含义" in requests[0].prompt + assert "视觉解释" in requests[0].prompt + + +def test_without_artifact_directory_native_chart_text_survives_and_images_are_warned( + tmp_path: Path, +) -> None: + path = tmp_path / "native-only.xlsx" + source_image = tmp_path / "source.png" + _write_png(source_image) + workbook = Workbook() + sheet = workbook.active + sheet.append(("Name", "Value")) + sheet.append(("A", 1)) + chart = PieChart() + chart.title = "Share" + chart.add_data(Reference(sheet, min_col=2, min_row=1, max_row=2), titles_from_data=True) + chart.set_categories(Reference(sheet, min_col=1, min_row=2, max_row=2)) + sheet.add_chart(chart, "D2") + sheet.add_image(Image(source_image), "J2") + workbook.save(path) + + document = extract_xlsx(path, preflight_xlsx(path)) + + assert not _visual_slots(document.sheets[0]) + native_text = repr(_native_slots(document.sheets[0])) + assert "Share" in native_text + assert "Series" in native_text + assert "Image name:" in native_text + assert [warning.code for warning in document.warnings].count( + "xlsx_visual_artifact_unavailable" + ) == 2 + + +def test_image_visual_preparation_reuses_shared_safety_helper(tmp_path: Path) -> None: + artifact_dir = tmp_path / "artifacts" + output_dir = tmp_path / "prepared" + artifact_dir.mkdir() + output_dir.mkdir() + disguised = artifact_dir / "xlsx-media.png" + image = PILImage.new("RGB", (16, 16), "red") + try: + image.save(disguised, format="JPEG") + finally: + image.close() + slot = XlsxImageSlot( + source_index=1, + anchor="A1", + artifact_name=disguised.name, + content_sha256="a" * 64, + ) + + with pytest.raises(DocumentTypeMismatchError, match="extension declares png"): + prepare_xlsx_visual_artifact(slot, artifact_dir, output_dir, "prepared") + + +def test_supported_chart_families_emit_native_type_facts(tmp_path: Path) -> None: + path = tmp_path / "chart-families.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Data" + sheet.append(("Category", "Value", "X", "Y")) + sheet.append(("A", 1, 1, 2)) + sheet.append(("B", 2, 2, 4)) + chart_types = (LineChart(), BarChart(), PieChart(), DoughnutChart()) + for index, chart in enumerate(chart_types, start=1): + chart.title = type(chart).__name__ + chart.add_data(Reference(sheet, min_col=2, min_row=1, max_row=3), titles_from_data=True) + chart.set_categories(Reference(sheet, min_col=1, min_row=2, max_row=3)) + sheet.add_chart(chart, f"F{index * 8}") + scatter = ScatterChart() + scatter.title = "ScatterChart" + scatter.series.append( + Series( + Reference(sheet, min_col=4, min_row=2, max_row=3), + Reference(sheet, min_col=3, min_row=2, max_row=3), + title="points", + ) + ) + sheet.add_chart(scatter, "F40") + workbook.save(path) + + document = extract_xlsx( + path, + preflight_xlsx(path), + artifact_dir=tmp_path / "artifacts", + ) + + chart_slots = [slot for slot in document.sheets[0].slots if isinstance(slot, XlsxChartSlot)] + assert len(chart_slots) == 5 + assert [ + next( + inline.text + for block in slot.blocks + if isinstance(block, ParagraphBlock) + for inline in block.inlines + if isinstance(inline, InlineText) and inline.text.startswith("Chart type:") + ) + for slot in chart_slots + ] == [ + "Chart type: line", + "Chart type: bar", + "Chart type: pie", + "Chart type: doughnut", + "Chart type: scatter", + ] + + +def test_duplicate_embedded_media_share_raw_digest_and_one_request(tmp_path: Path) -> None: + path = tmp_path / "duplicate-images.xlsx" + source_image = tmp_path / "shared.png" + image_bytes = _write_png(source_image, color="green") + workbook = Workbook() + first = workbook.active + first.title = "First" + first.add_image(Image(source_image), "B2") + second = workbook.create_sheet("Second") + second.add_image(Image(source_image), "H9") + workbook.save(path) + artifacts = tmp_path / "artifacts" + + document = extract_xlsx(path, preflight_xlsx(path), artifact_dir=artifacts) + image_slots = tuple( + slot for sheet in document.sheets for slot in sheet.slots if isinstance(slot, XlsxImageSlot) + ) + requests = build_xlsx_visual_requests(document, artifacts) + + assert [slot.anchor for slot in image_slots] == ["B2", "H9"] + assert image_slots[0].content_sha256 == image_slots[1].content_sha256 + assert image_slots[0].artifact_name == image_slots[1].artifact_name + assert (artifacts / image_slots[0].artifact_name).read_bytes() == image_bytes + assert ( + len([request for request in requests if request.digest == image_slots[0].content_sha256]) + == 1 + ) + + +def test_multilevel_category_cache_combines_levels_without_duplicate_index_failure( + tmp_path: Path, +) -> None: + path = tmp_path / "multilevel.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet.title = "Data" + sheet.append(("Category", "Value")) + sheet.append(("Jan", 1)) + sheet.append(("Feb", 2)) + chart = LineChart() + chart.add_data(Reference(sheet, min_col=2, min_row=1, max_row=3), titles_from_data=True) + chart.set_categories(Reference(sheet, min_col=1, min_row=2, max_row=3)) + sheet.add_chart(chart, "D2") + workbook.save(path) + with ZipFile(path) as archive: + chart_xml = archive.read("xl/charts/chart1.xml") + original = b"'Data'!$A$2:$A$3" + replacement = ( + b"'Data'!$A$2:$A$3" + b'Q1' + b'Q1' + b'JanFeb' + b"" + ) + assert original in chart_xml + chart_xml = chart_xml.replace(original, replacement).replace( + b"", + (b"<tx><strRef><f>'Data'!$A$1</f></strRef></tx>"), + ) + rewrite_xlsx(path, {"xl/charts/chart1.xml": chart_xml}) + + document = extract_xlsx( + path, + preflight_xlsx(path), + artifact_dir=tmp_path / "artifacts", + ) + + slot = next(slot for slot in document.sheets[0].slots if isinstance(slot, XlsxChartSlot)) + heading = next(block for block in slot.blocks if isinstance(block, HeadingBlock)) + table = next(block for block in slot.blocks if isinstance(block, TableBlock)) + assert heading.inlines == (InlineText("Category"),) + assert [row[3] for row in table.grid] == ["Q1 / Jan", "Q1 / Feb"] + + +def test_unsupported_chart_type_is_skipped_with_locatable_warning(tmp_path: Path) -> None: + path = tmp_path / "unsupported-chart.xlsx" + workbook = Workbook() + sheet = workbook.active + sheet["A1"] = "kept" + chart = LineChart() + chart.add_data(Reference(sheet, min_col=1, min_row=1, max_row=1)) + sheet.add_chart(chart, "D2") + workbook.save(path) + with ZipFile(path) as archive: + chart_xml = archive.read("xl/charts/chart1.xml") + assert b"" in chart_xml + chart_xml = chart_xml.replace(b"", b"").replace( + b"", b"" + ) + rewrite_xlsx(path, {"xl/charts/chart1.xml": chart_xml}) + + document = extract_xlsx(path, preflight_xlsx(path), artifact_dir=tmp_path / "artifacts") + + assert not any(isinstance(slot, XlsxChartSlot) for slot in document.sheets[0].slots) + warning = next( + warning for warning in document.warnings if warning.code == "xlsx_unsupported_object" + ) + assert warning.message.startswith("Sheet!D2:") + + +@pytest.mark.parametrize( + ("field", "limit", "message"), + [ + ("drawing_objects", media_module.MAX_DRAWING_OBJECTS, "drawing object"), + ("chart_cache_points", media_module.MAX_CHART_CACHE_POINTS, "chart cache point"), + ], +) +def test_visual_outer_limits_fail_before_semantic_preview( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + field: str, + limit: int, + message: str, +) -> None: + path = tmp_path / "bounded-visual.xlsx" + workbook = Workbook() + workbook.active["A1"] = "kept" + workbook.save(path) + preflight = preflight_xlsx(path) + accepted = replace( + preflight, + usage=replace(preflight.usage, **{field: limit}), + ) + bounded = replace( + preflight, + usage=replace(preflight.usage, **{field: limit + 1}), + ) + + def forbidden_preview(*args: object, **kwargs: object) -> bytes: + del args, kwargs + raise AssertionError("semantic preview must not run after an outer-limit failure") + + monkeypatch.setattr(media_module, "render_chart_semantic_preview", forbidden_preview) + extract_xlsx(path, accepted, artifact_dir=tmp_path / "accepted-artifacts") + with pytest.raises(LimitExceededError, match=message): + extract_xlsx(path, bounded, artifact_dir=tmp_path / "artifacts") + + assert not (tmp_path / "artifacts").exists() + + def test_extract_preserves_all_sheet_like_entries_states_and_empty_sheets( tmp_path: Path, ) -> None: @@ -129,7 +656,7 @@ def test_extract_preserves_all_sheet_like_entries_states_and_empty_sheets( assert isinstance(prelude.blocks[2], ParagraphBlock) assert len(_native_slots(document.sheets[1])) == 1 assert len(_native_slots(document.sheets[2])) == 1 - assert len(_native_slots(document.sheets[3])) == 1 + assert len(_native_slots(document.sheets[3])) == 2 def test_extract_builds_tables_regions_merges_and_ignores_style_only_cells( diff --git a/tests/test_xlsx_models.py b/tests/test_xlsx_models.py index 488c999..5934d79 100644 --- a/tests/test_xlsx_models.py +++ b/tests/test_xlsx_models.py @@ -54,6 +54,8 @@ def _document() -> XlsxDocument: artifact_name="xlsx-image-1.png", content_sha256="a" * 64, alt_text="diagram", + object_name="Picture 1", + title="Architecture", ), XlsxChartSlot( source_index=2, @@ -61,6 +63,9 @@ def _document() -> XlsxDocument: artifact_name="xlsx-chart-1.png", content_sha256="b" * 64, blocks=(TextBlock("Series: 1, 2, 3"),), + alt_text="trend chart", + object_name="Chart 1", + title="Trend", ), ), ), diff --git a/tests/test_xlsx_parser.py b/tests/test_xlsx_parser.py new file mode 100644 index 0000000..399207a --- /dev/null +++ b/tests/test_xlsx_parser.py @@ -0,0 +1,67 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest +from PIL import Image + +from opendocs._models import HeadingBlock, InlineText +from opendocs.parsers.xlsx.media import build_xlsx_visual_requests +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, +) +from opendocs.vision.base import VisionRequest, VisionResult, VisionTextElement + + +class RecordingVision: + def __init__(self) -> None: + self.requests: list[VisionRequest] = [] + + async def analyze(self, request: VisionRequest) -> VisionResult: + self.requests.append(request) + return VisionResult((VisionTextElement("视觉解释: 收入总体上升", request.source_index),)) + + +@pytest.mark.asyncio +async def test_xlsx_visual_request_seam_uses_fake_client_and_bounded_chart_prompt( + tmp_path: Path, +) -> None: + artifact_name = "chart.png" + image = Image.new("RGB", (32, 16), "white") + try: + image.save(tmp_path / artifact_name, "PNG") + finally: + image.close() + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Data", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxChartSlot( + 1, + "D2", + artifact_name, + "a" * 64, + (HeadingBlock(2, (InlineText("Revenue"),)),), + ), + ), + ), + ) + ) + specs = build_xlsx_visual_requests(document, tmp_path) + vision = RecordingVision() + + result = await vision.analyze(specs[0].to_vision_request()) + + assert result.elements == (VisionTextElement("视觉解释: 收入总体上升", 0),) + assert len(vision.requests) == 1 + assert vision.requests[0].image_path == tmp_path / artifact_name + assert "视觉解释" in vision.requests[0].prompt + assert "Excel 外观还原" in vision.requests[0].prompt From 4e8e97e6899756f49bc14731234810a427363cb6 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 16:32:23 +0800 Subject: [PATCH 06/12] =?UTF-8?q?=E8=AE=A9=20XLSX=20=E8=A7=86=E8=A7=89?= =?UTF-8?q?=E5=A4=B1=E8=B4=A5=E4=BB=8D=E8=BF=94=E5=9B=9E=E5=AE=8C=E6=95=B4?= =?UTF-8?q?=E5=8E=9F=E7=94=9F=E7=BB=93=E6=9E=9C?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Constraint: 原生事实始终优先,视觉结果只按对象位置追加;模型类失败不得中断 XLSX 文档。 Rejected: 不复用 Office 的致命视觉异常策略,也不把 sheet 映射为 page。 Confidence: high Scope-risk: 公共输入对等、资源实测和发布 smoke 留给后续单元。 Tested: 279 focused tests; ruff; format; ty; diff-check. Not-tested: real XLSX workbooks or live vision providers. --- src/opendocs/parsers/registry.py | 5 +- src/opendocs/parsers/xlsx/__init__.py | 22 +- src/opendocs/parsers/xlsx/merge.py | 204 +++++++++++++ src/opendocs/parsers/xlsx/parser.py | 311 ++++++++++++++++++++ tests/test_registry.py | 23 +- tests/test_xlsx_merge.py | 212 ++++++++++++++ tests/test_xlsx_parser.py | 405 +++++++++++++++++++++++++- tests/test_xlsx_preflight.py | 43 +-- 8 files changed, 1163 insertions(+), 62 deletions(-) create mode 100644 src/opendocs/parsers/xlsx/merge.py create mode 100644 src/opendocs/parsers/xlsx/parser.py create mode 100644 tests/test_xlsx_merge.py diff --git a/src/opendocs/parsers/registry.py b/src/opendocs/parsers/registry.py index 3088e6f..05f4581 100644 --- a/src/opendocs/parsers/registry.py +++ b/src/opendocs/parsers/registry.py @@ -84,5 +84,8 @@ def build_default_registry( DocumentType.PPTX, OfficeParser(DocumentType.PPTX, runtime, vision, vision_config, deadline=deadline), ) - registry.register(DocumentType.XLSX, XlsxParser()) + registry.register( + DocumentType.XLSX, + XlsxParser(runtime, vision, vision_config, deadline=deadline), + ) return registry diff --git a/src/opendocs/parsers/xlsx/__init__.py b/src/opendocs/parsers/xlsx/__init__.py index 6d9f5ba..9833ca9 100644 --- a/src/opendocs/parsers/xlsx/__init__.py +++ b/src/opendocs/parsers/xlsx/__init__.py @@ -1,21 +1,3 @@ -from __future__ import annotations +from opendocs.parsers.xlsx.parser import XlsxParser -import asyncio - -from opendocs._models import ParsedDocument -from opendocs.errors import UnsupportedDocumentError -from opendocs.options import ParseOptions -from opendocs.parsers.xlsx.preflight import preflight_xlsx -from opendocs.source import ResolvedSource - - -class XlsxParser: - async def parse( - self, - source: ResolvedSource, - *, - options: ParseOptions, - ) -> ParsedDocument: - del options - await asyncio.to_thread(preflight_xlsx, source.path) - raise UnsupportedDocumentError("XLSX content parsing is not implemented in this release") +__all__ = ["XlsxParser"] diff --git a/src/opendocs/parsers/xlsx/merge.py b/src/opendocs/parsers/xlsx/merge.py new file mode 100644 index 0000000..8800ba6 --- /dev/null +++ b/src/opendocs/parsers/xlsx/merge.py @@ -0,0 +1,204 @@ +from __future__ import annotations + +import re +from collections.abc import Mapping +from dataclasses import dataclass +from itertools import groupby + +from opendocs._models import ( + Block, + DocumentType, + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + ParsedDocument, + TableBlock, + TextBlock, + WarningRecord, +) +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetState, + XlsxSlot, +) +from opendocs.vision.base import VisionResult, VisionTableElement, VisionTextElement + +_A1_START = re.compile(r"^([A-Z]{1,3})([1-9][0-9]{0,6})") +_STATE_LABELS = { + XlsxSheetState.VISIBLE: "Visible", + XlsxSheetState.HIDDEN: "Hidden", + XlsxSheetState.VERY_HIDDEN: "Very Hidden", +} + + +@dataclass(frozen=True, slots=True) +class XlsxVisualOutcome: + result: VisionResult | None + warning_code: str | None = None + + def __post_init__(self) -> None: + if self.result is not None and not isinstance(self.result, VisionResult): + raise TypeError("result must be a VisionResult or None") + if self.warning_code is not None and not isinstance(self.warning_code, str): + raise TypeError("warning_code must be a str or None") + if self.warning_code == "": + raise ValueError("warning_code must not be empty") + + +def _column_number(label: str) -> int: + number = 0 + for character in label: + number = number * 26 + ord(character) - ord("A") + 1 + return number + + +def _anchor_position(anchor: str) -> tuple[int, int]: + match = _A1_START.match(anchor) + if match is None: + raise ValueError("XLSX slot anchor is invalid") + return int(match.group(2)), _column_number(match.group(1)) + + +def _slot_kind_rank(slot: XlsxSlot) -> int: + if isinstance(slot, XlsxNativeSlot): + return 0 + if isinstance(slot, XlsxChartSlot): + return 1 + return 2 + + +def _slot_sort_key(slot: XlsxSlot) -> tuple[int, int, int, int]: + row, column = _anchor_position(slot.anchor) + return row, column, _slot_kind_rank(slot), slot.source_index + + +def _is_extractor_prelude(slot: XlsxSlot, sheet_index: int) -> bool: + if not isinstance(slot, XlsxNativeSlot) or slot.source_index != 0: + return False + expected = f"" + return any( + isinstance(block, MarkdownBlock) and block.markdown == expected for block in slot.blocks + ) + + +def _sheet_prelude(sheet: XlsxSheet) -> tuple[Block, ...]: + return ( + MarkdownBlock(f""), + HeadingBlock(1, (InlineText(f"{sheet.name} ({_STATE_LABELS[sheet.state]})"),)), + ) + + +def _object_anchor(sheet: XlsxSheet, slot: XlsxImageSlot | XlsxChartSlot) -> MarkdownBlock: + return MarkdownBlock( + f"" + ) + + +def _metadata_blocks( + kind: str, + slot: XlsxImageSlot | XlsxChartSlot, +) -> tuple[Block, ...]: + labels = ( + (f"{kind} name", slot.object_name), + (f"{kind} description", slot.alt_text), + (f"{kind} title", slot.title), + ) + return ( + *((HeadingBlock(3, (InlineText("Embedded image"),)),) if kind == "Image" else ()), + *(ParagraphBlock((InlineText(f"{label}: {value}"),)) for label, value in labels if value), + ) + + +def _native_blocks(sheet: XlsxSheet, slot: XlsxSlot) -> tuple[Block, ...]: + if isinstance(slot, XlsxNativeSlot): + return slot.blocks + if isinstance(slot, XlsxChartSlot): + return ( + _object_anchor(sheet, slot), + *_metadata_blocks("Chart", slot), + *slot.blocks, + ) + return (_object_anchor(sheet, slot), *_metadata_blocks("Image", slot)) + + +def _vision_blocks(result: VisionResult) -> tuple[Block, ...]: + blocks: list[Block] = [HeadingBlock(3, (InlineText("Visual interpretation"),))] + for element in sorted(result.elements, key=lambda item: item.source_index): + if isinstance(element, VisionTextElement): + if element.text.strip(): + blocks.append(TextBlock(element.text.strip())) + elif isinstance(element, VisionTableElement): + blocks.append(TableBlock(element.grid, element.header_rows)) + return tuple(blocks) + + +def _visual_blocks( + slot: XlsxSlot, + visual_outcomes: Mapping[str, XlsxVisualOutcome], +) -> tuple[Block, ...]: + if not isinstance(slot, XlsxImageSlot | XlsxChartSlot): + return () + outcome = visual_outcomes.get(slot.content_sha256) + if outcome is None or outcome.result is None: + return () + return _vision_blocks(outcome.result) + + +def _visual_warning( + sheet: XlsxSheet, + slot: XlsxImageSlot | XlsxChartSlot, + code: str, +) -> WarningRecord: + kind = "chart" if isinstance(slot, XlsxChartSlot) else "image" + return WarningRecord( + code=code, + message=f"{sheet.name}!{slot.anchor}: {kind} visual interpretation was not completed", + ) + + +def merge_xlsx_document( + document: XlsxDocument, + visual_outcomes: Mapping[str, XlsxVisualOutcome], +) -> ParsedDocument: + if not isinstance(document, XlsxDocument): + raise TypeError("document must be an XlsxDocument") + for digest, outcome in visual_outcomes.items(): + if not isinstance(digest, str): + raise TypeError("visual outcome keys must be strings") + if not isinstance(outcome, XlsxVisualOutcome): + raise TypeError("visual outcomes must contain XlsxVisualOutcome values") + + blocks: list[Block] = [] + warnings = list(document.warnings) + for sheet in document.sheets: + blocks.extend(_sheet_prelude(sheet)) + slots = tuple( + sorted( + ( + slot + for slot in sheet.slots + if not _is_extractor_prelude(slot, sheet.sheet_index) + ), + key=_slot_sort_key, + ) + ) + for _position, positioned_slots in groupby( + slots, + key=lambda item: _anchor_position(item.anchor), + ): + group = tuple(positioned_slots) + for slot in group: + blocks.extend(_native_blocks(sheet, slot)) + for slot in group: + blocks.extend(_visual_blocks(slot, visual_outcomes)) + if isinstance(slot, XlsxImageSlot | XlsxChartSlot): + outcome = visual_outcomes.get(slot.content_sha256) + if outcome is not None and outcome.warning_code is not None: + warnings.append(_visual_warning(sheet, slot, outcome.warning_code)) + return ParsedDocument(DocumentType.XLSX, tuple(blocks), tuple(warnings)) diff --git a/src/opendocs/parsers/xlsx/parser.py b/src/opendocs/parsers/xlsx/parser.py new file mode 100644 index 0000000..f1b3973 --- /dev/null +++ b/src/opendocs/parsers/xlsx/parser.py @@ -0,0 +1,311 @@ +from __future__ import annotations + +import asyncio +from contextlib import suppress +from dataclasses import dataclass +from pathlib import Path +from typing import cast + +from opendocs._models import ParsedDocument +from opendocs._runtime import ParserRuntime +from opendocs.errors import DocumentTimeoutError, RuntimeDependencyError +from opendocs.options import ParseOptions, VisionConfig +from opendocs.parsers.xlsx.extract import extract_xlsx +from opendocs.parsers.xlsx.media import ( + XlsxVisualRequest, + build_xlsx_visual_requests, + prepare_xlsx_visual_artifact, +) +from opendocs.parsers.xlsx.merge import XlsxVisualOutcome, merge_xlsx_document +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + document_from_wire, + document_to_wire, +) +from opendocs.parsers.xlsx.preflight import preflight_xlsx +from opendocs.source import ResolvedSource +from opendocs.vision.base import VisionClient, VisionRequest, VisionResult, VisionTextElement +from opendocs.vision.images import ( + PreparedImage, + merge_tiled_results, + prepared_paths, + tile_prompt, +) + + +def _cleanup_worker_artifacts(workspace_path: Path) -> None: + for pattern in ("xlsx-media-*", "xlsx-chart-*", "xlsx-prepared-*"): + for artifact in workspace_path.glob(pattern): + with suppress(OSError): + artifact.unlink(missing_ok=True) + + +def _extract_xlsx_to_wire(path: Path, workspace_path: Path) -> dict[str, object]: + preflight = preflight_xlsx(path) + try: + document = extract_xlsx(path, preflight, artifact_dir=workspace_path) + return document_to_wire(document) + except BaseException: + _cleanup_worker_artifacts(workspace_path) + raise + + +def _artifact_slot_to_wire(slot: XlsxImageSlot | XlsxChartSlot) -> dict[str, object]: + return { + "type": "xlsx_visual_artifact", + "source_index": slot.source_index, + "anchor": slot.anchor, + "artifact_name": slot.artifact_name, + "content_sha256": slot.content_sha256, + "alt_text": slot.alt_text, + "object_name": slot.object_name, + "title": slot.title, + } + + +def _artifact_slot_from_wire(value: object) -> XlsxImageSlot: + if not isinstance(value, dict) or set(value) != { + "type", + "source_index", + "anchor", + "artifact_name", + "content_sha256", + "alt_text", + "object_name", + "title", + }: + raise ValueError("XLSX visual artifact wire is invalid") + if value.get("type") != "xlsx_visual_artifact": + raise ValueError("XLSX visual artifact wire is invalid") + payload = cast(dict[str, object], value) + return XlsxImageSlot( + source_index=cast(int, payload["source_index"]), + anchor=cast(str, payload["anchor"]), + artifact_name=cast(str, payload["artifact_name"]), + content_sha256=cast(str, payload["content_sha256"]), + alt_text=cast(str | None, payload["alt_text"]), + object_name=cast(str | None, payload["object_name"]), + title=cast(str | None, payload["title"]), + ) + + +def _prepare_xlsx_visual_to_wire( + slot_wire: dict[str, object], + artifact_dir: Path, + output_directory: Path, + output_stem: str, +) -> PreparedImage: + return prepare_xlsx_visual_artifact( + _artifact_slot_from_wire(slot_wire), + artifact_dir, + output_directory, + output_stem, + ) + + +def _visual_slots(document: XlsxDocument) -> tuple[XlsxImageSlot | XlsxChartSlot, ...]: + return tuple( + slot + for sheet in document.sheets + for slot in sheet.slots + if isinstance(slot, XlsxImageSlot | XlsxChartSlot) + ) + + +def _unique_requests( + document: XlsxDocument, + artifact_dir: Path, +) -> tuple[XlsxVisualRequest, ...]: + unique: list[XlsxVisualRequest] = [] + seen: set[str] = set() + for request in build_xlsx_visual_requests(document, artifact_dir): + if request.digest in seen: + continue + seen.add(request.digest) + unique.append(request) + return tuple(unique) + + +def _has_visual_content(result: VisionResult) -> bool: + return any( + not isinstance(element, VisionTextElement) or bool(element.text.strip()) + for element in result.elements + ) + + +@dataclass(frozen=True, slots=True) +class _PreparedVisual: + request: XlsxVisualRequest + prepared: PreparedImage + paths: tuple[Path, ...] + + +class XlsxParser: + def __init__( + self, + runtime: ParserRuntime, + vision: VisionClient | None, + vision_config: VisionConfig | None, + *, + deadline: float | None = None, + ) -> None: + if not isinstance(runtime, ParserRuntime): + raise TypeError("runtime must be a ParserRuntime") + if vision_config is not None and not isinstance(vision_config, VisionConfig): + raise TypeError("vision_config must be a VisionConfig or None") + self._runtime = runtime + self._vision = vision + self._vision_config = vision_config + self._deadline = deadline + + async def parse( + self, + source: ResolvedSource, + *, + options: ParseOptions, + ) -> ParsedDocument: + loop = asyncio.get_running_loop() + deadline = loop.time() + float(options.timeout) + if self._deadline is not None: + deadline = min(deadline, self._deadline) + try: + async with asyncio.timeout_at(deadline): + document = await self._extract(source) + visual_outcomes = await self._visual_outcomes(document) + except asyncio.CancelledError: + raise + except TimeoutError: + raise DocumentTimeoutError("XLSX parsing exceeded the document deadline") from None + return merge_xlsx_document(document, visual_outcomes) + + async def _extract(self, source: ResolvedSource) -> XlsxDocument: + wire = await self._runtime.run_native( + _extract_xlsx_to_wire, + source.path, + self._runtime.workspace.path, + ) + try: + return document_from_wire(wire) + except (TypeError, ValueError) as error: + _cleanup_worker_artifacts(self._runtime.workspace.path) + raise RuntimeDependencyError("native XLSX worker returned invalid data") from error + + async def _visual_outcomes( + self, + document: XlsxDocument, + ) -> dict[str, XlsxVisualOutcome]: + slots = _visual_slots(document) + if not slots: + return {} + representatives: dict[str, XlsxImageSlot | XlsxChartSlot] = {} + for slot in slots: + representatives.setdefault(slot.content_sha256, slot) + artifact_dir = self._runtime.workspace.path + raw_paths = {artifact_dir / slot.artifact_name for slot in slots} + prepared_items: list[_PreparedVisual] = [] + outcomes: dict[str, XlsxVisualOutcome] = {} + try: + if self._vision is None or self._vision_config is None: + return { + digest: XlsxVisualOutcome(None, "xlsx_vision_unavailable") + for digest in representatives + } + for request in _unique_requests(document, artifact_dir): + slot = representatives[request.digest] + try: + prepared = await self._runtime.run_native( + _prepare_xlsx_visual_to_wire, + _artifact_slot_to_wire(slot), + artifact_dir, + artifact_dir, + f"xlsx-prepared-{request.source_index}", + ) + paths = prepared_paths(prepared, artifact_dir) + if bool(prepared.get("skipped")) or not paths: + outcomes[request.digest] = XlsxVisualOutcome( + None, + "xlsx_vision_failed", + ) + for path in paths: + with suppress(OSError): + path.unlink(missing_ok=True) + continue + prepared_items.append(_PreparedVisual(request, prepared, paths)) + except asyncio.CancelledError: + raise + except BaseException: + outcomes[request.digest] = XlsxVisualOutcome(None, "xlsx_vision_failed") + + analyzed = await asyncio.gather( + *(self._analyze_prepared(item) for item in prepared_items), + return_exceptions=True, + ) + for item, outcome in zip(prepared_items, analyzed, strict=True): + if isinstance(outcome, asyncio.CancelledError): + raise outcome + if isinstance(outcome, BaseException): + outcomes[item.request.digest] = XlsxVisualOutcome( + None, + "xlsx_vision_failed", + ) + else: + outcomes[item.request.digest] = outcome + return outcomes + finally: + for item in prepared_items: + for path in item.paths: + with suppress(OSError): + path.unlink(missing_ok=True) + for path in raw_paths: + with suppress(OSError): + path.unlink(missing_ok=True) + _cleanup_worker_artifacts(artifact_dir) + + async def _analyze_prepared(self, item: _PreparedVisual) -> XlsxVisualOutcome: + vision = self._vision + config = self._vision_config + if vision is None or config is None: + return XlsxVisualOutcome(None, "xlsx_vision_unavailable") + + async def analyze_tile(request: VisionRequest) -> VisionResult | BaseException | object: + try: + return await vision.analyze(request) + except asyncio.CancelledError: + raise + except BaseException as error: + return error + + requests = tuple( + VisionRequest( + path, + tile_prompt(item.request.prompt, tile_index, len(item.paths)), + item.request.source_index * 10_000 + tile_index, + item.request.kind, + ) + for tile_index, path in enumerate(item.paths) + ) + try: + async with asyncio.timeout(float(config.timeout)): + results = await asyncio.gather(*(analyze_tile(request) for request in requests)) + except asyncio.CancelledError: + raise + except TimeoutError: + return XlsxVisualOutcome(None, "xlsx_vision_timeout") + if any(isinstance(result, TimeoutError) for result in results): + return XlsxVisualOutcome(None, "xlsx_vision_timeout") + if any(not isinstance(result, VisionResult) for result in results): + return XlsxVisualOutcome(None, "xlsx_vision_failed") + typed_results = cast(tuple[VisionResult, ...], results) + if any(not _has_visual_content(result) for result in typed_results): + return XlsxVisualOutcome(None, "xlsx_vision_failed") + try: + merged = merge_tiled_results(item.prepared, typed_results) + except asyncio.CancelledError: + raise + except BaseException: + return XlsxVisualOutcome(None, "xlsx_vision_failed") + if not _has_visual_content(merged): + return XlsxVisualOutcome(None, "xlsx_vision_failed") + return XlsxVisualOutcome(merged) diff --git a/tests/test_registry.py b/tests/test_registry.py index a60173d..e4d372e 100644 --- a/tests/test_registry.py +++ b/tests/test_registry.py @@ -1,11 +1,10 @@ from __future__ import annotations -from pathlib import Path from typing import Any, cast import pytest -from opendocs import CorruptDocumentError, ParseOptions, UnsupportedDocumentError +from opendocs import ParseOptions, UnsupportedDocumentError from opendocs._models import DocumentType, ParsedDocument, TextBlock from opendocs._runtime import ParserRuntime from opendocs.parsers.base import DocumentParser @@ -13,7 +12,6 @@ from opendocs.parsers.registry import ParserRegistry, build_default_registry from opendocs.parsers.xlsx import XlsxParser from opendocs.source import ParseWorkspace, ResolvedSource -from tests.xlsx_fixtures import write_xlsx class StubParser: @@ -170,22 +168,3 @@ def test_default_registry_without_runtime_keeps_binary_formats_unavailable() -> ): with pytest.raises(UnsupportedDocumentError, match=document_type.value): registry.get(document_type) - - -@pytest.mark.asyncio -async def test_xlsx_parser_seam_is_callable_and_prevalidates_the_package(tmp_path) -> None: - parser = XlsxParser() - valid = tmp_path / "valid.xlsx" - write_xlsx(valid) - - with pytest.raises(UnsupportedDocumentError, match="XLSX content parsing"): - await parser.parse(_resolved(valid), options=ParseOptions()) - - corrupt = tmp_path / "corrupt.xlsx" - corrupt.write_bytes(b"not-a-zip") - with pytest.raises(CorruptDocumentError): - await parser.parse(_resolved(corrupt), options=ParseOptions()) - - -def _resolved(path: Path) -> ResolvedSource: - return ResolvedSource(path=path, original_name="workbook.xlsx", owned=False) diff --git a/tests/test_xlsx_merge.py b/tests/test_xlsx_merge.py new file mode 100644 index 0000000..105e720 --- /dev/null +++ b/tests/test_xlsx_merge.py @@ -0,0 +1,212 @@ +from __future__ import annotations + +from opendocs._models import ( + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + TableBlock, + TextBlock, + WarningRecord, +) +from opendocs.parsers.xlsx.merge import XlsxVisualOutcome, merge_xlsx_document +from opendocs.parsers.xlsx.models import ( + XlsxChartSlot, + XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, + XlsxSheet, + XlsxSheetKind, + XlsxSheetState, +) +from opendocs.vision.base import VisionResult, VisionTextElement + + +def _prelude(sheet_index: int, name: str) -> XlsxNativeSlot: + return XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(f""), + HeadingBlock(1, (InlineText(name),)), + ParagraphBlock((InlineText("legacy state"),)), + ), + ) + + +def test_merge_orders_by_anchor_and_keeps_all_native_facts_before_same_anchor_vision() -> None: + chart_digest = "a" * 64 + image_digest = "b" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Data", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.HIDDEN, + ( + _prelude(1, "Data"), + XlsxNativeSlot(1, "A10", (TextBlock("late"),)), + XlsxChartSlot( + 3, + "B2", + "chart.png", + chart_digest, + (TextBlock("native chart facts"),), + ), + XlsxImageSlot( + 4, + "B2", + "image.png", + image_digest, + alt_text="Quarterly dashboard", + object_name="Picture 1", + ), + XlsxNativeSlot(2, "A2", (TextBlock("early"),)), + ), + ), + XlsxSheet( + 2, + "Empty", + XlsxSheetKind.CHARTSHEET, + XlsxSheetState.VERY_HIDDEN, + (_prelude(2, "Empty"),), + ), + ), + (WarningRecord("xlsx_unsupported_object", "Data!C4: unsupported control"),), + ) + outcomes = { + chart_digest: XlsxVisualOutcome( + VisionResult((VisionTextElement("chart interpretation", 0),)) + ), + image_digest: XlsxVisualOutcome( + VisionResult((VisionTextElement("image interpretation", 0),)) + ), + } + + merged = merge_xlsx_document(document, outcomes) + + assert merged.blocks[0:2] == ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Data (Hidden)"),)), + ) + assert merged.blocks[-2:] == ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Empty (Very Hidden)"),)), + ) + rendered_text = [block.text for block in merged.blocks if isinstance(block, TextBlock)] + assert rendered_text == ( + [ + "early", + "native chart facts", + "chart interpretation", + "image interpretation", + "late", + ] + ) + chart_native = merged.blocks.index(TextBlock("native chart facts")) + image_metadata = merged.blocks.index( + ParagraphBlock((InlineText("Image description: Quarterly dashboard"),)) + ) + chart_visual = merged.blocks.index(TextBlock("chart interpretation")) + image_visual = merged.blocks.index(TextBlock("image interpretation")) + assert chart_native < chart_visual + assert image_metadata < chart_visual + assert chart_visual < image_visual + assert merged.warnings == document.warnings + + +def test_merge_replays_digest_failure_for_every_occurrence_without_losing_metadata() -> None: + digest = "c" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "One", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + _prelude(1, "One"), + XlsxImageSlot(1, "C3", "same.png", digest, title="Logo"), + ), + ), + XlsxSheet( + 2, + "Two", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + _prelude(2, "Two"), + XlsxImageSlot(1, "D4", "same.png", digest, alt_text="Brand"), + ), + ), + ) + ) + + merged = merge_xlsx_document(document, {digest: XlsxVisualOutcome(None, "xlsx_vision_failed")}) + + assert ParagraphBlock((InlineText("Image title: Logo"),)) in merged.blocks + assert ParagraphBlock((InlineText("Image description: Brand"),)) in merged.blocks + assert [warning.code for warning in merged.warnings] == [ + "xlsx_vision_failed", + "xlsx_vision_failed", + ] + assert "One!C3" in merged.warnings[0].message + assert "Two!D4" in merged.warnings[1].message + + +def test_merge_keeps_valid_vision_tables_after_native_chart_data() -> None: + digest = "d" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Chart", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + _prelude(1, "Chart"), + XlsxChartSlot( + 1, + "A1", + "chart.png", + digest, + (TableBlock((("native", "1"),), 0),), + ), + ), + ), + ) + ) + outcome = XlsxVisualOutcome( + VisionResult((VisionTextElement("visual trend", 0),)), + ) + + merged = merge_xlsx_document(document, {digest: outcome}) + + assert merged.blocks.index(TableBlock((("native", "1"),), 0)) < merged.blocks.index( + TextBlock("visual trend") + ) + + +def test_merge_returns_headings_and_warnings_for_unsupported_only_workbook() -> None: + warning = WarningRecord("xlsx_unsupported_object", "Only!A1: unsupported object") + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Only", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + (_prelude(1, "Only"),), + ), + ), + (warning,), + ) + + merged = merge_xlsx_document(document, {}) + + assert merged.blocks == ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Only (Visible)"),)), + ) + assert merged.warnings == (warning,) diff --git a/tests/test_xlsx_parser.py b/tests/test_xlsx_parser.py index 399207a..3af78d9 100644 --- a/tests/test_xlsx_parser.py +++ b/tests/test_xlsx_parser.py @@ -1,31 +1,186 @@ from __future__ import annotations +import asyncio from pathlib import Path +from typing import Any, cast import pytest from PIL import Image -from opendocs._models import HeadingBlock, InlineText +import opendocs.parsers.xlsx.parser as parser_module +from opendocs._models import ( + DocumentType, + HeadingBlock, + InlineText, + MarkdownBlock, + ParagraphBlock, + TextBlock, +) +from opendocs._runtime import ParserRuntime +from opendocs.errors import ( + DocumentTimeoutError, + ModelAuthenticationError, + ModelInvalidRequestError, + ModelInvalidResponseError, + ModelPermissionError, + ModelUnavailableError, + RuntimeDependencyError, +) +from opendocs.options import ParseOptions, VisionConfig from opendocs.parsers.xlsx.media import build_xlsx_visual_requests from opendocs.parsers.xlsx.models import ( XlsxChartSlot, XlsxDocument, + XlsxImageSlot, + XlsxNativeSlot, XlsxSheet, XlsxSheetKind, XlsxSheetState, + document_to_wire, ) +from opendocs.parsers.xlsx.parser import XlsxParser, _extract_xlsx_to_wire +from opendocs.parsers.xlsx.preflight import XlsxPreflight +from opendocs.source import ParseWorkspace, ResolvedSource from opendocs.vision.base import VisionRequest, VisionResult, VisionTextElement +from tests.xlsx_fixtures import write_structured_xlsx class RecordingVision: - def __init__(self) -> None: + def __init__(self, result: object | BaseException | None = None) -> None: self.requests: list[VisionRequest] = [] + self.result = result async def analyze(self, request: VisionRequest) -> VisionResult: self.requests.append(request) + if isinstance(self.result, BaseException): + raise self.result + if self.result is not None: + return cast(VisionResult, self.result) return VisionResult((VisionTextElement("视觉解释: 收入总体上升", request.source_index),)) +class FatalVisionFailure(BaseException): + pass + + +def _document_with_images(*, duplicate: bool = False) -> XlsxDocument: + first = XlsxImageSlot(1, "B2", "same.png", "b" * 64, alt_text="First") + second = XlsxImageSlot(1, "C3", "same.png", "b" * 64, alt_text="Second") + sheets = [ + XlsxSheet( + 1, + "One", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("One"),)), + ), + ), + first, + ), + ) + ] + if duplicate: + sheets.append( + XlsxSheet( + 2, + "Two", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Two"),)), + ), + ), + second, + ), + ) + ) + return XlsxDocument(tuple(sheets)) + + +def _runtime_with_document( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + document: XlsxDocument | object, +) -> tuple[ParserRuntime, list[str]]: + workspace = tmp_path / "workspace" + workspace.mkdir() + artifact_names = ( + { + slot.artifact_name + for sheet in document.sheets + for slot in sheet.slots + if isinstance(slot, XlsxImageSlot | XlsxChartSlot) + } + if isinstance(document, XlsxDocument) + else set() + ) + for artifact_name in artifact_names or {"same.png"}: + (workspace / artifact_name).write_bytes(b"raw") + runtime = ParserRuntime(ParseWorkspace(workspace)) + calls: list[str] = [] + + async def run_native(function: Any, *args: object, **kwargs: object) -> object: + del kwargs + calls.append(function.__name__) + if function.__name__ == "_extract_xlsx_to_wire": + return document_to_wire(document) if isinstance(document, XlsxDocument) else document + if function.__name__ == "_prepare_xlsx_visual_to_wire": + output_directory = args[-2] + output_stem = args[-1] + assert isinstance(output_directory, Path) + assert isinstance(output_stem, str) + name = f"{output_stem}-0.png" + (output_directory / name).write_bytes(b"sanitized") + return { + "skipped": False, + "reason": None, + "width": 20, + "height": 10, + "parts": [ + { + "name": name, + "top": 0.0, + "bottom": 1.0, + "core_top": 0.0, + "core_bottom": 1.0, + "width": 20, + "height": 10, + } + ], + "facts": { + "alpha_coverage": 1.0, + "components": 1, + "edge_density": 0.5, + "color_count": 8, + "nearly_blank": False, + }, + } + raise AssertionError(f"unexpected native function: {function.__name__}") + + monkeypatch.setattr(runtime, "run_native", run_native) + return runtime, calls + + +async def _parse(parser: XlsxParser, tmp_path: Path, options: ParseOptions | None = None): + source = tmp_path / "source.xlsx" + source.write_bytes(b"source") + return await parser.parse( + ResolvedSource(source, source.name, False), + options=options or ParseOptions(), + ) + + @pytest.mark.asyncio async def test_xlsx_visual_request_seam_uses_fake_client_and_bounded_chart_prompt( tmp_path: Path, @@ -65,3 +220,249 @@ async def test_xlsx_visual_request_seam_uses_fake_client_and_bounded_chart_promp assert vision.requests[0].image_path == tmp_path / artifact_name assert "视觉解释" in vision.requests[0].prompt assert "Excel 外观还原" in vision.requests[0].prompt + + +def test_native_worker_preflights_before_extract_and_strictly_serializes( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + source = tmp_path / "source.xlsx" + write_structured_xlsx( + source, + sheets=(("Data", "worksheet", "visible", "A1", ("A1",)),), + ) + events: list[str] = [] + real_preflight = parser_module.preflight_xlsx + real_extract = parser_module.extract_xlsx + + def record_preflight(path: Path): + events.append("preflight") + return real_preflight(path) + + def record_extract(path: Path, preflight: XlsxPreflight, *, artifact_dir: Path): + events.append("extract") + return real_extract(path, preflight, artifact_dir=artifact_dir) + + monkeypatch.setattr(parser_module, "preflight_xlsx", record_preflight) + monkeypatch.setattr(parser_module, "extract_xlsx", record_extract) + + wire = _extract_xlsx_to_wire(source, tmp_path / "artifacts") + + assert wire["type"] == "xlsx_document" + assert events == ["preflight", "extract"] + + +@pytest.mark.asyncio +async def test_parser_deduplicates_visual_work_and_replays_success_per_occurrence( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + document = _document_with_images(duplicate=True) + runtime, calls = _runtime_with_document(monkeypatch, tmp_path, document) + vision = RecordingVision(VisionResult((VisionTextElement("visual", 0),))) + try: + result = await _parse( + XlsxParser(runtime, vision, VisionConfig("model")), + tmp_path, + ) + finally: + await runtime.aclose() + + assert [block for block in result.blocks if block == TextBlock("visual")] == [ + TextBlock("visual"), + TextBlock("visual"), + ] + assert len(vision.requests) == 1 + assert calls == ["_extract_xlsx_to_wire", "_prepare_xlsx_visual_to_wire"] + assert not (tmp_path / "workspace" / "same.png").exists() + assert not tuple((tmp_path / "workspace").glob("xlsx-prepared-*.png")) + + +@pytest.mark.asyncio +async def test_parser_without_vision_returns_native_and_occurrence_warning( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + runtime, calls = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + try: + result = await _parse(XlsxParser(runtime, None, None), tmp_path) + finally: + await runtime.aclose() + + assert result.document_type is DocumentType.XLSX + assert [warning.code for warning in result.warnings] == ["xlsx_vision_unavailable"] + assert "One!B2" in result.warnings[0].message + assert calls == ["_extract_xlsx_to_wire"] + assert not (tmp_path / "workspace" / "same.png").exists() + + +@pytest.mark.asyncio +async def test_parser_partial_failure_is_replayed_in_anchor_order_not_completion_order( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + slow_digest = "d" * 64 + failed_digest = "e" * 64 + document = XlsxDocument( + ( + XlsxSheet( + 1, + "Mixed", + XlsxSheetKind.WORKSHEET, + XlsxSheetState.VISIBLE, + ( + XlsxNativeSlot( + 0, + "A1", + ( + MarkdownBlock(""), + HeadingBlock(1, (InlineText("Mixed"),)), + ), + ), + XlsxImageSlot(1, "A5", "slow.png", slow_digest, alt_text="Slow"), + XlsxImageSlot(2, "A2", "failed.png", failed_digest, alt_text="Failed"), + ), + ), + ) + ) + + class ReverseCompletionVision: + async def analyze(self, request: VisionRequest) -> VisionResult: + if request.source_index == 0: + await asyncio.sleep(0.02) + return VisionResult((VisionTextElement("slow success", 0),)) + raise ModelUnavailableError("fast failure") + + runtime, calls = _runtime_with_document(monkeypatch, tmp_path, document) + try: + result = await _parse( + XlsxParser(runtime, ReverseCompletionVision(), VisionConfig("model")), + tmp_path, + ) + finally: + await runtime.aclose() + + failed_metadata = ParagraphBlock((InlineText("Image description: Failed"),)) + slow_metadata = ParagraphBlock((InlineText("Image description: Slow"),)) + assert result.blocks.index(failed_metadata) < result.blocks.index(slow_metadata) + assert TextBlock("slow success") in result.blocks + assert [warning.code for warning in result.warnings] == ["xlsx_vision_failed"] + assert "Mixed!A2" in result.warnings[0].message + assert calls == [ + "_extract_xlsx_to_wire", + "_prepare_xlsx_visual_to_wire", + "_prepare_xlsx_visual_to_wire", + ] + assert not tuple((tmp_path / "workspace").glob("*.png")) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "failure", + [ + ModelAuthenticationError("auth"), + ModelPermissionError("permission"), + ModelInvalidRequestError("invalid"), + ModelUnavailableError("provider"), + ModelInvalidResponseError("invalid response"), + RuntimeDependencyError("runtime"), + ValueError("plain provider error"), + FatalVisionFailure("base exception"), + object(), + VisionResult(()), + ], +) +async def test_parser_fails_open_for_every_non_timeout_visual_failure( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + failure: object, +) -> None: + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + vision = RecordingVision(failure) + try: + result = await _parse( + XlsxParser(runtime, vision, VisionConfig("model")), + tmp_path, + ) + finally: + await runtime.aclose() + + assert [warning.code for warning in result.warnings] == ["xlsx_vision_failed"] + assert HeadingBlock(1, (InlineText("One (Visible)"),)) in result.blocks + + +@pytest.mark.asyncio +async def test_parser_classifies_per_object_timeout_but_document_deadline_is_fatal( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + class SlowVision: + async def analyze(self, request: VisionRequest) -> VisionResult: + del request + await asyncio.sleep(1) + raise AssertionError("unreachable") + + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + try: + result = await _parse( + XlsxParser(runtime, SlowVision(), VisionConfig("model", timeout=0.01)), + tmp_path, + ParseOptions(timeout=1), + ) + assert [warning.code for warning in result.warnings] == ["xlsx_vision_timeout"] + finally: + await runtime.aclose() + + second_path = tmp_path / "deadline" + second_path.mkdir() + runtime, _ = _runtime_with_document(monkeypatch, second_path, _document_with_images()) + + async def slow_native(function: Any, *args: object, **kwargs: object) -> object: + del function, args, kwargs + await asyncio.sleep(1) + raise AssertionError("unreachable") + + monkeypatch.setattr(runtime, "run_native", slow_native) + try: + with pytest.raises(DocumentTimeoutError): + await _parse( + XlsxParser(runtime, None, None), + second_path, + ParseOptions(timeout=0.01), + ) + finally: + await runtime.aclose() + + +@pytest.mark.asyncio +async def test_parser_propagates_caller_cancellation( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, _document_with_images()) + + async def cancelled(function: Any, *args: object, **kwargs: object) -> object: + del function, args, kwargs + raise asyncio.CancelledError + + monkeypatch.setattr(runtime, "run_native", cancelled) + try: + with pytest.raises(asyncio.CancelledError): + await _parse(XlsxParser(runtime, None, None), tmp_path) + finally: + await runtime.aclose() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("wire", [{"type": "wrong"}, object()]) +async def test_parser_maps_invalid_native_wire_to_runtime_dependency( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + wire: object, +) -> None: + runtime, _ = _runtime_with_document(monkeypatch, tmp_path, wire) + try: + with pytest.raises(RuntimeDependencyError, match="invalid data"): + await _parse(XlsxParser(runtime, None, None), tmp_path) + finally: + await runtime.aclose() diff --git a/tests/test_xlsx_preflight.py b/tests/test_xlsx_preflight.py index d96208b..660a0ce 100644 --- a/tests/test_xlsx_preflight.py +++ b/tests/test_xlsx_preflight.py @@ -8,12 +8,14 @@ from openpyxl.chart import BarChart, Reference import opendocs.parsers.xlsx.preflight as preflight_module -from opendocs.errors import CorruptDocumentError, LimitExceededError, UnsupportedDocumentError +from opendocs._models import HeadingBlock, InlineText +from opendocs._runtime import ParserRuntime +from opendocs.errors import CorruptDocumentError, LimitExceededError from opendocs.options import ParseOptions from opendocs.parsers.xlsx import XlsxParser from opendocs.parsers.xlsx.models import XlsxSheetKind, XlsxSheetState from opendocs.parsers.xlsx.preflight import MAX_SHEETS, preflight_xlsx -from opendocs.source import ResolvedSource +from opendocs.source import ParseWorkspace, ResolvedSource from tests.xlsx_fixtures import rewrite_xlsx, write_structured_xlsx SHEET_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" @@ -129,16 +131,13 @@ def test_sparse_full_grid_dimension_fails_before_any_loader(tmp_path: Path) -> N async def test_parser_seam_preflights_limits_and_does_not_map_max_pages_to_sheets( tmp_path: Path, ) -> None: - parser = XlsxParser() accepted = tmp_path / "two-sheets.xlsx" rejected = tmp_path / "too-many-sheets.xlsx" - write_structured_xlsx( - accepted, - sheets=( - ("One", "worksheet", "visible", None, ()), - ("Two", "worksheet", "visible", None, ()), - ), - ) + workbook = Workbook() + workbook.active.title = "One" + workbook.create_sheet("Two") + workbook.save(accepted) + workbook.close() write_structured_xlsx( rejected, sheets=tuple( @@ -146,16 +145,26 @@ async def test_parser_seam_preflights_limits_and_does_not_map_max_pages_to_sheet ), ) - with pytest.raises(UnsupportedDocumentError, match="content parsing"): - await parser.parse( + workspace = tmp_path / "workspace" + workspace.mkdir() + runtime = ParserRuntime(ParseWorkspace(workspace)) + parser = XlsxParser(runtime, None, None) + try: + result = await parser.parse( ResolvedSource(accepted, "two-sheets.xlsx", False), options=ParseOptions(max_pages=1), ) - with pytest.raises(LimitExceededError, match="sheet count"): - await parser.parse( - ResolvedSource(rejected, "too-many-sheets.xlsx", False), - options=ParseOptions(max_pages=1), - ) + assert [block for block in result.blocks if isinstance(block, HeadingBlock)] == [ + HeadingBlock(1, (InlineText("One (Visible)"),)), + HeadingBlock(1, (InlineText("Two (Visible)"),)), + ] + with pytest.raises(LimitExceededError, match="sheet count"): + await parser.parse( + ResolvedSource(rejected, "too-many-sheets.xlsx", False), + options=ParseOptions(max_pages=1), + ) + finally: + await runtime.aclose() @pytest.mark.parametrize( From ccdc294133799c368406130d33c7c3c7dfd1f440 Mon Sep 17 00:00:00 2001 From: caichuanwang Date: Fri, 14 Aug 2026 16:51:46 +0800 Subject: [PATCH 07/12] =?UTF-8?q?=E8=AF=81=E6=98=8E=20XLSX=20=E5=85=AC?= =?UTF-8?q?=E5=85=B1=E5=85=A5=E5=8F=A3=E4=B8=8E=E8=B5=84=E6=BA=90=E8=BE=B9?= =?UTF-8?q?=E7=95=8C=E4=B8=80=E8=87=B4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Constraint: path、bytes、命名与匿名流在同步和异步 API 下必须走同一 Markdown 与错误契约。 Rejected: 不依赖临时文件扩展名驱动 openpyxl,也不把本机 RSS 测量写成硬保证。 Confidence: high Scope-risk: 发布文档与构建产物依赖白名单由 U8 收口。 Tested: 136 XLSX tests; 12 lifecycle/API tests; full suite 812 passed, 9 skipped, 3 known U8 failures; ruff; format; ty; diff-check. Not-tested: real XLSX workbooks, live vision providers, or cross-platform RSS. --- src/opendocs/parsers/xlsx/extract.py | 15 +- tests/native_worker_helpers.py | 4 + tests/test_api_xlsx.py | 440 +++++++++++++++++++++++++++ tests/test_xlsx_extract.py | 27 +- tests/test_xlsx_preflight.py | 127 ++++++++ tests/xlsx_fixtures.py | 35 +++ 6 files changed, 627 insertions(+), 21 deletions(-) create mode 100644 tests/test_api_xlsx.py diff --git a/src/opendocs/parsers/xlsx/extract.py b/src/opendocs/parsers/xlsx/extract.py index f371c79..13f362f 100644 --- a/src/opendocs/parsers/xlsx/extract.py +++ b/src/opendocs/parsers/xlsx/extract.py @@ -794,13 +794,14 @@ def extract_xlsx( f"object={warning.object_ordinal}: {warning.detail}" ), ) - workbook = openpyxl.load_workbook( - path, - read_only=False, - data_only=False, - rich_text=False, - keep_links=False, - ) + with path.open("rb") as package: + workbook = openpyxl.load_workbook( + package, + read_only=False, + data_only=False, + rich_text=False, + keep_links=False, + ) try: worksheets = {worksheet.title: worksheet for worksheet in workbook.worksheets} sheet_values: dict[int, tuple[dict[_Coordinate, str], set[_Coordinate]]] = {} diff --git a/tests/native_worker_helpers.py b/tests/native_worker_helpers.py index 0d4e4e0..a97890a 100644 --- a/tests/native_worker_helpers.py +++ b/tests/native_worker_helpers.py @@ -24,6 +24,10 @@ def sleep_and_echo(delay: float, value: object) -> object: return value +def hard_exit(status: int) -> None: + os._exit(status) + + def raise_corrupt(message: str) -> None: raise CorruptDocumentError(message) diff --git a/tests/test_api_xlsx.py b/tests/test_api_xlsx.py new file mode 100644 index 0000000..ba95e01 --- /dev/null +++ b/tests/test_api_xlsx.py @@ -0,0 +1,440 @@ +from __future__ import annotations + +import asyncio +import io +import warnings +from collections.abc import Callable +from pathlib import Path +from typing import Any, cast +from zipfile import ZipFile + +import pytest +from openpyxl import Workbook +from openpyxl.drawing.image import Image as SpreadsheetImage +from PIL import Image + +import opendocs.api as api_module +import opendocs.source as source_module +from opendocs import ( + DocumentTimeoutError, + LimitExceededError, + OpenDocsWarning, + ParseOptions, + RuntimeDependencyError, + VisionConfig, + aparse, + parse, +) +from opendocs._runtime import ParserRuntime +from opendocs.source import Source +from tests.native_worker_helpers import hard_exit, sleep_and_echo +from tests.xlsx_fixtures import rewrite_xlsx, write_public_contract_xlsx + + +class NamedBytesIO(io.BytesIO): + def __init__(self, data: bytes, name: str) -> None: + super().__init__(data) + self.name = name + + +def _source_factory(kind: str, path: Path, content: bytes) -> Callable[[], Source]: + if kind == "path": + return lambda: path + if kind == "bytes": + return lambda: content + if kind == "named_stream": + return lambda: NamedBytesIO(content, str(path)) + if kind == "unnamed_stream": + return lambda: io.BytesIO(content) + raise AssertionError(f"unknown source kind: {kind}") + + +def _warning_codes(captured: list[warnings.WarningMessage]) -> tuple[str, ...]: + return tuple( + item.message.code for item in captured if isinstance(item.message, OpenDocsWarning) + ) + + +def _invoke( + api_kind: str, + source: Source, + *, + options: ParseOptions | None = None, + vision: VisionConfig | None = None, +) -> tuple[str, tuple[str, ...], tuple[str, ...], tuple[str, ...]]: + with warnings.catch_warnings(record=True) as captured: + warnings.simplefilter("always", OpenDocsWarning) + result = ( + parse(source, options=options, vision=vision) + if api_kind == "parse" + else asyncio.run(aparse(source, options=options, vision=vision)) + ) + public_warnings = tuple(item for item in captured if isinstance(item.message, OpenDocsWarning)) + return ( + result, + _warning_codes(captured), + tuple(str(item.message) for item in public_warnings), + tuple(item.filename for item in public_warnings), + ) + + +def test_xlsx_public_api_has_eight_equivalent_input_and_api_combinations( + tmp_path: Path, +) -> None: + path = tmp_path / "contract.xlsx" + write_public_contract_xlsx(path) + content = path.read_bytes() + outcomes: list[tuple[str, tuple[str, ...], tuple[str, ...], tuple[str, ...]]] = [] + streams: list[Source] = [] + for source_kind in ("path", "bytes", "named_stream", "unnamed_stream"): + source_factory = _source_factory(source_kind, path, content) + for api_kind in ("parse", "aparse"): + source = source_factory() + streams.append(source) + outcomes.append(_invoke(api_kind, source)) + + assert len(outcomes) == 8 + assert [outcome[:3] for outcome in outcomes] == [outcomes[0][:3]] * 8 + result, warning_codes, _messages, filenames = outcomes[0] + assert warning_codes == ( + "xlsx_formula_cache_missing", + "xlsx_unsupported_number_format", + ) + assert filenames == (__file__, __file__) + assert "# Ledger (Visible)" in result + assert "$1,234.50" in result + assert "2026-08-14" in result + assert "=B2*2" in result + assert "# Hidden (Hidden)" in result + assert "# Very Hidden (Very Hidden)" in result + assert "# Empty (Visible)" in result + assert all(not source.closed for source in streams if hasattr(source, "closed")) + + +@pytest.mark.asyncio +async def test_xlsx_async_warning_emission_points_to_the_public_api_caller( + tmp_path: Path, +) -> None: + path = tmp_path / "warning-location.xlsx" + write_public_contract_xlsx(path) + + with warnings.catch_warnings(record=True) as captured: + warnings.simplefilter("always", OpenDocsWarning) + await aparse(path) + + public_warnings = tuple(item for item in captured if isinstance(item.message, OpenDocsWarning)) + assert tuple(item.filename for item in public_warnings) == (__file__, __file__) + + +def test_xlsx_public_output_is_repeatable_and_max_pages_does_not_limit_sheets( + tmp_path: Path, +) -> None: + path = tmp_path / "repeatable.xlsx" + write_public_contract_xlsx(path) + + outcomes = [_invoke("parse", path) for _ in range(3)] + assert outcomes == [outcomes[0]] * 3 + + limited = _invoke("parse", path, options=ParseOptions(max_pages=1)) + assert limited == outcomes[0] + assert limited[0].count("